Release v2.26.0 #284

Merged
TapTap merged 210 commits from dev into main 2026-09-18 19:05:52 +02:00
16 changed files with 1413 additions and 533 deletions
Showing only changes of commit 51e41dee2a - Show all commits
+34 -10
View File
@@ -329,10 +329,10 @@ static int config_add_remote_option(Config* config, const char* value, const cha
} }
/* Validate and append one --compare-dest/--copy-dest/--link-dest directory. /* Validate and append one --compare-dest/--copy-dest/--link-dest directory.
* The path is interpreted on the receiver relative to the destination root, * A relative path is interpreted on the receiver below the destination root; an
* so it must be a non-empty relative path with no "." / ".." components (an * absolute path is used verbatim on the receiver (matching rsync), still subject
* absolute or escaping path is rejected up front instead of failing on the * to the receiver's authorized-root confinement. Either way the path must be
* server). Returns 0 on success, -1 on error. */ * non-empty and traversal-free (no ".."). Returns 0 on success, -1 on error. */
static int set_basis_dest_option(Config* config, BasisDestType type, const char* value, static int set_basis_dest_option(Config* config, BasisDestType type, const char* value,
const char* option_name) { const char* option_name) {
if (!value || !value[0]) { if (!value || !value[0]) {
@@ -341,8 +341,9 @@ static int set_basis_dest_option(Config* config, BasisDestType type, const char*
} }
if (config_basis_append(config, type, value) != 0) { if (config_basis_append(config, type, value) != 0) {
log_message(LOG_LEVEL_ERROR, log_message(LOG_LEVEL_ERROR,
"%s requires a non-empty relative directory name with no '.', '..', or absolute " "%s requires a non-empty directory name with no '..' component "
"path (resolved below the destination root)", "(relative paths resolve below the destination root; absolute paths are used "
"verbatim)",
option_name); option_name);
return -1; return -1;
} }
@@ -657,16 +658,25 @@ static int config_add_pattern(char*** patterns, int* count, const char* value,
/* Validate and append one --filter=RULE string. Returns 0 on success, -1 on error. */ /* Validate and append one --filter=RULE string. Returns 0 on success, -1 on error. */
static int config_add_filter(Config* config, const char* rule) { static int config_add_filter(Config* config, const char* rule) {
char err[160]; char err[256];
FilterRule* parsed = filter_rule_parse(rule, err, sizeof(err)); /* Validate through the full list parser so clear/merge/dir-merge and the rule
if (!parsed) { modifiers are accepted (and a merge file is readable) at parse time. */
FilterParseOptions opts = {.delete_excluded = config->delete_excluded,
.cvs_exclude = config->cvs_exclude};
FilterRuleList* probe = filter_rule_list_create();
if (!probe) {
log_message(LOG_LEVEL_ERROR, "memory allocation failed for --filter");
return -1;
}
bool ok = filter_rule_list_parse_append(probe, rule, &opts, NULL, err, sizeof(err));
filter_rule_list_free(probe);
if (!ok) {
char* escaped = output_escape(rule, log_get_8_bit_output()); char* escaped = output_escape(rule, log_get_8_bit_output());
log_message(LOG_LEVEL_ERROR, "invalid --filter rule '%s': %s", log_message(LOG_LEVEL_ERROR, "invalid --filter rule '%s': %s",
escaped ? escaped : "<allocation failed>", err); escaped ? escaped : "<allocation failed>", err);
free(escaped); free(escaped);
return -1; return -1;
} }
filter_rule_free(parsed);
if (!config->filters) { if (!config->filters) {
config->filters = array_list_create(free); config->filters = array_list_create(free);
if (!config->filters) { if (!config->filters) {
@@ -1365,6 +1375,11 @@ static bool cli_handle_table_option(CliParseCtx* ctx) {
} }
if (entry->offset == offsetof(Config, eight_bit_output)) if (entry->offset == offsetof(Config, eight_bit_output))
protocol_set_8_bit_output(true); protocol_set_8_bit_output(true);
/* -F is repeatable: rsync's single -F transfers .rsync-filter files, a
repeated -FF excludes them. Count the occurrences so the scanner can
distinguish the two. */
if (entry->offset == offsetof(Config, per_dir_filter) && config->per_dir_filter_count < INT_MAX)
config->per_dir_filter_count++;
/* A delete-timing flag selects when --delete removes extras, so it /* A delete-timing flag selects when --delete removes extras, so it
implies --delete exactly like the rsync options do. */ implies --delete exactly like the rsync options do. */
if (entry->offset == offsetof(Config, delete_before) || if (entry->offset == offsetof(Config, delete_before) ||
@@ -2397,6 +2412,15 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool
config->preserve_times = true; config->preserve_times = true;
} }
/* --ignore-existing is a receiver-side existence policy: the receiver must
* answer "skip" BEFORE the sender transmits any payload, which only the
* per-file STATUS_CHECK handshake provides. Imply --incremental here (after
* the auto-preserve capture above, so a bare --ignore-existing does not gain
* -p/-t, which rsync likewise does not imply) so an existing destination is
* skipped on the wire instead of being streamed and discarded. */
if (config->ignore_existing)
config->use_incremental = true;
/* Derive the transport bit from the FINAL parsed flags. Every /* Derive the transport bit from the FINAL parsed flags. Every
* preservation/ownership option that needs the metadata frame (per-attribute * preservation/ownership option that needs the metadata frame (per-attribute
* perms/times/owner/group, atimes/crtimes, executability, xattrs/acls, * perms/times/owner/group, atimes/crtimes, executability, xattrs/acls,
+4 -1
View File
@@ -161,7 +161,8 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann
} }
if (rule_count > 0 || config->cvs_exclude) { if (rule_count > 0 || config->cvs_exclude) {
char err[160]; char err[160];
out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, err, sizeof(err)); out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude,
config->delete_excluded, err, sizeof(err));
free(texts); free(texts);
if (!out->base_filters) { if (!out->base_filters) {
log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err);
@@ -206,6 +207,8 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann
options->file_list = (const FileListSet*)config->files_from_set; options->file_list = (const FileListSet*)config->files_from_set;
options->base_filters = out->base_filters; options->base_filters = out->base_filters;
options->per_dir_filters = config->per_dir_filter; options->per_dir_filters = config->per_dir_filter;
options->delete_excluded = config->delete_excluded;
options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2;
options->dirs = config->dirs; options->dirs = config->dirs;
options->relative = config->relative; options->relative = config->relative;
/* -R/--relative outside --files-from reconstructs every destination path from /* -R/--relative outside --files-from reconstructs every destination path from
+150 -47
View File
@@ -51,29 +51,63 @@ static FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) {
return node; return node;
} }
/* Evaluate a rule chain for an entry inside the directory whose content /* Evaluate a rule chain for one entry. rsync precedence, highest first: the
* context is `node`. rsync precedence, highest first: the innermost (current) * innermost (current) directory's .rsync-filter rules, then each ancestor's,
* directory's .rsync-filter rules, then each ancestor's, then the root's, and * then the root's, and finally the command-line base rules (--filter/-C). The
* finally the command-line base rules (--filter/-C). A deeper per-directory * sender-side verdict decides whether the entry is hidden from the transfer;
* file therefore overrides a shallower one, and per-directory files override * the receiver-side verdict decides whether its destination mirror is protected
* the base rules by default. Returns FILTER_ACTION_NONE when nothing matched. */ * from --delete. Each side takes the FIRST matching rule independently. */
static FilterAction chain_rules_apply(const FilterRuleList* base, const FilterNode* node, typedef struct {
const char* rel, const char* leaf, bool is_dir) { bool hide; /* sender-side exclude matched */
if (node) { bool protect; /* receiver-side exclude matched */
FilterAction own_action = filter_rules_apply(node->own, rel, leaf, is_dir); } FilterOutcome;
if (own_action != FILTER_ACTION_NONE)
return own_action; static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel,
return chain_rules_apply(base, node->parent, rel, leaf, is_dir); const char* leaf, bool is_dir, FilterOutcome* out) {
memset(out, 0, sizeof(*out));
bool sender_decided = false;
bool receiver_decided = false;
const FilterNode* n = node;
while (!sender_decided || !receiver_decided) {
const FilterRuleList* list = n ? n->own : base;
if (list) {
if (!sender_decided) {
FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER);
if (action != FILTER_ACTION_NONE) {
out->hide = action == FILTER_ACTION_EXCLUDE;
sender_decided = true;
}
}
if (!receiver_decided) {
FilterAction action =
filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER);
if (action != FILTER_ACTION_NONE) {
out->protect = action == FILTER_ACTION_PROTECT;
receiver_decided = true;
}
}
}
if (!n)
break;
n = n->parent;
} }
return base ? filter_rules_apply(base, rel, leaf, is_dir) : FILTER_ACTION_NONE;
} }
static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel, static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel,
const char* leaf, bool is_dir, bool per_dir_filters) { const char* leaf, bool is_dir, bool exclude_filter_files,
/* -F: per-directory .rsync-filter files are never transferred. */ bool* protect_out) {
if (per_dir_filters && !is_dir && strcmp(leaf, ".rsync-filter") == 0) /* -FF: per-directory .rsync-filter files are never transferred (single -F
transfers them, matching rsync). */
if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) {
if (protect_out)
*protect_out = false;
return false; return false;
return chain_rules_apply(base, node, rel, leaf, is_dir) != FILTER_ACTION_EXCLUDE; }
FilterOutcome outcome;
chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome);
if (protect_out)
*protect_out = outcome.protect;
return !outcome.hide;
} }
static void dir_entry_destroy(void* item) { static void dir_entry_destroy(void* item) {
@@ -274,14 +308,19 @@ static char* scanner_prefix_send_path(const char* prefix, const char* rel) {
return path_cat(prefix, rel); return path_cat(prefix, rel);
} }
/* Apply the --files-from allow-set and the filter layer to one entry. */ /* Apply the --files-from allow-set and the filter layer to one entry. On
* return `*protect_out` is true when a receiver-side rule protects the entry's
* destination mirror from deletion. */
static bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, static bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base,
const FilterNode* node, const char* rel, const char* leaf, const FilterNode* node, const char* rel, const char* leaf,
bool is_dir, bool per_dir_filters) { bool is_dir, bool per_dir_filters, bool exclude_filter_files,
bool* protect_out) {
if (protect_out)
*protect_out = false;
if (file_list && !file_list_affects(file_list, rel)) if (file_list && !file_list_affects(file_list, rel))
return false; return false;
if (base || per_dir_filters) if (base || per_dir_filters)
return entry_allowed(base, node, rel, leaf, is_dir, per_dir_filters); return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out);
return true; return true;
} }
@@ -435,28 +474,77 @@ static bool scanner_record_synced_dir(const ScannerOptions* options, const char*
return ok; return ok;
} }
/* Merge the open directory's own .rsync-filter rules into the inherited /* Read every per-directory filter file that applies to `dir_path` (its
* context, returning the context used for this directory's entries. On a parse * .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a
* error the scanner is marked failed. Returns 0 on success, -1 on failure. */ * fresh list. Returns NULL on allocation/parse failure (message in `err`);
* returns an empty list (and *any_exists=false) when no file exists. */
static FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path,
const char* rel, bool* any_exists, char* err,
size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
const FilterRuleList* base = options->base_filters;
bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0);
if (any_exists)
*any_exists = false;
if (!have_names)
return NULL;
FilterRuleList* own = filter_rule_list_create();
if (!own) {
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false};
bool exists = false;
if (options->per_dir_filters) {
if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size))
goto fail;
if (exists && any_exists)
*any_exists = true;
}
if (base) {
for (int i = 0; i < base->dir_merge_count; i++) {
if (!filter_file_append(own, dir_path, base->dir_merge_names[i], rel, &opts, &exists, err,
err_size))
goto fail;
if (exists && any_exists)
*any_exists = true;
}
}
return own;
fail:
filter_rule_list_free(own);
return NULL;
}
/* Merge the open directory's own per-directory filter files (the default
* .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the
* base rule list) into the inherited context, returning the context used for
* this directory's entries. On a parse error the scanner is marked failed.
* Returns 0 on success, -1 on failure. */
static int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { static int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) {
if (!scanner->options.per_dir_filters) { char err[256];
bool any_exists = false;
FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path,
scanner->current_rel ? scanner->current_rel : "",
&any_exists, err, sizeof(err));
if (!own && any_exists) {
scanner->current_node = (FilterNode*)inherited; scanner->current_node = (FilterNode*)inherited;
return 0; return 0;
} }
char err[256];
bool exists = false;
FilterRuleList* own =
filter_file_read(scanner->current_path, scanner->current_rel ? scanner->current_rel : "",
&exists, err, sizeof(err));
if (!own) { if (!own) {
if (err[0] == '\0') {
scanner->current_node = (FilterNode*)inherited;
return 0;
}
char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output()); char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output());
log_message(LOG_LEVEL_ERROR, "invalid .rsync-filter in %s: %s", log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s",
escaped_path ? escaped_path : "<allocation failed>", err); escaped_path ? escaped_path : "<allocation failed>", err);
free(escaped_path); free(escaped_path);
scanner->failed = true; scanner->failed = true;
return -1; return -1;
} }
if (exists && own->count > 0) { if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) {
FilterNode* node = filter_node_alloc((FilterNode*)inherited, own); FilterNode* node = filter_node_alloc((FilterNode*)inherited, own);
if (!node || !array_list_add(scanner->filter_nodes, node)) { if (!node || !array_list_add(scanner->filter_nodes, node)) {
filter_node_destroy(node); filter_node_destroy(node);
@@ -1108,7 +1196,8 @@ static File* dirs_next_child(DirectoryScanner* scanner) {
return NULL; return NULL;
if (file && !entry_passes_selection(scanner->options.file_list, scanner->options.base_filters, if (file && !entry_passes_selection(scanner->options.file_list, scanner->options.base_filters,
NULL, entry->d_name, entry->d_name, file->is_dir, NULL, entry->d_name, entry->d_name, file->is_dir,
scanner->options.per_dir_filters)) { scanner->options.per_dir_filters,
scanner->options.exclude_per_dir_filter_files, NULL)) {
file_destroy(file); file_destroy(file);
continue; continue;
} }
@@ -1324,10 +1413,15 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) {
scanner->failed = true; scanner->failed = true;
break; break;
} }
bool protect = false;
bool passes_selection = entry_passes_selection( bool passes_selection = entry_passes_selection(
scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel,
entry->d_name, is_dir, scanner->options.per_dir_filters); entry->d_name, is_dir, scanner->options.per_dir_filters,
if (!passes_selection) { scanner->options.exclude_per_dir_filter_files, &protect);
/* A sender-side hide leaves the entry out of the transfer; an independent
receiver-side protect rule keeps a transferred entry's destination mirror
from being deleted. Both are recorded in the same protection set. */
if (!passes_selection || protect) {
/* --files-from subset pruning is not a filter exclusion: its delete /* --files-from subset pruning is not a filter exclusion: its delete
semantics stay keep-set-only (an unlisted source path is treated as semantics stay keep-set-only (an unlisted source path is treated as
absent, so its destination mirror is a deletable extra). A rule-based absent, so its destination mirror is a deletable extra). A rule-based
@@ -1751,15 +1845,17 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo
ps->failed = true; ps->failed = true;
return; return;
} }
bool protect = false;
bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel, bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel,
entry->d_name, is_dir, options->per_dir_filters); entry->d_name, is_dir, options->per_dir_filters,
options->exclude_per_dir_filter_files, &protect);
/* -R + --files-from: root-level files keep their bare relative send path. */ /* -R + --files-from: root-level files keep their bare relative send path. */
bool use_rel = options->relative && options->file_list != NULL; bool use_rel = options->relative && options->file_list != NULL;
if (!passes) { if (!passes || protect) {
/* --files-from subset pruning is not a filter exclusion; -R bare-wire-path /* --files-from subset pruning is not a filter exclusion; -R bare-wire-path
exclusions are never recorded (see ScannerOptions.excluded_paths). */ exclusions are never recorded (see ScannerOptions.excluded_paths). */
bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel); bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel);
if (!files_from_prune && !use_rel && options->excluded_paths) { if ((!files_from_prune && !use_rel) || protect) {
const char* rel_path = *cur_path == '/' ? cur_path + 1 : cur_path; const char* rel_path = *cur_path == '/' ? cur_path + 1 : cur_path;
char* prefixed = NULL; char* prefixed = NULL;
if (options->relative_prefix) { if (options->relative_prefix) {
@@ -1772,14 +1868,17 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo
} }
rel_path = prefixed; rel_path = prefixed;
} }
if (!excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) if (options->excluded_paths &&
!excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path))
ps->failed = true; ps->failed = true;
free(prefixed); free(prefixed);
} }
if (!passes) {
free(rel); free(rel);
free(cur_path); free(cur_path);
return; return;
} }
}
if (is_dir) { if (is_dir) {
if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) {
/* -x/--one-file-system: emit the mount-point directory entry (empty) but /* -x/--one-file-system: emit the mount-point directory entry (empty) but
@@ -2040,22 +2139,26 @@ ParallelScanner* parallel_scanner_create_with_options(const char* root_directory
root_dev = root_stats.st_dev; root_dev = root_stats.st_dev;
} }
/* Build the root directory's .rsync-filter context once; workers seed their /* Build the root directory's per-directory filter context once; workers seed
* scanners with it so per-dir rules behave identically to the sequential * their scanners with it so per-dir rules behave identically to the sequential
* scanner. */ * scanner. */
FilterNode* root_node = NULL; FilterNode* root_node = NULL;
if (options->per_dir_filters) { {
char err[256]; char err[256];
bool exists = false; bool any_exists = false;
FilterRuleList* own = filter_file_read(root_directory, "", &exists, err, sizeof(err)); FilterRuleList* own =
if (!own) { read_dir_filters(options, root_directory, "", &any_exists, err, sizeof(err));
log_message(LOG_LEVEL_ERROR, "invalid .rsync-filter in %s: %s", root_directory, err); if (!own && any_exists) {
/* no files exist: leave root_node NULL */
} else if (!own) {
if (err[0] != '\0') {
log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err);
array_list_delete(root_files); array_list_delete(root_files);
array_list_delete(subdirs); array_list_delete(subdirs);
parallel_scanner_destroy(ps); parallel_scanner_destroy(ps);
return NULL; return NULL;
} }
if (exists && own->count > 0) { } else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) {
root_node = filter_node_alloc(NULL, own); root_node = filter_node_alloc(NULL, own);
if (!root_node) { if (!root_node) {
filter_rule_list_free(own); filter_rule_list_free(own);
+6
View File
@@ -63,6 +63,12 @@ typedef struct {
const FileListSet* file_list; /* --files-from allow-set, or NULL */ const FileListSet* file_list; /* --files-from allow-set, or NULL */
const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */ const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */
bool per_dir_filters; /* -F: read .rsync-filter per directory */ bool per_dir_filters; /* -F: read .rsync-filter per directory */
/* --delete-excluded: per-directory plain rules become sender-only, so they no
longer protect the receiver from deletion. */
bool delete_excluded;
/* -FF: also exclude the per-directory filter files themselves from the
transfer (single -F transfers them). */
bool exclude_per_dir_filter_files;
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */ bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
bool relative; /* -R/--relative (dest rel paths, with --files-from) */ bool relative; /* -R/--relative (dest rel paths, with --files-from) */
/* -R/--relative outside --files-from: the destination-relative path prefix /* -R/--relative outside --files-from: the destination-relative path prefix
+5 -3
View File
@@ -114,10 +114,12 @@ void print_usage(void) {
printf(" --files-from <file> Read the source file list from FILE (paths relative to the " printf(" --files-from <file> Read the source file list from FILE (paths relative to the "
"source root)\n"); "source root)\n");
printf(" -0, --from0 Entries in --files-from are NUL-delimited\n"); printf(" -0, --from0 Entries in --files-from are NUL-delimited\n");
printf(" -f, --filter=RULE rsync-style filter rule (+/- include/exclude; repeatable;\n"); printf(" -f, --filter=RULE rsync-style filter rule: exclude/- include/+ hide/H show/S\n");
printf(" both --filter=RULE and the -f RULE / -f=RULE short forms work)\n"); printf(" protect/P risk/R merge/. dir-merge/: clear/! with modifiers\n");
printf(" (repeatable; --filter=RULE and -f RULE / -f=RULE both work)\n");
printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n"); printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n");
printf(" -F Apply per-directory .rsync-filter files during the scan\n"); printf(" -F Apply per-directory .rsync-filter files; repeated -FF also\n");
printf(" excludes the .rsync-filter files themselves\n");
printf(" --max-size <n> Skip files larger than n bytes\n"); printf(" --max-size <n> Skip files larger than n bytes\n");
printf(" --min-size <n> Skip files smaller than n bytes\n"); printf(" --min-size <n> Skip files smaller than n bytes\n");
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G; 0 = no limit,\n"); printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G; 0 = no limit,\n");
+20 -12
View File
@@ -329,30 +329,38 @@ bool config_has_basis(const Config* config) {
} }
/* A basis-dir path travels from the client to the receiver and is resolved /* A basis-dir path travels from the client to the receiver and is resolved
* below the destination root, so it must be a non-empty relative path with no * below the destination root when relative, or used verbatim when absolute
* "." or ".." component and no traversal: an absolute or escaping path would * (matching rsync). Either form must be non-empty, traversal-free (no "..")
* make the receiver read or link files outside its authorized root. * and free of "." components: an escaping path would make the receiver read or
* link files outside its authorized root. An absolute path is still subject to
* the receiver's root confinement at open time (file_open_secure_parent), so a
* basis outside the authorized root is simply not found rather than an escape.
* *
* Returns a malloc'd CANONICAL copy of an accepted path, or NULL when the path * Returns a malloc'd CANONICAL copy of an accepted path, or NULL when the path
* is rejected. Canonicalization collapses interior empty components ("a//b" -> * is rejected. Canonicalization collapses interior empty components ("a//b" ->
* "a/b"), drops "." components and trailing "/"s, so validation, the delete * "a/b"), drops "." components and trailing "/"s, and preserves a leading '/'
* walker prefix match and the receiver's basis lookup all agree on one form. * for absolute paths, so validation, the delete walker prefix match and the
* The normalizer is the single source of truth for both config_basis_path_valid * receiver's basis lookup all agree on one form. The normalizer is the single
* and config_basis_append. */ * source of truth for both config_basis_path_valid and config_basis_append. */
static char* basis_path_normalize(const char* path) { static char* basis_path_normalize(const char* path) {
if (!path || path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) if (!path || path[0] == '\0' || has_path_traversal(path))
return NULL; return NULL;
if (strcmp(path, ".") == 0) bool absolute = path[0] == '/';
if (!absolute && strcmp(path, ".") == 0)
return NULL;
if (absolute && strcmp(path, "/") == 0)
return NULL; return NULL;
char* dup = str_dup(path); char* dup = str_dup(path);
if (!dup) if (!dup)
return NULL; return NULL;
size_t out_len = 0; size_t out_len = 0;
char* out = malloc(strlen(path) + 1); char* out = malloc(strlen(path) + 2);
if (!out) { if (!out) {
free(dup); free(dup);
return NULL; return NULL;
} }
if (absolute)
out[out_len++] = '/';
char* saveptr = NULL; char* saveptr = NULL;
bool ok = true; bool ok = true;
for (char* part = strtok_r(dup, "/", &saveptr); part; part = strtok_r(NULL, "/", &saveptr)) { for (char* part = strtok_r(dup, "/", &saveptr); part; part = strtok_r(NULL, "/", &saveptr)) {
@@ -362,14 +370,14 @@ static char* basis_path_normalize(const char* path) {
} }
if (strcmp(part, ".") == 0) if (strcmp(part, ".") == 0)
continue; continue;
if (out_len > 0) if (out_len > 0 && out[out_len - 1] != '/')
out[out_len++] = '/'; out[out_len++] = '/';
size_t len = strlen(part); size_t len = strlen(part);
memcpy(out + out_len, part, len); memcpy(out + out_len, part, len);
out_len += len; out_len += len;
} }
free(dup); free(dup);
if (!ok || out_len == 0) { if (!ok || out_len == 0 || (absolute && out_len == 1)) {
free(out); free(out);
return NULL; return NULL;
} }
+4
View File
@@ -377,6 +377,10 @@ typedef struct Config {
bool from0; /* -0/--from0: NUL-delimited *-from files */ bool from0; /* -0/--from0: NUL-delimited *-from files */
bool cvs_exclude; /* -C/--cvs-exclude: standard CVS ignore set */ bool cvs_exclude; /* -C/--cvs-exclude: standard CVS ignore set */
bool per_dir_filter; /* -F: apply per-directory .rsync-filter files */ bool per_dir_filter; /* -F: apply per-directory .rsync-filter files */
/* -F click count. rsync's single -F means --filter='dir-merge
* /.rsync-filter' (the .rsync-filter files themselves are transferred); a
* repeated -F adds --filter='- .rsync-filter' so they are excluded too. */
int per_dir_filter_count;
bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */ bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */
/* --no-implied-dirs: client-only. With -R + --files-from, refuse to place a /* --no-implied-dirs: client-only. With -R + --files-from, refuse to place a
* listed file whose ancestor directory is not itself explicitly listed. */ * listed file whose ancestor directory is not itself explicitly listed. */
+21 -14
View File
@@ -40,19 +40,27 @@ static bool write_all(int fd, const void* data, unsigned long long size) {
} }
/* Preallocate `size` bytes on `fd` before any data is written (--preallocate). /* Preallocate `size` bytes on `fd` before any data is written (--preallocate).
* posix_fallocate reserves real disk blocks, so an out-of-space condition * fallocate(2) reserves real disk blocks, so an out-of-space condition
* (ENOSPC/EDQUOT) surfaces up front instead of partway through a transfer; * (ENOSPC/EDQUOT) surfaces up front instead of partway through a transfer;
* unavoidable fragmentation of a streamed file is also reduced. Some * unavoidable fragmentation of a streamed file is also reduced. rsync favors
* filesystems (e.g. tmpfs, ZFS) do not support it and return EOPNOTSUPP/ENOSYS, * the syscall over glibc posix_fallocate (whose emulation can be subtly
* where we fall back to ftruncate, which still extends the logical size so the * different), so try fallocate(2) first and only fall back to posix_fallocate,
* fail-fast/contiguity intent degrades gracefully but never fails. Genuine * then to ftruncate on filesystems (e.g. tmpfs, ZFS) that support neither. The
* allocation failures are propagated as the error code (caller fails the write). * logical size is always extended, so the fail-fast/contiguity intent degrades
* posix_fallocate leaves the fd's file offset unchanged, so the subsequent * gracefully but never fails on an unsupported filesystem; genuine allocation
* write_all at offset 0 is unaffected. Returns 0 on success (including the * failures are propagated as the error code (caller fails the write). Neither
* fallback) or a nonzero error code. */ * leaves the fd's file offset guaranteed, so the caller seeks back to 0 before
* writing. Returns 0 on success (including the fallback) or a nonzero error
* code. */
static int preallocate_fd(int fd, unsigned long long size) { static int preallocate_fd(int fd, unsigned long long size) {
if (size == 0) if (size == 0)
return 0; return 0;
#ifdef __linux__
if (fallocate(fd, 0, 0, (off_t)size) == 0)
return 0;
if (errno != EOPNOTSUPP && errno != ENOSYS && errno != EINVAL)
return errno;
#endif
int rc = posix_fallocate(fd, 0, (off_t)size); int rc = posix_fallocate(fd, 0, (off_t)size);
if (rc == EOPNOTSUPP || rc == ENOSYS) { if (rc == EOPNOTSUPP || rc == ENOSYS) {
if (ftruncate(fd, (off_t)size) == 0) if (ftruncate(fd, (off_t)size) == 0)
@@ -1066,11 +1074,10 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
} else { } else {
/* Preallocate the expected payload size before writing so an /* Preallocate the expected payload size before writing so an
out-of-space condition fails cleanly up front (--preallocate). out-of-space condition fails cleanly up front (--preallocate).
--sparse takes precedence: posix_fallocate would allocate every rsync lets --preallocate win over --sparse (the reserved blocks
block, defeating the holes the sparse writer would create, so the survive the sparse writer's seeks), so both flags can be active. */
two never combine here (the ftruncate presize below stays). */
int prealloc_rc = 0; int prealloc_rc = 0;
if (preallocate && !sparse && data_size > 0) { if (preallocate && data_size > 0) {
prealloc_rc = preallocate_fd(fd, data_size); prealloc_rc = preallocate_fd(fd, data_size);
if (prealloc_rc != 0) { if (prealloc_rc != 0) {
char* escaped_path = output_escape(path, log_get_8_bit_output()); char* escaped_path = output_escape(path, log_get_8_bit_output());
@@ -1196,7 +1203,7 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
if (fd < 0) if (fd < 0)
continue; /* EEXIST (or a transient open error): try a fresh name. */ continue; /* EEXIST (or a transient open error): try a fresh name. */
int prealloc_rc = 0; int prealloc_rc = 0;
if (preallocate && !sparse && data_size > 0) { if (preallocate && data_size > 0) {
prealloc_rc = preallocate_fd(fd, data_size); prealloc_rc = preallocate_fd(fd, data_size);
if (prealloc_rc != 0) { if (prealloc_rc != 0) {
char* escaped_path = output_escape(path, log_get_8_bit_output()); char* escaped_path = output_escape(path, log_get_8_bit_output());
+318 -128
View File
@@ -1,4 +1,5 @@
#include <errno.h> #include <errno.h>
#include <ctype.h>
#include <dirent.h> #include <dirent.h>
#include <fcntl.h> #include <fcntl.h>
#include <libgen.h> #include <libgen.h>
@@ -1335,7 +1336,11 @@ static bool basis_match_find(const Config* config, const char* check_path,
return false; return false;
for (int i = 0; i < config->basis_count; i++) { for (int i = 0; i < config->basis_count; i++) {
const BasisDest* entry = &config->basis_dirs[i]; const BasisDest* entry = &config->basis_dirs[i];
char* basis_dir = path_cat(config->receive_root_directory, entry->path); /* An absolute basis path is used verbatim (rsync semantics); a relative one
is resolved below the receive root. Both remain subject to the receiver's
authorized-root confinement inside file_open_secure_parent. */
char* basis_dir = entry->path[0] == '/' ? str_dup(entry->path)
: path_cat(config->receive_root_directory, entry->path);
if (!basis_dir) if (!basis_dir)
continue; continue;
char* candidate = path_cat(basis_dir, check_path); char* candidate = path_cat(basis_dir, check_path);
@@ -1397,7 +1402,8 @@ static bool basis_match_find(const Config* config, const char* check_path,
* transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a * transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a
* file. * file.
* *
* Similarity heuristic (deterministic, deliberately simpler than rsync's): * Similarity heuristic (rsync 3.4.1 parity, util1.c fuzzy_distance /
* find_filename_suffix + generator.c find_fuzzy):
* * candidates are the target's sibling entries in its destination * * candidates are the target's sibling entries in its destination
* directory, opened through the confined root (file_open_secure_parent + * directory, opened through the confined root (file_open_secure_parent +
* openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never * openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never
@@ -1406,12 +1412,15 @@ static bool basis_match_find(const Config* config, const char* check_path,
* temp scratch names are never candidates; * temp scratch names are never candidates;
* * size gate = the delta engine's own bounds (delta_should_attempt: both * * size gate = the delta engine's own bounds (delta_should_attempt: both
* files >= DELTA_MIN_FILE_SIZE, <= delta_max_file_size, ratio <= 10x), * files >= DELTA_MIN_FILE_SIZE, <= delta_max_file_size, ratio <= 10x),
* NOT rsync's ~1.5x size window; * because FastSync's delta engine cannot use a basis outside them;
* * name gate = Levenshtein edit distance between the basenames, accepted * * first pass = an exact size+mtime match wins regardless of name (rsync's
* only when distance <= half the length of the longer basename; * "fuzzy size/modtime match");
* * the single best candidate (smallest distance; tie-break: size closest * * otherwise the winner minimizes rsync's weighted Levenshtein distance
* to the incoming file, then lexicographically smaller basename) is read * (substitution ± byte difference, insertion UNIT+byte, 16.16 fixed point)
* and returned as the basis. * plus ten times the suffix distance, accepted only when <= 25*UNIT; the
* tie-break (smallest size gap, then lexical name) keeps the result
* deterministic across filesystem readdir order (rsync leaves equal
* distances to its file-list order).
* ------------------------------------------------------------------------- */ * ------------------------------------------------------------------------- */
/* A directory scan is linear in the number of entries; the fuzzy search stops /* A directory scan is linear in the number of entries; the fuzzy search stops
@@ -1429,107 +1438,108 @@ static bool basis_match_find(const Config* config, const char* check_path,
typedef struct { typedef struct {
char name[FUZZY_NAME_LIMIT + 1]; char name[FUZZY_NAME_LIMIT + 1];
unsigned long long size; unsigned long long size;
size_t distance; uint32_t distance;
unsigned long long size_gap; unsigned long long size_gap;
} FuzzyCandidate; } FuzzyCandidate;
/* Two-row DP scratch, allocated once per directory scan (not per candidate) so /* rsync's fuzzy distance is a weighted Levenshtein variant in 16.16 fixed point
* a 4096-entry directory never performs 4096 malloc/free pairs. */ * (util1.c fuzzy_distance): a substitution costs UNIT +/- the byte difference
typedef struct { * and an insertion costs UNIT + the inserted byte, so similar names score low.
size_t* prev; * The search keeps only distances <= 25*UNIT. Ported verbatim for parity. */
size_t* cur; #define FUZZY_DIST_UNIT (1u << 16)
} FuzzyEditBuffer; #define FUZZY_DIST_REJECT (0xFFFFu * FUZZY_DIST_UNIT + 1)
#define FUZZY_DIST_LIMIT (25u * FUZZY_DIST_UNIT)
static bool fuzzy_edit_buffer_init(FuzzyEditBuffer* buf) { static uint32_t fuzzy_distance(const char* s1, unsigned len1, const char* s2, unsigned len2,
buf->prev = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(size_t)); uint32_t upperlimit, uint32_t* scratch) {
buf->cur = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(size_t)); if ((len1 > len2 ? len1 - len2 : len2 - len1) * FUZZY_DIST_UNIT > upperlimit)
if (!buf->prev || !buf->cur) { return FUZZY_DIST_REJECT;
free(buf->prev); if (!len1 || !len2) {
free(buf->cur); if (!len1) {
buf->prev = NULL; s1 = s2;
buf->cur = NULL; len1 = len2;
return false;
} }
return true; uint32_t cost = 0;
for (unsigned i = 0; i < len1; i++)
cost += (uint8_t)s1[i];
return (uint32_t)len1 * FUZZY_DIST_UNIT + cost;
}
uint32_t* a = scratch;
for (unsigned i2 = 0; i2 < len2; i2++)
a[i2] = (i2 + 1) * FUZZY_DIST_UNIT;
for (unsigned i1 = 0; i1 < len1; i1++) {
uint32_t diag = i1 * FUZZY_DIST_UNIT;
uint32_t above = (i1 + 1) * FUZZY_DIST_UNIT;
for (unsigned i2 = 0; i2 < len2; i2++) {
uint32_t left = a[i2];
int32_t cost = (int32_t)(uint8_t)s1[i1] - (int32_t)(uint8_t)s2[i2];
if (cost != 0)
cost = cost < 0 ? (int32_t)(FUZZY_DIST_UNIT - (uint32_t)(-cost))
: (int32_t)(FUZZY_DIST_UNIT + (uint32_t)cost);
uint32_t diag_inc = diag + (uint32_t)cost;
uint32_t left_inc = left + FUZZY_DIST_UNIT + (uint8_t)s1[i1];
uint32_t above_inc = above + FUZZY_DIST_UNIT + (uint8_t)s2[i2];
a[i2] = above = left < above ? (left_inc < diag_inc ? left_inc : diag_inc)
: (above_inc < diag_inc ? above_inc : diag_inc);
diag = left;
}
}
return a[len2 - 1];
} }
static void fuzzy_edit_buffer_destroy(FuzzyEditBuffer* buf) { /* rsync's find_filename_suffix (util1.c): return the last significant filename
free(buf->prev); * suffix (its dot included). Leading dots are not a suffix; a trailing "~" is
free(buf->cur); * ignored; .bak/.old/.orig and a "~/<num>" backup marker are skipped. */
buf->prev = NULL; static const char* fuzzy_find_suffix(const char* fn, int fn_len, int* len_ptr) {
buf->cur = NULL; const char* suf;
const char* s;
bool had_tilde;
while (fn_len && *fn == '.') {
fn++;
fn_len--;
}
if (fn_len > 1 && fn[fn_len - 1] == '~') {
fn_len--;
had_tilde = true;
} else {
had_tilde = false;
}
suf = "";
*len_ptr = 0;
for (s = fn + fn_len; fn_len > 1;) {
int s_len;
while (--s != fn && *s != '.') {
}
if (s == fn)
break;
s_len = fn_len - (int)(s - fn);
fn_len = (int)(s - fn);
if (s_len == 4) {
if (strcmp(s + 1, "bak") == 0 || strcmp(s + 1, "old") == 0)
continue;
} else if (s_len == 5) {
if (strcmp(s + 1, "orig") == 0)
continue;
} else if (s_len > 2 && had_tilde && s[1] == '~' && isdigit((unsigned char)s[2])) {
continue;
}
*len_ptr = s_len;
suf = s;
if (s_len == 1)
break;
for (s++, s_len--; s_len > 0; s++, s_len--) {
if (!isdigit((unsigned char)*s))
return suf;
}
s = suf;
}
return suf;
} }
/* Cheap lower bounds used to reject a candidate BEFORE the DP: /* Deterministic ordering of two fuzzy candidates with equal rsync distance:
* - any edit script must at least absorb the length gap: d >= |la - lb|; * smallest size gap, then the lexical basename (rsync itself takes the last
* - any character of `a` that does not occur in `b` at all must be deleted or * equal-distance candidate in file-list order). */
* substituted at its own position: d >= (count of such characters).
* The acceptance gate is d*2 <= longer, so a candidate whose max of these two
* bounds already violates it can be skipped without computing the distance. */
static size_t fuzzy_absent_char_bound(const char* a, size_t la, const char* b, size_t lb) {
if (lb == 0)
return la;
bool present[256] = {false};
for (size_t i = 0; i < lb; i++)
present[(uint8_t)b[i]] = true;
size_t absent = 0;
for (size_t i = 0; i < la; i++)
if (!present[(uint8_t)a[i]])
absent++;
return absent;
}
/* Levenshtein edit distance between the two basenames. A shared prefix and a
* (non-overlapping) shared suffix can always be aligned at no cost, so the DP
* only runs over the differing middles; its two rows come from `buf` (allocated
* once by the caller). Callers enforce la, lb <= FUZZY_NAME_LIMIT. */
static size_t fuzzy_edit_distance(FuzzyEditBuffer* buf, const char* a, size_t la, const char* b,
size_t lb) {
size_t p = 0;
while (p < la && p < lb && a[p] == b[p])
p++;
/* Trim the common suffix (never overlapping the prefix). Working with two
moving end indices keeps the region arithmetic explicit and safe. */
size_t ae = la;
size_t be = lb;
while (ae > p && be > p && a[ae - 1] == b[be - 1]) {
ae--;
be--;
}
size_t ma = ae - p;
size_t mb = be - p;
/* cppcheck-suppress knownConditionTrueFalse -- the prefix/suffix trims above
only run while the corresponding ends match, so a middle can remain; the
analysis unsoundly concludes the trims always consume everything. */
if (ma == 0)
return mb;
if (mb == 0)
return ma;
const char* A = a + p;
const char* B = b + p;
size_t* prev = buf->prev;
size_t* cur = buf->cur;
for (size_t j = 0; j <= mb; j++)
prev[j] = j;
for (size_t i = 1; i <= ma; i++) {
cur[0] = i;
for (size_t j = 1; j <= mb; j++) {
size_t cost = A[i - 1] == B[j - 1] ? 0 : 1;
size_t del = prev[j] + 1;
size_t ins = cur[j - 1] + 1;
size_t sub = prev[j - 1] + cost;
size_t m = del < ins ? del : ins;
cur[j] = m < sub ? m : sub;
}
size_t* tmp = prev;
prev = cur;
cur = tmp;
}
return prev[mb];
}
/* Deterministic ordering of two fuzzy candidates: smallest edit distance,
* then the size closest to the incoming file, then the lexical basename. */
static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) { static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) {
if (!best->name[0]) if (!best->name[0])
return true; return true;
@@ -1546,8 +1556,8 @@ static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandid
* = 0) when no candidate qualifies, which means the caller performs the normal * = 0) when no candidate qualifies, which means the caller performs the normal
* whole-file transfer. */ * whole-file transfer. */
static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path, static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path,
unsigned long long check_size, unsigned long long check_size, time_t check_mtime,
unsigned long long* out_size) { long check_mtime_nsec, unsigned long long* out_size) {
*out_size = 0; *out_size = 0;
if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta ||
!check_path || check_size < DELTA_MIN_FILE_SIZE || check_size > config->delta_max_file_size || !check_path || check_size < DELTA_MIN_FILE_SIZE || check_size > config->delta_max_file_size ||
@@ -1590,18 +1600,28 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p
return NULL; return NULL;
} }
/* The DP scratch rows are allocated once per scan (not once per candidate). */ /* The weighted-distance scratch row is allocated once per scan (not once per
FuzzyEditBuffer ebuf; candidate). */
if (!fuzzy_edit_buffer_init(&ebuf)) { uint32_t* dist_scratch = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(uint32_t));
if (!dist_scratch) {
closedir(dir); closedir(dir);
close(dir_fd); close(dir_fd);
free(leaf); free(leaf);
free(full_path); free(full_path);
return NULL; return NULL;
} }
int fname_suf_len = 0;
const char* fname_suf = fuzzy_find_suffix(leaf, (int)target_len, &fname_suf_len);
FuzzyCandidate best; FuzzyCandidate best;
memset(&best, 0, sizeof(best)); memset(&best, 0, sizeof(best));
uint32_t lowest_dist = FUZZY_DIST_LIMIT;
/* rsync's fuzzy search runs an exact size+mtime pass before the name-distance
pass; such a candidate is almost certainly the same content and wins
regardless of how dissimilar its name is. The first one (directory order,
deterministic) is kept. */
FuzzyCandidate exact;
memset(&exact, 0, sizeof(exact));
const struct dirent* entry; const struct dirent* entry;
size_t scanned = 0; size_t scanned = 0;
/* readdir() yields entries in filesystem-dependent order, so the SET of /* readdir() yields entries in filesystem-dependent order, so the SET of
@@ -1621,22 +1641,32 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p
if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE || if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE ||
!delta_should_attempt(cand_size, check_size, config->delta_max_file_size)) !delta_should_attempt(cand_size, check_size, config->delta_max_file_size))
continue; continue;
/* Cheap pre-name gates run BEFORE the edit-distance DP. The edit distance long cand_nsec = 0;
is bounded below by the length gap |la-lb| and by the number of #ifdef __linux__
characters of one basename that are absent from the other (each such cand_nsec = st.st_mtim.tv_nsec;
position costs at least one op), so a candidate whose acceptance gate #endif
(distance*2 <= longer) already fails on the max of those bounds is if (!exact.name[0] && cand_size == check_size &&
skipped without running the DP. */ metadata_mtime_matches(st.st_mtime, cand_nsec, check_mtime, check_mtime_nsec,
size_t longer = target_len > name_len ? target_len : name_len; config->modify_window)) {
size_t bound = longer - (target_len < name_len ? target_len : name_len); memcpy(exact.name, name, name_len + 1);
size_t absent = fuzzy_absent_char_bound(leaf, target_len, name, name_len); exact.size = cand_size;
if (absent > bound) exact.size_gap = 0;
bound = absent;
if (bound * 2 > longer)
continue; continue;
size_t distance = fuzzy_edit_distance(&ebuf, leaf, target_len, name, name_len); }
if (distance * 2 > longer) /* rsync's name-distance pass: a weighted Levenshtein distance over the full
basenames, plus ten times the same distance over the filename suffixes,
accepted only when it does not exceed the running lowest distance. */
int name_suf_len = 0;
const char* name_suf = fuzzy_find_suffix(name, (int)name_len, &name_suf_len);
uint32_t distance = fuzzy_distance(name, (unsigned)name_len, leaf, (unsigned)target_len,
lowest_dist, dist_scratch);
if (distance < 0xFFFF0000U)
distance += fuzzy_distance(name_suf, (unsigned)name_suf_len, fname_suf,
(unsigned)fname_suf_len, 0xFFFF0000U, dist_scratch) *
10;
if (distance > lowest_dist)
continue; continue;
lowest_dist = distance;
FuzzyCandidate cand; FuzzyCandidate cand;
memcpy(cand.name, name, name_len + 1); memcpy(cand.name, name, name_len + 1);
cand.size = cand_size; cand.size = cand_size;
@@ -1647,7 +1677,11 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p
} }
closedir(dir); closedir(dir);
free(leaf); free(leaf);
fuzzy_edit_buffer_destroy(&ebuf); free(dist_scratch);
/* Prefer the exact size+mtime candidate over any name-distance winner. */
if (exact.name[0])
best = exact;
void* basis = NULL; void* basis = NULL;
if (best.name[0]) { if (best.name[0]) {
@@ -1764,6 +1798,7 @@ typedef struct {
long long check_mtime_nsec; long long check_mtime_nsec;
uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN]; uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN];
size_t check_digest_len; size_t check_digest_len;
bool dest_exists; /* any destination entry exists (lstat succeeded) */
bool has_old_file; bool has_old_file;
int old_fd; int old_fd;
struct stat old_st; struct stat old_st;
@@ -1860,6 +1895,9 @@ static IncrementalCheckOutcome incremental_check_open_destination(IncrementalChe
char* leaf = NULL; char* leaf = NULL;
int parent_fd = file_open_secure_parent(full_path, &leaf, false); int parent_fd = file_open_secure_parent(full_path, &leaf, false);
if (parent_fd >= 0) { if (parent_fd >= 0) {
struct stat dest_st;
if (fstatat(parent_fd, leaf, &dest_st, AT_SYMLINK_NOFOLLOW) == 0)
state->dest_exists = true;
/* O_NONBLOCK: an existing FIFO at the destination must not block this /* O_NONBLOCK: an existing FIFO at the destination must not block this
openat(); the S_ISREG gate below rejects the non-regular entry. */ openat(); the S_ISREG gate below rejects the non-regular entry. */
state->old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); state->old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK);
@@ -1903,6 +1941,82 @@ static IncrementalCheckOutcome incremental_check_report_dest_info(IncrementalChe
return INCREMENTAL_CONTINUE; return INCREMENTAL_CONTINUE;
} }
/* --ignore-existing short-circuit. The receiver must answer "skip" (STATUS_OK)
BEFORE the sender transmits any payload, otherwise the whole file crosses the
wire only to be discarded at write time. rsync skips an existing destination
entry regardless of its content or type, so the reply depends only on the
lstat existence probe; the ordinary --ignore-existing checks inside
file_receive remain as defense-in-depth for the frame types that have no
per-file check (directories/symlinks/specials/hard-links). */
static IncrementalCheckOutcome
incremental_check_ignore_existing(const IncrementalCheckState* state) {
if (!state->config->ignore_existing || !state->dest_exists)
return INCREMENTAL_CONTINUE;
if (!send_status(state->fd, STATUS_OK))
return INCREMENTAL_ERROR;
return INCREMENTAL_SKIP;
}
/* --link-dest relink of an already up-to-date destination. rsync hard-links a
destination entry to a matching basis even when the entry is already correct,
so a run over an existing tree still maximizes sharing with the basis. Only a
link-dest basis triggers this (copy-dest/compare-dest leave an up-to-date
destination untouched, matching rsync). The ordinary basis path further down
handles every not-up-to-date case, so this helper only adds the relink that
the quick-skip would otherwise short-circuit. */
static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalCheckState* state,
File** out_file) {
const Config* config = state->config;
if (!config_has_basis(config) || config->ignore_times || config->dry_run)
return INCREMENTAL_CONTINUE;
if (!state->has_old_file)
return INCREMENTAL_CONTINUE;
BasisMatch basis;
basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime,
(long)state->check_mtime_nsec, state->check_digest, state->check_digest_len,
true, true, &basis);
/* Only a link-dest hit relinks; a copy-dest/compare-dest hit (or a miss) lets
the up-to-date check below keep the existing destination. */
if (!basis.hit || basis.type != BASIS_DEST_LINK) {
basis_match_free(&basis);
return INCREMENTAL_CONTINUE;
}
/* Already the basis inode: nothing to do, leave the destination alone. */
if (basis.st.st_dev == state->old_st.st_dev && basis.st.st_ino == state->old_st.st_ino) {
basis_match_free(&basis);
return INCREMENTAL_CONTINUE;
}
File* materialized = file_create(state->check_path);
if (materialized && basis.content) {
data_destroy(materialized->data);
materialized->data = basis.content;
basis.content = NULL;
materialized->metadata = file_metadata_create(NULL, &basis.st, false, false);
materialized->skip = true;
materialized->basis_link = basis.basis_path;
basis.basis_path = NULL;
if (!materialized->metadata) {
file_destroy(materialized);
materialized = NULL;
}
} else {
file_destroy(materialized);
materialized = NULL;
}
if (materialized) {
if (!send_status(state->fd, STATUS_OK)) {
basis_match_free(&basis);
file_destroy(materialized);
return INCREMENTAL_ERROR;
}
basis_match_free(&basis);
*out_file = materialized;
return INCREMENTAL_FILE;
}
basis_match_free(&basis);
return INCREMENTAL_CONTINUE;
}
/* Metadata-only (and, when --checksum forces it, content) up-to-date decision. /* Metadata-only (and, when --checksum forces it, content) up-to-date decision.
Loads the old contents only when a checksum comparison or delta needs them. */ Loads the old contents only when a checksum comparison or delta needs them. */
static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckState* state, static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckState* state,
@@ -2289,8 +2403,9 @@ static IncrementalCheckOutcome incremental_check_try_fuzzy(IncrementalCheckState
if (!config->fuzzy || !config->use_delta) if (!config->fuzzy || !config->use_delta)
return INCREMENTAL_CONTINUE; return INCREMENTAL_CONTINUE;
unsigned long long fuzzy_size = 0; unsigned long long fuzzy_size = 0;
void* fuzzy_basis = void* fuzzy_basis = fuzzy_basis_find_and_load(config, state->check_path, state->check_size,
fuzzy_basis_find_and_load(config, state->check_path, state->check_size, &fuzzy_size); (time_t)state->check_mtime,
(long)state->check_mtime_nsec, &fuzzy_size);
if (fuzzy_basis != NULL) { if (fuzzy_basis != NULL) {
bool fuzzy_failed = false; bool fuzzy_failed = false;
File* fuzzy_file = receive_delta_file(state->fd, config, state->check_path, fuzzy_basis, File* fuzzy_file = receive_delta_file(state->fd, config, state->check_path, fuzzy_basis,
@@ -2351,6 +2466,24 @@ File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped,
if (outcome == INCREMENTAL_ERROR) if (outcome == INCREMENTAL_ERROR)
goto done; goto done;
/* --ignore-existing must answer before any data is requested; it takes
precedence over the metadata up-to-date check below. */
outcome = incremental_check_ignore_existing(&state);
if (outcome == INCREMENTAL_ERROR)
goto done;
if (outcome == INCREMENTAL_SKIP) {
*skipped = true;
goto done;
}
/* A --link-dest hit relinks even an already up-to-date destination before the
quick-skip can suppress it (rsync parity). */
outcome = incremental_check_link_dest_relink(&state, &result);
if (outcome == INCREMENTAL_ERROR)
goto done;
if (outcome == INCREMENTAL_FILE)
goto done;
outcome = incremental_check_quick_skip(&state, &try_delta); outcome = incremental_check_quick_skip(&state, &try_delta);
if (outcome == INCREMENTAL_ERROR) if (outcome == INCREMENTAL_ERROR)
goto done; goto done;
@@ -3012,6 +3145,29 @@ typedef struct {
bool limit_hit; bool limit_hit;
} DeleteBudgetState; } DeleteBudgetState;
/* Build the delete-walk protection prefix for one basis directory. The walker
compares paths relative to the receive root, so a relative entry is already
in the right form; an absolute entry that lies below the root is converted to
its root-relative form, and one outside the root returns NULL (the walk
cannot reach it, and it is not protected data beneath the root). */
static char* basis_delete_relative(const Config* config, const char* path) {
if (!path)
return NULL;
if (path[0] != '/')
return str_dup(path);
const char* root = config->receive_root_directory;
if (!root || root[0] != '/')
return NULL;
size_t root_len = strlen(root);
while (root_len > 1 && root[root_len - 1] == '/')
root_len--;
if (strncmp(path, root, root_len) != 0)
return NULL;
if (path[root_len] != '/')
return NULL; /* identical or a sibling sharing a name prefix */
return str_dup(path + root_len + 1);
}
/* Remove every destination entry under the receive root that is not in the /* Remove every destination entry under the receive root that is not in the
keep-set, bounded by the shared budget (a smaller client --max-delete=NUM keep-set, bounded by the shared budget (a smaller client --max-delete=NUM
replaces the server hard bound; rsync deletes up to the bound and skips the replaces the server hard bound; rsync deletes up to the bound and skips the
@@ -3042,10 +3198,16 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes
int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count +
(manifest->protected ? manifest->protected->size : 0); (manifest->protected ? manifest->protected->size : 0);
DeleteSkipEntry* skips = NULL; DeleteSkipEntry* skips = NULL;
char** owned_prefixes = NULL;
int used = 0;
if (skip_count > 0) { if (skip_count > 0) {
skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry));
if (!skips) owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*));
if (!skips || (config->basis_count > 0 && !owned_prefixes)) {
free(skips);
free(owned_prefixes);
return false; return false;
}
int idx = 0; int idx = 0;
if (config->delay_updates) { if (config->delay_updates) {
skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; skips[idx].prefix = DELAY_UPDATES_STAGING_DIR;
@@ -3053,7 +3215,13 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes
idx++; idx++;
} }
for (int i = 0; i < config->basis_count; i++) { for (int i = 0; i < config->basis_count; i++) {
skips[idx].prefix = config->basis_dirs[i].path; /* An absolute basis outside the receive root is unreachable by this walk,
so it contributes no protection prefix (and no slot). */
char* prefix = basis_delete_relative(config, config->basis_dirs[i].path);
if (!prefix)
continue;
owned_prefixes[i] = prefix;
skips[idx].prefix = prefix;
skips[idx].top_level_only = false; skips[idx].top_level_only = false;
idx++; idx++;
} }
@@ -3062,6 +3230,7 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes
skips[idx].top_level_only = false; skips[idx].top_level_only = false;
idx++; idx++;
} }
used = idx;
} }
/* Clamp rather than subtract: an accounting bug where deleted already exceeds /* Clamp rather than subtract: an accounting bug where deleted already exceeds
max_delete must never underflow into an effectively unlimited budget. */ max_delete must never underflow into an effectively unlimited budget. */
@@ -3076,7 +3245,12 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes
size_t skipped = 0; size_t skipped = 0;
DeleteWalkResult result = DeleteWalkResult result =
delete_extras_limited(config->receive_root_directory, manifest->keeps, manifest->dirs, delete_extras_limited(config->receive_root_directory, manifest->keeps, manifest->dirs,
remaining, skips, skip_count, &deleted, &skipped); remaining, skips, used, &deleted, &skipped);
if (owned_prefixes) {
for (int i = 0; i < config->basis_count; i++)
free(owned_prefixes[i]);
}
free(owned_prefixes);
free(skips); free(skips);
budget->deleted += deleted; budget->deleted += deleted;
budget->skipped += skipped; budget->skipped += skipped;
@@ -3113,10 +3287,16 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m
fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n");
int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count;
DeleteSkipEntry* skips = NULL; DeleteSkipEntry* skips = NULL;
char** owned_prefixes = NULL;
int used = 0;
if (skip_count > 0) { if (skip_count > 0) {
skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry));
if (!skips) owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*));
if (!skips || (config->basis_count > 0 && !owned_prefixes)) {
free(skips);
free(owned_prefixes);
return false; return false;
}
int idx = 0; int idx = 0;
if (config->delay_updates) { if (config->delay_updates) {
skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; skips[idx].prefix = DELAY_UPDATES_STAGING_DIR;
@@ -3124,10 +3304,15 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m
idx++; idx++;
} }
for (int i = 0; i < config->basis_count; i++) { for (int i = 0; i < config->basis_count; i++) {
skips[idx].prefix = config->basis_dirs[i].path; char* prefix = basis_delete_relative(config, config->basis_dirs[i].path);
if (!prefix)
continue;
owned_prefixes[i] = prefix;
skips[idx].prefix = prefix;
skips[idx].top_level_only = false; skips[idx].top_level_only = false;
idx++; idx++;
} }
used = idx;
} }
bool ok = true; bool ok = true;
for (int i = 0; i < manifest->missing->size; i++) { for (int i = 0; i < manifest->missing->size; i++) {
@@ -3140,7 +3325,7 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m
continue; continue;
} }
bool at_root = strchr(rel, '/') == NULL; bool at_root = strchr(rel, '/') == NULL;
if (path_under_skip_prefix(rel, at_root, skips, skip_count)) { if (path_under_skip_prefix(rel, at_root, skips, used)) {
char* escaped = output_escape(rel, log_get_8_bit_output()); char* escaped = output_escape(rel, log_get_8_bit_output());
log_message(LOG_LEVEL_WARNING, log_message(LOG_LEVEL_WARNING,
"missing-args path '%s' is protected (staging directory or basis snapshot); " "missing-args path '%s' is protected (staging directory or basis snapshot); "
@@ -3264,6 +3449,11 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m
if (!ok) if (!ok)
break; break;
} }
if (owned_prefixes) {
for (int i = 0; i < config->basis_count; i++)
free(owned_prefixes[i]);
}
free(owned_prefixes);
free(skips); free(skips);
return ok; return ok;
} }
+568 -223
View File
@@ -1,163 +1,14 @@
#include "filter.h" #include "filter.h"
#include "log.h" #include "log.h"
#include "utils.h" #include "utils.h"
#include <ctype.h>
#include <errno.h> #include <errno.h>
#include <limits.h> #include <limits.h>
#include <stdio.h> #include <stdio.h>
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
/* ---- Single rule parsing ---- */ /* ---- Ordered rule lists ---- */
static bool rule_text_is_unsupported_word(const char* p, size_t len) {
static const char* const words[] = {"merge", "dir-merge", "hide", "show",
"protect", "risk", "clear"};
for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) {
size_t wl = strlen(words[i]);
if (len == wl && strncmp(p, words[i], wl) == 0)
return true;
}
return false;
}
/* rsync include/exclude rule modifiers we do NOT implement. A rule whose +/- is
* immediately followed by one of these is rejected instead of being silently
* parsed as a literal pattern. */
static bool is_unsupported_rule_modifier(char c) {
return c == '!' || c == 'C' || c == 's' || c == 'r' || c == 'p' || c == 'x';
}
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
if (!line)
return NULL;
char* text = str_dup(line);
if (!text) {
if (err)
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
size_t len = strlen(text);
while (len > 0 && (text[len - 1] == '\n' || text[len - 1] == '\r'))
text[--len] = '\0';
const char* p = text;
while (*p == ' ' || *p == '\t')
p++;
if (*p == '\0') {
snprintf(err, err_size, "empty filter rule");
free(text);
return NULL;
}
FilterAction action = FILTER_ACTION_EXCLUDE;
if (*p == '+' || *p == '-') {
action = *p == '+' ? FILTER_ACTION_INCLUDE : FILTER_ACTION_EXCLUDE;
p++;
/* rsync attaches rule modifiers directly to the +/- (e.g. "-s foo"). Only
* the '/' anchor modifier is supported; anything else is a clear error
* rather than a silently-ignored literal. */
if (*p != ' ' && *p != '\t' && *p != '\0' && is_unsupported_rule_modifier(*p)) {
snprintf(err, err_size,
"filter rule modifier '%c' is not supported (only the '/' anchor after +/- "
"is implemented; put a space between +/- and the pattern)",
*p);
free(text);
return NULL;
}
while (*p == ' ' || *p == '\t')
p++;
} else {
/* ':' (dir-merge) and '.' (merge) are rsync filter-rule shorthands. At the
* start of a rule they mean "merge this file", so reject them instead of
* silently turning them into inert exclude patterns. */
if (*p == ':' || *p == '.' || *p == '!') {
snprintf(err, err_size,
"filter rule starting with '%c' is not supported (merge/dir-merge/list-clear "
"shorthands are not implemented; use +/- include/exclude rules)",
*p);
free(text);
return NULL;
}
const char* sp = p;
while (*sp != '\0' && *sp != ' ' && *sp != '\t')
sp++;
size_t word_len = (size_t)(sp - p);
if (rule_text_is_unsupported_word(p, word_len)) {
snprintf(err, err_size,
"'%.*s' filter directives are not supported (only +/- include/exclude rules "
"with an optional '/' anchor and trailing '/' dir marker)",
(int)word_len, p);
free(text);
return NULL;
}
if (word_len == strlen("include") && strncmp(p, "include", word_len) == 0) {
action = FILTER_ACTION_INCLUDE;
p = sp;
} else if (word_len == strlen("exclude") && strncmp(p, "exclude", word_len) == 0) {
action = FILTER_ACTION_EXCLUDE;
p = sp;
}
while (*p == ' ' || *p == '\t')
p++;
}
if (*p == '\0') {
snprintf(err, err_size, "filter rule has no pattern");
free(text);
return NULL;
}
/* A pattern beginning with '/' is anchored (either as "-/foo" or "- /foo"). */
bool anchored = false;
if (*p == '/') {
anchored = true;
p++;
while (*p == ' ' || *p == '\t')
p++;
}
if (*p == '\0') {
snprintf(err, err_size, "filter rule has no pattern after '/' anchor");
free(text);
return NULL;
}
/* Pattern runs to the end of the rule; a single trailing '/' marks dir-only. */
size_t pat_len = strlen(p);
bool dir_only = false;
if (pat_len > 1 && p[pat_len - 1] == '/') {
dir_only = true;
pat_len--;
} else if (pat_len == 1 && p[0] == '/') {
/* "//" anchored with nothing after: meaningless. */
snprintf(err, err_size, "filter rule has no pattern");
free(text);
return NULL;
}
FilterRule* rule = calloc(1, sizeof(FilterRule));
if (!rule) {
snprintf(err, err_size, "memory allocation failed");
free(text);
return NULL;
}
rule->pattern = malloc(pat_len + 1);
if (!rule->pattern) {
free(rule);
snprintf(err, err_size, "memory allocation failed");
free(text);
return NULL;
}
memcpy(rule->pattern, p, pat_len);
rule->pattern[pat_len] = '\0';
rule->action = action;
rule->anchored = anchored;
rule->dir_only = dir_only;
rule->owner = NULL;
free(text);
return rule;
}
void filter_rule_free(FilterRule* rule) { void filter_rule_free(FilterRule* rule) {
if (!rule) if (!rule)
@@ -167,8 +18,6 @@ void filter_rule_free(FilterRule* rule) {
free(rule); free(rule);
} }
/* ---- Ordered rule lists ---- */
FilterRuleList* filter_rule_list_create(void) { FilterRuleList* filter_rule_list_create(void) {
return calloc(1, sizeof(FilterRuleList)); return calloc(1, sizeof(FilterRuleList));
} }
@@ -190,28 +39,42 @@ bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) {
return true; return true;
} }
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err,
size_t err_size) {
FilterRule* rule = filter_rule_parse(line, err, err_size);
if (!rule)
return false;
if (!filter_rule_list_add(list, rule)) {
filter_rule_free(rule);
snprintf(err, err_size, "memory allocation failed");
return false;
}
return true;
}
void filter_rule_list_free(FilterRuleList* list) { void filter_rule_list_free(FilterRuleList* list) {
if (!list) if (!list)
return; return;
for (int i = 0; i < list->count; i++) for (int i = 0; i < list->count; i++)
filter_rule_free(list->items[i]); filter_rule_free(list->items[i]);
for (int i = 0; i < list->dir_merge_count; i++)
free(list->dir_merge_names[i]);
free(list->dir_merge_names);
free(list->items); free(list->items);
free(list); free(list);
} }
/* Register a per-directory merge-file basename (for "dir-merge NAME"/": NAME"
* and -F's .rsync-filter). Duplicate names are ignored. */
bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name) {
if (!list || !name || name[0] == '\0')
return false;
for (int i = 0; i < list->dir_merge_count; i++) {
if (strcmp(list->dir_merge_names[i], name) == 0)
return true;
}
if (list->dir_merge_count == list->dir_merge_capacity) {
int new_cap = list->dir_merge_capacity > 0 ? list->dir_merge_capacity * 2 : 4;
char** grown = realloc(list->dir_merge_names, (size_t)new_cap * sizeof(char*));
if (!grown)
return false;
list->dir_merge_names = grown;
list->dir_merge_capacity = new_cap;
}
char* dup = str_dup(name);
if (!dup)
return false;
list->dir_merge_names[list->dir_merge_count++] = dup;
return true;
}
static bool set_rule_owner(FilterRule* rule, const char* owner) { static bool set_rule_owner(FilterRule* rule, const char* owner) {
char* dup = str_dup(owner ? owner : ""); char* dup = str_dup(owner ? owner : "");
if (!dup) if (!dup)
@@ -221,7 +84,311 @@ static bool set_rule_owner(FilterRule* rule, const char* owner) {
return true; return true;
} }
/* ---- CVS default excludes (-C) ---- */ /* ---- Rule parsing ---- */
/* A short rule prefix is a single character; a long rule name is alphabetic
* (with '-'). `is_short` distinguishes the modifier-attachment rules. */
typedef enum {
RULE_KIND_EXCLUDE,
RULE_KIND_INCLUDE,
RULE_KIND_HIDE,
RULE_KIND_SHOW,
RULE_KIND_PROTECT,
RULE_KIND_RISK,
RULE_KIND_MERGE,
RULE_KIND_DIR_MERGE,
RULE_KIND_CLEAR,
RULE_KIND_UNKNOWN,
} RuleKind;
static bool short_rule_char(char c, RuleKind* kind) {
switch (c) {
case '-':
*kind = RULE_KIND_EXCLUDE;
return true;
case '+':
*kind = RULE_KIND_INCLUDE;
return true;
case 'H':
*kind = RULE_KIND_HIDE;
return true;
case 'S':
*kind = RULE_KIND_SHOW;
return true;
case 'P':
*kind = RULE_KIND_PROTECT;
return true;
case 'R':
*kind = RULE_KIND_RISK;
return true;
case '.':
*kind = RULE_KIND_MERGE;
return true;
case ':':
*kind = RULE_KIND_DIR_MERGE;
return true;
case '!':
*kind = RULE_KIND_CLEAR;
return true;
default:
return false;
}
}
static bool long_rule_name(const char* name, size_t len, RuleKind* kind) {
struct {
const char* word;
RuleKind kind;
} table[] = {
{"exclude", RULE_KIND_EXCLUDE}, {"include", RULE_KIND_INCLUDE},
{"hide", RULE_KIND_HIDE}, {"show", RULE_KIND_SHOW},
{"protect", RULE_KIND_PROTECT}, {"risk", RULE_KIND_RISK},
{"merge", RULE_KIND_MERGE}, {"dir-merge", RULE_KIND_DIR_MERGE},
{"clear", RULE_KIND_CLEAR},
};
for (size_t i = 0; i < sizeof(table) / sizeof(table[0]); i++) {
if (strlen(table[i].word) == len && strncmp(name, table[i].word, len) == 0) {
*kind = table[i].kind;
return true;
}
}
return false;
}
static bool is_modifier_char(char c) {
return c == 's' || c == 'r' || c == 'p' || c == 'x' || c == '/' || c == '!' || c == 'C';
}
/* Parse "RULE[,MODIFIERS] [PATTERN]". On success `kind`, `sides`,
* `sides_explicit`, `negate`, `anchored_mod`, `perishable`, `xattr`,
* `cvs_inject` and the pattern span (`pat_start`/`pat_len`, possibly 0 for
* merge/clear) are filled. Returns true on success. */
static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides,
bool* sides_explicit, bool* negate, bool* anchored_mod,
bool* perishable, bool* xattr, bool* cvs_inject,
const char** pat_start, size_t* pat_len) {
const char* p = text;
*sides = FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER;
*sides_explicit = false;
*negate = false;
*anchored_mod = false;
*perishable = false;
*xattr = false;
*cvs_inject = false;
*pat_start = NULL;
*pat_len = 0;
bool is_short = false;
if (short_rule_char(*p, kind)) {
is_short = true;
p++;
} else {
const char* name_start = p;
while (isalpha((unsigned char)*p) || *p == '-')
p++;
size_t name_len = (size_t)(p - name_start);
if (name_len == 0 || !long_rule_name(name_start, name_len, kind))
return false;
/* A long name must be followed by a separator, a comma or the end. */
if (*p != '\0' && *p != ',' && *p != ' ' && *p != '_')
return false;
}
/* Modifiers: long names require a comma; short names may attach directly.
Only commit a modifier run that terminates at a separator or the end, so a
pattern such as "*.tmp" written as "-*.tmp" is not mistaken for modifiers. */
const char* mod_start = p;
const char* mod_end = p;
if (*p == ',') {
p++;
mod_start = p;
while (is_modifier_char(*p))
p++;
mod_end = p;
} else if (is_short) {
const char* scan = p;
while (is_modifier_char(*scan))
scan++;
if (*scan == '\0' || *scan == ' ' || *scan == '_') {
mod_start = p;
mod_end = scan;
p = scan;
}
}
for (const char* m = mod_start; m < mod_end; m++) {
switch (*m) {
case 's':
*sides = FILTER_SIDE_SENDER;
*sides_explicit = true;
break;
case 'r':
*sides = FILTER_SIDE_RECEIVER;
*sides_explicit = true;
break;
case '!':
*negate = true;
break;
case '/':
*anchored_mod = true;
break;
case 'p':
*perishable = true;
break;
case 'x':
*xattr = true;
break;
case 'C':
*cvs_inject = true;
break;
default:
break;
}
}
/* A single space or underscore separates the rule/modifiers from the
pattern; further spaces/underscores belong to the pattern. */
const char* pat = p;
if (*pat == ' ' || *pat == '_')
pat++;
/* Trim a trailing newline/CR (the caller may pass a raw file line). */
*pat_start = pat;
*pat_len = strlen(pat);
while (*pat_len > 0 && (pat[*pat_len - 1] == '\n' || pat[*pat_len - 1] == '\r'))
(*pat_len)--;
return true;
}
FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err,
size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
if (!line)
return NULL;
const char* p = line;
while (*p == ' ' || *p == '\t')
p++;
if (*p == '\0' || *p == '\n' || *p == '\r') {
snprintf(err, err_size, "empty filter rule");
return NULL;
}
RuleKind kind = RULE_KIND_UNKNOWN;
unsigned sides;
bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject;
const char* pat;
size_t pat_len;
if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable,
&xattr, &cvs_inject, &pat, &pat_len)) {
snprintf(err, err_size, "unrecognized filter rule syntax");
return NULL;
}
if (cvs_inject) {
/* The C modifier expands to the CVS defaults in place; the rule itself
carries no pattern and is handled by the caller. */
snprintf(err, err_size, "the C modifier is handled by the rule-list parser");
return NULL;
}
if (kind == RULE_KIND_MERGE || kind == RULE_KIND_DIR_MERGE) {
snprintf(err, err_size, "merge/dir-merge rules are handled by the rule-list parser");
return NULL;
}
if (kind == RULE_KIND_CLEAR) {
if (pat_len != 0) {
snprintf(err, err_size, "clear takes no pattern");
return NULL;
}
FilterRule* rule = calloc(1, sizeof(FilterRule));
if (!rule) {
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
rule->action = FILTER_ACTION_NONE; /* clear marker: no pattern */
rule->sides = 0;
return rule;
}
FilterAction action;
switch (kind) {
case RULE_KIND_INCLUDE:
case RULE_KIND_SHOW:
case RULE_KIND_RISK:
action = FILTER_ACTION_INCLUDE;
break;
default:
action = FILTER_ACTION_EXCLUDE;
break;
}
if (kind == RULE_KIND_HIDE)
sides = FILTER_SIDE_SENDER;
else if (kind == RULE_KIND_SHOW)
sides = FILTER_SIDE_SENDER;
else if (kind == RULE_KIND_PROTECT)
sides = FILTER_SIDE_RECEIVER;
else if (kind == RULE_KIND_RISK)
sides = FILTER_SIDE_RECEIVER;
if (kind == RULE_KIND_HIDE || kind == RULE_KIND_SHOW || kind == RULE_KIND_PROTECT ||
kind == RULE_KIND_RISK)
sides_explicit = true;
/* --delete-excluded turns an unqualified (no explicit s/r) rule into a
sender-side-only rule, so it no longer protects the receiver. */
if (opts && opts->delete_excluded && !sides_explicit)
sides = FILTER_SIDE_SENDER;
if (pat_len == 0) {
snprintf(err, err_size, "filter rule has no pattern");
return NULL;
}
bool anchored = anchored_mod;
const char* pat_begin = pat;
if (*pat_begin == '/') {
anchored = true;
pat_begin++;
/* Drop the spaces that could follow the anchor in the "-/ foo" form. */
while (*pat_begin == ' ' || *pat_begin == '\t')
pat_begin++;
pat_len = strlen(pat_begin);
while (pat_len > 0 && (pat_begin[pat_len - 1] == '\n' || pat_begin[pat_len - 1] == '\r'))
pat_len--;
}
if (pat_len == 0) {
snprintf(err, err_size, "filter rule has no pattern after '/' anchor");
return NULL;
}
bool dir_only = false;
if (pat_len > 1 && pat_begin[pat_len - 1] == '/') {
dir_only = true;
pat_len--;
}
if (pat_len == 0) {
snprintf(err, err_size, "filter rule has no pattern");
return NULL;
}
FilterRule* rule = calloc(1, sizeof(FilterRule));
if (!rule) {
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
rule->pattern = malloc(pat_len + 1);
if (!rule->pattern) {
free(rule);
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
memcpy(rule->pattern, pat_begin, pat_len);
rule->pattern[pat_len] = '\0';
rule->action = action;
rule->sides = sides;
rule->anchored = anchored;
rule->dir_only = dir_only;
rule->negate = negate;
rule->perishable = perishable;
(void)xattr; /* xattr-name rules never match file/dir names; accepted/ignored */
return rule;
}
/* ---- CVS default excludes (-C and the C modifier) ---- */
typedef struct { typedef struct {
const char* pattern; const char* pattern;
@@ -240,12 +407,13 @@ static const CvsDefaultRule CVS_DEFAULTS[] = {
{".svn/", true}, {".git/", true}, {".hg/", true}, {".bzr/", true}, {".svn/", true}, {".git/", true}, {".hg/", true}, {".bzr/", true},
}; };
static bool cvs_rule_list_append(FilterRuleList* list) { static bool filter_list_append_cvs(FilterRuleList* list, unsigned sides) {
for (size_t i = 0; i < sizeof(CVS_DEFAULTS) / sizeof(CVS_DEFAULTS[0]); i++) { for (size_t i = 0; i < sizeof(CVS_DEFAULTS) / sizeof(CVS_DEFAULTS[0]); i++) {
FilterRule* rule = calloc(1, sizeof(FilterRule)); FilterRule* rule = calloc(1, sizeof(FilterRule));
if (!rule) if (!rule)
return false; return false;
rule->action = FILTER_ACTION_EXCLUDE; rule->action = FILTER_ACTION_EXCLUDE;
rule->sides = sides;
rule->dir_only = CVS_DEFAULTS[i].dir_only; rule->dir_only = CVS_DEFAULTS[i].dir_only;
size_t plen = strlen(CVS_DEFAULTS[i].pattern); size_t plen = strlen(CVS_DEFAULTS[i].pattern);
if (rule->dir_only && plen > 0 && CVS_DEFAULTS[i].pattern[plen - 1] == '/') if (rule->dir_only && plen > 0 && CVS_DEFAULTS[i].pattern[plen - 1] == '/')
@@ -269,8 +437,167 @@ static bool cvs_rule_list_append(FilterRuleList* list) {
return true; return true;
} }
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude, #define FILTER_MAX_MERGE_DEPTH 16
static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line,
const FilterParseOptions* opts, const char* base_dir,
int depth, char* err, size_t err_size);
/* Read a merge file and splice its rules into `list`. A relative path is
* resolved below `base_dir` when given, else used as-is (rsync resolves a
* command-line merge file relative to the current directory). */
static bool filter_list_merge_file(FilterRuleList* list, const char* name,
const FilterParseOptions* opts, const char* base_dir, int depth,
char* err, size_t err_size) { char* err, size_t err_size) {
if (name[0] == '\0') {
snprintf(err, err_size, "merge requires a filename");
return false;
}
char* path =
(base_dir && base_dir[0] && name[0] != '/') ? path_cat(base_dir, name) : str_dup(name);
if (!path) {
snprintf(err, err_size, "memory allocation failed");
return false;
}
FILE* fp = fopen(path, "r");
if (!fp) {
snprintf(err, err_size, "could not read merge file '%s': %s", path, strerror(errno));
free(path);
return false;
}
char* line = NULL;
size_t cap = 0;
bool ok = true;
while (true) {
ssize_t n = utils_getdelim_bounded(fp, &line, &cap, '\n', UTILS_MAX_LINE_LEN);
if (n < 0) {
snprintf(err, err_size, "error reading merge file '%s'", path);
ok = false;
break;
}
if (n == 0)
break;
const char* lp = line;
while (*lp == ' ' || *lp == '\t')
lp++;
if (*lp == '\0' || *lp == '\n' || *lp == '\r' || *lp == '#')
continue;
if (!filter_list_parse_append_depth(list, lp, opts, base_dir, depth + 1, err, err_size)) {
ok = false;
break;
}
}
free(line);
fclose(fp);
free(path);
return ok;
}
/* Parse one line and append/merge it into `list`. Handles clear, merge and
* dir-merge at the list level. */
static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line,
const FilterParseOptions* opts, const char* base_dir,
int depth, char* err, size_t err_size) {
if (depth > FILTER_MAX_MERGE_DEPTH) {
snprintf(err, err_size, "merge files nested too deeply");
return false;
}
const char* p = line;
while (*p == ' ' || *p == '\t')
p++;
if (*p == '\0' || *p == '\n' || *p == '\r')
return true;
RuleKind kind = RULE_KIND_UNKNOWN;
unsigned sides;
bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject;
const char* pat;
size_t pat_len;
if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable,
&xattr, &cvs_inject, &pat, &pat_len)) {
snprintf(err, err_size, "unrecognized filter rule syntax: %s", p);
return false;
}
(void)sides_explicit;
(void)negate;
(void)anchored_mod;
(void)perishable;
(void)xattr;
if (cvs_inject) {
/* "C" injects the CVS defaults in place; no pattern is expected. */
return filter_list_append_cvs(list, sides);
}
if (kind == RULE_KIND_CLEAR) {
if (pat_len != 0) {
snprintf(err, err_size, "clear takes no pattern");
return false;
}
for (int i = 0; i < list->count; i++)
filter_rule_free(list->items[i]);
list->count = 0;
return true;
}
if (kind == RULE_KIND_MERGE) {
if (pat_len == 0) {
snprintf(err, err_size, "merge requires a filename");
return false;
}
char* name = malloc(pat_len + 1);
if (!name) {
snprintf(err, err_size, "memory allocation failed");
return false;
}
memcpy(name, pat, pat_len);
name[pat_len] = '\0';
bool ok = filter_list_merge_file(list, name, opts, base_dir, depth, err, err_size);
free(name);
return ok;
}
if (kind == RULE_KIND_DIR_MERGE) {
if (pat_len == 0) {
snprintf(err, err_size, "dir-merge requires a filename");
return false;
}
char* name = malloc(pat_len + 1);
if (!name) {
snprintf(err, err_size, "memory allocation failed");
return false;
}
memcpy(name, pat, pat_len);
name[pat_len] = '\0';
bool ok = filter_rule_list_add_dir_merge(list, name);
free(name);
if (!ok) {
snprintf(err, err_size, "memory allocation failed");
return false;
}
return true;
}
FilterRule* rule = filter_rule_parse(p, opts, err, err_size);
if (!rule)
return false;
if (!filter_rule_list_add(list, rule)) {
filter_rule_free(rule);
snprintf(err, err_size, "memory allocation failed");
return false;
}
return true;
}
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line,
const FilterParseOptions* opts, const char* merge_base_dir,
char* err, size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
if (!list)
return false;
return filter_list_parse_append_depth(list, line, opts, merge_base_dir, 0, err, err_size);
}
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
bool delete_excluded, char* err, size_t err_size) {
if (err && err_size > 0) if (err && err_size > 0)
err[0] = '\0'; err[0] = '\0';
FilterRuleList* list = filter_rule_list_create(); FilterRuleList* list = filter_rule_list_create();
@@ -278,28 +605,16 @@ FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count,
snprintf(err, err_size, "memory allocation failed"); snprintf(err, err_size, "memory allocation failed");
return NULL; return NULL;
} }
FilterParseOptions opts = {.delete_excluded = delete_excluded, .cvs_exclude = cvs_exclude};
for (int i = 0; i < rule_count; i++) { for (int i = 0; i < rule_count; i++) {
if (!rule_texts || !rule_texts[i]) if (!rule_texts || !rule_texts[i])
continue; continue;
FilterRule* rule = filter_rule_parse(rule_texts[i], err, err_size); if (!filter_rule_list_parse_append(list, rule_texts[i], &opts, NULL, err, err_size)) {
if (!rule) {
filter_rule_list_free(list); filter_rule_list_free(list);
return NULL; return NULL;
} }
if (!set_rule_owner(rule, "")) {
filter_rule_free(rule);
filter_rule_list_free(list);
snprintf(err, err_size, "memory allocation failed");
return NULL;
} }
if (!filter_rule_list_add(list, rule)) { if (cvs_exclude && !filter_list_append_cvs(list, FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER)) {
filter_rule_free(rule);
filter_rule_list_free(list);
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
}
if (cvs_exclude && !cvs_rule_list_append(list)) {
filter_rule_list_free(list); filter_rule_list_free(list);
snprintf(err, err_size, "memory allocation failed"); snprintf(err, err_size, "memory allocation failed");
return NULL; return NULL;
@@ -307,38 +622,36 @@ FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count,
return list; return list;
} }
/* ---- Per-directory .rsync-filter files ---- */ /* ---- Per-directory merge files ---- */
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists, bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name,
const char* owner_rel, const FilterParseOptions* opts, bool* exists,
char* err, size_t err_size) { char* err, size_t err_size) {
if (err && err_size > 0) if (err && err_size > 0)
err[0] = '\0'; err[0] = '\0';
if (exists) if (exists)
*exists = false; *exists = false;
char* filter_path = path_cat(dir_path, ".rsync-filter"); if (!list)
return false;
char* filter_path = path_cat(dir_path, name);
if (!filter_path) { if (!filter_path) {
snprintf(err, err_size, "memory allocation failed"); snprintf(err, err_size, "memory allocation failed");
return NULL; return false;
} }
FILE* fp = fopen(filter_path, "r"); FILE* fp = fopen(filter_path, "r");
free(filter_path); free(filter_path);
if (!fp) { if (!fp) {
if (errno == ENOENT || errno == ENOTDIR) if (errno == ENOENT || errno == ENOTDIR)
return filter_rule_list_create(); return true;
char* escaped_dir = output_escape(dir_path, log_get_8_bit_output()); char* escaped_dir = output_escape(dir_path, log_get_8_bit_output());
log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s", log_message(LOG_LEVEL_WARNING, "Could not read %s in %s: %s", name,
escaped_dir ? escaped_dir : "<allocation failed>", strerror(errno)); escaped_dir ? escaped_dir : "<allocation failed>", strerror(errno));
free(escaped_dir); free(escaped_dir);
return filter_rule_list_create(); return true;
} }
if (exists) if (exists)
*exists = true; *exists = true;
FilterRuleList* list = filter_rule_list_create(); int rules_before = list->count;
if (!list) {
fclose(fp);
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
char* line = NULL; char* line = NULL;
size_t line_cap = 0; size_t line_cap = 0;
bool ok = true; bool ok = true;
@@ -346,9 +659,9 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, '\n', UTILS_MAX_LINE_LEN); ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, '\n', UTILS_MAX_LINE_LEN);
if (n < 0) { if (n < 0) {
if (errno == EFBIG) { if (errno == EFBIG) {
snprintf(err, err_size, "line in .rsync-filter exceeds %d bytes", (int)UTILS_MAX_LINE_LEN); snprintf(err, err_size, "line in %s exceeds %d bytes", name, (int)UTILS_MAX_LINE_LEN);
} else { } else {
snprintf(err, err_size, "error reading .rsync-filter: %s", strerror(errno)); snprintf(err, err_size, "error reading %s: %s", name, strerror(errno));
} }
ok = false; ok = false;
break; break;
@@ -360,20 +673,9 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
p++; p++;
if (*p == '\0' || *p == '\n' || *p == '\r' || *p == '#') if (*p == '\0' || *p == '\n' || *p == '\r' || *p == '#')
continue; continue;
FilterRule* rule = filter_rule_parse(p, err, err_size); /* Merge files inside a per-directory file resolve relative to that
if (!rule) { directory. */
ok = false; if (!filter_list_parse_append_depth(list, p, opts, dir_path, 0, err, err_size)) {
break;
}
if (!set_rule_owner(rule, owner_rel)) {
filter_rule_free(rule);
snprintf(err, err_size, "memory allocation failed");
ok = false;
break;
}
if (!filter_rule_list_add(list, rule)) {
filter_rule_free(rule);
snprintf(err, err_size, "memory allocation failed");
ok = false; ok = false;
break; break;
} }
@@ -381,12 +683,43 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
free(line); free(line);
fclose(fp); fclose(fp);
if (!ok) { if (!ok) {
/* Drop only the rules this file appended, leaving the caller's earlier
content untouched. */
for (int i = rules_before; i < list->count; i++)
filter_rule_free(list->items[i]);
list->count = rules_before;
return false;
}
for (int i = rules_before; i < list->count; i++) {
if (!set_rule_owner(list->items[i], owner_rel)) {
snprintf(err, err_size, "memory allocation failed");
return false;
}
}
return true;
}
FilterRuleList* filter_file_read_named(const char* dir_path, const char* name,
const char* owner_rel, const FilterParseOptions* opts,
bool* exists, char* err, size_t err_size) {
FilterRuleList* list = filter_rule_list_create();
if (!list) {
if (err && err_size > 0)
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
if (!filter_file_append(list, dir_path, name, owner_rel, opts, exists, err, err_size)) {
filter_rule_list_free(list); filter_rule_list_free(list);
return NULL; return NULL;
} }
return list; return list;
} }
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
char* err, size_t err_size) {
return filter_file_read_named(dir_path, ".rsync-filter", owner_rel, NULL, exists, err, err_size);
}
/* ---- Rule matching ---- */ /* ---- Rule matching ---- */
/* Match a pattern that contains '/' (non-anchored) against the end of the /* Match a pattern that contains '/' (non-anchored) against the end of the
@@ -402,10 +735,10 @@ static bool glob_suffix_match(const char* pattern, const char* str) {
} }
static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, const char* leaf, static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, const char* leaf,
bool is_dir) { bool is_dir, unsigned side) {
if (!rule || !rule->pattern) if (!rule || !rule->pattern)
return FILTER_ACTION_NONE; return FILTER_ACTION_NONE;
if (rule->dir_only && !is_dir) if (!(rule->sides & side))
return FILTER_ACTION_NONE; return FILTER_ACTION_NONE;
/* A rule applies only to entries below its owner directory. */ /* A rule applies only to entries below its owner directory. */
const char* rel2 = rel_path; const char* rel2 = rel_path;
@@ -420,24 +753,36 @@ static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, c
if (rel2[0] == '\0') if (rel2[0] == '\0')
return FILTER_ACTION_NONE; return FILTER_ACTION_NONE;
bool matched; bool matched;
if (rule->anchored) { if (rule->dir_only && !is_dir)
matched = false;
else if (rule->anchored)
matched = glob_match(rule->pattern, rel2); matched = glob_match(rule->pattern, rel2);
} else if (strchr(rule->pattern, '/') != NULL) { else if (strchr(rule->pattern, '/') != NULL)
matched = glob_suffix_match(rule->pattern, rel2); matched = glob_suffix_match(rule->pattern, rel2);
} else { else
matched = glob_match(rule->pattern, leaf); matched = glob_match(rule->pattern, leaf);
} if (rule->negate)
return matched ? rule->action : FILTER_ACTION_NONE; matched = !matched;
if (!matched)
return FILTER_ACTION_NONE;
if (side == FILTER_SIDE_RECEIVER)
return rule->action == FILTER_ACTION_EXCLUDE ? FILTER_ACTION_PROTECT : FILTER_ACTION_RISK;
return rule->action;
} }
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path,
bool is_dir) { const char* leaf, bool is_dir, unsigned side) {
if (!list) if (!list)
return FILTER_ACTION_NONE; return FILTER_ACTION_NONE;
for (int i = 0; i < list->count; i++) { for (int i = 0; i < list->count; i++) {
FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir); FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir, side);
if (action != FILTER_ACTION_NONE) if (action != FILTER_ACTION_NONE)
return action; return action;
} }
return FILTER_ACTION_NONE; return FILTER_ACTION_NONE;
} }
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
bool is_dir) {
return filter_rules_apply_side(list, rel_path, leaf, is_dir, FILTER_SIDE_SENDER);
}
+88 -34
View File
@@ -4,38 +4,51 @@
#include <stdbool.h> #include <stdbool.h>
#include <stddef.h> #include <stddef.h>
/* rsync-style filter rule engine (client-side file selection). /* rsync-style filter rule engine (client-side file selection and the
* receiver-side protection set it feeds).
* *
* Supported rule syntax (documented subset): * Rule syntax (see the rsync man page FILTER RULES section):
* [+|-] [anchored '/' prefix] pattern [trailing '/' for dir-only] * RULE [PATTERN_OR_FILENAME]
* * RULE,MODIFIERS [PATTERN_OR_FILENAME]
* "+ PATTERN" include rule (first match wins) * Short RULE names may attach MODIFIERS directly ("-sr foo"); the long name
* "- PATTERN" exclude rule * form requires the comma. The pattern/filename is separated from the rule by
* "PATTERN" implicit exclude rule (rsync default) * one space or underscore. Rule names:
* "include PATTERN" / "exclude PATTERN" word forms * exclude/- exclude (by default both sender-hide and receiver-protect)
* leading '/' after the +/- anchors the pattern to its owner directory * include/+ include (by default both sender-show and receiver-risk)
* (the transfer root for command-line/-C rules, the directory that * hide/H sender-only exclude
* contains a .rsync-filter file for per-directory rules) * show/S sender-only include
* a trailing '/' makes the rule match directories only * protect/P receiver-only exclude (protect from deletion)
* * risk/R receiver-only include (allow deletion)
* Rejected explicitly (no silent no-ops): the rsync merge/dir-merge/list-clear * merge/. read a client-side merge file for more rules
* shorthands written as a rule that starts with ':' or '.' or '!', the * dir-merge/: per-directory merge file (registered for the scanner)
* merge/dir-merge/hide/show/protect/risk/clear words, and every include/exclude * clear/! clear the current rule list (takes no argument)
* rule modifier other than '/' (! C s r p x). The pattern must be separated * Modifiers: '/' absolute anchor, '!' negate match, 'C' inject CVS defaults,
* from +/- by a space (or a single '/' anchor), exactly like rsync's * 's' sender side, 'r' receiver side, 'p' perishable, 'x' xattr name rule.
* "-s foo"/"-p ..." modifier syntax is refused. * A trailing '/' makes a pattern match directories only. A leading '/' anchors
* the pattern to its owner directory.
*/ */
typedef enum { typedef enum {
FILTER_ACTION_NONE = 0, /* no rule matched */ FILTER_ACTION_NONE = 0, /* no rule matched */
FILTER_ACTION_EXCLUDE = -1, FILTER_ACTION_EXCLUDE = -1,
FILTER_ACTION_INCLUDE = 1 FILTER_ACTION_INCLUDE = 1,
/* Receiver-side-only verdicts: the entry is transferred but its destination
* mirror is protected from --delete (PROTECT) or explicitly left at risk
* (RISK). */
FILTER_ACTION_PROTECT = 2,
FILTER_ACTION_RISK = 3,
} FilterAction; } FilterAction;
#define FILTER_SIDE_SENDER 1u
#define FILTER_SIDE_RECEIVER 2u
typedef struct { typedef struct {
FilterAction action; FilterAction action; /* EXCLUDE or INCLUDE (the base pattern action) */
unsigned sides; /* FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER */
bool anchored; /* pattern anchored to the rule's owner directory */ bool anchored; /* pattern anchored to the rule's owner directory */
bool dir_only; /* pattern had a trailing '/': matches directories only */ bool dir_only; /* pattern had a trailing '/': matches directories only */
bool negate; /* '!' modifier: match succeeds when the pattern does not */
bool perishable; /* 'p' modifier (ignored in deleted directories) */
char* owner; /* owning directory rel path ("" == transfer root) */ char* owner; /* owning directory rel path ("" == transfer root) */
char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */ char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */
} FilterRule; } FilterRule;
@@ -44,19 +57,41 @@ typedef struct {
FilterRule** items; /* owned array of rule pointers */ FilterRule** items; /* owned array of rule pointers */
int count; int count;
int capacity; int capacity;
/* Per-directory merge-file basenames registered by "dir-merge NAME"/": NAME"
* or by -F (.rsync-filter). Owned strings; the scanner reads each name in
* every directory it traverses. */
char** dir_merge_names;
int dir_merge_count;
int dir_merge_capacity;
} FilterRuleList; } FilterRuleList;
/* Context needed while parsing a rule list (merge files, --delete-excluded). */
typedef struct {
bool delete_excluded; /* --delete-excluded: default sides become sender-only */
bool cvs_exclude; /* -C: expand the CVS default excludes */
} FilterParseOptions;
/* Parse a single filter-rule line (no trailing newline required). Returns an /* Parse a single filter-rule line (no trailing newline required). Returns an
* owned rule, or NULL on unsupported/invalid syntax with a message in `err`. */ * owned rule, or NULL on unsupported/invalid syntax with a message in `err`.
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size); * `opts` may be NULL (no merge expansion / no delete-excluded). */
FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err,
size_t err_size);
void filter_rule_free(FilterRule* rule); void filter_rule_free(FilterRule* rule);
FilterRuleList* filter_rule_list_create(void); FilterRuleList* filter_rule_list_create(void);
/* Append a fully-parsed rule (takes ownership). Returns false on OOM. */ /* Append a fully-parsed rule (takes ownership). Returns false on OOM. */
bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule); bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule);
/* Parse `line` and append it. Returns false and fills `err` on bad syntax. */ /* Register a per-directory merge-file basename (idempotent). Returns false on
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err, * OOM. Used by the scanner to read custom "dir-merge" files. */
size_t err_size); bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name);
/* Parse `line` and append it. Handles "clear"/"!" (resets the list), "merge
* FILE"/". FILE" (splices the file's rules) and "dir-merge NAME"/": NAME"
* (registers a per-directory filename). Returns false and fills `err` on bad
* syntax or an unreadable merge file. `merge_base_dir` resolves a relative
* merge-file path (NULL means the process working directory). */
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line,
const FilterParseOptions* opts, const char* merge_base_dir,
char* err, size_t err_size);
void filter_rule_list_free(FilterRuleList* list); void filter_rule_list_free(FilterRuleList* list);
/* Build the command-line filter set: `rule_texts` (--filter=RULE in the order /* Build the command-line filter set: `rule_texts` (--filter=RULE in the order
@@ -64,19 +99,38 @@ void filter_rule_list_free(FilterRuleList* list);
* cvs_exclude is true. All rules are owned by "" (the transfer root). * cvs_exclude is true. All rules are owned by "" (the transfer root).
* Returns NULL on unsupported rule text (message in `err`). */ * Returns NULL on unsupported rule text (message in `err`). */
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude, FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
bool delete_excluded, char* err, size_t err_size);
/* Read "<dir_path>/<name>" and return its rules, each owned by `owner_rel`. A
* missing file yields an empty list with *exists=false; an unreadable file is
* treated as missing. Returns NULL only on parse or allocation failure
* (message in `err`). `opts` may be NULL. */
FilterRuleList* filter_file_read_named(const char* dir_path, const char* name,
const char* owner_rel, const FilterParseOptions* opts,
bool* exists, char* err, size_t err_size);
/* Append the rules of "<dir_path>/<name>" into an existing list (each owned by
* `owner_rel`). A missing file yields *exists=false and no error. Returns
* false only on parse/allocation failure (message in `err`). */
bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name,
const char* owner_rel, const FilterParseOptions* opts, bool* exists,
char* err, size_t err_size); char* err, size_t err_size);
/* Read "<dir_path>/.rsync-filter" and return its rules, each owned by /* filter_file_read_named with the default ".rsync-filter" name. */
* `owner_rel`. A missing file yields an empty list with *exists=false; an
* unreadable file is treated as missing. Returns NULL only on parse or
* allocation failure (message in `err`). */
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists, FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
char* err, size_t err_size); char* err, size_t err_size);
/* Evaluate an entry against one ordered rule list. Returns FILTER_ACTION_NONE /* Evaluate an entry against one ordered rule list for one side. Returns
* when no rule matched, otherwise the first matching rule's action. * FILTER_ACTION_NONE when no rule matched, otherwise the first matching rule's
* `rel_path` is the entry's path relative to the transfer root ("" == root), * action (for the receiver side an EXCLUDE is reported as
* `leaf` its final name, `is_dir` whether it is a directory. */ * FILTER_ACTION_PROTECT and an INCLUDE as FILTER_ACTION_RISK). `rel_path` is
* the entry's path relative to the transfer root ("" == root), `leaf` its final
* name, `is_dir` whether it is a directory. */
FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path,
const char* leaf, bool is_dir, unsigned side);
/* Sender-side convenience wrapper (kept for callers/tests that only need the
* transfer decision). */
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
bool is_dir); bool is_dir);
+64 -17
View File
@@ -589,6 +589,67 @@ static int identity_split_chown(const char* value, char** puser, char** pgroup)
return 0; return 0;
} }
/* --chown is rsync's shorthand for "--usermap=*:USER --groupmap=*:GROUP", so a
* name TO value must be resolved on the RECEIVER, not on the sender. Append the
* equivalent map rule (FROM matches every id). The numeric/'*' forms are stored
* numerically exactly as rsync's id_parse/user_to_uid would. Returns 0 on
* success, -1 on a malformed numeric token or allocation failure. */
static int identity_append_chown_rule(Config* config, bool is_group, const char* token) {
IdentityMap rule;
memset(&rule, 0, sizeof(rule));
rule.from = IDENTITY_MATCH_ANY;
rule.from_hi = IDENTITY_MATCH_ANY;
if (strcmp(token, "*") == 0) {
rule.to = IDENTITY_CURRENT;
} else if (identity_all_digits(token[0] == '@' ? token + 1 : token)) {
if (identity_resolve_token(token, is_group, &rule.to) != 0) {
log_message(LOG_LEVEL_ERROR, "--chown numeric id is out of range: %s", token);
return -1;
}
} else {
rule.to = 0;
rule.to_name = str_dup(token);
if (!rule.to_name)
return -1;
}
if (identity_append_rule(is_group ? &config->groupmap : &config->usermap,
is_group ? &config->groupmap_count : &config->usermap_count,
&rule) != 0) {
free(rule.to_name);
log_message(LOG_LEVEL_ERROR, "--chown has too many rules (max %d)", MAX_IDENTITY_MAP);
return -1;
}
return 0;
}
/* Resolve/record one --chown side. The source-side numeric value is kept in
* chown_uid/chown_gid purely as a fallback (the appended map rule resolves the
* name on the receiver and wins); a name that does not exist on the sender is
* accepted and left to receiver-side resolution, matching rsync. */
static int identity_parse_chown_side(Config* config, bool is_group, const char* token) {
if (identity_append_chown_rule(config, is_group, token) != 0)
return -1;
bool numeric = identity_all_digits(token[0] == '@' ? token + 1 : token);
int32_t resolved;
if (identity_resolve_token(token, is_group, &resolved) == 0) {
if (is_group) {
config->chown_gid = resolved;
config->chown_gid_set = true;
} else {
config->chown_uid = resolved;
config->chown_uid_set = true;
}
return 0;
}
if (numeric) {
log_message(LOG_LEVEL_ERROR, "--chown could not resolve numeric id '%s'", token);
return -1;
}
/* Unknown sender-side name: rsync accepts it and resolves it (or warns) on
* the receiver; do the same instead of failing the whole run. */
return 0;
}
int identity_parse_chown(Config* config, const char* value) { int identity_parse_chown(Config* config, const char* value) {
if (!config || !value || *value == '\0') { if (!config || !value || *value == '\0') {
log_message(LOG_LEVEL_ERROR, "--chown requires a value (USER:GROUP, USER, or :GROUP)"); log_message(LOG_LEVEL_ERROR, "--chown requires a value (USER:GROUP, USER, or :GROUP)");
@@ -627,33 +688,19 @@ int identity_parse_chown(Config* config, const char* value) {
if (*user == '\0') { if (*user == '\0') {
log_message(LOG_LEVEL_ERROR, "--chown requires a user or group (got '%s')", value); log_message(LOG_LEVEL_ERROR, "--chown requires a user or group (got '%s')", value);
ret = -1; ret = -1;
} else if (identity_resolve_token(user, false, &config->chown_uid) != 0) { } else if (identity_parse_chown_side(config, false, user) != 0) {
log_message(LOG_LEVEL_ERROR,
"--chown could not resolve user '%s' (use a name that exists "
"on the source, '*', or @N)",
value);
ret = -1; ret = -1;
} else {
config->chown_uid_set = true;
} }
} else { } else {
/* --chown=USER:GROUP, --chown=:GROUP, --chown=USER: */ /* --chown=USER:GROUP, --chown=:GROUP, --chown=USER: */
if (*user != '\0') { if (*user != '\0' && identity_parse_chown_side(config, false, user) != 0) {
if (identity_resolve_token(user, false, &config->chown_uid) != 0) {
log_message(LOG_LEVEL_ERROR, "--chown could not resolve user '%s'", value);
ret = -1; ret = -1;
goto done; goto done;
} }
config->chown_uid_set = true; if (*group != '\0' && identity_parse_chown_side(config, true, group) != 0) {
}
if (*group != '\0') {
if (identity_resolve_token(group, true, &config->chown_gid) != 0) {
log_message(LOG_LEVEL_ERROR, "--chown could not resolve group '%s'", value);
ret = -1; ret = -1;
goto done; goto done;
} }
config->chown_gid_set = true;
}
if (!*user && !*group) { if (!*user && !*group) {
log_message(LOG_LEVEL_ERROR, "--chown must set a user, a group, or both (got '%s')", value); log_message(LOG_LEVEL_ERROR, "--chown must set a user, a group, or both (got '%s')", value);
ret = -1; ret = -1;
+31 -4
View File
@@ -4963,15 +4963,20 @@ class TestFuzzy:
"no-candidate fuzzy run should have sent the whole file" "no-candidate fuzzy run should have sent the whole file"
def test_dissimilar_sibling_is_not_used(self, shared_server): def test_dissimilar_sibling_is_not_used(self, shared_server):
# The destination holds a large sibling whose basename is too different # A sibling whose basename is too different from the incoming name is
# from the incoming name; the name gate must reject it and fall back to # rejected by rsync's fuzzy distance window (the length gap exceeds
# a whole-file transfer. # 25), so the run falls back to a whole-file transfer. A distinct
# mtime keeps rsync's exact size+mtime first pass from accepting it.
source, dest = self._prepare("dissim") source, dest = self._prepare("dissim")
old_bytes, new_bytes = _random_payloads() old_bytes, new_bytes = _random_payloads()
self._seed_dest(source, dest, {"totally-unrelated-notes.bin": old_bytes}, long_name = "totally-unrelated-notes-with-a-very-long-name.bin"
self._seed_dest(source, dest, {long_name: old_bytes},
shared_server.port) shared_server.port)
with open(os.path.join(source, self.NEW_NAME), "wb") as fh: with open(os.path.join(source, self.NEW_NAME), "wb") as fh:
fh.write(new_bytes) fh.write(new_bytes)
received_dir = get_dest_received_dir(dest, source)
os.utime(os.path.join(received_dir, long_name), (self.TS, self.TS))
os.utime(os.path.join(source, self.NEW_NAME), (self.TS + 100000, self.TS + 100000))
result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port)
assert result.returncode == 0, \ assert result.returncode == 0, \
f"--fuzzy dissimilar-sibling run failed: {(result.stderr or result.stdout)[:300]}" f"--fuzzy dissimilar-sibling run failed: {(result.stderr or result.stdout)[:300]}"
@@ -4980,6 +4985,28 @@ class TestFuzzy:
assert proxy.client_to_server > len(new_bytes) // 2, \ assert proxy.client_to_server > len(new_bytes) // 2, \
"a dissimilar-named sibling must not be used as a fuzzy basis" "a dissimilar-named sibling must not be used as a fuzzy basis"
def test_exact_size_mtime_sibling_is_used(self, shared_server):
# rsync's fuzzy first pass accepts a sibling with an exact size+mtime
# match regardless of how unrelated its name is (its content is almost
# certainly the same).
source, dest = self._prepare("exact")
old_bytes, new_bytes = _random_payloads()
self._seed_dest(source, dest, {"unrelated-blob.bin": old_bytes},
shared_server.port)
with open(os.path.join(source, self.NEW_NAME), "wb") as fh:
fh.write(new_bytes)
received_dir = get_dest_received_dir(dest, source)
ts = 1600000000
os.utime(os.path.join(received_dir, "unrelated-blob.bin"), (ts, ts))
os.utime(os.path.join(source, self.NEW_NAME), (ts, ts))
result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port)
assert result.returncode == 0, \
f"--fuzzy exact size+mtime run failed: {(result.stderr or result.stdout)[:300]}"
received = get_dest_received_dir(dest, source)
assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes
assert proxy.client_to_server < len(new_bytes) // 4, \
"an exact size+mtime sibling should be used as a fuzzy basis"
def test_fuzzy_helps_when_dest_holds_an_unsuitable_file(self, shared_server): def test_fuzzy_helps_when_dest_holds_an_unsuitable_file(self, shared_server):
# The destination DOES hold the exact new name, but it is a tiny stale # The destination DOES hold the exact new name, but it is a tiny stale
# file (below the delta engine's minimum, ratio far outside its window), # file (below the delta engine's minimum, ratio far outside its window),
+61 -11
View File
@@ -1044,10 +1044,11 @@ static void test_parse_args_basis_dirs() {
config_delete(cfg); config_delete(cfg);
} }
/* Absolute, escaping, or degenerate basis-dir values must be rejected up /* Escaping or degenerate basis-dir values must be rejected up front (they would
front: they would resolve outside the destination root on the receiver. */ resolve outside the destination root on the receiver); an absolute path is
accepted (rsync parity) and canonicalized with its leading '/' preserved. */
static void test_parse_args_basis_invalid_paths() { static void test_parse_args_basis_invalid_paths() {
static const char* const invalid[] = {"/abs", "..", "a/../b", "."}; static const char* const invalid[] = {"..", "a/../b", ".", "/", ""};
for (size_t i = 0; i < sizeof(invalid) / sizeof(invalid[0]); i++) { for (size_t i = 0; i < sizeof(invalid) / sizeof(invalid[0]); i++) {
Config* cfg = config_create(); Config* cfg = config_create();
char* argv[] = {"fastsync", "--link-dest", (char*)invalid[i], "/src", "/dst"}; char* argv[] = {"fastsync", "--link-dest", (char*)invalid[i], "/src", "/dst"};
@@ -1056,6 +1057,15 @@ static void test_parse_args_basis_invalid_paths() {
EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1);
config_delete(cfg); config_delete(cfg);
} }
Config* cfg = config_create();
char* argv[] = {"fastsync", "--link-dest=/abs/dir", "/src", "/dst"};
int positional_args[2];
int positional_count = 0;
EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0);
EXPECT_EQ_INT(cfg->basis_count, 1);
EXPECT_EQ_STR(cfg->basis_dirs[0].path, "/abs/dir");
config_delete(cfg);
} }
/* Basis dirs require the per-file incremental handshake, which -s disables. */ /* Basis dirs require the per-file incremental handshake, which -s disables. */
@@ -2551,15 +2561,35 @@ static void test_parse_args_filter_rules() {
EXPECT_EQ_INT(parse_args(cfg, 4, missing_argv, positional_args, &positional_count), -1); EXPECT_EQ_INT(parse_args(cfg, 4, missing_argv, positional_args, &positional_count), -1);
config_delete(cfg); config_delete(cfg);
/* rsync shorthands/modifiers we do not support are rejected instead of being /* Full rsync grammar (rule words, modifiers, clear) is supported. */
* silently parsed as literal patterns. */
static const char* const unsupported[] = {
": .rsync-filter", ". /tmp/rules", "-s foo", "-p bar", "-C", "-! *.o", "!",
};
for (size_t i = 0; i < sizeof(unsupported) / sizeof(unsupported[0]); i++) {
cfg = config_create(); cfg = config_create();
positional_count = 0; positional_count = 0;
char* rule_argv[] = {"fastsync", "--filter", (char*)unsupported[i], "/src", "/dst"}; char* grammar_argv[] = {"fastsync",
"--filter=hide *.tmp",
"--filter=show *.txt",
"--filter=protect *.bak",
"--filter=risk *.o",
"--filter=-s foo",
"--filter=-p bar",
"--filter=-! *.o",
"--filter=dir-merge .rules",
"--filter=!",
"/src",
"/dst"};
EXPECT_EQ_INT(parse_args(cfg, 11, grammar_argv, positional_args, &positional_count), 0);
config_delete(cfg);
/* Genuinely malformed rules are still rejected. */
static const char* const malformed[] = {
"merge", /* merge requires a filename */
"dir-merge", /* dir-merge requires a filename */
"clear extra", /* clear takes no pattern */
"no-such-rule x", /* unknown rule word */
};
for (size_t i = 0; i < sizeof(malformed) / sizeof(malformed[0]); i++) {
cfg = config_create();
positional_count = 0;
char* rule_argv[] = {"fastsync", "--filter", (char*)malformed[i], "/src", "/dst"};
EXPECT_EQ_INT(parse_args(cfg, 5, rule_argv, positional_args, &positional_count), -1); EXPECT_EQ_INT(parse_args(cfg, 5, rule_argv, positional_args, &positional_count), -1);
config_delete(cfg); config_delete(cfg);
} }
@@ -3146,6 +3176,27 @@ static void test_parse_args_chown() {
EXPECT_TRUE(cfg->chown_gid_set); EXPECT_TRUE(cfg->chown_gid_set);
EXPECT_EQ_INT(cfg->chown_gid, IDENTITY_CURRENT); EXPECT_EQ_INT(cfg->chown_gid, IDENTITY_CURRENT);
config_delete(cfg); config_delete(cfg);
/* A --chown NAME is converted to the equivalent receiver-resolved map rule
* (rsync implements --chown as --usermap=*:USER --groupmap=*:GROUP), so the
* name is carried on the wire as to_name instead of being resolved on the
* sender. A name that does not exist on the sender is accepted and left for
* the receiver to resolve (or warn about), matching rsync. */
cfg = config_create();
positional_count = 0;
char* argv5[] = {"fastsync", "--chown=no_such_user_zzz:no_such_group_zzz", "/src", "/dst"};
EXPECT_EQ_INT(parse_args(cfg, 4, argv5, positional_args, &positional_count), 0);
EXPECT_EQ_INT(cfg->usermap_count, 1);
EXPECT_NOT_NULL(cfg->usermap[0].to_name);
if (cfg->usermap[0].to_name)
EXPECT_EQ_STR(cfg->usermap[0].to_name, "no_such_user_zzz");
EXPECT_FALSE(cfg->chown_uid_set);
EXPECT_EQ_INT(cfg->groupmap_count, 1);
EXPECT_NOT_NULL(cfg->groupmap[0].to_name);
if (cfg->groupmap[0].to_name)
EXPECT_EQ_STR(cfg->groupmap[0].to_name, "no_such_group_zzz");
EXPECT_FALSE(cfg->chown_gid_set);
config_delete(cfg);
} }
/* --copy-as=USER[:GROUP] (P7 Wave E): resolve the user/group against the local /* --copy-as=USER[:GROUP] (P7 Wave E): resolve the user/group against the local
@@ -3220,7 +3271,6 @@ static void test_parse_args_rejects_malformed_identity() {
{"--groupmap", "@1"}, {"--groupmap", "@1"},
{"--groupmap", "no_such_group_qqq:x"}, {"--groupmap", "no_such_group_qqq:x"},
{"--chown", "a:b:c"}, {"--chown", "a:b:c"},
{"--chown", "no_such_user_zzz:"},
{"--copy-as", ""}, {"--copy-as", ""},
{"--copy-as", ":"}, {"--copy-as", ":"},
{"--copy-as", "a:b:c"}, {"--copy-as", "a:b:c"},
+10 -2
View File
@@ -1192,7 +1192,10 @@ static void test_config_basis_wire_rejects_escaping() {
c->basis_dirs = calloc(1, sizeof(BasisDest)); c->basis_dirs = calloc(1, sizeof(BasisDest));
c->basis_dirs[0].type = BASIS_DEST_LINK; c->basis_dirs[0].type = BASIS_DEST_LINK;
c->basis_dirs[0].path = str_dup("/abs"); c->basis_dirs[0].path = str_dup("/abs");
EXPECT_FALSE(roundtrip_config_ok(c)); /* An absolute basis dir is accepted (rsync parity); it is only usable when it
lies within the receiver's authorized root, which file_open_secure_parent
enforces at lookup time. */
EXPECT_TRUE(roundtrip_config_ok(c));
config_delete(c); config_delete(c);
/* A well-formed list still round-trips even with a manually built struct. */ /* A well-formed list still round-trips even with a manually built struct. */
@@ -1225,8 +1228,13 @@ static void test_config_basis_normalization() {
/* Degenerate values that normalize away to nothing stay rejected. */ /* Degenerate values that normalize away to nothing stay rejected. */
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "."), -1); EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "."), -1);
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ".."), -1); EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ".."), -1);
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/abs"), -1); /* An absolute path is canonicalized (leading '/' preserved) and accepted. */
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/abs"), 0);
EXPECT_EQ_STR(c->basis_dirs[c->basis_count - 1].path, "/abs");
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/a//b/"), 0);
EXPECT_EQ_STR(c->basis_dirs[c->basis_count - 1].path, "/a/b");
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "a/../b"), -1); EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "a/../b"), -1);
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/"), -1);
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ""), -1); EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ""), -1);
config_delete(c); config_delete(c);
} }
+8 -6
View File
@@ -855,7 +855,7 @@ static void test_filter_rules(bool parallel) {
/* - *.tmp excludes only the tmp file; other files remain (default include). */ /* - *.tmp excludes only the tmp file; other files remain (default include). */
const char* exclude_only[] = {"- *.tmp"}; const char* exclude_only[] = {"- *.tmp"};
char err[160]; char err[160];
FilterRuleList* base = filter_base_build(exclude_only, 1, false, err, sizeof(err)); FilterRuleList* base = filter_base_build(exclude_only, 1, false, false, err, sizeof(err));
EXPECT_NOT_NULL(base); EXPECT_NOT_NULL(base);
ScannerOptions options = {0}; ScannerOptions options = {0};
options.base_filters = base; options.base_filters = base;
@@ -875,7 +875,7 @@ static void test_filter_rules(bool parallel) {
/* Anchored include then exclude-all: only root-level keep* survives. */ /* Anchored include then exclude-all: only root-level keep* survives. */
const char* anchored[] = {"+ /a.txt", "- *"}; const char* anchored[] = {"+ /a.txt", "- *"};
base = filter_base_build(anchored, 2, false, err, sizeof(err)); base = filter_base_build(anchored, 2, false, false, err, sizeof(err));
EXPECT_NOT_NULL(base); EXPECT_NOT_NULL(base);
options.base_filters = base; options.base_filters = base;
rc = parallel ? collect_files_parallel(root, &options, &paths, &count) rc = parallel ? collect_files_parallel(root, &options, &paths, &count)
@@ -889,7 +889,7 @@ static void test_filter_rules(bool parallel) {
/* The common include idiom (the exact rule order the CLI compiles from /* The common include idiom (the exact rule order the CLI compiles from
* --include='*.txt' --exclude='*'): only .txt files survive. */ * --include='*.txt' --exclude='*'): only .txt files survive. */
const char* idiom[] = {"+ *.txt", "- *"}; const char* idiom[] = {"+ *.txt", "- *"};
base = filter_base_build(idiom, 2, false, err, sizeof(err)); base = filter_base_build(idiom, 2, false, false, err, sizeof(err));
EXPECT_NOT_NULL(base); EXPECT_NOT_NULL(base);
options.base_filters = base; options.base_filters = base;
rc = parallel ? collect_files_parallel(root, &options, &paths, &count) rc = parallel ? collect_files_parallel(root, &options, &paths, &count)
@@ -905,7 +905,7 @@ static void test_filter_rules(bool parallel) {
/* An include rule alone is NOT a mandatory whitelist (rsync semantics): only /* An include rule alone is NOT a mandatory whitelist (rsync semantics): only
* the matching file is affected, everything else is still transferred. */ * the matching file is affected, everything else is still transferred. */
const char* include_alone[] = {"+ *.txt"}; const char* include_alone[] = {"+ *.txt"};
base = filter_base_build(include_alone, 1, false, err, sizeof(err)); base = filter_base_build(include_alone, 1, false, false, err, sizeof(err));
EXPECT_NOT_NULL(base); EXPECT_NOT_NULL(base);
options.base_filters = base; options.base_filters = base;
rc = parallel ? collect_files_parallel(root, &options, &paths, &count) rc = parallel ? collect_files_parallel(root, &options, &paths, &count)
@@ -932,7 +932,7 @@ static void test_filter_dir_only_and_anchored(bool parallel) {
const char* rules[] = {"- /sub/"}; const char* rules[] = {"- /sub/"};
char err[160]; char err[160];
FilterRuleList* base = filter_base_build(rules, 1, false, err, sizeof(err)); FilterRuleList* base = filter_base_build(rules, 1, false, false, err, sizeof(err));
EXPECT_NOT_NULL(base); EXPECT_NOT_NULL(base);
ScannerOptions options = {0}; ScannerOptions options = {0};
options.base_filters = base; options.base_filters = base;
@@ -966,7 +966,7 @@ static void test_cvs_defaults(bool parallel) {
create_test_file("test_scan_cvs/keep.txt", "keep"); create_test_file("test_scan_cvs/keep.txt", "keep");
char err[160]; char err[160];
FilterRuleList* base = filter_base_build(NULL, 0, true, err, sizeof(err)); FilterRuleList* base = filter_base_build(NULL, 0, true, false, err, sizeof(err));
EXPECT_NOT_NULL(base); EXPECT_NOT_NULL(base);
ScannerOptions options = {0}; ScannerOptions options = {0};
options.base_filters = base; options.base_filters = base;
@@ -1005,6 +1005,7 @@ static void test_per_dir_filter(bool parallel) {
ScannerOptions options = {0}; ScannerOptions options = {0};
options.per_dir_filters = true; options.per_dir_filters = true;
options.exclude_per_dir_filter_files = true; /* -FF */
if (parallel) if (parallel)
options.num_threads = 2; options.num_threads = 2;
char** paths = NULL; char** paths = NULL;
@@ -1137,6 +1138,7 @@ static void test_per_dir_filter_override(bool parallel) {
ScannerOptions options = {0}; ScannerOptions options = {0};
options.per_dir_filters = true; options.per_dir_filters = true;
options.exclude_per_dir_filter_files = true; /* -FF */
if (parallel) if (parallel)
options.num_threads = 2; options.num_threads = 2;
char** paths = NULL; char** paths = NULL;