diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 71088e8..ab7a1d9 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -329,10 +329,10 @@ static int config_add_remote_option(Config* config, const char* value, const cha } /* Validate and append one --compare-dest/--copy-dest/--link-dest directory. - * The path is interpreted on the receiver relative to the destination root, - * so it must be a non-empty relative path with no "." / ".." components (an - * absolute or escaping path is rejected up front instead of failing on the - * server). Returns 0 on success, -1 on error. */ + * A relative path is interpreted on the receiver below the destination root; an + * absolute path is used verbatim on the receiver (matching rsync), still subject + * to the receiver's authorized-root confinement. Either way the path must be + * non-empty and traversal-free (no ".."). Returns 0 on success, -1 on error. */ static int set_basis_dest_option(Config* config, BasisDestType type, const char* value, const char* option_name) { if (!value || !value[0]) { @@ -341,8 +341,9 @@ static int set_basis_dest_option(Config* config, BasisDestType type, const char* } if (config_basis_append(config, type, value) != 0) { log_message(LOG_LEVEL_ERROR, - "%s requires a non-empty relative directory name with no '.', '..', or absolute " - "path (resolved below the destination root)", + "%s requires a non-empty directory name with no '..' component " + "(relative paths resolve below the destination root; absolute paths are used " + "verbatim)", option_name); return -1; } @@ -657,16 +658,25 @@ static int config_add_pattern(char*** patterns, int* count, const char* value, /* Validate and append one --filter=RULE string. Returns 0 on success, -1 on error. */ static int config_add_filter(Config* config, const char* rule) { - char err[160]; - FilterRule* parsed = filter_rule_parse(rule, err, sizeof(err)); - if (!parsed) { + char err[256]; + /* Validate through the full list parser so clear/merge/dir-merge and the rule + modifiers are accepted (and a merge file is readable) at parse time. */ + FilterParseOptions opts = {.delete_excluded = config->delete_excluded, + .cvs_exclude = config->cvs_exclude}; + FilterRuleList* probe = filter_rule_list_create(); + if (!probe) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for --filter"); + return -1; + } + bool ok = filter_rule_list_parse_append(probe, rule, &opts, NULL, err, sizeof(err)); + filter_rule_list_free(probe); + if (!ok) { char* escaped = output_escape(rule, log_get_8_bit_output()); log_message(LOG_LEVEL_ERROR, "invalid --filter rule '%s': %s", escaped ? escaped : "", err); free(escaped); return -1; } - filter_rule_free(parsed); if (!config->filters) { config->filters = array_list_create(free); if (!config->filters) { @@ -1365,6 +1375,11 @@ static bool cli_handle_table_option(CliParseCtx* ctx) { } if (entry->offset == offsetof(Config, eight_bit_output)) protocol_set_8_bit_output(true); + /* -F is repeatable: rsync's single -F transfers .rsync-filter files, a + repeated -FF excludes them. Count the occurrences so the scanner can + distinguish the two. */ + if (entry->offset == offsetof(Config, per_dir_filter) && config->per_dir_filter_count < INT_MAX) + config->per_dir_filter_count++; /* A delete-timing flag selects when --delete removes extras, so it implies --delete exactly like the rsync options do. */ if (entry->offset == offsetof(Config, delete_before) || @@ -2397,6 +2412,15 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool config->preserve_times = true; } + /* --ignore-existing is a receiver-side existence policy: the receiver must + * answer "skip" BEFORE the sender transmits any payload, which only the + * per-file STATUS_CHECK handshake provides. Imply --incremental here (after + * the auto-preserve capture above, so a bare --ignore-existing does not gain + * -p/-t, which rsync likewise does not imply) so an existing destination is + * skipped on the wire instead of being streamed and discarded. */ + if (config->ignore_existing) + config->use_incremental = true; + /* Derive the transport bit from the FINAL parsed flags. Every * preservation/ownership option that needs the metadata frame (per-attribute * perms/times/owner/group, atimes/crtimes, executability, xattrs/acls, diff --git a/src/client/client_send.c b/src/client/client_send.c index 76d1bd1..cdc1cb3 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -161,7 +161,8 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann } if (rule_count > 0 || config->cvs_exclude) { char err[160]; - out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, err, sizeof(err)); + out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, + config->delete_excluded, err, sizeof(err)); free(texts); if (!out->base_filters) { log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); @@ -206,6 +207,8 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann options->file_list = (const FileListSet*)config->files_from_set; options->base_filters = out->base_filters; options->per_dir_filters = config->per_dir_filter; + options->delete_excluded = config->delete_excluded; + options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2; options->dirs = config->dirs; options->relative = config->relative; /* -R/--relative outside --files-from reconstructs every destination path from diff --git a/src/client/scanner.c b/src/client/scanner.c index 8770757..ce27f2c 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -51,29 +51,63 @@ static FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) { return node; } -/* Evaluate a rule chain for an entry inside the directory whose content - * context is `node`. rsync precedence, highest first: the innermost (current) - * directory's .rsync-filter rules, then each ancestor's, then the root's, and - * finally the command-line base rules (--filter/-C). A deeper per-directory - * file therefore overrides a shallower one, and per-directory files override - * the base rules by default. Returns FILTER_ACTION_NONE when nothing matched. */ -static FilterAction chain_rules_apply(const FilterRuleList* base, const FilterNode* node, - const char* rel, const char* leaf, bool is_dir) { - if (node) { - FilterAction own_action = filter_rules_apply(node->own, rel, leaf, is_dir); - if (own_action != FILTER_ACTION_NONE) - return own_action; - return chain_rules_apply(base, node->parent, rel, leaf, is_dir); +/* Evaluate a rule chain for one entry. rsync precedence, highest first: the + * innermost (current) directory's .rsync-filter rules, then each ancestor's, + * then the root's, and finally the command-line base rules (--filter/-C). The + * sender-side verdict decides whether the entry is hidden from the transfer; + * the receiver-side verdict decides whether its destination mirror is protected + * from --delete. Each side takes the FIRST matching rule independently. */ +typedef struct { + bool hide; /* sender-side exclude matched */ + bool protect; /* receiver-side exclude matched */ +} FilterOutcome; + +static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel, + const char* leaf, bool is_dir, FilterOutcome* out) { + memset(out, 0, sizeof(*out)); + bool sender_decided = false; + bool receiver_decided = false; + const FilterNode* n = node; + while (!sender_decided || !receiver_decided) { + const FilterRuleList* list = n ? n->own : base; + if (list) { + if (!sender_decided) { + FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER); + if (action != FILTER_ACTION_NONE) { + out->hide = action == FILTER_ACTION_EXCLUDE; + sender_decided = true; + } + } + if (!receiver_decided) { + FilterAction action = + filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER); + if (action != FILTER_ACTION_NONE) { + out->protect = action == FILTER_ACTION_PROTECT; + receiver_decided = true; + } + } + } + if (!n) + break; + n = n->parent; } - return base ? filter_rules_apply(base, rel, leaf, is_dir) : FILTER_ACTION_NONE; } static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel, - const char* leaf, bool is_dir, bool per_dir_filters) { - /* -F: per-directory .rsync-filter files are never transferred. */ - if (per_dir_filters && !is_dir && strcmp(leaf, ".rsync-filter") == 0) + const char* leaf, bool is_dir, bool exclude_filter_files, + bool* protect_out) { + /* -FF: per-directory .rsync-filter files are never transferred (single -F + transfers them, matching rsync). */ + if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) { + if (protect_out) + *protect_out = false; return false; - return chain_rules_apply(base, node, rel, leaf, is_dir) != FILTER_ACTION_EXCLUDE; + } + FilterOutcome outcome; + chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome); + if (protect_out) + *protect_out = outcome.protect; + return !outcome.hide; } static void dir_entry_destroy(void* item) { @@ -274,14 +308,19 @@ static char* scanner_prefix_send_path(const char* prefix, const char* rel) { return path_cat(prefix, rel); } -/* Apply the --files-from allow-set and the filter layer to one entry. */ +/* Apply the --files-from allow-set and the filter layer to one entry. On + * return `*protect_out` is true when a receiver-side rule protects the entry's + * destination mirror from deletion. */ static bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, const FilterNode* node, const char* rel, const char* leaf, - bool is_dir, bool per_dir_filters) { + bool is_dir, bool per_dir_filters, bool exclude_filter_files, + bool* protect_out) { + if (protect_out) + *protect_out = false; if (file_list && !file_list_affects(file_list, rel)) return false; if (base || per_dir_filters) - return entry_allowed(base, node, rel, leaf, is_dir, per_dir_filters); + return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out); return true; } @@ -435,28 +474,77 @@ static bool scanner_record_synced_dir(const ScannerOptions* options, const char* return ok; } -/* Merge the open directory's own .rsync-filter rules into the inherited - * context, returning the context used for this directory's entries. On a parse - * error the scanner is marked failed. Returns 0 on success, -1 on failure. */ +/* Read every per-directory filter file that applies to `dir_path` (its + * .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a + * fresh list. Returns NULL on allocation/parse failure (message in `err`); + * returns an empty list (and *any_exists=false) when no file exists. */ +static FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path, + const char* rel, bool* any_exists, char* err, + size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + const FilterRuleList* base = options->base_filters; + bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0); + if (any_exists) + *any_exists = false; + if (!have_names) + return NULL; + FilterRuleList* own = filter_rule_list_create(); + if (!own) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false}; + bool exists = false; + if (options->per_dir_filters) { + if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size)) + goto fail; + if (exists && any_exists) + *any_exists = true; + } + if (base) { + for (int i = 0; i < base->dir_merge_count; i++) { + if (!filter_file_append(own, dir_path, base->dir_merge_names[i], rel, &opts, &exists, err, + err_size)) + goto fail; + if (exists && any_exists) + *any_exists = true; + } + } + return own; +fail: + filter_rule_list_free(own); + return NULL; +} + +/* Merge the open directory's own per-directory filter files (the default + * .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the + * base rule list) into the inherited context, returning the context used for + * this directory's entries. On a parse error the scanner is marked failed. + * Returns 0 on success, -1 on failure. */ static int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { - if (!scanner->options.per_dir_filters) { + char err[256]; + bool any_exists = false; + FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path, + scanner->current_rel ? scanner->current_rel : "", + &any_exists, err, sizeof(err)); + if (!own && any_exists) { scanner->current_node = (FilterNode*)inherited; return 0; } - char err[256]; - bool exists = false; - FilterRuleList* own = - filter_file_read(scanner->current_path, scanner->current_rel ? scanner->current_rel : "", - &exists, err, sizeof(err)); if (!own) { + if (err[0] == '\0') { + scanner->current_node = (FilterNode*)inherited; + return 0; + } char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "invalid .rsync-filter in %s: %s", + log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", escaped_path ? escaped_path : "", err); free(escaped_path); scanner->failed = true; return -1; } - if (exists && own->count > 0) { + if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { FilterNode* node = filter_node_alloc((FilterNode*)inherited, own); if (!node || !array_list_add(scanner->filter_nodes, node)) { filter_node_destroy(node); @@ -1108,7 +1196,8 @@ static File* dirs_next_child(DirectoryScanner* scanner) { return NULL; if (file && !entry_passes_selection(scanner->options.file_list, scanner->options.base_filters, NULL, entry->d_name, entry->d_name, file->is_dir, - scanner->options.per_dir_filters)) { + scanner->options.per_dir_filters, + scanner->options.exclude_per_dir_filter_files, NULL)) { file_destroy(file); continue; } @@ -1324,10 +1413,15 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { scanner->failed = true; break; } + bool protect = false; bool passes_selection = entry_passes_selection( scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, - entry->d_name, is_dir, scanner->options.per_dir_filters); - if (!passes_selection) { + entry->d_name, is_dir, scanner->options.per_dir_filters, + scanner->options.exclude_per_dir_filter_files, &protect); + /* A sender-side hide leaves the entry out of the transfer; an independent + receiver-side protect rule keeps a transferred entry's destination mirror + from being deleted. Both are recorded in the same protection set. */ + if (!passes_selection || protect) { /* --files-from subset pruning is not a filter exclusion: its delete semantics stay keep-set-only (an unlisted source path is treated as absent, so its destination mirror is a deletable extra). A rule-based @@ -1751,15 +1845,17 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo ps->failed = true; return; } + bool protect = false; bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel, - entry->d_name, is_dir, options->per_dir_filters); + entry->d_name, is_dir, options->per_dir_filters, + options->exclude_per_dir_filter_files, &protect); /* -R + --files-from: root-level files keep their bare relative send path. */ bool use_rel = options->relative && options->file_list != NULL; - if (!passes) { + if (!passes || protect) { /* --files-from subset pruning is not a filter exclusion; -R bare-wire-path exclusions are never recorded (see ScannerOptions.excluded_paths). */ bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel); - if (!files_from_prune && !use_rel && options->excluded_paths) { + if ((!files_from_prune && !use_rel) || protect) { const char* rel_path = *cur_path == '/' ? cur_path + 1 : cur_path; char* prefixed = NULL; if (options->relative_prefix) { @@ -1772,13 +1868,16 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo } rel_path = prefixed; } - if (!excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) + if (options->excluded_paths && + !excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) ps->failed = true; free(prefixed); } - free(rel); - free(cur_path); - return; + if (!passes) { + free(rel); + free(cur_path); + return; + } } if (is_dir) { if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { @@ -2040,22 +2139,26 @@ ParallelScanner* parallel_scanner_create_with_options(const char* root_directory root_dev = root_stats.st_dev; } - /* Build the root directory's .rsync-filter context once; workers seed their - * scanners with it so per-dir rules behave identically to the sequential + /* Build the root directory's per-directory filter context once; workers seed + * their scanners with it so per-dir rules behave identically to the sequential * scanner. */ FilterNode* root_node = NULL; - if (options->per_dir_filters) { + { char err[256]; - bool exists = false; - FilterRuleList* own = filter_file_read(root_directory, "", &exists, err, sizeof(err)); - if (!own) { - log_message(LOG_LEVEL_ERROR, "invalid .rsync-filter in %s: %s", root_directory, err); - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - if (exists && own->count > 0) { + bool any_exists = false; + FilterRuleList* own = + read_dir_filters(options, root_directory, "", &any_exists, err, sizeof(err)); + if (!own && any_exists) { + /* no files exist: leave root_node NULL */ + } else if (!own) { + if (err[0] != '\0') { + log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + } else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { root_node = filter_node_alloc(NULL, own); if (!root_node) { filter_rule_list_free(own); diff --git a/src/client/scanner.h b/src/client/scanner.h index 7676a41..8788a34 100644 --- a/src/client/scanner.h +++ b/src/client/scanner.h @@ -63,8 +63,14 @@ typedef struct { const FileListSet* file_list; /* --files-from allow-set, or NULL */ const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */ bool per_dir_filters; /* -F: read .rsync-filter per directory */ - bool dirs; /* -d/--dirs: transfer dir entries, no recursion */ - bool relative; /* -R/--relative (dest rel paths, with --files-from) */ + /* --delete-excluded: per-directory plain rules become sender-only, so they no + longer protect the receiver from deletion. */ + bool delete_excluded; + /* -FF: also exclude the per-directory filter files themselves from the + transfer (single -F transfers them). */ + bool exclude_per_dir_filter_files; + bool dirs; /* -d/--dirs: transfer dir entries, no recursion */ + bool relative; /* -R/--relative (dest rel paths, with --files-from) */ /* -R/--relative outside --files-from: the destination-relative path prefix * reconstructed from the source spec (rsync's '/./' cut point), or NULL when * -R is off or --files-from is in use (the bare-relative path then comes from diff --git a/src/client/usage.c b/src/client/usage.c index 6a79632..eee2ec4 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -114,10 +114,12 @@ void print_usage(void) { printf(" --files-from Read the source file list from FILE (paths relative to the " "source root)\n"); printf(" -0, --from0 Entries in --files-from are NUL-delimited\n"); - printf(" -f, --filter=RULE rsync-style filter rule (+/- include/exclude; repeatable;\n"); - printf(" both --filter=RULE and the -f RULE / -f=RULE short forms work)\n"); + printf(" -f, --filter=RULE rsync-style filter rule: exclude/- include/+ hide/H show/S\n"); + printf(" protect/P risk/R merge/. dir-merge/: clear/! with modifiers\n"); + printf(" (repeatable; --filter=RULE and -f RULE / -f=RULE both work)\n"); printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n"); - printf(" -F Apply per-directory .rsync-filter files during the scan\n"); + printf(" -F Apply per-directory .rsync-filter files; repeated -FF also\n"); + printf(" excludes the .rsync-filter files themselves\n"); printf(" --max-size Skip files larger than n bytes\n"); printf(" --min-size Skip files smaller than n bytes\n"); printf(" --max-alloc Maximum single allocation (default: 1G; 0 = no limit,\n"); diff --git a/src/shared/config.c b/src/shared/config.c index 153f94d..fc3ba3f 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -329,30 +329,38 @@ bool config_has_basis(const Config* config) { } /* A basis-dir path travels from the client to the receiver and is resolved - * below the destination root, so it must be a non-empty relative path with no - * "." or ".." component and no traversal: an absolute or escaping path would - * make the receiver read or link files outside its authorized root. + * below the destination root when relative, or used verbatim when absolute + * (matching rsync). Either form must be non-empty, traversal-free (no "..") + * and free of "." components: an escaping path would make the receiver read or + * link files outside its authorized root. An absolute path is still subject to + * the receiver's root confinement at open time (file_open_secure_parent), so a + * basis outside the authorized root is simply not found rather than an escape. * * Returns a malloc'd CANONICAL copy of an accepted path, or NULL when the path * is rejected. Canonicalization collapses interior empty components ("a//b" -> - * "a/b"), drops "." components and trailing "/"s, so validation, the delete - * walker prefix match and the receiver's basis lookup all agree on one form. - * The normalizer is the single source of truth for both config_basis_path_valid - * and config_basis_append. */ + * "a/b"), drops "." components and trailing "/"s, and preserves a leading '/' + * for absolute paths, so validation, the delete walker prefix match and the + * receiver's basis lookup all agree on one form. The normalizer is the single + * source of truth for both config_basis_path_valid and config_basis_append. */ static char* basis_path_normalize(const char* path) { - if (!path || path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) + if (!path || path[0] == '\0' || has_path_traversal(path)) return NULL; - if (strcmp(path, ".") == 0) + bool absolute = path[0] == '/'; + if (!absolute && strcmp(path, ".") == 0) + return NULL; + if (absolute && strcmp(path, "/") == 0) return NULL; char* dup = str_dup(path); if (!dup) return NULL; size_t out_len = 0; - char* out = malloc(strlen(path) + 1); + char* out = malloc(strlen(path) + 2); if (!out) { free(dup); return NULL; } + if (absolute) + out[out_len++] = '/'; char* saveptr = NULL; bool ok = true; for (char* part = strtok_r(dup, "/", &saveptr); part; part = strtok_r(NULL, "/", &saveptr)) { @@ -362,14 +370,14 @@ static char* basis_path_normalize(const char* path) { } if (strcmp(part, ".") == 0) continue; - if (out_len > 0) + if (out_len > 0 && out[out_len - 1] != '/') out[out_len++] = '/'; size_t len = strlen(part); memcpy(out + out_len, part, len); out_len += len; } free(dup); - if (!ok || out_len == 0) { + if (!ok || out_len == 0 || (absolute && out_len == 1)) { free(out); return NULL; } diff --git a/src/shared/config.h b/src/shared/config.h index 6762acc..1f3f126 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -377,6 +377,10 @@ typedef struct Config { bool from0; /* -0/--from0: NUL-delimited *-from files */ bool cvs_exclude; /* -C/--cvs-exclude: standard CVS ignore set */ bool per_dir_filter; /* -F: apply per-directory .rsync-filter files */ + /* -F click count. rsync's single -F means --filter='dir-merge + * /.rsync-filter' (the .rsync-filter files themselves are transferred); a + * repeated -F adds --filter='- .rsync-filter' so they are excluded too. */ + int per_dir_filter_count; bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */ /* --no-implied-dirs: client-only. With -R + --files-from, refuse to place a * listed file whose ancestor directory is not itself explicitly listed. */ diff --git a/src/shared/file.c b/src/shared/file.c index 27084d6..b8c7028 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -40,19 +40,27 @@ static bool write_all(int fd, const void* data, unsigned long long size) { } /* Preallocate `size` bytes on `fd` before any data is written (--preallocate). - * posix_fallocate reserves real disk blocks, so an out-of-space condition + * fallocate(2) reserves real disk blocks, so an out-of-space condition * (ENOSPC/EDQUOT) surfaces up front instead of partway through a transfer; - * unavoidable fragmentation of a streamed file is also reduced. Some - * filesystems (e.g. tmpfs, ZFS) do not support it and return EOPNOTSUPP/ENOSYS, - * where we fall back to ftruncate, which still extends the logical size so the - * fail-fast/contiguity intent degrades gracefully but never fails. Genuine - * allocation failures are propagated as the error code (caller fails the write). - * posix_fallocate leaves the fd's file offset unchanged, so the subsequent - * write_all at offset 0 is unaffected. Returns 0 on success (including the - * fallback) or a nonzero error code. */ + * unavoidable fragmentation of a streamed file is also reduced. rsync favors + * the syscall over glibc posix_fallocate (whose emulation can be subtly + * different), so try fallocate(2) first and only fall back to posix_fallocate, + * then to ftruncate on filesystems (e.g. tmpfs, ZFS) that support neither. The + * logical size is always extended, so the fail-fast/contiguity intent degrades + * gracefully but never fails on an unsupported filesystem; genuine allocation + * failures are propagated as the error code (caller fails the write). Neither + * leaves the fd's file offset guaranteed, so the caller seeks back to 0 before + * writing. Returns 0 on success (including the fallback) or a nonzero error + * code. */ static int preallocate_fd(int fd, unsigned long long size) { if (size == 0) return 0; +#ifdef __linux__ + if (fallocate(fd, 0, 0, (off_t)size) == 0) + return 0; + if (errno != EOPNOTSUPP && errno != ENOSYS && errno != EINVAL) + return errno; +#endif int rc = posix_fallocate(fd, 0, (off_t)size); if (rc == EOPNOTSUPP || rc == ENOSYS) { if (ftruncate(fd, (off_t)size) == 0) @@ -1066,11 +1074,10 @@ static bool file_to_disk_secure_impl(const char* path, const void* data, } else { /* Preallocate the expected payload size before writing so an out-of-space condition fails cleanly up front (--preallocate). - --sparse takes precedence: posix_fallocate would allocate every - block, defeating the holes the sparse writer would create, so the - two never combine here (the ftruncate presize below stays). */ + rsync lets --preallocate win over --sparse (the reserved blocks + survive the sparse writer's seeks), so both flags can be active. */ int prealloc_rc = 0; - if (preallocate && !sparse && data_size > 0) { + if (preallocate && data_size > 0) { prealloc_rc = preallocate_fd(fd, data_size); if (prealloc_rc != 0) { char* escaped_path = output_escape(path, log_get_8_bit_output()); @@ -1196,7 +1203,7 @@ static bool file_to_disk_secure_impl(const char* path, const void* data, if (fd < 0) continue; /* EEXIST (or a transient open error): try a fresh name. */ int prealloc_rc = 0; - if (preallocate && !sparse && data_size > 0) { + if (preallocate && data_size > 0) { prealloc_rc = preallocate_fd(fd, data_size); if (prealloc_rc != 0) { char* escaped_path = output_escape(path, log_get_8_bit_output()); diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index 6da4325..dfd554a 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -1,4 +1,5 @@ #include +#include #include #include #include @@ -1335,7 +1336,11 @@ static bool basis_match_find(const Config* config, const char* check_path, return false; for (int i = 0; i < config->basis_count; i++) { const BasisDest* entry = &config->basis_dirs[i]; - char* basis_dir = path_cat(config->receive_root_directory, entry->path); + /* An absolute basis path is used verbatim (rsync semantics); a relative one + is resolved below the receive root. Both remain subject to the receiver's + authorized-root confinement inside file_open_secure_parent. */ + char* basis_dir = entry->path[0] == '/' ? str_dup(entry->path) + : path_cat(config->receive_root_directory, entry->path); if (!basis_dir) continue; char* candidate = path_cat(basis_dir, check_path); @@ -1397,7 +1402,8 @@ static bool basis_match_find(const Config* config, const char* check_path, * transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a * file. * - * Similarity heuristic (deterministic, deliberately simpler than rsync's): + * Similarity heuristic (rsync 3.4.1 parity, util1.c fuzzy_distance / + * find_filename_suffix + generator.c find_fuzzy): * * candidates are the target's sibling entries in its destination * directory, opened through the confined root (file_open_secure_parent + * openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never @@ -1406,12 +1412,15 @@ static bool basis_match_find(const Config* config, const char* check_path, * temp scratch names are never candidates; * * size gate = the delta engine's own bounds (delta_should_attempt: both * files >= DELTA_MIN_FILE_SIZE, <= delta_max_file_size, ratio <= 10x), - * NOT rsync's ~1.5x size window; - * * name gate = Levenshtein edit distance between the basenames, accepted - * only when distance <= half the length of the longer basename; - * * the single best candidate (smallest distance; tie-break: size closest - * to the incoming file, then lexicographically smaller basename) is read - * and returned as the basis. + * because FastSync's delta engine cannot use a basis outside them; + * * first pass = an exact size+mtime match wins regardless of name (rsync's + * "fuzzy size/modtime match"); + * * otherwise the winner minimizes rsync's weighted Levenshtein distance + * (substitution ± byte difference, insertion UNIT+byte, 16.16 fixed point) + * plus ten times the suffix distance, accepted only when <= 25*UNIT; the + * tie-break (smallest size gap, then lexical name) keeps the result + * deterministic across filesystem readdir order (rsync leaves equal + * distances to its file-list order). * ------------------------------------------------------------------------- */ /* A directory scan is linear in the number of entries; the fuzzy search stops @@ -1429,107 +1438,108 @@ static bool basis_match_find(const Config* config, const char* check_path, typedef struct { char name[FUZZY_NAME_LIMIT + 1]; unsigned long long size; - size_t distance; + uint32_t distance; unsigned long long size_gap; } FuzzyCandidate; -/* Two-row DP scratch, allocated once per directory scan (not per candidate) so - * a 4096-entry directory never performs 4096 malloc/free pairs. */ -typedef struct { - size_t* prev; - size_t* cur; -} FuzzyEditBuffer; +/* rsync's fuzzy distance is a weighted Levenshtein variant in 16.16 fixed point + * (util1.c fuzzy_distance): a substitution costs UNIT +/- the byte difference + * and an insertion costs UNIT + the inserted byte, so similar names score low. + * The search keeps only distances <= 25*UNIT. Ported verbatim for parity. */ +#define FUZZY_DIST_UNIT (1u << 16) +#define FUZZY_DIST_REJECT (0xFFFFu * FUZZY_DIST_UNIT + 1) +#define FUZZY_DIST_LIMIT (25u * FUZZY_DIST_UNIT) -static bool fuzzy_edit_buffer_init(FuzzyEditBuffer* buf) { - buf->prev = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(size_t)); - buf->cur = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(size_t)); - if (!buf->prev || !buf->cur) { - free(buf->prev); - free(buf->cur); - buf->prev = NULL; - buf->cur = NULL; - return false; - } - return true; -} - -static void fuzzy_edit_buffer_destroy(FuzzyEditBuffer* buf) { - free(buf->prev); - free(buf->cur); - buf->prev = NULL; - buf->cur = NULL; -} - -/* Cheap lower bounds used to reject a candidate BEFORE the DP: - * - any edit script must at least absorb the length gap: d >= |la - lb|; - * - any character of `a` that does not occur in `b` at all must be deleted or - * substituted at its own position: d >= (count of such characters). - * The acceptance gate is d*2 <= longer, so a candidate whose max of these two - * bounds already violates it can be skipped without computing the distance. */ -static size_t fuzzy_absent_char_bound(const char* a, size_t la, const char* b, size_t lb) { - if (lb == 0) - return la; - bool present[256] = {false}; - for (size_t i = 0; i < lb; i++) - present[(uint8_t)b[i]] = true; - size_t absent = 0; - for (size_t i = 0; i < la; i++) - if (!present[(uint8_t)a[i]]) - absent++; - return absent; -} - -/* Levenshtein edit distance between the two basenames. A shared prefix and a - * (non-overlapping) shared suffix can always be aligned at no cost, so the DP - * only runs over the differing middles; its two rows come from `buf` (allocated - * once by the caller). Callers enforce la, lb <= FUZZY_NAME_LIMIT. */ -static size_t fuzzy_edit_distance(FuzzyEditBuffer* buf, const char* a, size_t la, const char* b, - size_t lb) { - size_t p = 0; - while (p < la && p < lb && a[p] == b[p]) - p++; - /* Trim the common suffix (never overlapping the prefix). Working with two - moving end indices keeps the region arithmetic explicit and safe. */ - size_t ae = la; - size_t be = lb; - while (ae > p && be > p && a[ae - 1] == b[be - 1]) { - ae--; - be--; - } - size_t ma = ae - p; - size_t mb = be - p; - /* cppcheck-suppress knownConditionTrueFalse -- the prefix/suffix trims above - only run while the corresponding ends match, so a middle can remain; the - analysis unsoundly concludes the trims always consume everything. */ - if (ma == 0) - return mb; - if (mb == 0) - return ma; - const char* A = a + p; - const char* B = b + p; - size_t* prev = buf->prev; - size_t* cur = buf->cur; - for (size_t j = 0; j <= mb; j++) - prev[j] = j; - for (size_t i = 1; i <= ma; i++) { - cur[0] = i; - for (size_t j = 1; j <= mb; j++) { - size_t cost = A[i - 1] == B[j - 1] ? 0 : 1; - size_t del = prev[j] + 1; - size_t ins = cur[j - 1] + 1; - size_t sub = prev[j - 1] + cost; - size_t m = del < ins ? del : ins; - cur[j] = m < sub ? m : sub; +static uint32_t fuzzy_distance(const char* s1, unsigned len1, const char* s2, unsigned len2, + uint32_t upperlimit, uint32_t* scratch) { + if ((len1 > len2 ? len1 - len2 : len2 - len1) * FUZZY_DIST_UNIT > upperlimit) + return FUZZY_DIST_REJECT; + if (!len1 || !len2) { + if (!len1) { + s1 = s2; + len1 = len2; } - size_t* tmp = prev; - prev = cur; - cur = tmp; + uint32_t cost = 0; + for (unsigned i = 0; i < len1; i++) + cost += (uint8_t)s1[i]; + return (uint32_t)len1 * FUZZY_DIST_UNIT + cost; } - return prev[mb]; + uint32_t* a = scratch; + for (unsigned i2 = 0; i2 < len2; i2++) + a[i2] = (i2 + 1) * FUZZY_DIST_UNIT; + for (unsigned i1 = 0; i1 < len1; i1++) { + uint32_t diag = i1 * FUZZY_DIST_UNIT; + uint32_t above = (i1 + 1) * FUZZY_DIST_UNIT; + for (unsigned i2 = 0; i2 < len2; i2++) { + uint32_t left = a[i2]; + int32_t cost = (int32_t)(uint8_t)s1[i1] - (int32_t)(uint8_t)s2[i2]; + if (cost != 0) + cost = cost < 0 ? (int32_t)(FUZZY_DIST_UNIT - (uint32_t)(-cost)) + : (int32_t)(FUZZY_DIST_UNIT + (uint32_t)cost); + uint32_t diag_inc = diag + (uint32_t)cost; + uint32_t left_inc = left + FUZZY_DIST_UNIT + (uint8_t)s1[i1]; + uint32_t above_inc = above + FUZZY_DIST_UNIT + (uint8_t)s2[i2]; + a[i2] = above = left < above ? (left_inc < diag_inc ? left_inc : diag_inc) + : (above_inc < diag_inc ? above_inc : diag_inc); + diag = left; + } + } + return a[len2 - 1]; } -/* Deterministic ordering of two fuzzy candidates: smallest edit distance, - * then the size closest to the incoming file, then the lexical basename. */ +/* rsync's find_filename_suffix (util1.c): return the last significant filename + * suffix (its dot included). Leading dots are not a suffix; a trailing "~" is + * ignored; .bak/.old/.orig and a "~/" backup marker are skipped. */ +static const char* fuzzy_find_suffix(const char* fn, int fn_len, int* len_ptr) { + const char* suf; + const char* s; + bool had_tilde; + + while (fn_len && *fn == '.') { + fn++; + fn_len--; + } + if (fn_len > 1 && fn[fn_len - 1] == '~') { + fn_len--; + had_tilde = true; + } else { + had_tilde = false; + } + suf = ""; + *len_ptr = 0; + for (s = fn + fn_len; fn_len > 1;) { + int s_len; + while (--s != fn && *s != '.') { + } + if (s == fn) + break; + s_len = fn_len - (int)(s - fn); + fn_len = (int)(s - fn); + if (s_len == 4) { + if (strcmp(s + 1, "bak") == 0 || strcmp(s + 1, "old") == 0) + continue; + } else if (s_len == 5) { + if (strcmp(s + 1, "orig") == 0) + continue; + } else if (s_len > 2 && had_tilde && s[1] == '~' && isdigit((unsigned char)s[2])) { + continue; + } + *len_ptr = s_len; + suf = s; + if (s_len == 1) + break; + for (s++, s_len--; s_len > 0; s++, s_len--) { + if (!isdigit((unsigned char)*s)) + return suf; + } + s = suf; + } + return suf; +} + +/* Deterministic ordering of two fuzzy candidates with equal rsync distance: + * smallest size gap, then the lexical basename (rsync itself takes the last + * equal-distance candidate in file-list order). */ static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) { if (!best->name[0]) return true; @@ -1546,8 +1556,8 @@ static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandid * = 0) when no candidate qualifies, which means the caller performs the normal * whole-file transfer. */ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path, - unsigned long long check_size, - unsigned long long* out_size) { + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, unsigned long long* out_size) { *out_size = 0; if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || !check_path || check_size < DELTA_MIN_FILE_SIZE || check_size > config->delta_max_file_size || @@ -1590,18 +1600,28 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p return NULL; } - /* The DP scratch rows are allocated once per scan (not once per candidate). */ - FuzzyEditBuffer ebuf; - if (!fuzzy_edit_buffer_init(&ebuf)) { + /* The weighted-distance scratch row is allocated once per scan (not once per + candidate). */ + uint32_t* dist_scratch = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(uint32_t)); + if (!dist_scratch) { closedir(dir); close(dir_fd); free(leaf); free(full_path); return NULL; } + int fname_suf_len = 0; + const char* fname_suf = fuzzy_find_suffix(leaf, (int)target_len, &fname_suf_len); FuzzyCandidate best; memset(&best, 0, sizeof(best)); + uint32_t lowest_dist = FUZZY_DIST_LIMIT; + /* rsync's fuzzy search runs an exact size+mtime pass before the name-distance + pass; such a candidate is almost certainly the same content and wins + regardless of how dissimilar its name is. The first one (directory order, + deterministic) is kept. */ + FuzzyCandidate exact; + memset(&exact, 0, sizeof(exact)); const struct dirent* entry; size_t scanned = 0; /* readdir() yields entries in filesystem-dependent order, so the SET of @@ -1621,22 +1641,32 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE || !delta_should_attempt(cand_size, check_size, config->delta_max_file_size)) continue; - /* Cheap pre-name gates run BEFORE the edit-distance DP. The edit distance - is bounded below by the length gap |la-lb| and by the number of - characters of one basename that are absent from the other (each such - position costs at least one op), so a candidate whose acceptance gate - (distance*2 <= longer) already fails on the max of those bounds is - skipped without running the DP. */ - size_t longer = target_len > name_len ? target_len : name_len; - size_t bound = longer - (target_len < name_len ? target_len : name_len); - size_t absent = fuzzy_absent_char_bound(leaf, target_len, name, name_len); - if (absent > bound) - bound = absent; - if (bound * 2 > longer) + long cand_nsec = 0; +#ifdef __linux__ + cand_nsec = st.st_mtim.tv_nsec; +#endif + if (!exact.name[0] && cand_size == check_size && + metadata_mtime_matches(st.st_mtime, cand_nsec, check_mtime, check_mtime_nsec, + config->modify_window)) { + memcpy(exact.name, name, name_len + 1); + exact.size = cand_size; + exact.size_gap = 0; continue; - size_t distance = fuzzy_edit_distance(&ebuf, leaf, target_len, name, name_len); - if (distance * 2 > longer) + } + /* rsync's name-distance pass: a weighted Levenshtein distance over the full + basenames, plus ten times the same distance over the filename suffixes, + accepted only when it does not exceed the running lowest distance. */ + int name_suf_len = 0; + const char* name_suf = fuzzy_find_suffix(name, (int)name_len, &name_suf_len); + uint32_t distance = fuzzy_distance(name, (unsigned)name_len, leaf, (unsigned)target_len, + lowest_dist, dist_scratch); + if (distance < 0xFFFF0000U) + distance += fuzzy_distance(name_suf, (unsigned)name_suf_len, fname_suf, + (unsigned)fname_suf_len, 0xFFFF0000U, dist_scratch) * + 10; + if (distance > lowest_dist) continue; + lowest_dist = distance; FuzzyCandidate cand; memcpy(cand.name, name, name_len + 1); cand.size = cand_size; @@ -1647,7 +1677,11 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p } closedir(dir); free(leaf); - fuzzy_edit_buffer_destroy(&ebuf); + free(dist_scratch); + + /* Prefer the exact size+mtime candidate over any name-distance winner. */ + if (exact.name[0]) + best = exact; void* basis = NULL; if (best.name[0]) { @@ -1764,6 +1798,7 @@ typedef struct { long long check_mtime_nsec; uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN]; size_t check_digest_len; + bool dest_exists; /* any destination entry exists (lstat succeeded) */ bool has_old_file; int old_fd; struct stat old_st; @@ -1860,6 +1895,9 @@ static IncrementalCheckOutcome incremental_check_open_destination(IncrementalChe char* leaf = NULL; int parent_fd = file_open_secure_parent(full_path, &leaf, false); if (parent_fd >= 0) { + struct stat dest_st; + if (fstatat(parent_fd, leaf, &dest_st, AT_SYMLINK_NOFOLLOW) == 0) + state->dest_exists = true; /* O_NONBLOCK: an existing FIFO at the destination must not block this openat(); the S_ISREG gate below rejects the non-regular entry. */ state->old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); @@ -1903,6 +1941,82 @@ static IncrementalCheckOutcome incremental_check_report_dest_info(IncrementalChe return INCREMENTAL_CONTINUE; } +/* --ignore-existing short-circuit. The receiver must answer "skip" (STATUS_OK) + BEFORE the sender transmits any payload, otherwise the whole file crosses the + wire only to be discarded at write time. rsync skips an existing destination + entry regardless of its content or type, so the reply depends only on the + lstat existence probe; the ordinary --ignore-existing checks inside + file_receive remain as defense-in-depth for the frame types that have no + per-file check (directories/symlinks/specials/hard-links). */ +static IncrementalCheckOutcome +incremental_check_ignore_existing(const IncrementalCheckState* state) { + if (!state->config->ignore_existing || !state->dest_exists) + return INCREMENTAL_CONTINUE; + if (!send_status(state->fd, STATUS_OK)) + return INCREMENTAL_ERROR; + return INCREMENTAL_SKIP; +} + +/* --link-dest relink of an already up-to-date destination. rsync hard-links a + destination entry to a matching basis even when the entry is already correct, + so a run over an existing tree still maximizes sharing with the basis. Only a + link-dest basis triggers this (copy-dest/compare-dest leave an up-to-date + destination untouched, matching rsync). The ordinary basis path further down + handles every not-up-to-date case, so this helper only adds the relink that + the quick-skip would otherwise short-circuit. */ +static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalCheckState* state, + File** out_file) { + const Config* config = state->config; + if (!config_has_basis(config) || config->ignore_times || config->dry_run) + return INCREMENTAL_CONTINUE; + if (!state->has_old_file) + return INCREMENTAL_CONTINUE; + BasisMatch basis; + basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, + true, true, &basis); + /* Only a link-dest hit relinks; a copy-dest/compare-dest hit (or a miss) lets + the up-to-date check below keep the existing destination. */ + if (!basis.hit || basis.type != BASIS_DEST_LINK) { + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; + } + /* Already the basis inode: nothing to do, leave the destination alone. */ + if (basis.st.st_dev == state->old_st.st_dev && basis.st.st_ino == state->old_st.st_ino) { + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; + } + File* materialized = file_create(state->check_path); + if (materialized && basis.content) { + data_destroy(materialized->data); + materialized->data = basis.content; + basis.content = NULL; + materialized->metadata = file_metadata_create(NULL, &basis.st, false, false); + materialized->skip = true; + materialized->basis_link = basis.basis_path; + basis.basis_path = NULL; + if (!materialized->metadata) { + file_destroy(materialized); + materialized = NULL; + } + } else { + file_destroy(materialized); + materialized = NULL; + } + if (materialized) { + if (!send_status(state->fd, STATUS_OK)) { + basis_match_free(&basis); + file_destroy(materialized); + return INCREMENTAL_ERROR; + } + basis_match_free(&basis); + *out_file = materialized; + return INCREMENTAL_FILE; + } + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; +} + /* Metadata-only (and, when --checksum forces it, content) up-to-date decision. Loads the old contents only when a checksum comparison or delta needs them. */ static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckState* state, @@ -2289,8 +2403,9 @@ static IncrementalCheckOutcome incremental_check_try_fuzzy(IncrementalCheckState if (!config->fuzzy || !config->use_delta) return INCREMENTAL_CONTINUE; unsigned long long fuzzy_size = 0; - void* fuzzy_basis = - fuzzy_basis_find_and_load(config, state->check_path, state->check_size, &fuzzy_size); + void* fuzzy_basis = fuzzy_basis_find_and_load(config, state->check_path, state->check_size, + (time_t)state->check_mtime, + (long)state->check_mtime_nsec, &fuzzy_size); if (fuzzy_basis != NULL) { bool fuzzy_failed = false; File* fuzzy_file = receive_delta_file(state->fd, config, state->check_path, fuzzy_basis, @@ -2351,6 +2466,24 @@ File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, if (outcome == INCREMENTAL_ERROR) goto done; + /* --ignore-existing must answer before any data is requested; it takes + precedence over the metadata up-to-date check below. */ + outcome = incremental_check_ignore_existing(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_SKIP) { + *skipped = true; + goto done; + } + + /* A --link-dest hit relinks even an already up-to-date destination before the + quick-skip can suppress it (rsync parity). */ + outcome = incremental_check_link_dest_relink(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + outcome = incremental_check_quick_skip(&state, &try_delta); if (outcome == INCREMENTAL_ERROR) goto done; @@ -3012,6 +3145,29 @@ typedef struct { bool limit_hit; } DeleteBudgetState; +/* Build the delete-walk protection prefix for one basis directory. The walker + compares paths relative to the receive root, so a relative entry is already + in the right form; an absolute entry that lies below the root is converted to + its root-relative form, and one outside the root returns NULL (the walk + cannot reach it, and it is not protected data beneath the root). */ +static char* basis_delete_relative(const Config* config, const char* path) { + if (!path) + return NULL; + if (path[0] != '/') + return str_dup(path); + const char* root = config->receive_root_directory; + if (!root || root[0] != '/') + return NULL; + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(path, root, root_len) != 0) + return NULL; + if (path[root_len] != '/') + return NULL; /* identical or a sibling sharing a name prefix */ + return str_dup(path + root_len + 1); +} + /* Remove every destination entry under the receive root that is not in the keep-set, bounded by the shared budget (a smaller client --max-delete=NUM replaces the server hard bound; rsync deletes up to the bound and skips the @@ -3042,10 +3198,16 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + (manifest->protected ? manifest->protected->size : 0); DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; if (skip_count > 0) { skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - if (!skips) + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); return false; + } int idx = 0; if (config->delay_updates) { skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; @@ -3053,7 +3215,13 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes idx++; } for (int i = 0; i < config->basis_count; i++) { - skips[idx].prefix = config->basis_dirs[i].path; + /* An absolute basis outside the receive root is unreachable by this walk, + so it contributes no protection prefix (and no slot). */ + char* prefix = basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; skips[idx].top_level_only = false; idx++; } @@ -3062,6 +3230,7 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes skips[idx].top_level_only = false; idx++; } + used = idx; } /* Clamp rather than subtract: an accounting bug where deleted already exceeds max_delete must never underflow into an effectively unlimited budget. */ @@ -3076,7 +3245,12 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes size_t skipped = 0; DeleteWalkResult result = delete_extras_limited(config->receive_root_directory, manifest->keeps, manifest->dirs, - remaining, skips, skip_count, &deleted, &skipped); + remaining, skips, used, &deleted, &skipped); + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); free(skips); budget->deleted += deleted; budget->skipped += skipped; @@ -3113,10 +3287,16 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; if (skip_count > 0) { skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - if (!skips) + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); return false; + } int idx = 0; if (config->delay_updates) { skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; @@ -3124,10 +3304,15 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m idx++; } for (int i = 0; i < config->basis_count; i++) { - skips[idx].prefix = config->basis_dirs[i].path; + char* prefix = basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; skips[idx].top_level_only = false; idx++; } + used = idx; } bool ok = true; for (int i = 0; i < manifest->missing->size; i++) { @@ -3140,7 +3325,7 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m continue; } bool at_root = strchr(rel, '/') == NULL; - if (path_under_skip_prefix(rel, at_root, skips, skip_count)) { + if (path_under_skip_prefix(rel, at_root, skips, used)) { char* escaped = output_escape(rel, log_get_8_bit_output()); log_message(LOG_LEVEL_WARNING, "missing-args path '%s' is protected (staging directory or basis snapshot); " @@ -3264,6 +3449,11 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m if (!ok) break; } + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); free(skips); return ok; } diff --git a/src/shared/filter.c b/src/shared/filter.c index d93d7d6..1033962 100644 --- a/src/shared/filter.c +++ b/src/shared/filter.c @@ -1,163 +1,14 @@ #include "filter.h" #include "log.h" #include "utils.h" +#include #include #include #include #include #include -/* ---- Single rule parsing ---- */ - -static bool rule_text_is_unsupported_word(const char* p, size_t len) { - static const char* const words[] = {"merge", "dir-merge", "hide", "show", - "protect", "risk", "clear"}; - for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) { - size_t wl = strlen(words[i]); - if (len == wl && strncmp(p, words[i], wl) == 0) - return true; - } - return false; -} - -/* rsync include/exclude rule modifiers we do NOT implement. A rule whose +/- is - * immediately followed by one of these is rejected instead of being silently - * parsed as a literal pattern. */ -static bool is_unsupported_rule_modifier(char c) { - return c == '!' || c == 'C' || c == 's' || c == 'r' || c == 'p' || c == 'x'; -} - -FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size) { - if (err && err_size > 0) - err[0] = '\0'; - if (!line) - return NULL; - char* text = str_dup(line); - if (!text) { - if (err) - snprintf(err, err_size, "memory allocation failed"); - return NULL; - } - size_t len = strlen(text); - while (len > 0 && (text[len - 1] == '\n' || text[len - 1] == '\r')) - text[--len] = '\0'; - - const char* p = text; - while (*p == ' ' || *p == '\t') - p++; - if (*p == '\0') { - snprintf(err, err_size, "empty filter rule"); - free(text); - return NULL; - } - - FilterAction action = FILTER_ACTION_EXCLUDE; - if (*p == '+' || *p == '-') { - action = *p == '+' ? FILTER_ACTION_INCLUDE : FILTER_ACTION_EXCLUDE; - p++; - /* rsync attaches rule modifiers directly to the +/- (e.g. "-s foo"). Only - * the '/' anchor modifier is supported; anything else is a clear error - * rather than a silently-ignored literal. */ - if (*p != ' ' && *p != '\t' && *p != '\0' && is_unsupported_rule_modifier(*p)) { - snprintf(err, err_size, - "filter rule modifier '%c' is not supported (only the '/' anchor after +/- " - "is implemented; put a space between +/- and the pattern)", - *p); - free(text); - return NULL; - } - while (*p == ' ' || *p == '\t') - p++; - } else { - /* ':' (dir-merge) and '.' (merge) are rsync filter-rule shorthands. At the - * start of a rule they mean "merge this file", so reject them instead of - * silently turning them into inert exclude patterns. */ - if (*p == ':' || *p == '.' || *p == '!') { - snprintf(err, err_size, - "filter rule starting with '%c' is not supported (merge/dir-merge/list-clear " - "shorthands are not implemented; use +/- include/exclude rules)", - *p); - free(text); - return NULL; - } - const char* sp = p; - while (*sp != '\0' && *sp != ' ' && *sp != '\t') - sp++; - size_t word_len = (size_t)(sp - p); - if (rule_text_is_unsupported_word(p, word_len)) { - snprintf(err, err_size, - "'%.*s' filter directives are not supported (only +/- include/exclude rules " - "with an optional '/' anchor and trailing '/' dir marker)", - (int)word_len, p); - free(text); - return NULL; - } - if (word_len == strlen("include") && strncmp(p, "include", word_len) == 0) { - action = FILTER_ACTION_INCLUDE; - p = sp; - } else if (word_len == strlen("exclude") && strncmp(p, "exclude", word_len) == 0) { - action = FILTER_ACTION_EXCLUDE; - p = sp; - } - while (*p == ' ' || *p == '\t') - p++; - } - - if (*p == '\0') { - snprintf(err, err_size, "filter rule has no pattern"); - free(text); - return NULL; - } - - /* A pattern beginning with '/' is anchored (either as "-/foo" or "- /foo"). */ - bool anchored = false; - if (*p == '/') { - anchored = true; - p++; - while (*p == ' ' || *p == '\t') - p++; - } - if (*p == '\0') { - snprintf(err, err_size, "filter rule has no pattern after '/' anchor"); - free(text); - return NULL; - } - - /* Pattern runs to the end of the rule; a single trailing '/' marks dir-only. */ - size_t pat_len = strlen(p); - bool dir_only = false; - if (pat_len > 1 && p[pat_len - 1] == '/') { - dir_only = true; - pat_len--; - } else if (pat_len == 1 && p[0] == '/') { - /* "//" anchored with nothing after: meaningless. */ - snprintf(err, err_size, "filter rule has no pattern"); - free(text); - return NULL; - } - - FilterRule* rule = calloc(1, sizeof(FilterRule)); - if (!rule) { - snprintf(err, err_size, "memory allocation failed"); - free(text); - return NULL; - } - rule->pattern = malloc(pat_len + 1); - if (!rule->pattern) { - free(rule); - snprintf(err, err_size, "memory allocation failed"); - free(text); - return NULL; - } - memcpy(rule->pattern, p, pat_len); - rule->pattern[pat_len] = '\0'; - rule->action = action; - rule->anchored = anchored; - rule->dir_only = dir_only; - rule->owner = NULL; - free(text); - return rule; -} +/* ---- Ordered rule lists ---- */ void filter_rule_free(FilterRule* rule) { if (!rule) @@ -167,8 +18,6 @@ void filter_rule_free(FilterRule* rule) { free(rule); } -/* ---- Ordered rule lists ---- */ - FilterRuleList* filter_rule_list_create(void) { return calloc(1, sizeof(FilterRuleList)); } @@ -190,28 +39,42 @@ bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) { return true; } -bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err, - size_t err_size) { - FilterRule* rule = filter_rule_parse(line, err, err_size); - if (!rule) - return false; - if (!filter_rule_list_add(list, rule)) { - filter_rule_free(rule); - snprintf(err, err_size, "memory allocation failed"); - return false; - } - return true; -} - void filter_rule_list_free(FilterRuleList* list) { if (!list) return; for (int i = 0; i < list->count; i++) filter_rule_free(list->items[i]); + for (int i = 0; i < list->dir_merge_count; i++) + free(list->dir_merge_names[i]); + free(list->dir_merge_names); free(list->items); free(list); } +/* Register a per-directory merge-file basename (for "dir-merge NAME"/": NAME" + * and -F's .rsync-filter). Duplicate names are ignored. */ +bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name) { + if (!list || !name || name[0] == '\0') + return false; + for (int i = 0; i < list->dir_merge_count; i++) { + if (strcmp(list->dir_merge_names[i], name) == 0) + return true; + } + if (list->dir_merge_count == list->dir_merge_capacity) { + int new_cap = list->dir_merge_capacity > 0 ? list->dir_merge_capacity * 2 : 4; + char** grown = realloc(list->dir_merge_names, (size_t)new_cap * sizeof(char*)); + if (!grown) + return false; + list->dir_merge_names = grown; + list->dir_merge_capacity = new_cap; + } + char* dup = str_dup(name); + if (!dup) + return false; + list->dir_merge_names[list->dir_merge_count++] = dup; + return true; +} + static bool set_rule_owner(FilterRule* rule, const char* owner) { char* dup = str_dup(owner ? owner : ""); if (!dup) @@ -221,7 +84,311 @@ static bool set_rule_owner(FilterRule* rule, const char* owner) { return true; } -/* ---- CVS default excludes (-C) ---- */ +/* ---- Rule parsing ---- */ + +/* A short rule prefix is a single character; a long rule name is alphabetic + * (with '-'). `is_short` distinguishes the modifier-attachment rules. */ +typedef enum { + RULE_KIND_EXCLUDE, + RULE_KIND_INCLUDE, + RULE_KIND_HIDE, + RULE_KIND_SHOW, + RULE_KIND_PROTECT, + RULE_KIND_RISK, + RULE_KIND_MERGE, + RULE_KIND_DIR_MERGE, + RULE_KIND_CLEAR, + RULE_KIND_UNKNOWN, +} RuleKind; + +static bool short_rule_char(char c, RuleKind* kind) { + switch (c) { + case '-': + *kind = RULE_KIND_EXCLUDE; + return true; + case '+': + *kind = RULE_KIND_INCLUDE; + return true; + case 'H': + *kind = RULE_KIND_HIDE; + return true; + case 'S': + *kind = RULE_KIND_SHOW; + return true; + case 'P': + *kind = RULE_KIND_PROTECT; + return true; + case 'R': + *kind = RULE_KIND_RISK; + return true; + case '.': + *kind = RULE_KIND_MERGE; + return true; + case ':': + *kind = RULE_KIND_DIR_MERGE; + return true; + case '!': + *kind = RULE_KIND_CLEAR; + return true; + default: + return false; + } +} + +static bool long_rule_name(const char* name, size_t len, RuleKind* kind) { + struct { + const char* word; + RuleKind kind; + } table[] = { + {"exclude", RULE_KIND_EXCLUDE}, {"include", RULE_KIND_INCLUDE}, + {"hide", RULE_KIND_HIDE}, {"show", RULE_KIND_SHOW}, + {"protect", RULE_KIND_PROTECT}, {"risk", RULE_KIND_RISK}, + {"merge", RULE_KIND_MERGE}, {"dir-merge", RULE_KIND_DIR_MERGE}, + {"clear", RULE_KIND_CLEAR}, + }; + for (size_t i = 0; i < sizeof(table) / sizeof(table[0]); i++) { + if (strlen(table[i].word) == len && strncmp(name, table[i].word, len) == 0) { + *kind = table[i].kind; + return true; + } + } + return false; +} + +static bool is_modifier_char(char c) { + return c == 's' || c == 'r' || c == 'p' || c == 'x' || c == '/' || c == '!' || c == 'C'; +} + +/* Parse "RULE[,MODIFIERS] [PATTERN]". On success `kind`, `sides`, + * `sides_explicit`, `negate`, `anchored_mod`, `perishable`, `xattr`, + * `cvs_inject` and the pattern span (`pat_start`/`pat_len`, possibly 0 for + * merge/clear) are filled. Returns true on success. */ +static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides, + bool* sides_explicit, bool* negate, bool* anchored_mod, + bool* perishable, bool* xattr, bool* cvs_inject, + const char** pat_start, size_t* pat_len) { + const char* p = text; + *sides = FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER; + *sides_explicit = false; + *negate = false; + *anchored_mod = false; + *perishable = false; + *xattr = false; + *cvs_inject = false; + *pat_start = NULL; + *pat_len = 0; + + bool is_short = false; + if (short_rule_char(*p, kind)) { + is_short = true; + p++; + } else { + const char* name_start = p; + while (isalpha((unsigned char)*p) || *p == '-') + p++; + size_t name_len = (size_t)(p - name_start); + if (name_len == 0 || !long_rule_name(name_start, name_len, kind)) + return false; + /* A long name must be followed by a separator, a comma or the end. */ + if (*p != '\0' && *p != ',' && *p != ' ' && *p != '_') + return false; + } + + /* Modifiers: long names require a comma; short names may attach directly. + Only commit a modifier run that terminates at a separator or the end, so a + pattern such as "*.tmp" written as "-*.tmp" is not mistaken for modifiers. */ + const char* mod_start = p; + const char* mod_end = p; + if (*p == ',') { + p++; + mod_start = p; + while (is_modifier_char(*p)) + p++; + mod_end = p; + } else if (is_short) { + const char* scan = p; + while (is_modifier_char(*scan)) + scan++; + if (*scan == '\0' || *scan == ' ' || *scan == '_') { + mod_start = p; + mod_end = scan; + p = scan; + } + } + for (const char* m = mod_start; m < mod_end; m++) { + switch (*m) { + case 's': + *sides = FILTER_SIDE_SENDER; + *sides_explicit = true; + break; + case 'r': + *sides = FILTER_SIDE_RECEIVER; + *sides_explicit = true; + break; + case '!': + *negate = true; + break; + case '/': + *anchored_mod = true; + break; + case 'p': + *perishable = true; + break; + case 'x': + *xattr = true; + break; + case 'C': + *cvs_inject = true; + break; + default: + break; + } + } + + /* A single space or underscore separates the rule/modifiers from the + pattern; further spaces/underscores belong to the pattern. */ + const char* pat = p; + if (*pat == ' ' || *pat == '_') + pat++; + /* Trim a trailing newline/CR (the caller may pass a raw file line). */ + *pat_start = pat; + *pat_len = strlen(pat); + while (*pat_len > 0 && (pat[*pat_len - 1] == '\n' || pat[*pat_len - 1] == '\r')) + (*pat_len)--; + return true; +} + +FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err, + size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + if (!line) + return NULL; + const char* p = line; + while (*p == ' ' || *p == '\t') + p++; + if (*p == '\0' || *p == '\n' || *p == '\r') { + snprintf(err, err_size, "empty filter rule"); + return NULL; + } + + RuleKind kind = RULE_KIND_UNKNOWN; + unsigned sides; + bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject; + const char* pat; + size_t pat_len; + if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable, + &xattr, &cvs_inject, &pat, &pat_len)) { + snprintf(err, err_size, "unrecognized filter rule syntax"); + return NULL; + } + if (cvs_inject) { + /* The C modifier expands to the CVS defaults in place; the rule itself + carries no pattern and is handled by the caller. */ + snprintf(err, err_size, "the C modifier is handled by the rule-list parser"); + return NULL; + } + if (kind == RULE_KIND_MERGE || kind == RULE_KIND_DIR_MERGE) { + snprintf(err, err_size, "merge/dir-merge rules are handled by the rule-list parser"); + return NULL; + } + if (kind == RULE_KIND_CLEAR) { + if (pat_len != 0) { + snprintf(err, err_size, "clear takes no pattern"); + return NULL; + } + FilterRule* rule = calloc(1, sizeof(FilterRule)); + if (!rule) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + rule->action = FILTER_ACTION_NONE; /* clear marker: no pattern */ + rule->sides = 0; + return rule; + } + + FilterAction action; + switch (kind) { + case RULE_KIND_INCLUDE: + case RULE_KIND_SHOW: + case RULE_KIND_RISK: + action = FILTER_ACTION_INCLUDE; + break; + default: + action = FILTER_ACTION_EXCLUDE; + break; + } + if (kind == RULE_KIND_HIDE) + sides = FILTER_SIDE_SENDER; + else if (kind == RULE_KIND_SHOW) + sides = FILTER_SIDE_SENDER; + else if (kind == RULE_KIND_PROTECT) + sides = FILTER_SIDE_RECEIVER; + else if (kind == RULE_KIND_RISK) + sides = FILTER_SIDE_RECEIVER; + if (kind == RULE_KIND_HIDE || kind == RULE_KIND_SHOW || kind == RULE_KIND_PROTECT || + kind == RULE_KIND_RISK) + sides_explicit = true; + /* --delete-excluded turns an unqualified (no explicit s/r) rule into a + sender-side-only rule, so it no longer protects the receiver. */ + if (opts && opts->delete_excluded && !sides_explicit) + sides = FILTER_SIDE_SENDER; + + if (pat_len == 0) { + snprintf(err, err_size, "filter rule has no pattern"); + return NULL; + } + + bool anchored = anchored_mod; + const char* pat_begin = pat; + if (*pat_begin == '/') { + anchored = true; + pat_begin++; + /* Drop the spaces that could follow the anchor in the "-/ foo" form. */ + while (*pat_begin == ' ' || *pat_begin == '\t') + pat_begin++; + pat_len = strlen(pat_begin); + while (pat_len > 0 && (pat_begin[pat_len - 1] == '\n' || pat_begin[pat_len - 1] == '\r')) + pat_len--; + } + if (pat_len == 0) { + snprintf(err, err_size, "filter rule has no pattern after '/' anchor"); + return NULL; + } + bool dir_only = false; + if (pat_len > 1 && pat_begin[pat_len - 1] == '/') { + dir_only = true; + pat_len--; + } + if (pat_len == 0) { + snprintf(err, err_size, "filter rule has no pattern"); + return NULL; + } + + FilterRule* rule = calloc(1, sizeof(FilterRule)); + if (!rule) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + rule->pattern = malloc(pat_len + 1); + if (!rule->pattern) { + free(rule); + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + memcpy(rule->pattern, pat_begin, pat_len); + rule->pattern[pat_len] = '\0'; + rule->action = action; + rule->sides = sides; + rule->anchored = anchored; + rule->dir_only = dir_only; + rule->negate = negate; + rule->perishable = perishable; + (void)xattr; /* xattr-name rules never match file/dir names; accepted/ignored */ + return rule; +} + +/* ---- CVS default excludes (-C and the C modifier) ---- */ typedef struct { const char* pattern; @@ -240,12 +407,13 @@ static const CvsDefaultRule CVS_DEFAULTS[] = { {".svn/", true}, {".git/", true}, {".hg/", true}, {".bzr/", true}, }; -static bool cvs_rule_list_append(FilterRuleList* list) { +static bool filter_list_append_cvs(FilterRuleList* list, unsigned sides) { for (size_t i = 0; i < sizeof(CVS_DEFAULTS) / sizeof(CVS_DEFAULTS[0]); i++) { FilterRule* rule = calloc(1, sizeof(FilterRule)); if (!rule) return false; rule->action = FILTER_ACTION_EXCLUDE; + rule->sides = sides; rule->dir_only = CVS_DEFAULTS[i].dir_only; size_t plen = strlen(CVS_DEFAULTS[i].pattern); if (rule->dir_only && plen > 0 && CVS_DEFAULTS[i].pattern[plen - 1] == '/') @@ -269,8 +437,167 @@ static bool cvs_rule_list_append(FilterRuleList* list) { return true; } +#define FILTER_MAX_MERGE_DEPTH 16 + +static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line, + const FilterParseOptions* opts, const char* base_dir, + int depth, char* err, size_t err_size); + +/* Read a merge file and splice its rules into `list`. A relative path is + * resolved below `base_dir` when given, else used as-is (rsync resolves a + * command-line merge file relative to the current directory). */ +static bool filter_list_merge_file(FilterRuleList* list, const char* name, + const FilterParseOptions* opts, const char* base_dir, int depth, + char* err, size_t err_size) { + if (name[0] == '\0') { + snprintf(err, err_size, "merge requires a filename"); + return false; + } + char* path = + (base_dir && base_dir[0] && name[0] != '/') ? path_cat(base_dir, name) : str_dup(name); + if (!path) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + FILE* fp = fopen(path, "r"); + if (!fp) { + snprintf(err, err_size, "could not read merge file '%s': %s", path, strerror(errno)); + free(path); + return false; + } + char* line = NULL; + size_t cap = 0; + bool ok = true; + while (true) { + ssize_t n = utils_getdelim_bounded(fp, &line, &cap, '\n', UTILS_MAX_LINE_LEN); + if (n < 0) { + snprintf(err, err_size, "error reading merge file '%s'", path); + ok = false; + break; + } + if (n == 0) + break; + const char* lp = line; + while (*lp == ' ' || *lp == '\t') + lp++; + if (*lp == '\0' || *lp == '\n' || *lp == '\r' || *lp == '#') + continue; + if (!filter_list_parse_append_depth(list, lp, opts, base_dir, depth + 1, err, err_size)) { + ok = false; + break; + } + } + free(line); + fclose(fp); + free(path); + return ok; +} + +/* Parse one line and append/merge it into `list`. Handles clear, merge and + * dir-merge at the list level. */ +static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line, + const FilterParseOptions* opts, const char* base_dir, + int depth, char* err, size_t err_size) { + if (depth > FILTER_MAX_MERGE_DEPTH) { + snprintf(err, err_size, "merge files nested too deeply"); + return false; + } + const char* p = line; + while (*p == ' ' || *p == '\t') + p++; + if (*p == '\0' || *p == '\n' || *p == '\r') + return true; + + RuleKind kind = RULE_KIND_UNKNOWN; + unsigned sides; + bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject; + const char* pat; + size_t pat_len; + if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable, + &xattr, &cvs_inject, &pat, &pat_len)) { + snprintf(err, err_size, "unrecognized filter rule syntax: %s", p); + return false; + } + (void)sides_explicit; + (void)negate; + (void)anchored_mod; + (void)perishable; + (void)xattr; + + if (cvs_inject) { + /* "C" injects the CVS defaults in place; no pattern is expected. */ + return filter_list_append_cvs(list, sides); + } + if (kind == RULE_KIND_CLEAR) { + if (pat_len != 0) { + snprintf(err, err_size, "clear takes no pattern"); + return false; + } + for (int i = 0; i < list->count; i++) + filter_rule_free(list->items[i]); + list->count = 0; + return true; + } + if (kind == RULE_KIND_MERGE) { + if (pat_len == 0) { + snprintf(err, err_size, "merge requires a filename"); + return false; + } + char* name = malloc(pat_len + 1); + if (!name) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + memcpy(name, pat, pat_len); + name[pat_len] = '\0'; + bool ok = filter_list_merge_file(list, name, opts, base_dir, depth, err, err_size); + free(name); + return ok; + } + if (kind == RULE_KIND_DIR_MERGE) { + if (pat_len == 0) { + snprintf(err, err_size, "dir-merge requires a filename"); + return false; + } + char* name = malloc(pat_len + 1); + if (!name) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + memcpy(name, pat, pat_len); + name[pat_len] = '\0'; + bool ok = filter_rule_list_add_dir_merge(list, name); + free(name); + if (!ok) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + return true; + } + + FilterRule* rule = filter_rule_parse(p, opts, err, err_size); + if (!rule) + return false; + if (!filter_rule_list_add(list, rule)) { + filter_rule_free(rule); + snprintf(err, err_size, "memory allocation failed"); + return false; + } + return true; +} + +bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, + const FilterParseOptions* opts, const char* merge_base_dir, + char* err, size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + if (!list) + return false; + return filter_list_parse_append_depth(list, line, opts, merge_base_dir, 0, err, err_size); +} + FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude, - char* err, size_t err_size) { + bool delete_excluded, char* err, size_t err_size) { if (err && err_size > 0) err[0] = '\0'; FilterRuleList* list = filter_rule_list_create(); @@ -278,28 +605,16 @@ FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, snprintf(err, err_size, "memory allocation failed"); return NULL; } + FilterParseOptions opts = {.delete_excluded = delete_excluded, .cvs_exclude = cvs_exclude}; for (int i = 0; i < rule_count; i++) { if (!rule_texts || !rule_texts[i]) continue; - FilterRule* rule = filter_rule_parse(rule_texts[i], err, err_size); - if (!rule) { + if (!filter_rule_list_parse_append(list, rule_texts[i], &opts, NULL, err, err_size)) { filter_rule_list_free(list); return NULL; } - if (!set_rule_owner(rule, "")) { - filter_rule_free(rule); - filter_rule_list_free(list); - snprintf(err, err_size, "memory allocation failed"); - return NULL; - } - if (!filter_rule_list_add(list, rule)) { - filter_rule_free(rule); - filter_rule_list_free(list); - snprintf(err, err_size, "memory allocation failed"); - return NULL; - } } - if (cvs_exclude && !cvs_rule_list_append(list)) { + if (cvs_exclude && !filter_list_append_cvs(list, FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER)) { filter_rule_list_free(list); snprintf(err, err_size, "memory allocation failed"); return NULL; @@ -307,38 +622,36 @@ FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, return list; } -/* ---- Per-directory .rsync-filter files ---- */ +/* ---- Per-directory merge files ---- */ -FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists, - char* err, size_t err_size) { +bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name, + const char* owner_rel, const FilterParseOptions* opts, bool* exists, + char* err, size_t err_size) { if (err && err_size > 0) err[0] = '\0'; if (exists) *exists = false; - char* filter_path = path_cat(dir_path, ".rsync-filter"); + if (!list) + return false; + char* filter_path = path_cat(dir_path, name); if (!filter_path) { snprintf(err, err_size, "memory allocation failed"); - return NULL; + return false; } FILE* fp = fopen(filter_path, "r"); free(filter_path); if (!fp) { if (errno == ENOENT || errno == ENOTDIR) - return filter_rule_list_create(); + return true; char* escaped_dir = output_escape(dir_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s", + log_message(LOG_LEVEL_WARNING, "Could not read %s in %s: %s", name, escaped_dir ? escaped_dir : "", strerror(errno)); free(escaped_dir); - return filter_rule_list_create(); + return true; } if (exists) *exists = true; - FilterRuleList* list = filter_rule_list_create(); - if (!list) { - fclose(fp); - snprintf(err, err_size, "memory allocation failed"); - return NULL; - } + int rules_before = list->count; char* line = NULL; size_t line_cap = 0; bool ok = true; @@ -346,9 +659,9 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, '\n', UTILS_MAX_LINE_LEN); if (n < 0) { if (errno == EFBIG) { - snprintf(err, err_size, "line in .rsync-filter exceeds %d bytes", (int)UTILS_MAX_LINE_LEN); + snprintf(err, err_size, "line in %s exceeds %d bytes", name, (int)UTILS_MAX_LINE_LEN); } else { - snprintf(err, err_size, "error reading .rsync-filter: %s", strerror(errno)); + snprintf(err, err_size, "error reading %s: %s", name, strerror(errno)); } ok = false; break; @@ -360,20 +673,9 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo p++; if (*p == '\0' || *p == '\n' || *p == '\r' || *p == '#') continue; - FilterRule* rule = filter_rule_parse(p, err, err_size); - if (!rule) { - ok = false; - break; - } - if (!set_rule_owner(rule, owner_rel)) { - filter_rule_free(rule); - snprintf(err, err_size, "memory allocation failed"); - ok = false; - break; - } - if (!filter_rule_list_add(list, rule)) { - filter_rule_free(rule); - snprintf(err, err_size, "memory allocation failed"); + /* Merge files inside a per-directory file resolve relative to that + directory. */ + if (!filter_list_parse_append_depth(list, p, opts, dir_path, 0, err, err_size)) { ok = false; break; } @@ -381,12 +683,43 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo free(line); fclose(fp); if (!ok) { + /* Drop only the rules this file appended, leaving the caller's earlier + content untouched. */ + for (int i = rules_before; i < list->count; i++) + filter_rule_free(list->items[i]); + list->count = rules_before; + return false; + } + for (int i = rules_before; i < list->count; i++) { + if (!set_rule_owner(list->items[i], owner_rel)) { + snprintf(err, err_size, "memory allocation failed"); + return false; + } + } + return true; +} + +FilterRuleList* filter_file_read_named(const char* dir_path, const char* name, + const char* owner_rel, const FilterParseOptions* opts, + bool* exists, char* err, size_t err_size) { + FilterRuleList* list = filter_rule_list_create(); + if (!list) { + if (err && err_size > 0) + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + if (!filter_file_append(list, dir_path, name, owner_rel, opts, exists, err, err_size)) { filter_rule_list_free(list); return NULL; } return list; } +FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists, + char* err, size_t err_size) { + return filter_file_read_named(dir_path, ".rsync-filter", owner_rel, NULL, exists, err, err_size); +} + /* ---- Rule matching ---- */ /* Match a pattern that contains '/' (non-anchored) against the end of the @@ -402,10 +735,10 @@ static bool glob_suffix_match(const char* pattern, const char* str) { } static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, const char* leaf, - bool is_dir) { + bool is_dir, unsigned side) { if (!rule || !rule->pattern) return FILTER_ACTION_NONE; - if (rule->dir_only && !is_dir) + if (!(rule->sides & side)) return FILTER_ACTION_NONE; /* A rule applies only to entries below its owner directory. */ const char* rel2 = rel_path; @@ -420,24 +753,36 @@ static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, c if (rel2[0] == '\0') return FILTER_ACTION_NONE; bool matched; - if (rule->anchored) { + if (rule->dir_only && !is_dir) + matched = false; + else if (rule->anchored) matched = glob_match(rule->pattern, rel2); - } else if (strchr(rule->pattern, '/') != NULL) { + else if (strchr(rule->pattern, '/') != NULL) matched = glob_suffix_match(rule->pattern, rel2); - } else { + else matched = glob_match(rule->pattern, leaf); - } - return matched ? rule->action : FILTER_ACTION_NONE; + if (rule->negate) + matched = !matched; + if (!matched) + return FILTER_ACTION_NONE; + if (side == FILTER_SIDE_RECEIVER) + return rule->action == FILTER_ACTION_EXCLUDE ? FILTER_ACTION_PROTECT : FILTER_ACTION_RISK; + return rule->action; } -FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, - bool is_dir) { +FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path, + const char* leaf, bool is_dir, unsigned side) { if (!list) return FILTER_ACTION_NONE; for (int i = 0; i < list->count; i++) { - FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir); + FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir, side); if (action != FILTER_ACTION_NONE) return action; } return FILTER_ACTION_NONE; } + +FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, + bool is_dir) { + return filter_rules_apply_side(list, rel_path, leaf, is_dir, FILTER_SIDE_SENDER); +} diff --git a/src/shared/filter.h b/src/shared/filter.h index 8c43f27..97a373f 100644 --- a/src/shared/filter.h +++ b/src/shared/filter.h @@ -4,79 +4,133 @@ #include #include -/* rsync-style filter rule engine (client-side file selection). +/* rsync-style filter rule engine (client-side file selection and the + * receiver-side protection set it feeds). * - * Supported rule syntax (documented subset): - * [+|-] [anchored '/' prefix] pattern [trailing '/' for dir-only] - * - * "+ PATTERN" include rule (first match wins) - * "- PATTERN" exclude rule - * "PATTERN" implicit exclude rule (rsync default) - * "include PATTERN" / "exclude PATTERN" word forms - * leading '/' after the +/- anchors the pattern to its owner directory - * (the transfer root for command-line/-C rules, the directory that - * contains a .rsync-filter file for per-directory rules) - * a trailing '/' makes the rule match directories only - * - * Rejected explicitly (no silent no-ops): the rsync merge/dir-merge/list-clear - * shorthands written as a rule that starts with ':' or '.' or '!', the - * merge/dir-merge/hide/show/protect/risk/clear words, and every include/exclude - * rule modifier other than '/' (! C s r p x). The pattern must be separated - * from +/- by a space (or a single '/' anchor), exactly like rsync's - * "-s foo"/"-p ..." modifier syntax is refused. + * Rule syntax (see the rsync man page FILTER RULES section): + * RULE [PATTERN_OR_FILENAME] + * RULE,MODIFIERS [PATTERN_OR_FILENAME] + * Short RULE names may attach MODIFIERS directly ("-sr foo"); the long name + * form requires the comma. The pattern/filename is separated from the rule by + * one space or underscore. Rule names: + * exclude/- exclude (by default both sender-hide and receiver-protect) + * include/+ include (by default both sender-show and receiver-risk) + * hide/H sender-only exclude + * show/S sender-only include + * protect/P receiver-only exclude (protect from deletion) + * risk/R receiver-only include (allow deletion) + * merge/. read a client-side merge file for more rules + * dir-merge/: per-directory merge file (registered for the scanner) + * clear/! clear the current rule list (takes no argument) + * Modifiers: '/' absolute anchor, '!' negate match, 'C' inject CVS defaults, + * 's' sender side, 'r' receiver side, 'p' perishable, 'x' xattr name rule. + * A trailing '/' makes a pattern match directories only. A leading '/' anchors + * the pattern to its owner directory. */ typedef enum { FILTER_ACTION_NONE = 0, /* no rule matched */ FILTER_ACTION_EXCLUDE = -1, - FILTER_ACTION_INCLUDE = 1 + FILTER_ACTION_INCLUDE = 1, + /* Receiver-side-only verdicts: the entry is transferred but its destination + * mirror is protected from --delete (PROTECT) or explicitly left at risk + * (RISK). */ + FILTER_ACTION_PROTECT = 2, + FILTER_ACTION_RISK = 3, } FilterAction; +#define FILTER_SIDE_SENDER 1u +#define FILTER_SIDE_RECEIVER 2u + typedef struct { - FilterAction action; - bool anchored; /* pattern anchored to the rule's owner directory */ - bool dir_only; /* pattern had a trailing '/': matches directories only */ - char* owner; /* owning directory rel path ("" == transfer root) */ - char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */ + FilterAction action; /* EXCLUDE or INCLUDE (the base pattern action) */ + unsigned sides; /* FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER */ + bool anchored; /* pattern anchored to the rule's owner directory */ + bool dir_only; /* pattern had a trailing '/': matches directories only */ + bool negate; /* '!' modifier: match succeeds when the pattern does not */ + bool perishable; /* 'p' modifier (ignored in deleted directories) */ + char* owner; /* owning directory rel path ("" == transfer root) */ + char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */ } FilterRule; typedef struct { FilterRule** items; /* owned array of rule pointers */ int count; int capacity; + /* Per-directory merge-file basenames registered by "dir-merge NAME"/": NAME" + * or by -F (.rsync-filter). Owned strings; the scanner reads each name in + * every directory it traverses. */ + char** dir_merge_names; + int dir_merge_count; + int dir_merge_capacity; } FilterRuleList; +/* Context needed while parsing a rule list (merge files, --delete-excluded). */ +typedef struct { + bool delete_excluded; /* --delete-excluded: default sides become sender-only */ + bool cvs_exclude; /* -C: expand the CVS default excludes */ +} FilterParseOptions; + /* Parse a single filter-rule line (no trailing newline required). Returns an - * owned rule, or NULL on unsupported/invalid syntax with a message in `err`. */ -FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size); + * owned rule, or NULL on unsupported/invalid syntax with a message in `err`. + * `opts` may be NULL (no merge expansion / no delete-excluded). */ +FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err, + size_t err_size); void filter_rule_free(FilterRule* rule); FilterRuleList* filter_rule_list_create(void); /* Append a fully-parsed rule (takes ownership). Returns false on OOM. */ bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule); -/* Parse `line` and append it. Returns false and fills `err` on bad syntax. */ -bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err, - size_t err_size); +/* Register a per-directory merge-file basename (idempotent). Returns false on + * OOM. Used by the scanner to read custom "dir-merge" files. */ +bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name); +/* Parse `line` and append it. Handles "clear"/"!" (resets the list), "merge + * FILE"/". FILE" (splices the file's rules) and "dir-merge NAME"/": NAME" + * (registers a per-directory filename). Returns false and fills `err` on bad + * syntax or an unreadable merge file. `merge_base_dir` resolves a relative + * merge-file path (NULL means the process working directory). */ +bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, + const FilterParseOptions* opts, const char* merge_base_dir, + char* err, size_t err_size); void filter_rule_list_free(FilterRuleList* list); /* Build the command-line filter set: `rule_texts` (--filter=RULE in the order * given, 0..rule_count) followed by the -C CVS default excludes when - * cvs_exclude is true. All rules are owned by "" (the transfer root). + * cvs_exclude is true. All rules are owned by "" (the transfer root). * Returns NULL on unsupported rule text (message in `err`). */ FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude, - char* err, size_t err_size); + bool delete_excluded, char* err, size_t err_size); -/* Read "/.rsync-filter" and return its rules, each owned by - * `owner_rel`. A missing file yields an empty list with *exists=false; an - * unreadable file is treated as missing. Returns NULL only on parse or - * allocation failure (message in `err`). */ +/* Read "/" and return its rules, each owned by `owner_rel`. A + * missing file yields an empty list with *exists=false; an unreadable file is + * treated as missing. Returns NULL only on parse or allocation failure + * (message in `err`). `opts` may be NULL. */ +FilterRuleList* filter_file_read_named(const char* dir_path, const char* name, + const char* owner_rel, const FilterParseOptions* opts, + bool* exists, char* err, size_t err_size); + +/* Append the rules of "/" into an existing list (each owned by + * `owner_rel`). A missing file yields *exists=false and no error. Returns + * false only on parse/allocation failure (message in `err`). */ +bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name, + const char* owner_rel, const FilterParseOptions* opts, bool* exists, + char* err, size_t err_size); + +/* filter_file_read_named with the default ".rsync-filter" name. */ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists, char* err, size_t err_size); -/* Evaluate an entry against one ordered rule list. Returns FILTER_ACTION_NONE - * when no rule matched, otherwise the first matching rule's action. - * `rel_path` is the entry's path relative to the transfer root ("" == root), - * `leaf` its final name, `is_dir` whether it is a directory. */ +/* Evaluate an entry against one ordered rule list for one side. Returns + * FILTER_ACTION_NONE when no rule matched, otherwise the first matching rule's + * action (for the receiver side an EXCLUDE is reported as + * FILTER_ACTION_PROTECT and an INCLUDE as FILTER_ACTION_RISK). `rel_path` is + * the entry's path relative to the transfer root ("" == root), `leaf` its final + * name, `is_dir` whether it is a directory. */ +FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path, + const char* leaf, bool is_dir, unsigned side); + +/* Sender-side convenience wrapper (kept for callers/tests that only need the + * transfer decision). */ FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, bool is_dir); diff --git a/src/shared/identity.c b/src/shared/identity.c index c806789..1f717d5 100644 --- a/src/shared/identity.c +++ b/src/shared/identity.c @@ -589,6 +589,67 @@ static int identity_split_chown(const char* value, char** puser, char** pgroup) return 0; } +/* --chown is rsync's shorthand for "--usermap=*:USER --groupmap=*:GROUP", so a + * name TO value must be resolved on the RECEIVER, not on the sender. Append the + * equivalent map rule (FROM matches every id). The numeric/'*' forms are stored + * numerically exactly as rsync's id_parse/user_to_uid would. Returns 0 on + * success, -1 on a malformed numeric token or allocation failure. */ +static int identity_append_chown_rule(Config* config, bool is_group, const char* token) { + IdentityMap rule; + memset(&rule, 0, sizeof(rule)); + rule.from = IDENTITY_MATCH_ANY; + rule.from_hi = IDENTITY_MATCH_ANY; + if (strcmp(token, "*") == 0) { + rule.to = IDENTITY_CURRENT; + } else if (identity_all_digits(token[0] == '@' ? token + 1 : token)) { + if (identity_resolve_token(token, is_group, &rule.to) != 0) { + log_message(LOG_LEVEL_ERROR, "--chown numeric id is out of range: %s", token); + return -1; + } + } else { + rule.to = 0; + rule.to_name = str_dup(token); + if (!rule.to_name) + return -1; + } + if (identity_append_rule(is_group ? &config->groupmap : &config->usermap, + is_group ? &config->groupmap_count : &config->usermap_count, + &rule) != 0) { + free(rule.to_name); + log_message(LOG_LEVEL_ERROR, "--chown has too many rules (max %d)", MAX_IDENTITY_MAP); + return -1; + } + return 0; +} + +/* Resolve/record one --chown side. The source-side numeric value is kept in + * chown_uid/chown_gid purely as a fallback (the appended map rule resolves the + * name on the receiver and wins); a name that does not exist on the sender is + * accepted and left to receiver-side resolution, matching rsync. */ +static int identity_parse_chown_side(Config* config, bool is_group, const char* token) { + if (identity_append_chown_rule(config, is_group, token) != 0) + return -1; + bool numeric = identity_all_digits(token[0] == '@' ? token + 1 : token); + int32_t resolved; + if (identity_resolve_token(token, is_group, &resolved) == 0) { + if (is_group) { + config->chown_gid = resolved; + config->chown_gid_set = true; + } else { + config->chown_uid = resolved; + config->chown_uid_set = true; + } + return 0; + } + if (numeric) { + log_message(LOG_LEVEL_ERROR, "--chown could not resolve numeric id '%s'", token); + return -1; + } + /* Unknown sender-side name: rsync accepts it and resolves it (or warns) on + * the receiver; do the same instead of failing the whole run. */ + return 0; +} + int identity_parse_chown(Config* config, const char* value) { if (!config || !value || *value == '\0') { log_message(LOG_LEVEL_ERROR, "--chown requires a value (USER:GROUP, USER, or :GROUP)"); @@ -627,32 +688,18 @@ int identity_parse_chown(Config* config, const char* value) { if (*user == '\0') { log_message(LOG_LEVEL_ERROR, "--chown requires a user or group (got '%s')", value); ret = -1; - } else if (identity_resolve_token(user, false, &config->chown_uid) != 0) { - log_message(LOG_LEVEL_ERROR, - "--chown could not resolve user '%s' (use a name that exists " - "on the source, '*', or @N)", - value); + } else if (identity_parse_chown_side(config, false, user) != 0) { ret = -1; - } else { - config->chown_uid_set = true; } } else { /* --chown=USER:GROUP, --chown=:GROUP, --chown=USER: */ - if (*user != '\0') { - if (identity_resolve_token(user, false, &config->chown_uid) != 0) { - log_message(LOG_LEVEL_ERROR, "--chown could not resolve user '%s'", value); - ret = -1; - goto done; - } - config->chown_uid_set = true; + if (*user != '\0' && identity_parse_chown_side(config, false, user) != 0) { + ret = -1; + goto done; } - if (*group != '\0') { - if (identity_resolve_token(group, true, &config->chown_gid) != 0) { - log_message(LOG_LEVEL_ERROR, "--chown could not resolve group '%s'", value); - ret = -1; - goto done; - } - config->chown_gid_set = true; + if (*group != '\0' && identity_parse_chown_side(config, true, group) != 0) { + ret = -1; + goto done; } if (!*user && !*group) { log_message(LOG_LEVEL_ERROR, "--chown must set a user, a group, or both (got '%s')", value); diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index d9270e1..3784601 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -4963,15 +4963,20 @@ class TestFuzzy: "no-candidate fuzzy run should have sent the whole file" def test_dissimilar_sibling_is_not_used(self, shared_server): - # The destination holds a large sibling whose basename is too different - # from the incoming name; the name gate must reject it and fall back to - # a whole-file transfer. + # A sibling whose basename is too different from the incoming name is + # rejected by rsync's fuzzy distance window (the length gap exceeds + # 25), so the run falls back to a whole-file transfer. A distinct + # mtime keeps rsync's exact size+mtime first pass from accepting it. source, dest = self._prepare("dissim") old_bytes, new_bytes = _random_payloads() - self._seed_dest(source, dest, {"totally-unrelated-notes.bin": old_bytes}, + long_name = "totally-unrelated-notes-with-a-very-long-name.bin" + self._seed_dest(source, dest, {long_name: old_bytes}, shared_server.port) with open(os.path.join(source, self.NEW_NAME), "wb") as fh: fh.write(new_bytes) + received_dir = get_dest_received_dir(dest, source) + os.utime(os.path.join(received_dir, long_name), (self.TS, self.TS)) + os.utime(os.path.join(source, self.NEW_NAME), (self.TS + 100000, self.TS + 100000)) result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) assert result.returncode == 0, \ f"--fuzzy dissimilar-sibling run failed: {(result.stderr or result.stdout)[:300]}" @@ -4980,6 +4985,28 @@ class TestFuzzy: assert proxy.client_to_server > len(new_bytes) // 2, \ "a dissimilar-named sibling must not be used as a fuzzy basis" + def test_exact_size_mtime_sibling_is_used(self, shared_server): + # rsync's fuzzy first pass accepts a sibling with an exact size+mtime + # match regardless of how unrelated its name is (its content is almost + # certainly the same). + source, dest = self._prepare("exact") + old_bytes, new_bytes = _random_payloads() + self._seed_dest(source, dest, {"unrelated-blob.bin": old_bytes}, + shared_server.port) + with open(os.path.join(source, self.NEW_NAME), "wb") as fh: + fh.write(new_bytes) + received_dir = get_dest_received_dir(dest, source) + ts = 1600000000 + os.utime(os.path.join(received_dir, "unrelated-blob.bin"), (ts, ts)) + os.utime(os.path.join(source, self.NEW_NAME), (ts, ts)) + result, proxy = self._run_measured(source, dest, ["--fuzzy"], shared_server.port) + assert result.returncode == 0, \ + f"--fuzzy exact size+mtime run failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.NEW_NAME)) == new_bytes + assert proxy.client_to_server < len(new_bytes) // 4, \ + "an exact size+mtime sibling should be used as a fuzzy basis" + def test_fuzzy_helps_when_dest_holds_an_unsuitable_file(self, shared_server): # The destination DOES hold the exact new name, but it is a tiny stale # file (below the delta engine's minimum, ratio far outside its window), diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index 66c6f3d..5503814 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -1044,10 +1044,11 @@ static void test_parse_args_basis_dirs() { config_delete(cfg); } -/* Absolute, escaping, or degenerate basis-dir values must be rejected up - front: they would resolve outside the destination root on the receiver. */ +/* Escaping or degenerate basis-dir values must be rejected up front (they would + resolve outside the destination root on the receiver); an absolute path is + accepted (rsync parity) and canonicalized with its leading '/' preserved. */ static void test_parse_args_basis_invalid_paths() { - static const char* const invalid[] = {"/abs", "..", "a/../b", "."}; + static const char* const invalid[] = {"..", "a/../b", ".", "/", ""}; for (size_t i = 0; i < sizeof(invalid) / sizeof(invalid[0]); i++) { Config* cfg = config_create(); char* argv[] = {"fastsync", "--link-dest", (char*)invalid[i], "/src", "/dst"}; @@ -1056,6 +1057,15 @@ static void test_parse_args_basis_invalid_paths() { EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); config_delete(cfg); } + + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--link-dest=/abs/dir", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->basis_count, 1); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "/abs/dir"); + config_delete(cfg); } /* Basis dirs require the per-file incremental handshake, which -s disables. */ @@ -2551,15 +2561,35 @@ static void test_parse_args_filter_rules() { EXPECT_EQ_INT(parse_args(cfg, 4, missing_argv, positional_args, &positional_count), -1); config_delete(cfg); - /* rsync shorthands/modifiers we do not support are rejected instead of being - * silently parsed as literal patterns. */ - static const char* const unsupported[] = { - ": .rsync-filter", ". /tmp/rules", "-s foo", "-p bar", "-C", "-! *.o", "!", + /* Full rsync grammar (rule words, modifiers, clear) is supported. */ + cfg = config_create(); + positional_count = 0; + char* grammar_argv[] = {"fastsync", + "--filter=hide *.tmp", + "--filter=show *.txt", + "--filter=protect *.bak", + "--filter=risk *.o", + "--filter=-s foo", + "--filter=-p bar", + "--filter=-! *.o", + "--filter=dir-merge .rules", + "--filter=!", + "/src", + "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 11, grammar_argv, positional_args, &positional_count), 0); + config_delete(cfg); + + /* Genuinely malformed rules are still rejected. */ + static const char* const malformed[] = { + "merge", /* merge requires a filename */ + "dir-merge", /* dir-merge requires a filename */ + "clear extra", /* clear takes no pattern */ + "no-such-rule x", /* unknown rule word */ }; - for (size_t i = 0; i < sizeof(unsupported) / sizeof(unsupported[0]); i++) { + for (size_t i = 0; i < sizeof(malformed) / sizeof(malformed[0]); i++) { cfg = config_create(); positional_count = 0; - char* rule_argv[] = {"fastsync", "--filter", (char*)unsupported[i], "/src", "/dst"}; + char* rule_argv[] = {"fastsync", "--filter", (char*)malformed[i], "/src", "/dst"}; EXPECT_EQ_INT(parse_args(cfg, 5, rule_argv, positional_args, &positional_count), -1); config_delete(cfg); } @@ -3146,6 +3176,27 @@ static void test_parse_args_chown() { EXPECT_TRUE(cfg->chown_gid_set); EXPECT_EQ_INT(cfg->chown_gid, IDENTITY_CURRENT); config_delete(cfg); + + /* A --chown NAME is converted to the equivalent receiver-resolved map rule + * (rsync implements --chown as --usermap=*:USER --groupmap=*:GROUP), so the + * name is carried on the wire as to_name instead of being resolved on the + * sender. A name that does not exist on the sender is accepted and left for + * the receiver to resolve (or warn about), matching rsync. */ + cfg = config_create(); + positional_count = 0; + char* argv5[] = {"fastsync", "--chown=no_such_user_zzz:no_such_group_zzz", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv5, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->usermap_count, 1); + EXPECT_NOT_NULL(cfg->usermap[0].to_name); + if (cfg->usermap[0].to_name) + EXPECT_EQ_STR(cfg->usermap[0].to_name, "no_such_user_zzz"); + EXPECT_FALSE(cfg->chown_uid_set); + EXPECT_EQ_INT(cfg->groupmap_count, 1); + EXPECT_NOT_NULL(cfg->groupmap[0].to_name); + if (cfg->groupmap[0].to_name) + EXPECT_EQ_STR(cfg->groupmap[0].to_name, "no_such_group_zzz"); + EXPECT_FALSE(cfg->chown_gid_set); + config_delete(cfg); } /* --copy-as=USER[:GROUP] (P7 Wave E): resolve the user/group against the local @@ -3220,7 +3271,6 @@ static void test_parse_args_rejects_malformed_identity() { {"--groupmap", "@1"}, {"--groupmap", "no_such_group_qqq:x"}, {"--chown", "a:b:c"}, - {"--chown", "no_such_user_zzz:"}, {"--copy-as", ""}, {"--copy-as", ":"}, {"--copy-as", "a:b:c"}, diff --git a/tests/test_config.c b/tests/test_config.c index e8ed56b..a5fe1c7 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -1192,7 +1192,10 @@ static void test_config_basis_wire_rejects_escaping() { c->basis_dirs = calloc(1, sizeof(BasisDest)); c->basis_dirs[0].type = BASIS_DEST_LINK; c->basis_dirs[0].path = str_dup("/abs"); - EXPECT_FALSE(roundtrip_config_ok(c)); + /* An absolute basis dir is accepted (rsync parity); it is only usable when it + lies within the receiver's authorized root, which file_open_secure_parent + enforces at lookup time. */ + EXPECT_TRUE(roundtrip_config_ok(c)); config_delete(c); /* A well-formed list still round-trips even with a manually built struct. */ @@ -1225,8 +1228,13 @@ static void test_config_basis_normalization() { /* Degenerate values that normalize away to nothing stay rejected. */ EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "."), -1); EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ".."), -1); - EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/abs"), -1); + /* An absolute path is canonicalized (leading '/' preserved) and accepted. */ + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/abs"), 0); + EXPECT_EQ_STR(c->basis_dirs[c->basis_count - 1].path, "/abs"); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/a//b/"), 0); + EXPECT_EQ_STR(c->basis_dirs[c->basis_count - 1].path, "/a/b"); EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "a/../b"), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/"), -1); EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ""), -1); config_delete(c); } diff --git a/tests/test_scanner.c b/tests/test_scanner.c index 755ed5d..933372d 100644 --- a/tests/test_scanner.c +++ b/tests/test_scanner.c @@ -855,7 +855,7 @@ static void test_filter_rules(bool parallel) { /* - *.tmp excludes only the tmp file; other files remain (default include). */ const char* exclude_only[] = {"- *.tmp"}; char err[160]; - FilterRuleList* base = filter_base_build(exclude_only, 1, false, err, sizeof(err)); + FilterRuleList* base = filter_base_build(exclude_only, 1, false, false, err, sizeof(err)); EXPECT_NOT_NULL(base); ScannerOptions options = {0}; options.base_filters = base; @@ -875,7 +875,7 @@ static void test_filter_rules(bool parallel) { /* Anchored include then exclude-all: only root-level keep* survives. */ const char* anchored[] = {"+ /a.txt", "- *"}; - base = filter_base_build(anchored, 2, false, err, sizeof(err)); + base = filter_base_build(anchored, 2, false, false, err, sizeof(err)); EXPECT_NOT_NULL(base); options.base_filters = base; rc = parallel ? collect_files_parallel(root, &options, &paths, &count) @@ -889,7 +889,7 @@ static void test_filter_rules(bool parallel) { /* The common include idiom (the exact rule order the CLI compiles from * --include='*.txt' --exclude='*'): only .txt files survive. */ const char* idiom[] = {"+ *.txt", "- *"}; - base = filter_base_build(idiom, 2, false, err, sizeof(err)); + base = filter_base_build(idiom, 2, false, false, err, sizeof(err)); EXPECT_NOT_NULL(base); options.base_filters = base; rc = parallel ? collect_files_parallel(root, &options, &paths, &count) @@ -905,7 +905,7 @@ static void test_filter_rules(bool parallel) { /* An include rule alone is NOT a mandatory whitelist (rsync semantics): only * the matching file is affected, everything else is still transferred. */ const char* include_alone[] = {"+ *.txt"}; - base = filter_base_build(include_alone, 1, false, err, sizeof(err)); + base = filter_base_build(include_alone, 1, false, false, err, sizeof(err)); EXPECT_NOT_NULL(base); options.base_filters = base; rc = parallel ? collect_files_parallel(root, &options, &paths, &count) @@ -932,7 +932,7 @@ static void test_filter_dir_only_and_anchored(bool parallel) { const char* rules[] = {"- /sub/"}; char err[160]; - FilterRuleList* base = filter_base_build(rules, 1, false, err, sizeof(err)); + FilterRuleList* base = filter_base_build(rules, 1, false, false, err, sizeof(err)); EXPECT_NOT_NULL(base); ScannerOptions options = {0}; options.base_filters = base; @@ -966,7 +966,7 @@ static void test_cvs_defaults(bool parallel) { create_test_file("test_scan_cvs/keep.txt", "keep"); char err[160]; - FilterRuleList* base = filter_base_build(NULL, 0, true, err, sizeof(err)); + FilterRuleList* base = filter_base_build(NULL, 0, true, false, err, sizeof(err)); EXPECT_NOT_NULL(base); ScannerOptions options = {0}; options.base_filters = base; @@ -1005,6 +1005,7 @@ static void test_per_dir_filter(bool parallel) { ScannerOptions options = {0}; options.per_dir_filters = true; + options.exclude_per_dir_filter_files = true; /* -FF */ if (parallel) options.num_threads = 2; char** paths = NULL; @@ -1137,6 +1138,7 @@ static void test_per_dir_filter_override(bool parallel) { ScannerOptions options = {0}; options.per_dir_filters = true; + options.exclude_per_dir_filter_files = true; /* -FF */ if (parallel) options.num_threads = 2; char** paths = NULL;