diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 5526c39..fa9370a 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -210,9 +210,9 @@ order-independent because it runs over the fully parsed config. |------|-------------------|-----------------|-------| | `--checksum` | Skip based on checksum | ✅ Implemented | With `--incremental`, compares xxHash64 content checksums; `-c` remains compression | | `--checksum-choice=STR` | Choose checksum algorithm | ❌ Not Implemented | xxHash used internally | -| `--compare-dest=DIR` | Compare dest files relative to DIR | ❌ Not Implemented | Removed because it had no effect | -| `--copy-dest=DIR` | Include copies of unchanged files | ❌ Not Implemented | Removed because it had no effect | -| `--link-dest=DIR` | Hardlink to files when unchanged | ❌ Not Implemented | Removed because it had no effect | +| `--compare-dest=DIR` | Compare dest files relative to DIR | ✅ Implemented | DIR is a receiver-side basis relative to the destination root (confined below it; absolute/`..`/`.` rejected, `//` collapsed and trailing `/` dropped). On the receiver's per-file check (implies `--incremental`) an exact match = same size + mtime (unless `--size-only`; `-I` disables matching) **and** equal xxHash64 of the sender's file; a match suppresses the data transfer. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Divergences: when the destination already holds a *different* version rsync deletes it but FastSync instead transfers the data (keeps the mirror complete; never deletes without `--delete`); attribute-only differences on a match are not re-applied (data is skipped so the sender never sends metadata); content is verified by xxHash64, stricter than rsync's default quick check. Sizing: FastSync's whole-file payload limit is 256 MiB on **every** transfer path (not basis-specific); rsync applies basis dirs to arbitrary sizes, so FastSync refuses a basis run whose source contains a larger file up front with a clear error before any transfer. Wire: a basis-count field is always present on the config frame (protocol bumped to 2.8.0, so clients and servers must both be 2.8.0) | +| `--copy-dest=DIR` | Include copies of unchanged files | ✅ Implemented | Same basis rules as `--compare-dest`, but an exact match materializes a **local copy** of the DIR file into the destination (via the normal atomic temp+rename store path, so `--existing`/`--ignore-existing`/`--update`/`--backup`/`--delay-updates` all still apply) instead of transferring data. Repeatable; command-line order = priority. Content is xxHash64-verified before the copy. Divergences: a basis-hit destination keeps the basis file's own mode/uid/gid and mtime (the sender sends no metadata on a skip), so with `--size-only` its mtime can differ from the source and attribute-only differences are copied with the basis attributes rather than rsync's "copy + fix attributes". Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.8.0 | +| `--link-dest=DIR` | Hardlink to files when unchanged | ✅ Implemented | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy, never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Content is xxHash64-verified before linking. Divergences and caveats: an already up-to-date destination file is not re-linked to a basis file (only files that would otherwise be written are linked); a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); with `--size-only` the linked mtime can differ from the source; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.8.0 | | `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ❌ Not Implemented | | ## 12. Compression diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 86c7995..5778f41 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -114,6 +114,27 @@ static int set_nonneg_int_option(int* dest, const char* value, const char* optio return 0; } +/* Validate and append one --compare-dest/--copy-dest/--link-dest directory. + * The path is interpreted on the receiver relative to the destination root, + * so it must be a non-empty relative path with no "." / ".." components (an + * absolute or escaping path is rejected up front instead of failing on the + * server). Returns 0 on success, -1 on error. */ +static int set_basis_dest_option(Config* config, BasisDestType type, const char* value, + const char* option_name) { + if (!value || !value[0]) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", option_name); + return -1; + } + if (config_basis_append(config, type, value) != 0) { + log_message(LOG_LEVEL_ERROR, + "%s requires a non-empty relative directory name with no '.', '..', or absolute " + "path (resolved below the destination root)", + option_name); + return -1; + } + return 0; +} + static int set_stderr_mode(const char* value) { if (strcmp(value, "errors") == 0 || strcmp(value, "e") == 0) log_set_stderr_mode(LOG_STDERR_ERRORS); @@ -974,6 +995,36 @@ int parse_args(Config* config, int argc, char* argv[], int* positional_args, } log_message(LOG_LEVEL_ERROR, "%s is not supported yet (xxHash64 is used)", argv[i]); return -1; + } else if (strncmp(argv[i], "--compare-dest=", 15) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[i] + 15, "--compare-dest") != 0) + return -1; + } else if (opt_is(argv[i], "--compare-dest", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_basis_dest_option(config, BASIS_DEST_COMPARE, argv[++i], "--compare-dest") != 0) + return -1; + } else if (strncmp(argv[i], "--copy-dest=", 12) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[i] + 12, "--copy-dest") != 0) + return -1; + } else if (opt_is(argv[i], "--copy-dest", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_basis_dest_option(config, BASIS_DEST_COPY, argv[++i], "--copy-dest") != 0) + return -1; + } else if (strncmp(argv[i], "--link-dest=", 12) == 0) { + if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[i] + 12, "--link-dest") != 0) + return -1; + } else if (opt_is(argv[i], "--link-dest", NULL)) { + if (i + 1 >= argc) { + log_message(LOG_LEVEL_ERROR, "missing argument for %s", argv[i]); + return -1; + } + if (set_basis_dest_option(config, BASIS_DEST_LINK, argv[++i], "--link-dest") != 0) + return -1; } else if (argv[i][0] == '-') { char* escaped = output_escape(argv[i], false); fprintf(stderr, "Unknown option: %s\n", escaped ? escaped : ""); @@ -1010,6 +1061,13 @@ int parse_args(Config* config, int argc, char* argv[], int* positional_args, config->files_from_set = set; } + /* The "unchanged" decision for --compare-dest/--copy-dest/--link-dest must + * be made on the receiver against the basis directories, which requires the + * per-file STATUS_CHECK handshake: basis-dir options therefore imply + * --incremental (and, via the block below, metadata) on the sender. */ + if (config_has_basis(config)) + config->use_incremental = true; + /* Incremental and delta transfers need metadata unless the user disabled it. */ if ((config->use_incremental || config->use_delta) && !config->use_metadata && !config->metadata_explicitly_disabled) { diff --git a/src/client/client_send.c b/src/client/client_send.c index 0c33c45..db9bb71 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -218,6 +218,49 @@ static bool files_from_list_valid(const Config* config) { return no_implied_dirs_files_from_valid(config); } +/* Basis directories are honored by the receiver's per-file incremental check, + which (like every whole-file payload path in FastSync) is bounded by + MAX_RECEIVE_WHOLE_FILE_SIZE. rsync would apply basis dirs to files of any + size; FastSync cannot, so when basis dirs are requested this preflight scan + refuses the run up front with a clear diagnostic instead of letting the + receiver abort the whole transfer mid-stream with no client explanation. + Returns true when the tree can be transferred. */ +static bool basis_oversize_preflight(const Config* config) { + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) + return false; + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + prepared_scanner_destroy(&prepared); + if (!scanner) + return false; + bool ok = true; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (f == NULL || f->is_dir || f->data == NULL || f->data->size <= MAX_RECEIVE_WHOLE_FILE_SIZE) + continue; + char* escaped = output_escape(file_wire_path(f), config->eight_bit_output); + log_message(LOG_LEVEL_ERROR, + "%s is %llu bytes, larger than the %llu-byte whole-file transfer limit; " + "--compare-dest/--copy-dest/--link-dest cannot sync files above this limit", + escaped ? escaped : "", (unsigned long long)f->data->size, + (unsigned long long)MAX_RECEIVE_WHOLE_FILE_SIZE); + free(escaped); + ok = false; + break; + } + chunk_destroy(chunk); + if (!ok) + break; + } + if (directory_scanner_failed(scanner)) + ok = false; + directory_scanner_destroy(scanner); + return ok; +} + /* Select the configured transport for both transfer execution paths. */ static Client* connect_transfer_client(const Config* config) { if (config->transport == TRANSPORT_SSH) { @@ -662,7 +705,10 @@ static int incremental_check(Client* client, File* file, const Config* config, return -1; if (!send_n_data(client->file_descriptor, &mtime_nsec, sizeof(mtime_nsec))) return -1; - if (config->checksum) { + /* With alternate basis directories the receiver must be able to verify the + * content of every candidate basis file, so the sender supplies its xxHash64 + * for every file even when --checksum was not requested. */ + if (config->checksum || config_has_basis(config)) { uint64_t checksum; if (!file_checksum(file, &checksum) || !send_n_data(client->file_descriptor, &checksum, sizeof(checksum))) @@ -1216,6 +1262,8 @@ int send_files(Config* config) { return send_dry_run_manifest(config); if (!files_from_list_valid(config)) return 1; + if (config_has_basis(config) && !basis_oversize_preflight(config)) + return 1; Client* client = connect_transfer_client(config); if (!client) { @@ -1389,6 +1437,8 @@ int send_files_multithreaded(Config** config_ptr) { return send_dry_run_manifest(config); if (!files_from_list_valid(config)) return 1; + if (config_has_basis(config) && !basis_oversize_preflight(config)) + return 1; long pages = sysconf(_SC_AVPHYS_PAGES); long page_size = sysconf(_SC_PAGE_SIZE); diff --git a/src/client/client_validation.c b/src/client/client_validation.c index d27fded..f068699 100644 --- a/src/client/client_validation.c +++ b/src/client/client_validation.c @@ -11,6 +11,12 @@ bool validate_config(const Config* config) { print_usage(); return false; } + if (config_has_basis(config) && config->use_chunk_serialization) { + log_message(LOG_LEVEL_ERROR, + "--compare-dest/--copy-dest/--link-dest require per-file incremental checks and " + "cannot be combined with -s (chunk serialization)"); + return false; + } if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) { log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s " "(chunk serialization)"); diff --git a/src/client/usage.c b/src/client/usage.c index 6d18531..0cb0d76 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -70,6 +70,13 @@ void print_usage(void) { printf(" -@, --modify-window Modification time tolerance\n"); printf(" -u, --update Skip files newer than the source on receiver\n"); printf(" --existing Skip files not already present at destination\n"); + printf(" --compare-dest Treat DIR (relative to destination root) as an extra\n"); + printf(" comparison basis: unchanged files are not transferred\n"); + printf(" (requires --incremental, which is implied)\n"); + printf(" --copy-dest Like --compare-dest, but copies the unchanged file from DIR\n"); + printf(" into the destination instead of transferring its data\n"); + printf(" --link-dest Like --copy-dest, but hard-links the unchanged file from DIR\n"); + printf(" into the destination (repeatable; earlier DIRs win)\n"); printf(" --checksum-choice, --cc Checksum algorithm (not supported yet; xxHash64 is " "used)\n"); printf(" --delta Delta transfer for changed files (requires --incremental)\n"); diff --git a/src/server/receiver.c b/src/server/receiver.c index e83c50f..384f186 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -302,7 +302,7 @@ static bool receiver_save_file(File* file, void* context_pointer) { result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config); } if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir && - !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { + !file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { file_destroy(file); return false; } diff --git a/src/shared/config.c b/src/shared/config.c index 6a2aea0..b839744 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -109,9 +109,8 @@ static void config_set_defaults(Config* config) { config->rsync_path = NULL; config->old_args = false; config->temp_dir = NULL; - config->compare_dest = NULL; - config->copy_dest = NULL; - config->link_dest = NULL; + config->basis_dirs = NULL; + config->basis_count = 0; config->partial_dir = NULL; config->suffix = NULL; config->delete_before = false; @@ -212,6 +211,87 @@ bool config_has_valid_delete_timing(const Config* config) { return timing_count <= 1; } +bool config_has_basis(const Config* config) { + return config && config->basis_count > 0; +} + +/* A basis-dir path travels from the client to the receiver and is resolved + * below the destination root, so it must be a non-empty relative path with no + * "." or ".." component and no traversal: an absolute or escaping path would + * make the receiver read or link files outside its authorized root. + * + * Returns a malloc'd CANONICAL copy of an accepted path, or NULL when the path + * is rejected. Canonicalization collapses interior empty components ("a//b" -> + * "a/b"), drops "." components and trailing "/"s, so validation, the delete + * walker prefix match and the receiver's basis lookup all agree on one form. + * The normalizer is the single source of truth for both config_basis_path_valid + * and config_basis_append. */ +static char* basis_path_normalize(const char* path) { + if (!path || path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) + return NULL; + if (strcmp(path, ".") == 0) + return NULL; + char* dup = str_dup(path); + if (!dup) + return NULL; + size_t out_len = 0; + char* out = malloc(strlen(path) + 1); + if (!out) { + free(dup); + return NULL; + } + char* saveptr = NULL; + bool ok = true; + for (char* part = strtok_r(dup, "/", &saveptr); part; part = strtok_r(NULL, "/", &saveptr)) { + if (strcmp(part, "..") == 0) { + ok = false; + break; + } + if (strcmp(part, ".") == 0) + continue; + if (out_len > 0) + out[out_len++] = '/'; + size_t len = strlen(part); + memcpy(out + out_len, part, len); + out_len += len; + } + free(dup); + if (!ok || out_len == 0) { + free(out); + return NULL; + } + out[out_len] = '\0'; + return out; +} + +bool config_basis_path_valid(const char* path) { + char* normalized = basis_path_normalize(path); + if (!normalized) + return false; + free(normalized); + return true; +} + +int config_basis_append(Config* config, BasisDestType type, const char* path) { + if (!config || + (type != BASIS_DEST_COMPARE && type != BASIS_DEST_COPY && type != BASIS_DEST_LINK) || + config->basis_count >= MAX_BASIS_DIRS) + return -1; + char* normalized = basis_path_normalize(path); + if (!normalized) + return -1; + BasisDest* grown = realloc(config->basis_dirs, (config->basis_count + 1) * sizeof(BasisDest)); + if (!grown) { + free(normalized); + return -1; + } + config->basis_dirs = grown; + config->basis_dirs[config->basis_count].type = type; + config->basis_dirs[config->basis_count].path = normalized; + config->basis_count++; + return 0; +} + bool config_is_remote_dest(const char* s) { if (s == NULL) return false; @@ -268,9 +348,13 @@ void config_delete(Config* config) { free(config->rsh_command); free(config->rsync_path); free(config->temp_dir); - free(config->compare_dest); - free(config->copy_dest); - free(config->link_dest); + for (int i = 0; i < config->basis_count; i++) { + free(config->basis_dirs[i].path); + config->basis_dirs[i].path = NULL; + } + free(config->basis_dirs); + config->basis_dirs = NULL; + config->basis_count = 0; free(config->partial_dir); free(config->suffix); free(config->address); @@ -358,6 +442,17 @@ static bool send_resume_options(int fd, const Config* c) { send_str(fd, c->chmod_spec ? c->chmod_spec : "") && send_skip_compress_options(fd, c); } +static bool send_basis_options(int fd, const Config* c) { + if (!send_int(fd, c->basis_count)) + return false; + for (int i = 0; i < c->basis_count; i++) { + if (!send_int(fd, (int)c->basis_dirs[i].type) || + !send_str(fd, c->basis_dirs[i].path ? c->basis_dirs[i].path : "")) + return false; + } + return true; +} + static bool receive_core_fields(int fd, Config* c) { int value; if (!receive_wire_bool(fd, &c->eight_bit_output)) @@ -506,12 +601,35 @@ static bool receive_resume_options(int fd, Config* c) { return true; } +static bool receive_basis_options(int fd, Config* c) { + int count; + if (!receive_int(fd, &count)) + return false; + if (count < 0 || count > MAX_BASIS_DIRS) + return false; + for (int i = 0; i < count; i++) { + int type; + if (!receive_int(fd, &type) || type <= BASIS_DEST_NONE || type > BASIS_DEST_LINK) + return false; + char* path = receive_str(fd); + if (!path) + return false; + /* config_basis_append validates and canonicalizes the path; a rejected + path (absolute / traversal / empty) drops the whole connection. */ + bool ok = config_basis_append(c, (BasisDestType)type, path) == 0; + free(path); + if (!ok) + return false; + } + return true; +} + bool config_send(int file_descriptor, const Config* config) { protocol_session_set_max_alloc(NULL, config->max_alloc); if (!send_core_fields(file_descriptor, config) || !send_delta_fields(file_descriptor, config) || !send_file_options(file_descriptor, config) || !send_selection_options(file_descriptor, config) || - !send_resume_options(file_descriptor, config)) + !send_resume_options(file_descriptor, config) || !send_basis_options(file_descriptor, config)) return false; Status status; if (!receive_status(file_descriptor, &status)) @@ -543,7 +661,8 @@ Config* config_receive(int file_descriptor) { !receive_delta_fields(file_descriptor, config) || !receive_file_options(file_descriptor, config) || !receive_selection_options(file_descriptor, config) || - !receive_resume_options(file_descriptor, config)) + !receive_resume_options(file_descriptor, config) || + !receive_basis_options(file_descriptor, config)) goto error; if (config->compress_choice[0] != '\0' && strcmp(config->compress_choice, "zstd") != 0 && strcmp(config->compress_choice, "none") != 0) { diff --git a/src/shared/config.h b/src/shared/config.h index 406060b..4ee1caa 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -12,6 +12,22 @@ typedef enum { TRANSPORT_TCP, TRANSPORT_SSH } TransportType; Config can carry it; the concrete type lives in delay_updates.h. */ typedef struct DelayUpdatesContext DelayUpdatesContext; +/* Alternate basis-directory modes (--compare-dest / --copy-dest / + * --link-dest). Each flag adds one entry to the ordered Config->basis_dirs + * list; the receiver consults entries in command-line order and stops at the + * first exact match, mirroring rsync's basis-dir priority rules. */ +typedef enum { + BASIS_DEST_NONE = 0, + BASIS_DEST_COMPARE, /* compare only: never copies, never materializes */ + BASIS_DEST_COPY, /* local copy of the matched basis file */ + BASIS_DEST_LINK /* hard link to the matched basis file */ +} BasisDestType; + +typedef struct BasisDest { + BasisDestType type; + char* path; /* relative to the destination root (receiver-confined) */ +} BasisDest; + typedef struct Config { char* version; char* send_directory; @@ -134,9 +150,12 @@ typedef struct Config { char* rsync_path; bool old_args; char* temp_dir; - char* compare_dest; - char* copy_dest; - char* link_dest; + /* Alternate basis directories, ordered by command-line appearance. Each + * entry's type selects compare/copy/link behavior on an exact match. These + * cross the wire so the receiver can consult them; they are interpreted + * relative to the destination root and confined there. */ + BasisDest* basis_dirs; + int basis_count; // PR #174: Partial transfer resumption char* partial_dir; @@ -189,6 +208,8 @@ typedef struct Config { #define PROTOCOL_VERSION "2.8.0" #define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024) +/* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */ +#define MAX_BASIS_DIRS 64 Config* config_create(void); void config_delete(Config* config); @@ -208,5 +229,11 @@ bool config_delete_timing_early(const Config* config); * set (none = the default delete-after commit timing); without deletion no * timing flag may be set (each timing flag implies --delete). */ bool config_has_valid_delete_timing(const Config* config); +/* True when at least one --compare-dest/--copy-dest/--link-dest was set. */ +bool config_has_basis(const Config* config); +/* Append one basis-dir entry. Returns 0 on success, -1 on allocation failure. */ +int config_basis_append(Config* config, BasisDestType type, const char* path); +/* Validate a client-provided basis-dir path (relative, confined, non-empty). */ +bool config_basis_path_valid(const char* path); #endif diff --git a/src/shared/file.c b/src/shared/file.c index 8e08ae2..46a0770 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -83,6 +83,7 @@ File* file_create(const char* path) { file->metadata = NULL; file->skip = false; file->is_dir = false; + file->basis_link = NULL; return file; } @@ -98,6 +99,8 @@ void file_destroy(void* item) { file->path = NULL; free(file->send_path); file->send_path = NULL; + free(file->basis_link); + file->basis_link = NULL; free(file); } @@ -630,6 +633,106 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data, preserve_executability, false, true, false, temp_dir); } +/* Atomic --link-dest install. The destination is replaced (via a temporary + * name and a final rename) with a hard link to `basis_path`. When a hard + * link cannot be created (the basis lives on a different filesystem, the + * filesystem refuses hard links, ...) the install falls back to writing a + * local copy from `data`/`data_size`, which the caller has already verified is + * byte-identical to the basis file. `metadata` is only applied on that copy + * fallback; a successful hard link keeps the basis inode's own attributes + * (applying metadata through the shared inode would mutate the basis file). + * Returns false only when both the link and the copy fallback fail. */ +bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data, + unsigned long long data_size, const FileMetadata* metadata, + bool preserve_executability, bool use_fsync, const char* temp_dir) { + if (!path || !basis_path) + return false; + char* leaf = NULL; + int dirfd = file_open_secure_parent(path, &leaf, true); + if (dirfd < 0) + return false; + + int scratch_dirfd = -1; + if (temp_dir) { + scratch_dirfd = file_open_private_dir(temp_dir); + if (scratch_dirfd < 0) { + int saved_errno = errno; + log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s", temp_dir, + strerror(saved_errno)); + close(dirfd); + free(leaf); + return false; + } + } + + char* basis_leaf = NULL; + int basis_dirfd = file_open_secure_parent(basis_path, &basis_leaf, false); + bool linked = false; + if (basis_dirfd >= 0 && basis_leaf != NULL) { + int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), ~0ULL); + char* tmp = NULL; + if (tmp_size >= 0) + tmp = malloc((size_t)tmp_size + 1); + if (!tmp) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed while hard-linking basis file"); + } else { + for (unsigned int i = 0; i < 100 && !linked; ++i) { + if (scratch_dirfd >= 0) + snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), + next_temp_sequence()); + else + snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i); + if (linkat(basis_dirfd, basis_leaf, scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, 0) == + 0) { + linked = true; + break; + } + if (errno != EEXIST) + break; /* EXDEV / EPERM / ...: give up and fall back to a copy */ + } + if (linked) { + int target_dirfd = scratch_dirfd >= 0 ? scratch_dirfd : dirfd; + if (use_fsync) { + int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC); + if (tfd < 0 || fsync(tfd) != 0) { + linked = false; + if (tfd >= 0) + close(tfd); + } else { + close(tfd); + } + } + if (linked && renameat(target_dirfd, tmp, dirfd, leaf) != 0) + linked = false; + if (!linked) + unlinkat(target_dirfd, tmp, 0); + } + free(tmp); + } + } + if (basis_dirfd >= 0) + close(basis_dirfd); + free(basis_leaf); + basis_leaf = NULL; + + if (!linked) { + if (scratch_dirfd >= 0) + close(scratch_dirfd); + close(dirfd); + free(leaf); + /* The basis file could not be linked in (missing, cross-device, refused + by the filesystem). Write a byte-identical local copy instead. */ + return file_to_disk_secure_with_fsync(path, data, data_size, false, false, metadata, + preserve_executability, use_fsync, temp_dir); + } + + if (scratch_dirfd >= 0) + close(scratch_dirfd); + close(dirfd); + free(leaf); + return true; +} + bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size, bool inplace, bool sparse) { if (!path || (!data && data_size != 0) || has_path_traversal(path)) diff --git a/src/shared/file.h b/src/shared/file.h index 38c9f01..25ea5d4 100644 --- a/src/shared/file.h +++ b/src/shared/file.h @@ -63,5 +63,12 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data, unsigned long long data_size, bool sparse, const FileMetadata* metadata, bool preserve_executability, const char* temp_dir); +/* Atomic --link-dest install: replace `path` with a hard link to `basis_path` + (via a temp name + rename); fall back to a byte-identical local copy from + `data` when the link is impossible (EXDEV/EPERM/unsupported filesystem). + `metadata` is applied only on the copy fallback. */ +bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data, + unsigned long long data_size, const FileMetadata* metadata, + bool preserve_executability, bool use_fsync, const char* temp_dir); #endif diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index 5ccea6b..0e63375 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -75,10 +75,17 @@ static FileSaveResult file_stage_delayed_update(const char* root_directory, /* The staged location is brand new (stale leftovers from a prior crash were wiped by prepare), so the plain atomic temp+rename engine installs the complete file there. --temp-dir scratch is deliberately not layered on - top of the delay-updates staging tree. */ - bool ok = - file_to_disk_secure_with_fsync(staged_path, file->data->data, file->data->size, false, sparse, - metadata, preserve_executability, config->use_fsync, NULL); + top of the delay-updates staging tree. A --link-dest basis file is hard + linked into the staging tree (so publication's rename keeps the link). */ + bool ok; + if (file->basis_link) { + ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size, + metadata, preserve_executability, config->use_fsync, NULL); + } else { + ok = file_to_disk_secure_with_fsync(staged_path, file->data->data, file->data->size, false, + sparse, metadata, preserve_executability, config->use_fsync, + NULL); + } if (!ok) { free(staged_path); return FILE_SAVE_ERROR; @@ -272,16 +279,27 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi while (temp_len > 1 && confined_temp[temp_len - 1] == '/') confined_temp[--temp_len] = '\0'; } - bool ok = - config && config->ignore_existing - ? file_to_disk_secure_no_replace(disk_path, file->data->data, file->data->size, sparse, - metadata, preserve_executability, confined_temp) - : config && config->update - ? file_to_disk_secure_update(disk_path, file->data->data, file->data->size, inplace, - sparse, metadata, preserve_executability, confined_temp) - : file_to_disk_secure_with_fsync(disk_path, file->data->data, file->data->size, inplace, - sparse, metadata, preserve_executability, - config && config->use_fsync, confined_temp); + /* A --link-dest basis hit installs an atomic hard link (with a byte-copy + fallback); --inplace and the update/no-replace write variants do not + apply to a fresh hard link, whose inode attributes already match. The + existing/ignore-existing/update/backup preamble above has already made the + policy decision. */ + bool ok; + if (config && file->basis_link) { + ok = file_to_disk_secure_link(disk_path, file->basis_link, file->data->data, file->data->size, + metadata, preserve_executability, config->use_fsync, + confined_temp); + } else { + ok = config && config->ignore_existing + ? file_to_disk_secure_no_replace(disk_path, file->data->data, file->data->size, sparse, + metadata, preserve_executability, confined_temp) + : config && config->update + ? file_to_disk_secure_update(disk_path, file->data->data, file->data->size, inplace, + sparse, metadata, preserve_executability, confined_temp) + : file_to_disk_secure_with_fsync(disk_path, file->data->data, file->data->size, + inplace, sparse, metadata, preserve_executability, + config && config->use_fsync, confined_temp); + } free(confined_temp); confined_temp = NULL; if (!ok) @@ -505,6 +523,143 @@ static File* receive_delta_file(int fd, const Config* config, const char* check_ return NULL; } +/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ---- + * The receiver consults the ordered basis-dir list only when the destination + * entry is NOT already up to date. An "exact match" requires an equal size, + * an equal mtime (unless --size-only), and an equal content xxHash64, so a + * hard link / local copy is only ever made from byte-identical content. */ + +typedef struct BasisMatch { + bool hit; + BasisDestType type; + char* basis_path; /* owned absolute path of the matched basis file */ + struct stat st; /* fstat() of the matched basis file */ + Data* content; /* owned basis bytes (or empty Data), NULL when not loaded */ +} BasisMatch; + +static void basis_match_free(BasisMatch* match) { + if (!match) + return; + free(match->basis_path); + match->basis_path = NULL; + data_destroy(match->content); + match->content = NULL; + match->hit = false; + match->type = BASIS_DEST_NONE; +} + +/* Open `path` (via the secure, root-confined primitives) and require it to be + a regular file of exactly `expected_size` bytes. Returns an open read-only + descriptor and its fstat on success. */ +static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd, + struct stat* out_st) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return false; + int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW); + free(leaf); + close(parent_fd); + if (fd < 0) + return false; + struct stat st; + if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) || + (unsigned long long)st.st_size != expected_size) { + close(fd); + return false; + } + *out_fd = fd; + *out_st = st; + return true; +} + +/* Read the whole remaining content of an open descriptor. A zero-length file + yields an empty Data (data pointer NULL). */ +static Data* basis_read_content(int fd, unsigned long long size) { + if (size == 0) + return data_create_reserve(0); + if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) + return NULL; + void* buf = protocol_alloc((size_t)size); + if (!buf) + return NULL; + size_t got = 0; + while (got < (size_t)size) { + ssize_t n = read(fd, (char*)buf + got, (size_t)size - got); + if (n <= 0) { + free(buf); + return NULL; + } + got += (size_t)n; + } + return data_create(buf, (size_t)size); +} + +/* --ignore-times forces every file to be updated, so no basis hit is ever + declared (matching rsync, where -I prevents link-dest from linking). */ +static bool basis_quick_matches(const Config* config, const struct stat* st, time_t check_mtime, + long check_mtime_nsec) { + if (config->size_only) + return true; + long mtime_nsec = 0; +#ifdef __linux__ + mtime_nsec = st->st_mtim.tv_nsec; +#endif + return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec, + config->modify_window); +} + +/* Search the basis-dir list in command-line order and return the first exact + match. When load_content is true the matched bytes are kept in out->content + so the caller can materialize the file without re-reading it. */ +static bool basis_match_find(const Config* config, const char* check_path, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, uint64_t check_checksum, bool load_content, + BasisMatch* out) { + memset(out, 0, sizeof(*out)); + if (!config || !config_has_basis(config) || config->ignore_times) + return false; + for (int i = 0; i < config->basis_count; i++) { + const BasisDest* entry = &config->basis_dirs[i]; + char* basis_dir = path_cat(config->receive_root_directory, entry->path); + if (!basis_dir) + continue; + char* candidate = path_cat(basis_dir, check_path); + free(basis_dir); + if (!candidate) + continue; + + int fd; + struct stat st; + if (basis_open_regular(candidate, check_size, &fd, &st)) { + if (basis_quick_matches(config, &st, check_mtime, check_mtime_nsec)) { + Data* content = basis_read_content(fd, check_size); + if (content) { + uint64_t basis_hash = check_size == 0 ? delta_xxhash64("", 0) + : content->data ? delta_xxhash64(content->data, content->size) + : 0; + if (basis_hash == check_checksum) { + out->hit = true; + out->type = entry->type; + out->basis_path = candidate; + candidate = NULL; /* ownership transferred to out */ + out->st = st; + out->content = load_content ? content : NULL; + if (!load_content) + data_destroy(content); + close(fd); + return true; + } + } + data_destroy(content); + } + close(fd); + } + free(candidate); + } + return false; +} + File* receive_incremental_check(int fd, const Config* config, bool* skipped) { if (!config || !skipped) { send_status(fd, STATUS_ERROR); @@ -531,7 +686,8 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) { send_status(fd, STATUS_ERROR); return NULL; } - if (config->checksum && !receive_n_data(fd, &check_checksum, sizeof(check_checksum))) { + if ((config->checksum || config_has_basis(config)) && + !receive_n_data(fd, &check_checksum, sizeof(check_checksum))) { free(check_path); return NULL; } @@ -642,6 +798,78 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped) { return NULL; } + /* ---- Alternate basis directories ---- */ + if (config_has_basis(config)) { + BasisMatch basis; + basis_match_find(config, check_path, check_size, (time_t)check_mtime, (long)check_mtime_nsec, + check_checksum, true, &basis); + if (basis.hit) { + if (basis.type == BASIS_DEST_COMPARE) { + /* compare-dest never copies: an exact match only suppresses the data + for a file the destination does not already hold (sparse backup). + When the destination holds a DIFFERENT version FastSync falls back to + a normal transfer rather than deleting the stale entry the way rsync + does (see RSYNC_COMPAT.md). */ + basis_match_free(&basis); + if (!has_old_file) { + if (!send_status(fd, STATUS_OK)) { + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + free(old_data); + close(old_fd); + free(full_path); + free(check_path); + *skipped = true; + return NULL; + } + } else { + /* copy-dest / link-dest: materialize the unchanged file locally so the + sender can skip the data. The store engine re-applies the normal + existing/ignore-existing/update/backup/delay-updates policy. */ + File* materialized = file_create(check_path); + if (materialized && basis.content) { + materialized->data = basis.content; + basis.content = NULL; + materialized->metadata = file_metadata_create(&basis.st); + materialized->skip = true; /* receiver must not ack this as a data file */ + if (basis.type == BASIS_DEST_LINK) { + materialized->basis_link = basis.basis_path; + basis.basis_path = NULL; + } + if (!materialized->metadata) { + file_destroy(materialized); + materialized = NULL; + } + } else { + file_destroy(materialized); + materialized = NULL; + } + if (materialized) { + if (!send_status(fd, STATUS_OK)) { + basis_match_free(&basis); + file_destroy(materialized); + close(old_fd); + free(full_path); + free(check_path); + return NULL; + } + basis_match_free(&basis); + free(old_data); + close(old_fd); + free(full_path); + free(check_path); + *skipped = false; + return materialized; + } + /* Materialization setup failed: fall through to the normal transfer. */ + } + } + basis_match_free(&basis); + } + if (try_delta && old_data != NULL) { bool delta_failed = false; File* delta_file = @@ -839,7 +1067,36 @@ bool manifest_delete_extras(const Config* config, ArrayList* manifest) { if (!config || !manifest) return false; fprintf(stderr, "Deleting files not in manifest...\n"); - const char* skip_staging = config->delay_updates ? DELAY_UPDATES_STAGING_DIR : NULL; - return delete_extras_limited(config->receive_root_directory, manifest, MAX_SERVER_DELETE_COUNT, - skip_staging); + /* With --delay-updates the staged (not yet published) files live directly + under the receive root in the staging directory; the delete walker must + not treat them as extras or it would remove every staged file before it + can be published. That staging name is protected only as a DIRECT child + of the receive root so a nested destination directory that happens to be + named .fastsync-stage is still ordinary content. Alternate basis + directories (--compare-dest / --copy-dest / --link-dest) are excluded at + any depth: they are extra comparison snapshots the user pointed at, not + destination content, and deleting them would destroy the very files a + --link-dest run just linked into place. */ + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; + DeleteSkipEntry* skips = NULL; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + if (!skips) + return false; + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + skips[idx].prefix = config->basis_dirs[i].path; + skips[idx].top_level_only = false; + idx++; + } + } + bool deletion_ok = delete_extras_limited(config->receive_root_directory, manifest, + MAX_SERVER_DELETE_COUNT, skips, skip_count); + free(skips); + return deletion_ok; } diff --git a/src/shared/file_types.h b/src/shared/file_types.h index c831813..a116916 100644 --- a/src/shared/file_types.h +++ b/src/shared/file_types.h @@ -29,6 +29,12 @@ typedef struct { /* True when this entry is an explicit directory entry (--dirs mode): the * receiver creates the directory instead of writing a regular file. */ bool is_dir; + /* Receiver-only, --link-dest: when set, install the destination entry as a + * hard link to this absolute (root-confined) path instead of writing + * `data`. The matching code has already verified the link target's content + * equals the incoming file, and `data` is kept as the cross-filesystem + * fallback (a local copy) if the hard link cannot be created. */ + char* basis_link; } File; /* The path that should be sent on the wire and used for the receiver-side diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index 0ec5cd5..38636d4 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -302,7 +302,7 @@ int write_thread(void* pipeline_context) { /* Record the per-file outcome so a --remove-source-files sender learns which sources were actually written versus skipped on the receiver. Explicit directory entries have no source and are never acknowledged. */ - if (context->config->remove_source_files && !file->is_dir && + if (context->config->remove_source_files && !file->is_dir && !file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) { file_destroy(file); pipeline_context_receiver_note_bytes_released(context, file_bytes); diff --git a/src/shared/utils.c b/src/shared/utils.c index bcff6c0..616ea66 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -204,9 +204,26 @@ static bool is_dir_in_manifest(const char* rel_path, ArrayList* manifest) { return false; } +/* True when child_rel is, or lies below, a protected entry. A prefix "a" + therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only + set only protect DIRECT children of the receive root (at_root); nested + directories that share such a name stay ordinary destination content. */ +static bool path_under_skip_prefix(const char* child_rel, bool at_root, + const DeleteSkipEntry* skips, int skip_count) { + for (int i = 0; i < skip_count; i++) { + if (skips[i].top_level_only && !at_root) + continue; + size_t prefix_len = strlen(skips[i].prefix); + if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 && + (child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/')) + return true; + } + return false; +} + static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, - size_t max_delete, size_t* deleted_count, - const char* skip_root_child) { + size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips, + int skip_count) { int scanfd = dup(dirfd); if (scanfd < 0) return false; @@ -220,18 +237,22 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes while ((entry = readdir(dir)) != NULL) { if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) continue; - /* A --delay-updates run keeps its staging directory as a direct child of - the receive root. Its contents are not manifest entries yet (they are - published after deletion), so descending into it would delete every - staged file as an "extra". Skip only the top-level staging name; nested - directories with the same name are ordinary destination content. */ - if (rel_path[0] == '\0' && skip_root_child && strcmp(entry->d_name, skip_root_child) == 0) - continue; char* child_rel = path_cat((char*)rel_path, entry->d_name); if (!child_rel) { operation_ok = false; continue; } + /* A --delay-updates run keeps its staging directory as a direct child of + the receive root, and basis-dir snapshots live below it too. Their + contents are not manifest entries, so descending into them would delete + every staged / basis file as an "extra". Only the staging name (a + top-level-only prefix) and the basis prefixes are protected: a nested + destination directory that happens to be called .fastsync-stage is + ordinary content. */ + if (path_under_skip_prefix(child_rel, rel_path[0] == '\0', skips, skip_count)) { + free(child_rel); + continue; + } struct stat st; if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { if (errno != ENOENT) @@ -249,7 +270,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes bool child_removed = false; if (childfd >= 0) { child_removed = delete_extras_fd(childfd, child_rel, manifest, max_delete, deleted_count, - skip_root_child); + skips, skip_count); if (!child_removed) operation_ok = false; close(childfd); @@ -301,7 +322,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes } bool delete_extras_limited(const char* dest_root, ArrayList* manifest, size_t max_delete, - const char* skip_root_child) { + const DeleteSkipEntry* skips, int skip_count) { if (!manifest) return false; int rootfd; @@ -318,14 +339,14 @@ bool delete_extras_limited(const char* dest_root, ArrayList* manifest, size_t ma if (rootfd < 0) return false; size_t deleted_count = 0; - bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skip_root_child); + bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skips, skip_count); if (close(rootfd) != 0) ok = false; return ok; } bool delete_extras(const char* dest_root, ArrayList* manifest) { - return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL); + return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL, 0); } bool has_path_traversal(const char* path) { diff --git a/src/shared/utils.h b/src/shared/utils.h index 50e2dd3..4928e1e 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -10,12 +10,21 @@ char* output_escape(const char* string, bool eight_bit_output); char* path_cat(const char* path1, const char* path2); bool glob_match(const char* pattern, const char* str); bool delete_extras(const char* dest_root, ArrayList* manifest); -/* Remove files/dirs under dest_root that are not listed in manifest. When - skip_root_child is non-NULL, a direct child of dest_root with that exact - name is left untouched (used to protect the --delay-updates staging - directory, which holds files that are still to be published). */ +/* One protected entry for the delete walker. When top_level_only is true the + prefix is skipped only as a DIRECT child of dest_root (the --delay-updates + staging directory, which must not hide genuine extras inside a nested + destination directory that happens to share the staging name); otherwise the + prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest + basis trees, which the transfer links from and are never destination + content). */ +typedef struct { + const char* prefix; + bool top_level_only; +} DeleteSkipEntry; +/* Remove files/dirs under dest_root that are not listed in manifest without + ever descending into a protected prefix (see DeleteSkipEntry). */ bool delete_extras_limited(const char* dest_root, ArrayList* manifest, size_t max_delete, - const char* skip_root_child); + const DeleteSkipEntry* skips, int skip_count); bool utils_set_authorized_root(int fd, const char* canonical_path); /* The fd-only compatibility form is fail-closed for path-based operations; * callers should use utils_set_authorized_root with the canonical identity. */ diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index 191d92d..a42bf29 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -147,7 +147,10 @@ class TestRemoveSourceFiles: with open(source_file, "wb") as f: f.write(b"keep after skip") - result, _ = run_client(source, dest, port=shared_server.port) + # The seed run preserves timestamps (-M) so the destination copy has the + # source's exact mtime; otherwise the incremental skip would depend on + # both writes landing in the same whole second (a race). + result, _ = run_client(source, dest, flags=["-M"], port=shared_server.port) assert result.returncode == 0 result, _ = run_client(source, dest, flags=["--remove-source-files", "--incremental"], @@ -2102,3 +2105,377 @@ class TestDeleteTiming: assert result.returncode == 0, \ f"--delete-before against a refuse-delete server failed: {result.stderr[:300]}" assert os.path.exists(extra), "unauthorized delete removed an extra file" + + +def _pin_mtime(path, ts): + os.utime(path, (ts, ts)) + + +class TestBasisDestDirs: + """--compare-dest / --copy-dest / --link-dest alternate basis directories. + + FastSync's basis directories are relative to the destination root and are + confined below it. The "unchanged" decision is receiver-side and requires + the per-file --incremental handshake (implied by these flags), so the basis + snapshot must reproduce the exact destination-relative mirror path of the + incoming files. + """ + + STAGING = ".fastsync-stage" + TS = 1577836800 # 2020-01-01 00:00:00 UTC, used to pin matching mtimes + + # fixture files: source and basis share the mtime pin, so a basis "match" + # is decided purely by content (xxHash). unchanged.txt is byte-identical; + # changed.txt is byte-DIFFERENT but has the SAME SIZE as the source (and + # the same pinned mtime), which is what forces the content-hash gate; + # added.txt does not exist in the basis at all. + UNCHANGED = "unchanged.txt" + CHANGED = "changed.txt" + ADDED = "added.txt" + + def _make_source(self, name, source_files): + src = os.path.join(TEST_DATA_DIR, name) + clean_dir(src) + for rel, content in source_files.items(): + full = os.path.join(src, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + _pin_mtime(full, self.TS) + return src + + def _seed_basis_file(self, dest, source, basis_dir, rel, content, ts=None): + base = os.path.join(dest, basis_dir, os.path.relpath( + get_dest_received_dir(dest, source), dest)) + full = os.path.join(base, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(content) + _pin_mtime(full, self.TS if ts is None else ts) + return full + + def _seed_basis(self, dest, source, basis_dir, basis_files): + for rel, content in basis_files.items(): + self._seed_basis_file(dest, source, basis_dir, rel, content) + return os.path.join(dest, basis_dir, os.path.relpath( + get_dest_received_dir(dest, source), dest)) + + def _source_tree(self, prefix): + return { + self.UNCHANGED: b"stable content v1\n", + self.CHANGED: b"changed content now\n", + self.ADDED: b"brand new content\n", + } + + def _basis_tree(self, prefix): + # unchanged.txt is identical to the source; changed.txt has the SAME + # byte size and pinned mtime but a different body (equal size forces + # the xxHash gate); added.txt is missing from the basis. + return { + self.UNCHANGED: b"stable content v1\n", + self.CHANGED: b"CHANGED CONTENT NOW\n", + } + + def test_same_size_different_content_is_not_a_basis_match(self, shared_server): + # Core safety property: equal size + pinned mtime but different content + # must NEVER be hard-linked or copied from the basis -- the xxHash gate + # rejects it and the sender's data is transferred instead. + for flag, basis_dir in (("--link-dest", "szlb"), ("--copy-dest", "szcp"), + ("--compare-dest", "szcmp")): + source = self._make_source("basis_same_size_src", + {self.UNCHANGED: b"same length body\n"}) + dest = os.path.join(TEST_DATA_DIR, f"basis_same_size_dst_{basis_dir}") + clean_dir(dest) + basis_file = self._seed_basis_file(dest, source, basis_dir, self.UNCHANGED, + b"SAME LENGTH BODY!") + result, _ = run_client(source, dest, flags=[f"{flag}={basis_dir}"], + port=shared_server.port) + assert result.returncode == 0, \ + f"{flag} same-size mismatch failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + dest_file = os.path.join(received, self.UNCHANGED) + assert _read_file(dest_file) == b"same length body\n", \ + f"{flag}: basis content leaked into the destination on a hash mismatch" + if flag != "--compare-dest": + assert os.stat(dest_file).st_ino != os.stat(basis_file).st_ino, \ + f"{flag}: linked/copied from a content-mismatched basis file" + + def test_compare_dest_skips_matching_and_transfers_missing(self, shared_server): + source = self._make_source("basis_compare_src", self._source_tree("c")) + dest = os.path.join(TEST_DATA_DIR, "basis_compare_dst") + clean_dir(dest) + self._seed_basis(dest, source, "cbasis", self._basis_tree("c")) + result, _ = run_client(source, dest, + flags=["--compare-dest=cbasis"], + port=shared_server.port) + assert result.returncode == 0, f"compare-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + # compare-dest never copies: an exact basis match is skipped, leaving a + # sparse destination (rsync parity). + assert not os.path.exists(os.path.join(received, self.UNCHANGED)), \ + "compare-dest materialized the unchanged file" + # Files the destination lacks AND the basis cannot satisfy are still + # transferred normally. + assert _read_file(os.path.join(received, self.CHANGED)) == \ + self._source_tree("c")[self.CHANGED], "changed file not transferred" + assert _read_file(os.path.join(received, self.ADDED)) == \ + self._source_tree("c")[self.ADDED], "added file not transferred" + + def test_compare_dest_content_mismatch_forces_transfer(self, shared_server): + # The basis holds a file with a DIFFERENT body: even though it shares + # the mtime pin, the xxHash check fails and the data must be sent. + source = self._make_source("basis_compare_mismatch_src", {self.UNCHANGED: b"real data\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_compare_mismatch_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "cbasis", {self.UNCHANGED: b"stale data!!\n"}) + result, _ = run_client(source, dest, flags=["--compare-dest=cbasis"], + port=shared_server.port) + assert result.returncode == 0, f"compare-dest mismatch failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.UNCHANGED)) == b"real data\n", \ + "content mismatch did not fall back to a normal transfer" + assert os.stat(os.path.join(received, self.UNCHANGED)).st_ino != \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + + def test_copy_dest_copies_unchanged_and_transfers_changed(self, shared_server): + source = self._make_source("basis_copy_src", self._source_tree("cp")) + dest = os.path.join(TEST_DATA_DIR, "basis_copy_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "cpbasis", self._basis_tree("cp")) + result, _ = run_client(source, dest, flags=["--copy-dest=cpbasis"], + port=shared_server.port) + assert result.returncode == 0, f"copy-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + unchanged = os.path.join(received, self.UNCHANGED) + assert _read_file(unchanged) == b"stable content v1\n", "unchanged file not materialized" + # A real local copy, NOT a hard link to the basis file. + assert os.stat(unchanged).st_ino != os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + # Equal-size/different-content basis file falls back to the sender data. + assert _read_file(os.path.join(received, self.CHANGED)) == \ + self._source_tree("cp")[self.CHANGED] + assert _read_file(os.path.join(received, self.ADDED)) == \ + self._source_tree("cp")[self.ADDED] + + def test_link_dest_hardlinks_and_falls_back(self, shared_server): + source = self._make_source("basis_link_src", self._source_tree("ln")) + dest = os.path.join(TEST_DATA_DIR, "basis_link_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "lnbasis", self._basis_tree("ln")) + result, _ = run_client(source, dest, flags=["--link-dest=lnbasis"], + port=shared_server.port) + assert result.returncode == 0, f"link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + unchanged = os.path.join(received, self.UNCHANGED) + basis_file = os.path.join(basis, self.UNCHANGED) + # Real hard link: same inode as the DIR file, nlink >= 2, no data copy. + assert os.path.exists(unchanged) + assert os.stat(unchanged).st_ino == os.stat(basis_file).st_ino, \ + "link-dest did not produce a hard link" + assert os.stat(unchanged).st_nlink >= 2 + # Equal-size/different-content basis file must fall back to a plain + # transfer (not a link). + changed = os.path.join(received, self.CHANGED) + assert _read_file(changed) == self._source_tree("ln")[self.CHANGED] + assert os.stat(changed).st_ino != os.stat(os.path.join(basis, self.CHANGED)).st_ino + + @pytest.mark.parametrize("flag", ["--compare-dest", "--copy-dest", "--link-dest"]) + def test_basis_dir_missing_is_a_clean_noop(self, shared_server, flag): + # A basis directory that does not exist must simply transfer everything. + source = self._make_source("basis_missing_src", {self.UNCHANGED: b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_missing_dst") + clean_dir(dest) + result, _ = run_client(source, dest, flags=[f"{flag}=nope"], + port=shared_server.port) + assert result.returncode == 0, f"{flag} with missing dir failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert _read_file(os.path.join(received, self.UNCHANGED)) == b"content\n" + + def test_link_dest_multithreaded(self, shared_server): + source = self._make_source("basis_link_mt_src", self._source_tree("mt")) + dest = os.path.join(TEST_DATA_DIR, "basis_link_mt_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "mtbasis", self._basis_tree("mt")) + result, _ = run_client(source, dest, flags=["--link-dest=mtbasis", "-m"], + port=shared_server.port) + assert result.returncode == 0, f"-m link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, self.UNCHANGED)).st_ino == \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + assert _read_file(os.path.join(received, self.ADDED)) == \ + self._source_tree("mt")[self.ADDED] + + def test_link_dest_with_delay_updates_stages_and_publishes_link(self, shared_server): + source = self._make_source("basis_link_delay_src", {self.UNCHANGED: b"v1\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_link_delay_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "delaybasis", {self.UNCHANGED: b"v1\n"}) + result, _ = run_client(source, dest, + flags=["--link-dest=delaybasis", "--delay-updates"], + port=shared_server.port) + assert result.returncode == 0, f"delay-updates link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + unchanged = os.path.join(received, self.UNCHANGED) + assert os.stat(unchanged).st_ino == \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "delay-updates staging tree was not cleaned up" + + def test_delete_does_not_touch_basis_dir(self): + """--delete removes genuine extras but must never treat a basis-dir + snapshot (which a --link-dest run just linked from) as destination + content.""" + source = self._make_source("basis_delete_src", {self.UNCHANGED: b"v1\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_delete_dst") + clean_dir(dest) + basis = self._seed_basis(dest, source, "delbasis", {self.UNCHANGED: b"v1\n"}) + received = get_dest_received_dir(dest, source) + os.makedirs(received, exist_ok=True) + extra = os.path.join(received, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"extra") + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, flags=["--link-dest=delbasis", "--delete"], + port=server.port) + assert result.returncode == 0, \ + f"delete+link-dest failed: {result.stderr[:300]}" + assert not os.path.exists(extra), "genuine extra file was not deleted" + assert _read_file(os.path.join(received, self.UNCHANGED)) == b"v1\n" + assert os.path.exists(os.path.join(basis, self.UNCHANGED)), \ + "basis directory was deleted by --delete" + assert os.stat(os.path.join(received, self.UNCHANGED)).st_ino == \ + os.stat(os.path.join(basis, self.UNCHANGED)).st_ino + + def test_delay_delete_keeps_nested_staging_named_dir_as_content(self): + # The real --delay-updates staging directory is protected from --delete + # only as a DIRECT child of the receive root. A nested destination + # directory that merely shares the staging name is ordinary content, so + # its extras must still be deleted (regression guard for the walker). + source = self._make_source("basis_nested_stage_src", + {"top.txt": b"top\n", "sub/real.txt": b"real\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_nested_stage_dst") + clean_dir(dest) + self._seed_basis(dest, source, "nstbasis", + {"top.txt": b"top\n", "sub/real.txt": b"real\n"}) + received = get_dest_received_dir(dest, source) + nested = os.path.join(received, "sub", self.STAGING) + os.makedirs(nested, exist_ok=True) + extra = os.path.join(nested, "extra.txt") + with open(extra, "wb") as fh: + fh.write(b"nested extra") + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, + flags=["--link-dest=nstbasis", "--delete", + "--delay-updates"], + port=server.port) + assert result.returncode == 0, \ + f"delay-delete nested staging failed: {result.stderr[:300]}" + assert not os.path.exists(extra), \ + "extra inside a nested .fastsync-stage dir was not deleted" + assert not os.path.isdir(nested), \ + "nested .fastsync-stage dir should have been removed after its extra" + assert _read_file(os.path.join(received, "sub", "real.txt")) == b"real\n" + assert not os.path.isdir(os.path.join(dest, self.STAGING)), \ + "real delay-updates staging tree was not cleaned up" + + def test_basis_priority_first_match_wins(self, shared_server): + # Two link-dest dirs both hold the exact file: the FIRST (command-line + # order) basis directory must win and supply the hard link. + source = self._make_source("basis_prio_src", {"f.txt": b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_prio_dst") + clean_dir(dest) + first = self._seed_basis_file(dest, source, "b1", "f.txt", b"content\n") + self._seed_basis_file(dest, source, "b2", "f.txt", b"content\n") + result, _ = run_client(source, dest, flags=["--link-dest=b1", "--link-dest=b2"], + port=shared_server.port) + assert result.returncode == 0, f"link-dest priority failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, "f.txt")).st_ino == os.stat(first).st_ino, \ + "first basis dir did not win over the second" + + def test_basis_priority_across_compare_and_link(self, shared_server): + # A compare-dest entry listed BEFORE a link-dest entry shadows it (the + # exact match is found first and nothing is materialized); reversing the + # order lets the link-dest entry win and materialize a hard link. + source = self._make_source("basis_prio_mixed_src", {"f.txt": b"content\n"}) + + dest = os.path.join(TEST_DATA_DIR, "basis_prio_mixed_dst") + clean_dir(dest) + self._seed_basis_file(dest, source, "cmpb", "f.txt", b"content\n") + self._seed_basis_file(dest, source, "lnb", "f.txt", b"content\n") + result, _ = run_client(source, dest, + flags=["--compare-dest=cmpb", "--link-dest=lnb"], + port=shared_server.port) + assert result.returncode == 0, \ + f"mixed priority (compare first) failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert not os.path.exists(os.path.join(received, "f.txt")), \ + "compare-dest matched first, so the file must stay sparse (no link-dest materialize)" + + dest = os.path.join(TEST_DATA_DIR, "basis_prio_mixed_dst2") + clean_dir(dest) + self._seed_basis_file(dest, source, "cmpb", "f.txt", b"content\n") + linkb2 = self._seed_basis_file(dest, source, "lnb", "f.txt", b"content\n") + result, _ = run_client(source, dest, + flags=["--link-dest=lnb", "--compare-dest=cmpb"], + port=shared_server.port) + assert result.returncode == 0, \ + f"mixed priority (link first) failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, "f.txt")).st_ino == os.stat(linkb2).st_ino, \ + "link-dest did not materialize when listed before compare-dest" + + def test_link_dest_size_only_ignores_mtime(self, shared_server): + # --size-only drops the mtime leg of the quick check: a basis file with + # the SAME content but a DIFFERENT mtime is still an exact match. + source = self._make_source("basis_sizeonly_src", {"f.txt": b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_sizeonly_dst") + clean_dir(dest) + basis_file = self._seed_basis_file(dest, source, "sob", "f.txt", b"content\n", + ts=self.TS + 500) + result, _ = run_client(source, dest, flags=["--link-dest=sob", "--size-only"], + port=shared_server.port) + assert result.returncode == 0, f"size-only link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert os.stat(os.path.join(received, "f.txt")).st_ino == os.stat(basis_file).st_ino, \ + "--size-only should link a basis file whose mtime differs" + + def test_link_dest_ignore_times_never_links(self, shared_server): + # -I/--ignore-times forces every file to be updated, so a basis dir is + # never used to hard-link (rsync parity). The file is transferred and + # stored as a fresh inode even though it matches the basis exactly. + source = self._make_source("basis_igntimes_src", {"f.txt": b"content\n"}) + dest = os.path.join(TEST_DATA_DIR, "basis_igntimes_dst") + clean_dir(dest) + basis_file = self._seed_basis_file(dest, source, "itb", "f.txt", b"content\n") + result, _ = run_client(source, dest, flags=["--link-dest=itb", "--ignore-times"], + port=shared_server.port) + assert result.returncode == 0, f"ignore-times link-dest failed: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + dest_file = os.path.join(received, "f.txt") + assert _read_file(dest_file) == b"content\n" + assert os.stat(dest_file).st_ino != os.stat(basis_file).st_ino, \ + "--ignore-times must not hard-link to a basis file" + + def test_basis_refuses_file_above_whole_file_limit(self, shared_server): + # Every whole-file payload path in FastSync (basis dirs included) is + # bounded by MAX_RECEIVE_WHOLE_FILE_SIZE. rsync supports basis dirs for + # arbitrary sizes; FastSync refuses such a run up front with a clear + # diagnostic instead of letting the receiver abort the whole transfer + # mid-stream with no client-side explanation. + source = self._make_source("basis_oversize_src", {"small.txt": b"ok\n"}) + big = os.path.join(source, "huge.bin") + with open(big, "wb") as fh: + os.ftruncate(fh.fileno(), 256 * 1024 * 1024 + 4096) + dest = os.path.join(TEST_DATA_DIR, "basis_oversize_dst") + clean_dir(dest) + result, _ = run_client(source, dest, flags=["--link-dest=nope"], + port=shared_server.port) + assert result.returncode != 0, \ + "basis run with an over-limit file unexpectedly succeeded" + assert "larger than" in result.stderr, \ + f"no clear over-limit diagnostic: {result.stderr[:300]}" + received = get_dest_received_dir(dest, source) + assert not os.path.exists(received), \ + "over-limit basis run transferred files before failing" diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index 250e687..3d21fc6 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -595,6 +595,86 @@ static void test_parse_args_relative_no_implied_mkpath() { config_delete(cfg); } +/* Parse --compare-dest/--copy-dest/--link-dest, including the =value and + separate-argument forms, and verify the ordered (repeatable) basis list. */ +static void test_parse_args_basis_dirs() { + Config* cfg = config_create(); + int positional_args[2]; + int positional_count = 0; + char* argv[] = {"fastsync", "--link-dest=prior", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(config_has_basis(cfg)); + EXPECT_EQ_INT(cfg->basis_count, 1); + EXPECT_EQ_INT(cfg->basis_dirs[0].type, BASIS_DEST_LINK); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "prior"); + /* Basis dirs are honored by the receiver-side per-file check, so they imply + --incremental (and, unless disabled, metadata) on the sender. */ + EXPECT_TRUE(cfg->use_incremental); + EXPECT_TRUE(cfg->use_metadata); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv2[] = {"fastsync", "--compare-dest", "cmp", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->basis_count, 1); + EXPECT_EQ_INT(cfg->basis_dirs[0].type, BASIS_DEST_COMPARE); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "cmp"); + config_delete(cfg); + + /* Repetition is supported: entries keep command-line order and type. */ + cfg = config_create(); + positional_count = 0; + char* argv3[] = {"fastsync", "--link-dest=a", "--compare-dest=b", + "--link-dest=c", "--copy-dest=d", "/src", + "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 7, argv3, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->basis_count, 4); + EXPECT_EQ_INT(cfg->basis_dirs[0].type, BASIS_DEST_LINK); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "a"); + EXPECT_EQ_INT(cfg->basis_dirs[1].type, BASIS_DEST_COMPARE); + EXPECT_EQ_STR(cfg->basis_dirs[1].path, "b"); + EXPECT_EQ_INT(cfg->basis_dirs[2].type, BASIS_DEST_LINK); + EXPECT_EQ_STR(cfg->basis_dirs[2].path, "c"); + EXPECT_EQ_INT(cfg->basis_dirs[3].type, BASIS_DEST_COPY); + EXPECT_EQ_STR(cfg->basis_dirs[3].path, "d"); + config_delete(cfg); + + /* Nested relative basis dirs are allowed (they resolve below the root). */ + cfg = config_create(); + positional_count = 0; + char* argv4[] = {"fastsync", "--copy-dest=snap/2026-01", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv4, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->basis_count, 1); + EXPECT_EQ_STR(cfg->basis_dirs[0].path, "snap/2026-01"); + config_delete(cfg); +} + +/* Absolute, escaping, or degenerate basis-dir values must be rejected up + front: they would resolve outside the destination root on the receiver. */ +static void test_parse_args_basis_invalid_paths() { + static const char* const invalid[] = {"/abs", "..", "a/../b", "."}; + for (size_t i = 0; i < sizeof(invalid) / sizeof(invalid[0]); i++) { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--link-dest", (char*)invalid[i], "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + } +} + +/* Basis dirs require the per-file incremental handshake, which -s disables. */ +static void test_validate_config_basis_rejects_chunk_serialization() { + Config* cfg = valid_client_config(); + EXPECT_EQ_INT(config_basis_append(cfg, BASIS_DEST_LINK, "prior"), 0); + cfg->use_chunk_serialization = true; + EXPECT_FALSE(validate_config(cfg)); + cfg->use_chunk_serialization = false; + EXPECT_TRUE(validate_config(cfg)); + config_delete(cfg); +} + /* --del is accepted as the rsync alias for --delete-during: it enables * deletion with the during (early) timing. */ static void test_parse_args_delete_during_alias() { @@ -701,9 +781,6 @@ static void test_parse_args_rejects_unimplemented_options() { "-e", "--rsh", "--rsync-path", - "--compare-dest", - "--copy-dest", - "--link-dest", "--address", "--bind-address", "--ipv6", @@ -1660,4 +1737,7 @@ void test_client_cli() { test_parse_args_files_from(); test_parse_args_filter_rules(); test_parse_args_from0_cvs_filter_file_flags(); + test_parse_args_basis_dirs(); + test_parse_args_basis_invalid_paths(); + test_validate_config_basis_rejects_chunk_serialization(); } diff --git a/tests/test_config.c b/tests/test_config.c index 557879b..b0c8ceb 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -561,6 +561,119 @@ static void test_config_delete_timing_conflict_rejected() { config_delete(c); } +/* Basis-dir lists survive the config wire: each entry's type and path must + round-trip unchanged. */ +static void test_config_basis_roundtrip() { + if (is_running_under_valgrind()) + return; + Config* send_cfg = config_create(); + EXPECT_NOT_NULL(send_cfg); + send_cfg->send_directory = str_dup("/send/src"); + send_cfg->receive_root_directory = str_dup("/send/dst"); + EXPECT_EQ_INT(config_basis_append(send_cfg, BASIS_DEST_LINK, "prior"), 0); + EXPECT_EQ_INT(config_basis_append(send_cfg, BASIS_DEST_COMPARE, "snap/2026-01"), 0); + EXPECT_EQ_INT(config_basis_append(send_cfg, BASIS_DEST_COPY, "copy"), 0); + + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + Config* recv = config_receive(p[0]); + bool ok = recv != NULL && recv->basis_count == 3 && recv->basis_dirs != NULL; + if (ok) { + ok = recv->basis_dirs[0].type == BASIS_DEST_LINK && + strcmp(recv->basis_dirs[0].path, "prior") == 0; + ok = ok && recv->basis_dirs[1].type == BASIS_DEST_COMPARE && + strcmp(recv->basis_dirs[1].path, "snap/2026-01") == 0; + ok = ok && recv->basis_dirs[2].type == BASIS_DEST_COPY && + strcmp(recv->basis_dirs[2].path, "copy") == 0; + } + config_delete(recv); + close(p[0]); + close(p[1]); + _exit(ok ? 0 : 1); + } else { + close(p[0]); + io_set_fds(p[1], p[1]); + bool sent = config_send(p[1], send_cfg); + int status; + waitpid(pid, &status, 0); + close(p[1]); + config_delete(send_cfg); + EXPECT_TRUE(sent); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); + } +} + +/* The receiver must reject a basis-dir path that would escape the destination + root. The values are injected directly (bypassing the client-side append + validator) so the receiver-side wire validation is what is exercised. */ +static void test_config_basis_wire_rejects_escaping() { + if (is_running_under_valgrind()) + return; + Config* c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->basis_count = 1; + c->basis_dirs = calloc(1, sizeof(BasisDest)); + c->basis_dirs[0].type = BASIS_DEST_LINK; + c->basis_dirs[0].path = str_dup("../../etc"); + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->basis_count = 1; + c->basis_dirs = calloc(1, sizeof(BasisDest)); + c->basis_dirs[0].type = BASIS_DEST_LINK; + c->basis_dirs[0].path = str_dup("/abs"); + EXPECT_FALSE(roundtrip_config_ok(c)); + config_delete(c); + + /* A well-formed list still round-trips even with a manually built struct. */ + c = config_create(); + EXPECT_NOT_NULL(c); + c->send_directory = str_dup("/src"); + c->receive_root_directory = str_dup("/dst"); + c->basis_count = 1; + c->basis_dirs = calloc(1, sizeof(BasisDest)); + c->basis_dirs[0].type = BASIS_DEST_COPY; + c->basis_dirs[0].path = str_dup("safe"); + EXPECT_TRUE(roundtrip_config_ok(c)); + config_delete(c); +} + +/* Basis-dir paths are canonicalized on the way in: trailing slashes and + interior empty / "." components are dropped so validation, the delete-walker + prefix and the receiver lookup all agree on one stored form. */ +static void test_config_basis_normalization() { + Config* c = config_create(); + EXPECT_NOT_NULL(c); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "prior/"), 0); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "a//b"), 0); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "./x/./y/"), 0); + EXPECT_EQ_INT(c->basis_count, 3); + EXPECT_EQ_STR(c->basis_dirs[0].path, "prior"); + EXPECT_EQ_STR(c->basis_dirs[1].path, "a/b"); + EXPECT_EQ_STR(c->basis_dirs[2].path, "x/y"); + + /* Degenerate values that normalize away to nothing stay rejected. */ + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "."), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ".."), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "/abs"), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "a/../b"), -1); + EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, ""), -1); + config_delete(c); +} + static void test_config_is_remote_dest() { /* Valid SSH-style destinations */ EXPECT_TRUE(config_is_remote_dest("user@host:/path")); @@ -599,6 +712,9 @@ void test_config() { test_config_delay_updates_reserved_backup_rejected(); test_config_delete_timing_wire_roundtrip(); test_config_delete_timing_conflict_rejected(); + test_config_basis_roundtrip(); + test_config_basis_wire_rejects_escaping(); + test_config_basis_normalization(); } test_config_delete_timing_early_helper(); test_config_is_remote_dest();