From 5597e74f6a1fa550c21f124b2472abd942dfb1bc Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 01:23:49 +0200 Subject: [PATCH 01/68] docs(handoff): v2.28.0 released to main (PR #304, tag v2.28.0) --- HANDOFF.md | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/HANDOFF.md b/HANDOFF.md index 43b23f0..c479f73 100644 --- a/HANDOFF.md +++ b/HANDOFF.md @@ -9,8 +9,11 @@ `project(FastFileTransfer VERSION 2.28.0)`. The cycle batched all wire changes (stats counters, filter-rule block, `--verify-basis`) under the one bump. -- **`main` = `ef76c90`** (tag `v2.26.0`); the 2.27.0/2.28.0 work is on `dev` - and not yet released. A `dev -> main` v2.28.0 release PR is the next step. +- **Release `v2.28.0` tagged and merged to `main`** via PR #304 + (`b4d54504`); tag `v2.28.0`. Main push CI run **585** fully green (lint, + build-and-test, parity-full, ASan, UBSan, fuzz-build, coverage, valgrind). + Gitea release `v2.28.0` published. `dev` and `main` are at the release + content. - Parity matrix: **116 ✅ / 14 ⚠️ / 27 ❌ = 157** (was 111/13/33 at cycle start). - Working tree clean; feature branch deleted; no scratch trees or worktrees. -- 2.54.0 From 558782d339a7a9c0c6ae125bbd52d83db2986825 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 12:09:21 +0200 Subject: [PATCH 02/68] test: eliminate fork/write race in incremental-check server tests The parent sends the file data body after the receiver's STATUS_NEXT, but the forked child exited as soon as receive_incremental_check returned. The parent's send_data could then race the child's exit into a spurious EPIPE (seen in the coverage job as test_server.c:1222), or the reverse: the parent could be descheduled past the child's exit. Keep the child alive until the parent closes its write end (drain to EOF), and close the parent's write end before waitpid so the child can observe EOF. Applied to the three tests sharing the pattern: size-mismatch, FIFO destination, and FIFO basis. Child exit status remains the authoritative assertion. --- tests/test_server.c | 24 +++++++++++++++++++++--- 1 file changed, 21 insertions(+), 3 deletions(-) diff --git a/tests/test_server.c b/tests/test_server.c index 31c7eb5..849dff2 100644 --- a/tests/test_server.c +++ b/tests/test_server.c @@ -458,6 +458,11 @@ static void test_incremental_check_size_mismatch_full_transfer() { File* file = receive_incremental_check(p[0], cfg, &skipped); bool ok = file != NULL && !skipped && file->path != NULL && strcmp(file->path, "file.txt") == 0; file_destroy(file); + /* Stay alive until the parent closes its write end so its send_data can + never race this exit into a spurious EPIPE. */ + char drain; + while (read(p[0], &drain, 1) > 0) { + } config_delete(cfg); close(p[0]); _exit(ok ? 0 : 1); @@ -487,9 +492,9 @@ static void test_incremental_check_size_mismatch_full_transfer() { EXPECT_TRUE(send_data(p[1], body)); data_destroy(body); + close(p[1]); int status; waitpid(pid, &status, 0); - close(p[1]); config_delete(cfg); unlink(path); rmdir(root); @@ -1025,6 +1030,11 @@ static void test_incremental_check_fifo_destination_does_not_hang() { File* file = receive_incremental_check(p[0], cfg, &skipped); bool ok = file != NULL && !skipped; file_destroy(file); + /* Stay alive until the parent closes its write end so its send_data can + never race this exit into a spurious EPIPE. */ + char drain; + while (read(p[0], &drain, 1) > 0) { + } config_delete(cfg); close(p[0]); _exit(ok ? 0 : 1); @@ -1051,9 +1061,9 @@ static void test_incremental_check_fifo_destination_does_not_hang() { EXPECT_TRUE(send_data(p[1], body)); data_destroy(body); + close(p[1]); int status; waitpid(pid, &status, 0); - close(p[1]); config_delete(cfg); unlink(path); rmdir(root); @@ -1191,6 +1201,12 @@ static void test_incremental_check_basis_fifo_does_not_hang() { File* file = receive_incremental_check(p[0], cfg, &skipped); bool ok = file != NULL && !skipped; file_destroy(file); + /* The parent sends the data body after the check reply and only then closes + its write end. Stay alive until that EOF so the parent's send_data can + never race this exit into a spurious EPIPE. */ + char drain; + while (read(p[0], &drain, 1) > 0) { + } config_delete(cfg); close(p[0]); _exit(ok ? 0 : 1); @@ -1222,9 +1238,11 @@ static void test_incremental_check_basis_fifo_does_not_hang() { EXPECT_TRUE(send_data(p[1], body)); data_destroy(body); + /* Close the write end before waiting: this hands the child EOF so it can + exit, and guarantees the child was still alive for the data write. */ + close(p[1]); int status; waitpid(pid, &status, 0); - close(p[1]); config_delete(cfg); unlink(basis_path); rmdir(basis_dir); -- 2.54.0 From 9691dba6f0a58b6c81f17b0fa78072641838ebf8 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 13:15:04 +0200 Subject: [PATCH 03/68] delete: reproduce rsync traversal order for extras removal Collect each directory's entries up front and process extraneous subdirectories first (descending name, depth-first), then extraneous files (descending name), then descend into kept subdirectories in ascending order. Emit a trailing slash for deleted directories in observers/dry-run output. This matches rsync's delete order for --delete-before/--delete-after/--delete-delay and for dry-run listings. --- src/shared/delete_plan.c | 155 +++++++----- src/shared/utils.c | 515 +++++++++++++++++++++++++++------------ src/shared/utils.h | 22 ++ 3 files changed, 485 insertions(+), 207 deletions(-) diff --git a/src/shared/delete_plan.c b/src/shared/delete_plan.c index 8aa96f7..644fb4c 100644 --- a/src/shared/delete_plan.c +++ b/src/shared/delete_plan.c @@ -471,6 +471,24 @@ static void notify_deleted(DeletePlanSession* session, const char* rel) { session->observer(session->observer_context, rel); } +/* A removed directory is reported with rsync's trailing slash (`deleting dir/`) + while files keep their bare path. */ +static void notify_deleted_dir(DeletePlanSession* session, const char* rel) { + if (!session || !session->observer || !rel) + return; + size_t len = strlen(rel); + char* with_slash = malloc(len + 2); + if (!with_slash) { + session->observer(session->observer_context, rel); + return; + } + memcpy(with_slash, rel, len); + with_slash[len] = '/'; + with_slash[len + 1] = '\0'; + session->observer(session->observer_context, with_slash); + free(with_slash); +} + DeletePlanSession* delete_plan_session_create(const Config* config) { if (!config) return NULL; @@ -696,7 +714,7 @@ static bool process_extra_dir(int dirfd, const char* name, const char* child_rel session->deleted++; session->planned++; log_deleted(child_rel); - notify_deleted(session, child_rel); + notify_deleted_dir(session, child_rel); *removed = true; return true; } @@ -733,81 +751,108 @@ static bool process_children(int dirfd, const char* dir_rel, const ArrayList* ke const ArrayList* keep_files, bool at_root, bool force_now, const PlanSkips* skips, DeletePlanSession* session, bool* survives) { *survives = false; - int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - if (scanfd < 0) + DeleteDirEntry* entries = NULL; + size_t count = 0; + bool collect_ok = true; + if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) return false; - DIR* dir = fdopendir(scanfd); - if (!dir) { - close(scanfd); + bool operation_ok = collect_ok; + bool local_survives = false; + bool* shielded = calloc(count ? count : 1, sizeof(bool)); + bool* is_extra = calloc(count ? count : 1, sizeof(bool)); + bool* force = calloc(count ? count : 1, sizeof(bool)); + if (!shielded || !is_extra || !force) { + free(shielded); + free(is_extra); + free(force); + delete_dir_entries_free(entries, count); return false; } - bool operation_ok = true; - bool local_survives = false; - const struct dirent* entry; - while ((entry = readdir(dir)) != NULL) { - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; + + /* rsync's order: extraneous subdirectories in descending name order, then + extraneous files in descending name order (kept entries survive and are not + touched here — a kept subdirectory gets its own per-directory plan). */ + if (count > 1) + qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); + size_t dir_count = 0; + while (dir_count < count && entries[dir_count].is_dir) + dir_count++; + + for (size_t i = 0; i < count; i++) { char* child_rel = - (strcmp(dir_rel, ".") == 0) ? str_dup(entry->d_name) : path_cat(dir_rel, entry->d_name); + (strcmp(dir_rel, ".") == 0) ? str_dup(entries[i].name) : path_cat(dir_rel, entries[i].name); if (!child_rel) { operation_ok = false; continue; } if (path_under_skip_prefix(child_rel, at_root, skips->entries, skips->count)) { + shielded[i] = true; local_survives = true; free(child_rel); continue; } - struct stat st; - if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { - if (errno != ENOENT) - operation_ok = false; - free(child_rel); - continue; - } - bool is_dir = S_ISDIR(st.st_mode); - bool in_keep_dirs = is_dir && list_contains_str(keep_dirs, entry->d_name); - bool in_keep_files = !is_dir && list_contains_str(keep_files, entry->d_name); + bool is_dir = entries[i].is_dir; + bool in_keep_dirs = is_dir && list_contains_str(keep_dirs, entries[i].name); + bool in_keep_files = !is_dir && list_contains_str(keep_files, entries[i].name); bool rule_protected = skips->protect_rules && - filter_rules_apply_side(skips->protect_rules, child_rel, entry->d_name, is_dir, + filter_rules_apply_side(skips->protect_rules, child_rel, entries[i].name, is_dir, FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT; - if (in_keep_dirs) { + if (in_keep_dirs || in_keep_files || rule_protected) { + shielded[i] = true; local_survives = true; - } else if (keep_dirs && !is_dir && list_contains_str(keep_dirs, entry->d_name)) { - /* Destination file blocks a source directory: clear it now, whatever the - delete timing, so the directory can be created. */ - if (!process_extra_file(dirfd, entry->d_name, child_rel, true, session)) - operation_ok = false; - } else if (in_keep_files) { - local_survives = true; - } else if (keep_files && is_dir && list_contains_str(keep_files, entry->d_name)) { - /* Destination directory blocks a source file: remove it now. */ - bool removed = false; - if (!process_extra_dir(dirfd, entry->d_name, child_rel, true, skips, session, &removed)) - operation_ok = false; - else if (!removed) - local_survives = true; } else if (is_dir) { - if (rule_protected) { - local_survives = true; - } else { - bool removed = false; - if (!process_extra_dir(dirfd, entry->d_name, child_rel, force_now, skips, session, - &removed)) - operation_ok = false; - else if (!removed) - local_survives = true; - } - } else if (rule_protected) { - local_survives = true; + /* A destination directory blocks a source file of the same name: remove + it now, whatever the delete timing, so the file can be created. */ + is_extra[i] = true; + force[i] = keep_files && list_contains_str(keep_files, entries[i].name); } else { - if (!process_extra_file(dirfd, entry->d_name, child_rel, force_now, session)) - operation_ok = false; + /* A destination file blocks a source directory of the same name: clear it + now so the directory can be created. */ + is_extra[i] = true; + force[i] = keep_dirs && list_contains_str(keep_dirs, entries[i].name); } free(child_rel); } - closedir(dir); + + /* Pass 1: extraneous subdirectories, descending. */ + for (size_t i = 0; i < dir_count; i++) { + if (!is_extra[i]) + continue; + char* child_rel = + (strcmp(dir_rel, ".") == 0) ? str_dup(entries[i].name) : path_cat(dir_rel, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + bool removed = false; + if (!process_extra_dir(dirfd, entries[i].name, child_rel, force[i] || force_now, skips, session, + &removed)) + operation_ok = false; + else if (!removed) + local_survives = true; + free(child_rel); + } + + /* Pass 2: extraneous files, descending. */ + for (size_t i = dir_count; i < count; i++) { + if (!is_extra[i]) + continue; + char* child_rel = + (strcmp(dir_rel, ".") == 0) ? str_dup(entries[i].name) : path_cat(dir_rel, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + if (!process_extra_file(dirfd, entries[i].name, child_rel, force[i] || force_now, session)) + operation_ok = false; + free(child_rel); + } + + free(shielded); + free(is_extra); + free(force); + delete_dir_entries_free(entries, count); *survives = local_survives; return operation_ok; } @@ -981,7 +1026,7 @@ static bool apply_deferred_path(DeletePlanSession* session, const Config* config session->deleted++; session->planned++; log_deleted(rel); - notify_deleted(session, rel); + notify_deleted_dir(session, rel); } else if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) { close(parent_fd); free(leaf); diff --git a/src/shared/utils.c b/src/shared/utils.c index 52bb5b2..d028e36 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -672,22 +672,24 @@ static bool is_synced_dir(const PathIndex* dirs, const char* rel) { return path_index_contains(dirs, rel[0] == '\0' ? "." : rel); } -/* Remove the extras directly inside the directory open on `dirfd`, recursing - into every child directory so kept content below a synchronized prefix is - reached. `all_removed` reports whether every child entry was removed (so the - caller may rmdir this directory). A child directory is never removed when it - is itself a synchronized directory or holds kept content; with a dirs index - supplied, direct children of a non-synchronized directory are never extras at - all (they are left in place but still descended into). Symlinks are unlinked - like any other non-directory extra (never followed). */ -static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, - const PathIndex* dirs, DeleteBudget* budget, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, bool parent_deletable, - bool* all_removed, DeletePathObserver observer, - void* observer_context) { - /* openat(dirfd, ".") opens an independent file description: a dup() would - share dirfd's file offset and a prior pass could leave the stream drained. */ +/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed + strcmp would order bytes >= 0x80 differently). */ +static int delete_name_cmp(const char* a, const char* b) { + const unsigned char* pa = (const unsigned char*)a; + const unsigned char* pb = (const unsigned char*)b; + while (*pa != '\0' && *pa == *pb) { + pa++; + pb++; + } + return (int)*pa - (int)*pb; +} + +bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, + bool* operation_ok) { + *out = NULL; + *count = 0; + if (operation_ok) + *operation_ok = true; int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); if (scanfd < 0) return false; @@ -696,17 +698,127 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k close(scanfd); return false; } - bool operation_ok = true; - bool local_survives = false; - /* A directory is deletable when it or ANY ancestor is synchronized; the - `parent_deletable` flag carries that down the recursion so dest-only - directories below a synchronized root are removed wholesale. */ - bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); + DeleteDirEntry* entries = NULL; + size_t used = 0; + size_t capacity = 0; + bool ok = true; const struct dirent* entry; while ((entry = readdir(dir)) != NULL) { if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) continue; - char* child_rel = path_cat((char*)rel_path, entry->d_name); + struct stat st; + if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { + if (errno != ENOENT && operation_ok) + *operation_ok = false; + continue; + } + if (used == capacity) { + size_t next = capacity == 0 ? 16 : capacity * 2; + DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown)); + if (!grown) { + ok = false; + break; + } + entries = grown; + capacity = next; + } + entries[used].name = str_dup(entry->d_name); + if (!entries[used].name) { + ok = false; + break; + } + entries[used].is_dir = S_ISDIR(st.st_mode); + used++; + } + closedir(dir); + if (!ok) { + delete_dir_entries_free(entries, used); + return false; + } + *out = entries; + *count = used; + return true; +} + +void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) { + if (!entries) + return; + for (size_t i = 0; i < count; i++) + free(entries[i].name); + free(entries); +} + +/* rsync's extraneous-entry order: subdirectories before files, each group in + descending name order. */ +int delete_dir_entry_cmp_desc(const void* a, const void* b) { + const DeleteDirEntry* ea = a; + const DeleteDirEntry* eb = b; + if (ea->is_dir != eb->is_dir) + return ea->is_dir ? -1 : 1; + return -delete_name_cmp(ea->name, eb->name); +} + +/* rsync's kept-subdirectory order: plain ascending name. */ +int delete_dir_entry_cmp_asc(const void* a, const void* b) { + const DeleteDirEntry* ea = a; + const DeleteDirEntry* eb = b; + return delete_name_cmp(ea->name, eb->name); +} + +/* Remove the extras directly inside the directory open on `dirfd`, recursing + into every child directory so kept content below a synchronized prefix is + reached. `all_removed` reports whether every child entry was removed (so the + caller may rmdir this directory). A child directory is never removed when it + is itself a synchronized directory or holds kept content; with a dirs index + supplied, direct children of a non-synchronized directory are never extras at + all (they are left in place but still descended into). Symlinks are unlinked + like any other non-directory extra (never followed). + + Entries are processed in rsync's order (extraneous subdirectories in + descending name order, then extraneous files, then kept subdirectories in + ascending order) rather than readdir() order, so `--max-delete` leaves the + same survivors and the `--info=del`/dry-run line order matches rsync. */ +static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, + const PathIndex* dirs, DeleteBudget* budget, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, bool parent_deletable, + bool* all_removed, DeletePathObserver observer, + void* observer_context) { + DeleteDirEntry* entries = NULL; + size_t count = 0; + bool collect_ok = true; + if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) + return false; + bool operation_ok = collect_ok; + bool local_survives = false; + bool* shielded = calloc(count ? count : 1, sizeof(bool)); + bool* is_extra = calloc(count ? count : 1, sizeof(bool)); + if (!shielded || !is_extra) { + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); + return false; + } + /* A directory is deletable when it or ANY ancestor is synchronized; the + `parent_deletable` flag carries that down the recursion so dest-only + directories below a synchronized root are removed wholesale. */ + bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); + bool at_root = rel_path[0] == '\0'; + + /* Reproduce rsync's traversal order: extraneous subdirectories in descending + name order, then extraneous files in descending name order, and kept + subdirectories only afterwards (ascending). Sorting up front also fixes the + identity of the survivors under a partial --max-delete. */ + if (count > 1) + qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); + size_t dir_count = 0; + while (dir_count < count && entries[dir_count].is_dir) + dir_count++; + + /* Classify every entry up front (the verdict does not depend on processing + order) so the ordered passes below can act on it. */ + for (size_t i = 0; i < count; i++) { + char* child_rel = path_cat((char*)rel_path, entries[i].name); if (!child_rel) { operation_ok = false; continue; @@ -718,93 +830,142 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k top-level-only prefix) and the basis prefixes are protected: a nested destination directory that happens to be called .fastsync-stage is ordinary content. */ - if (path_under_skip_prefix(child_rel, rel_path[0] == '\0', skips, skip_count)) { + if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { + shielded[i] = true; local_survives = true; - free(child_rel); - continue; - } - struct stat st; - if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { - if (errno != ENOENT) - operation_ok = false; - free(child_rel); - continue; - } - bool is_dir = S_ISDIR(st.st_mode); - if (protect_rules && filter_rules_apply_side(protect_rules, child_rel, entry->d_name, is_dir, - FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { + } else if (protect_rules && + filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, + FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { /* A first-match protect rule shields the extra; for a directory the whole subtree is shielded (rsync prunes an excluded directory), so do not descend. */ + shielded[i] = true; local_survives = true; - free(child_rel); + } else if (entries[i].is_dir) { + bool child_synced = dirs && path_index_contains(dirs, child_rel); + is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } else { + is_extra[i] = deletable && !keep_is_file(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } + free(child_rel); + } + + /* Pass 1: extraneous subdirectories, descending. */ + for (size_t i = 0; i < dir_count; i++) { + if (!is_extra[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; continue; } - if (is_dir) { - int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, - protect_rules, deletable, &child_all_removed, observer, - observer_context)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, + protect_rules, deletable, &child_all_removed, observer, + observer_context)) operation_ok = false; - } - bool child_synced = dirs && path_index_contains(dirs, child_rel); - if (child_synced || keep_is_dir(keep, child_rel)) { - /* A synchronized directory and a directory holding kept content are - never removed. */ - local_survives = true; - } else if (child_all_removed && deletable) { - if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - local_survives = true; - } else if (unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0) { - /* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory - still holds entries the walker leaves in place (a protected - excluded prefix, a kept file the manifest protects, a symlink); - rsync leaves such a directory behind, so this is not an error. - Only genuine I/O failures abort the deletion. */ - if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) - operation_ok = false; - local_survives = true; - } else { - budget->deleted++; - if (observer) - observer(observer_context, child_rel); - } - } else { - local_survives = true; - } - } else { - bool found = keep_is_file(keep, child_rel); - if (found || !deletable) { - /* Kept file, or a child of a directory that is not synchronized: never - an extra for this run. */ - local_survives = true; - } else if (budget->deleted >= budget->max_delete) { + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (child_all_removed && deletable) { + if (budget->deleted >= budget->max_delete) { budget->limit_hit = true; budget->skipped++; local_survives = true; - } else if (unlinkat(dirfd, entry->d_name, 0) != 0) { - if (errno != ENOENT) + } else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) { + /* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still + holds entries the walker leaves in place (a protected excluded + prefix, a kept file the manifest protects, a symlink); rsync leaves + such a directory behind, so this is not an error. Only genuine I/O + failures abort the deletion. */ + if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) operation_ok = false; local_survives = true; } else { budget->deleted++; + /* rsync reports a removed directory with a trailing slash. */ + if (observer) { + size_t len = strlen(child_rel); + char* with_slash = malloc(len + 2); + if (with_slash) { + memcpy(with_slash, child_rel, len); + with_slash[len] = '/'; + with_slash[len + 1] = '\0'; + observer(observer_context, with_slash); + free(with_slash); + } else { + observer(observer_context, child_rel); + } + } + } + } else { + local_survives = true; + } + free(child_rel); + } + + /* Pass 2: extraneous files, descending. */ + for (size_t i = dir_count; i < count; i++) { + if (!is_extra[i]) + continue; + if (budget->deleted >= budget->max_delete) { + budget->limit_hit = true; + budget->skipped++; + local_survives = true; + } else if (unlinkat(dirfd, entries[i].name, 0) != 0) { + if (errno != ENOENT) + operation_ok = false; + local_survives = true; + } else { + budget->deleted++; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (child_rel) { if (observer) observer(observer_context, child_rel); char* escaped_path = output_escape(child_rel, log_get_8_bit_output()); fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : ""); free(escaped_path); } + free(child_rel); } + } + + /* Pass 3: kept subdirectories, ascending (rsync descends into these only + after the parent's own extras have been handled). */ + for (size_t i = dir_count; i-- > 0;) { + if (is_extra[i] || shielded[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, + protect_rules, deletable, &child_all_removed, observer, + observer_context)) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + /* A kept/synchronized directory is never removed. */ + local_survives = true; free(child_rel); } - closedir(dir); + + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); *all_removed = !local_survives; return operation_ok; } @@ -817,97 +978,147 @@ static bool list_extras_fd(int dirfd, const char* rel_path, const PathIndex* kee const DeleteSkipEntry* skips, int skip_count, const FilterRuleList* protect_rules, bool parent_deletable, bool* all_removed) { - int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - if (scanfd < 0) + DeleteDirEntry* entries = NULL; + size_t count = 0; + bool collect_ok = true; + if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) return false; - DIR* dir = fdopendir(scanfd); - if (!dir) { - close(scanfd); + bool operation_ok = collect_ok; + bool local_survives = false; + bool* shielded = calloc(count ? count : 1, sizeof(bool)); + bool* is_extra = calloc(count ? count : 1, sizeof(bool)); + if (!shielded || !is_extra) { + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); return false; } - bool operation_ok = true; - bool local_survives = false; bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); - const struct dirent* entry; - while ((entry = readdir(dir)) != NULL) { - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; - char* child_rel = path_cat((char*)rel_path, entry->d_name); + bool at_root = rel_path[0] == '\0'; + + /* Mirror the delete walk's rsync order (extraneous subdirectories descending, + then extraneous files descending, then kept subdirectories ascending). */ + if (count > 1) + qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); + size_t dir_count = 0; + while (dir_count < count && entries[dir_count].is_dir) + dir_count++; + + for (size_t i = 0; i < count; i++) { + char* child_rel = path_cat((char*)rel_path, entries[i].name); if (!child_rel) { operation_ok = false; continue; } - if (path_under_skip_prefix(child_rel, rel_path[0] == '\0', skips, skip_count)) { + if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { + shielded[i] = true; local_survives = true; - free(child_rel); - continue; - } - struct stat st; - if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { - if (errno != ENOENT) - operation_ok = false; - free(child_rel); - continue; - } - bool is_dir = S_ISDIR(st.st_mode); - if (protect_rules && filter_rules_apply_side(protect_rules, child_rel, entry->d_name, is_dir, - FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { + } else if (protect_rules && + filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, + FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { /* Mirror the delete walk: a protected entry is never reported as a would-delete and a protected directory's subtree is not enumerated. */ + shielded[i] = true; local_survives = true; - free(child_rel); + } else if (entries[i].is_dir) { + bool child_synced = dirs && path_index_contains(dirs, child_rel); + is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } else { + is_extra[i] = deletable && !keep_is_file(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } + free(child_rel); + } + + /* Pass 1: extraneous subdirectories, descending (recorded after contents). */ + for (size_t i = 0; i < dir_count; i++) { + if (!is_extra[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; continue; } - if (is_dir) { - int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count, - protect_rules, deletable, &child_all_removed)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count, + protect_rules, deletable, &child_all_removed)) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (child_all_removed && deletable) { + size_t len = strlen(child_rel); + char* copy = malloc(len + 2); + if (!copy) { operation_ok = false; - } - bool child_synced = dirs && path_index_contains(dirs, child_rel); - if (child_synced || keep_is_dir(keep, child_rel)) { - local_survives = true; - } else if (child_all_removed && deletable) { - size_t len = strlen(child_rel); - char* copy = malloc(len + 2); - if (!copy) { - operation_ok = false; - } else { - memcpy(copy, child_rel, len); - copy[len] = '/'; - copy[len + 1] = '\0'; - if (!array_list_add(out, copy)) { - free(copy); - operation_ok = false; - } else { - (*recorded)++; - } - } } else { - local_survives = true; - } - } else { - bool found = keep_is_file(keep, child_rel); - if (found || !deletable) { - local_survives = true; - } else { - char* copy = str_dup(child_rel); - if (!copy || !array_list_add(out, copy)) { + memcpy(copy, child_rel, len); + copy[len] = '/'; + copy[len + 1] = '\0'; + if (!array_list_add(out, copy)) { free(copy); operation_ok = false; } else { (*recorded)++; } } + } else { + local_survives = true; } free(child_rel); } - closedir(dir); + + /* Pass 2: extraneous files, descending. */ + for (size_t i = dir_count; i < count; i++) { + if (!is_extra[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + char* copy = str_dup(child_rel); + if (!copy || !array_list_add(out, copy)) { + free(copy); + operation_ok = false; + } else { + (*recorded)++; + } + free(child_rel); + } + + /* Pass 3: kept subdirectories, ascending. */ + for (size_t i = dir_count; i-- > 0;) { + if (is_extra[i] || shielded[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count, + protect_rules, deletable, &child_all_removed)) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + local_survives = true; + free(child_rel); + } + + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); *all_removed = !local_survives; return operation_ok; } diff --git a/src/shared/utils.h b/src/shared/utils.h index 5d746f6..7bf89d8 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -134,6 +134,28 @@ typedef struct { only DIRECT children of the destination root, i.e. child_rel has no '/'). */ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, int skip_count); +/* One destination-directory entry collected up front so the delete walkers can + reproduce rsync's traversal order instead of readdir() order. rsync processes + a directory's extraneous subdirectories first (descending name, depth-first), + then its extraneous files (descending name), and only afterwards descends into + its kept subdirectories (ascending name). */ +typedef struct { + char* name; + bool is_dir; +} DeleteDirEntry; +/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."), + stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of + *count entries whose names the caller frees with delete_dir_entries_free(). + Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is + skipped, any other stat failure is reported through *operation_ok while the + walk continues. */ +bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok); +void delete_dir_entries_free(DeleteDirEntry* entries, size_t count); +/* Sort comparators: `_desc` orders subdirectories before files and each group by + descending name (rsync's extraneous-entry order); `_asc` orders plain ascending + name (rsync's kept-subdirectory order). */ +int delete_dir_entry_cmp_desc(const void* a, const void* b); +int delete_dir_entry_cmp_asc(const void* a, const void* b); /* Remove files/dirs/symlinks under dest_root that are not listed in manifest without ever descending into a protected prefix (see DeleteSkipEntry). When `synced_dirs` is non-NULL, extras are only removed directly inside a directory -- 2.54.0 From 402cae80ad57b6148317824972e22a13b1c2f357 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 13:39:04 +0200 Subject: [PATCH 04/68] parity: rsync-exact relative basis-dir resolution and fuzzy eligibility Resolve a relative --compare-dest/--copy-dest/--link-dest DIR against the destination directory and append the file's transfer-relative name, as rsync 3.4.1 does, instead of appending FastSync's source-mirrored wire path (the historical spelling stays as a fallback for existing layouts). Stop inheriting the ordinary delta engine's 16 KiB minimum and 10x size ratio in the -y/--fuzzy candidate search: rsync's find_fuzzy has no delta-size gate, so an oversized or sub-16-KiB sibling is now reused. The ordinary delta path's bounds are unchanged. --- src/shared/file_receive.c | 127 ++++++++---- tests/integration/test_parity_basis_fuzzy.py | 198 +++++++++++++++++++ tests/integration/test_parity_quickwins.py | 38 ++-- 3 files changed, 304 insertions(+), 59 deletions(-) create mode 100644 tests/integration/test_parity_basis_fuzzy.py diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index b724c4f..d41ca1d 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -1389,6 +1389,43 @@ bool file_basis_content_required(const Config* config) { return config != NULL && config->verify_basis; } +/* Probe one candidate basis file: open it (confined, O_NOFOLLOW) and apply + rsync's metadata quick-check; under --verify-basis also hash its bytes and + require the sender's digest. On a hit record `candidate` in `out` and return + true. The caller retains ownership of `candidate`. */ +static bool basis_match_probe(const Config* config, const char* candidate, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, const uint8_t* check_digest, + size_t check_digest_len, BasisDestType type, BasisMatch* out) { + int fd; + struct stat st; + if (!basis_open_regular(candidate, check_size, &fd, &st)) + return false; + bool hit = false; + if (file_basis_quick_match(config, &st, check_mtime, check_mtime_nsec)) { + hit = true; + if (file_basis_content_required(config)) { + uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t basis_len = 0; + bool hashed = checksum_digest_fd((ChecksumAlgo)config->checksum_algo, config->checksum_seed, + fd, basis_digest, sizeof(basis_digest), &basis_len); + hit = hashed && basis_len == check_digest_len && check_digest_len > 0 && + memcmp(basis_digest, check_digest, check_digest_len) == 0; + } + } + close(fd); + if (!hit) + return false; + char* owned = str_dup(candidate); + if (!owned) + return false; + out->hit = true; + out->type = type; + out->basis_path = owned; + out->st = st; + return true; +} + /* Search the basis-dir list in command-line order and return the first match. By default (no --verify-basis) rsync's metadata quick-check is sufficient: basis_open_regular has already required an equal size, and @@ -1403,7 +1440,22 @@ bool file_basis_content_required(const Config* config) { --dry-run passes false because hashing a basis against a client-supplied digest would be a 1-bit content oracle. Without --verify-basis a dry-run can still confirm the metadata-only hit without reading any basis bytes, matching - rsync's read-only quick-check. */ + rsync's read-only quick-check. + + Path resolution (rsync 3.4.1 parity): rsync resolves a relative + --compare-dest/--copy-dest/--link-dest DIR against the destination directory + (the receiver's cwd) and appends the file's TRANSFER-RELATIVE name, e.g. + `--compare-dest=basis` with `rsync src/ dst/` probes `dst/basis/`. + FastSync's receive root IS the destination directory, but its default transfer + mirrors the absolute source path below that root, so check_path carries the + source-root scaffolding rsync would not append. Recover rsync's spelling with + utils_strip_transfer_root for a relative DIR; under -R/--files-from the wire + path is already transfer-relative, so it is used as-is. A relative DIR also + probes the historical mirror-appended spelling as a fallback, so existing + FastSync-laid-out snapshot trees keep resolving. An absolute DIR is used + verbatim and keeps appending the destination-relative check_path (FastSync's + mirrored layout). Every candidate stays confined to the authorized root by + file_open_secure_parent. */ static bool basis_match_find(const Config* config, const char* check_path, unsigned long long check_size, time_t check_mtime, long check_mtime_nsec, const uint8_t* check_digest, @@ -1416,47 +1468,39 @@ static bool basis_match_find(const Config* config, const char* check_path, the basis bytes. */ if (file_basis_content_required(config) && !hash_content) return false; + const char* transfer_rel = check_path; + if (!config->relative && config->files_from_set == NULL) + transfer_rel = utils_strip_transfer_root(check_path, config->send_directory); for (int i = 0; i < config->basis_count; i++) { const BasisDest* entry = &config->basis_dirs[i]; /* An absolute basis path is used verbatim (rsync semantics); a relative one is resolved below the receive root. Both remain subject to the receiver's authorized-root confinement inside file_open_secure_parent. */ - char* basis_dir = entry->path[0] == '/' ? str_dup(entry->path) - : path_cat(config->receive_root_directory, entry->path); + bool absolute = entry->path[0] == '/'; + char* basis_dir = + absolute ? str_dup(entry->path) : path_cat(config->receive_root_directory, entry->path); if (!basis_dir) continue; - char* candidate = path_cat(basis_dir, check_path); - free(basis_dir); - if (!candidate) - continue; - - int fd; - struct stat st; - if (basis_open_regular(candidate, check_size, &fd, &st)) { - if (file_basis_quick_match(config, &st, check_mtime, check_mtime_nsec)) { - bool hit = true; - if (file_basis_content_required(config)) { - uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN]; - size_t basis_len = 0; - bool hashed = - checksum_digest_fd((ChecksumAlgo)config->checksum_algo, config->checksum_seed, fd, - basis_digest, sizeof(basis_digest), &basis_len); - hit = hashed && basis_len == check_digest_len && check_digest_len > 0 && - memcmp(basis_digest, check_digest, check_digest_len) == 0; - } - if (hit) { - out->hit = true; - out->type = entry->type; - out->basis_path = candidate; - candidate = NULL; /* ownership transferred to out */ - out->st = st; - close(fd); - return true; - } - } - close(fd); + const char* names[2]; + int name_count = 0; + if (absolute) + names[name_count++] = check_path; + else + names[name_count++] = transfer_rel; + if (!absolute && strcmp(transfer_rel, check_path) != 0) + names[name_count++] = check_path; /* historical mirror-appended spelling */ + bool found = false; + for (int n = 0; n < name_count && !found; n++) { + char* candidate = path_cat(basis_dir, names[n]); + if (!candidate) + continue; + found = basis_match_probe(config, candidate, check_size, check_mtime, check_mtime_nsec, + check_digest, check_digest_len, entry->type, out); + free(candidate); } - free(candidate); + free(basis_dir); + if (found) + return true; } return false; } @@ -1489,9 +1533,12 @@ static bool basis_match_find(const Config* config, const char* check_path, * followed and nothing outside the destination root is ever read; * * dotfiles, directories, the target's own name, and the .fastsync-stage / * temp scratch names are never candidates; - * * size gate = the delta engine's own bounds (delta_should_attempt: both - * files >= DELTA_MIN_FILE_SIZE, <= delta_max_file_size, ratio <= 10x), - * because FastSync's delta engine cannot use a basis outside them; + * * size gate = rsync's, NOT the ordinary delta engine's bounds: any + * non-empty regular sibling up to the receiver's whole-file buffer cap is + * eligible, regardless of the 16 KiB delta minimum or the 10x delta size + * ratio (rsync's find_fuzzy has no delta-size gate at all). The delta + * engine consumes the fuzzy basis through the same signature handshake + * whether or not it is inside delta_should_attempt's window; * * first pass = an exact size+mtime match wins regardless of name (rsync's * "fuzzy size/modtime match"); * * otherwise the winner minimizes rsync's weighted Levenshtein distance @@ -1639,8 +1686,7 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p long check_mtime_nsec, unsigned long long* out_size) { *out_size = 0; if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || - !check_path || check_size < DELTA_MIN_FILE_SIZE || check_size > config->delta_max_file_size || - check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) + !check_path || check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) return NULL; char* full_path = path_cat(config->receive_root_directory, check_path); @@ -1717,8 +1763,7 @@ static void* fuzzy_basis_find_and_load(const Config* config, const char* check_p if (fstatat(dir_fd, name, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISREG(st.st_mode)) continue; unsigned long long cand_size = (unsigned long long)st.st_size; - if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE || - !delta_should_attempt(cand_size, check_size, config->delta_max_file_size)) + if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE) continue; long cand_nsec = 0; #ifdef __linux__ diff --git a/tests/integration/test_parity_basis_fuzzy.py b/tests/integration/test_parity_basis_fuzzy.py new file mode 100644 index 0000000..9f4a72f --- /dev/null +++ b/tests/integration/test_parity_basis_fuzzy.py @@ -0,0 +1,198 @@ +"""Differential rsync-parity coverage for two residuals closed on this branch. + +* A4 -- ``--compare-dest``/``--copy-dest``/``--link-dest`` relative-DIR + resolution: rsync resolves a relative DIR against the destination directory + and appends the file's TRANSFER-RELATIVE name. FastSync's default transfer + mirrors the absolute source path below its receive root, so a naive relative + DIR used to probe a different tree. These tests seed the basis at rsync's + spelling and assert FastSync finds it (byte-exact / hard-linked / sparse), + matching real rsync 3.4.1. + +* A5 -- ``-y``/``--fuzzy`` candidate eligibility: rsync's ``find_fuzzy`` has no + delta-size gate, so it reuses an oversized (>10x) or sub-16-KiB sibling; + FastSync used to decline both. These tests assert FastSync now uses the same + sibling as rsync (observable as ``Matched data``) with a byte-exact result. + +Every test skips cleanly when rsync is absent. +""" +import os +import shutil +import subprocess +import sys + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import ( # noqa: E402 + TEST_DATA_DIR, + clean_dir, + get_dest_received_dir, + run_client, +) + +RSYNC = shutil.which("rsync") +requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed") + +OLD_MTIME = 1_500_000_000 + + +def _write(path, content, mtime=None): + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "wb") as fh: + fh.write(content) + if mtime is not None: + os.utime(path, (mtime, mtime)) + + +def _read(path): + with open(path, "rb") as fh: + return fh.read() + + +def _rsync(args): + env = dict(os.environ, LC_ALL="C") + return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120) + + +def _stat_bytes(text, label): + for line in text.splitlines(): + if line.startswith(label + ":"): + return int(line.split(":", 1)[1].strip().split()[0].replace(",", "")) + return None + + +class TestRelativeBasisDirResolution: + """A4: a relative basis DIR must resolve to the same tree as rsync's.""" + + _FILES = { + "root.txt": b"root-basis-content\n", + "sub/nested.txt": b"nested-basis-content\n", + } + + def _seed_source(self, source): + clean_dir(source) + for rel, data in self._FILES.items(): + _write(os.path.join(source, rel), data, OLD_MTIME) + return self._FILES + + @requires_rsync + @pytest.mark.parametrize("flag", ["--compare-dest", "--link-dest"]) + def test_relative_dir_resolves_like_rsync(self, shared_server, flag): + tag = flag.lstrip("-") + source = os.path.join(TEST_DATA_DIR, f"relbasis_{tag}_src") + rdst = os.path.join(TEST_DATA_DIR, f"relbasis_{tag}_rdst") + fdst = os.path.join(TEST_DATA_DIR, f"relbasis_{tag}_fdst") + self._seed_source(source) + + # rsync: relative DIR -> dest/basis/. + clean_dir(rdst) + for rel, data in self._FILES.items(): + _write(os.path.join(rdst, "basis", rel), data, OLD_MTIME) + rs = _rsync(["-a", f"{flag}=basis", source + "/", rdst + "/"]) + assert rs.returncode == 0, rs.stderr + + # FastSync: the SAME relative spelling seeded at the SAME + # transfer-relative location under its destination root. + clean_dir(fdst) + for rel, data in self._FILES.items(): + _write(os.path.join(fdst, "basis", rel), data, OLD_MTIME) + result, _ = run_client(source, fdst, + flags=["-a", f"{flag}=basis", "--incremental"], + port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + received = get_dest_received_dir(fdst, source) + + for rel, data in self._FILES.items(): + rfile = os.path.join(rdst, rel) + ffile = os.path.join(received, rel) + basis = os.path.join(fdst, "basis", rel) + if flag == "--compare-dest": + # compare-dest never copies: both destinations stay sparse. + assert not os.path.exists(rfile), f"rsync copied {rel}" + assert not os.path.exists(ffile), ( + f"FastSync did not resolve the relative basis DIR at {basis!r} " + f"(expected {rel!r} to stay sparse like rsync)") + else: + # link-dest hard-links; a basis miss would transfer a new file. + assert os.path.exists(ffile), f"FastSync lost {rel}" + assert _read(ffile) == data + assert os.stat(ffile).st_ino == os.stat(basis).st_ino, ( + f"FastSync did not hard-link {rel!r} to the relative basis at " + f"{basis!r} (basis not resolved like rsync)") + + @requires_rsync + def test_relative_dir_copy_dest_content(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "relbasis_copy_src") + fdst = os.path.join(TEST_DATA_DIR, "relbasis_copy_fdst") + self._seed_source(source) + clean_dir(fdst) + for rel, data in self._FILES.items(): + _write(os.path.join(fdst, "basis", rel), data, OLD_MTIME) + result, _ = run_client(source, fdst, + flags=["-a", "--copy-dest=basis", "--incremental"], + port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + received = get_dest_received_dir(fdst, source) + for rel, data in self._FILES.items(): + ffile = os.path.join(received, rel) + assert os.path.exists(ffile), f"copy-dest did not materialize {rel}" + assert _read(ffile) == data + assert os.stat(ffile).st_ino != os.stat(os.path.join(fdst, "basis", rel)).st_ino + + +class TestFuzzyEligibilityWindow: + """A5: --fuzzy candidate eligibility must match rsync's uncapped window.""" + + BASE = b"the quick brown fox jumps over the lazy dog\n" * 4000 + + def _run_pair(self, shared_server, tag, payload, sibling): + source = os.path.join(TEST_DATA_DIR, f"fzw_{tag}_src") + dest = os.path.join(TEST_DATA_DIR, f"fzw_{tag}_dst") + rdst = os.path.join(TEST_DATA_DIR, f"fzw_{tag}_rdst") + clean_dir(source) + clean_dir(dest) + clean_dir(rdst) + _write(os.path.join(source, "report_v2.txt"), payload) + for root in (rdst, get_dest_received_dir(dest, source)): + _write(os.path.join(root, "report_v1.txt"), sibling) + + rs = _rsync(["-a", "--no-whole-file", "--fuzzy", "--stats", + source + "/", rdst + "/"]) + assert rs.returncode == 0, rs.stderr + result, _ = run_client( + source, dest, + flags=["-a", "--incremental", "--delta", "--fuzzy", "--stats"], + port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + + # The reconstructed file is byte-exact in every case. + assert _read(os.path.join(get_dest_received_dir(dest, source), + "report_v2.txt")) == payload + return rs, result + + @requires_rsync + def test_oversized_sibling_eligible_like_rsync(self, shared_server): + """A sibling 20x the source is used by rsync; FastSync must too (its old + 10x delta-size gate declined it).""" + n = 65536 + payload = (self.BASE * ((n // len(self.BASE)) + 1))[:n] + sibling = (self.BASE * 200)[: n * 20] + rs, result = self._run_pair(shared_server, "big", payload, sibling) + assert _stat_bytes(rs.stdout, "Matched data") > 0, \ + "rsync should use a >10x fuzzy basis" + assert _stat_bytes(result.stdout, "Matched data") > 0, ( + "FastSync's fuzzy eligibility must accept a >10x sibling like rsync " + f"(Matched data={_stat_bytes(result.stdout, 'Matched data')})") + + @requires_rsync + def test_small_source_sibling_eligible_like_rsync(self, shared_server): + """A sub-16-KiB source with an identical sibling is used by rsync; + FastSync's old 16 KiB delta minimum declined it.""" + n = 8192 + payload = (self.BASE * ((n // len(self.BASE)) + 1))[:n] + rs, result = self._run_pair(shared_server, "small", payload, payload) + assert _stat_bytes(rs.stdout, "Matched data") > 0, \ + "rsync applies --fuzzy below 16 KiB" + assert _stat_bytes(result.stdout, "Matched data") > 0, ( + "FastSync's fuzzy eligibility must accept a sub-16-KiB source like " + f"rsync (Matched data={_stat_bytes(result.stdout, 'Matched data')})") diff --git a/tests/integration/test_parity_quickwins.py b/tests/integration/test_parity_quickwins.py index 596b468..d2e13b5 100644 --- a/tests/integration/test_parity_quickwins.py +++ b/tests/integration/test_parity_quickwins.py @@ -861,11 +861,11 @@ class TestFuzzy: byte-exact result. FastSync ports rsync 3.4.1's weighted-Levenshtein name heuristic, so where both delta engines admit the candidate the tools pick the same basis (the ``fuzzy_basis`` differential asserts the tree and the - Matched/Literal counters match with the block size pinned). The residual is - candidate ELIGIBILITY: FastSync's delta size gate (both files >= 16 KiB and - a <= 10x size ratio) is narrower than rsync's, which empirically uses a - fuzzy basis well beyond 10x and below 16 KiB. These tests pin the window - boundary and prove the byte-exact fallback on both sides of it.""" + Matched/Literal counters match with the block size pinned). Candidate + ELIGIBILITY is now rsync's too: the fuzzy search no longer inherits the + ordinary delta engine's 16 KiB minimum or 10x size-ratio bound, so an + oversized or sub-16-KiB sibling is reused exactly as rsync reuses it. + These tests pin that window on both sides.""" _BASE = b"the quick brown fox jumps over the lazy dog\n" * 4000 @@ -900,9 +900,10 @@ class TestFuzzy: return rs, result @requires_rsync - def test_fuzzy_above_size_window_declines_but_tree_exact(self, shared_server): - """A sibling >10x the source is used by rsync but declined by FastSync's - delta size-ratio gate; both destinations stay byte-identical.""" + def test_fuzzy_above_size_window_matches_rsync(self, shared_server): + """A sibling >10x the source is used by rsync and by FastSync: fuzzy + eligibility is rsync's, not the ordinary delta size-ratio gate; both + destinations stay byte-identical and both reuse the basis.""" n = 65536 payload = (self._BASE * ((n // len(self._BASE)) + 1))[:n] sibling = (self._BASE * 200)[: n * 20] @@ -911,15 +912,16 @@ class TestFuzzy: rs, result = self._run_both(shared_server, source, dest, rdst, payload, sibling) assert _stat_bytes(rs.stdout, "Matched data") > 0, \ - "rsync should still use a >10x fuzzy basis" - assert _stat_bytes(result.stdout, "Matched data") == 0, \ - "FastSync's 10x delta size-ratio gate must decline the oversized basis" - assert _stat_bytes(result.stdout, "Literal data") == n + "rsync should use a >10x fuzzy basis" + assert _stat_bytes(result.stdout, "Matched data") > 0, \ + "FastSync must accept a >10x fuzzy basis like rsync" + assert _stat_bytes(result.stdout, "Literal data") < n @requires_rsync - def test_fuzzy_below_delta_minimum_declines_but_tree_exact(self, shared_server): - """A sibling below the 16 KiB delta minimum is used by rsync but never - enters FastSync's delta/fuzzy path; both trees stay byte-identical.""" + def test_fuzzy_below_delta_minimum_matches_rsync(self, shared_server): + """A sibling below the 16 KiB delta minimum is used by rsync and by + FastSync: fuzzy eligibility no longer inherits the delta engine's + minimum; both trees stay byte-identical and both reuse the basis.""" n = 8192 payload = (self._BASE * ((n // len(self._BASE)) + 1))[:n] source, dest, rdst = (self._src("small"), self._dst("small"), @@ -928,9 +930,9 @@ class TestFuzzy: payload, payload) assert _stat_bytes(rs.stdout, "Matched data") > 0, \ "rsync applies --fuzzy below 16 KiB" - assert _stat_bytes(result.stdout, "Matched data") == 0, \ - "FastSync's 16 KiB delta minimum must bypass the fuzzy basis" - assert _stat_bytes(result.stdout, "Literal data") == n + assert _stat_bytes(result.stdout, "Matched data") > 0, \ + "FastSync must apply --fuzzy below 16 KiB like rsync" + assert _stat_bytes(result.stdout, "Literal data") < n class TestIgnoreExistingShortCircuit: -- 2.54.0 From 79a28cdb966dce2db1f2ab243c1638460fded325 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 13:49:04 +0200 Subject: [PATCH 05/68] scanner: emit entries in rsync's sorted depth-first flist order Buffer and sort each directory's inspected entries (non-directories ascending, then directories ascending) and walk them depth-first via a LIFO directory stack, so the sequential scanner's stream matches rsync 3.4.1's flist order. This makes the --info=name transfer order and the --delete-during/--delete-delay deletion sequence byte-identical to rsync (differential tests in test_parity_order.py); --threads stays unordered (no rsync analogue) and is documented as such. Adds LIFO queue_push/queue_pop over the existing ring buffer. --- src/client/scanner.c | 221 +++++++++++++++++++------ src/client/scanner.h | 12 ++ src/shared/queue.c | 17 ++ src/shared/queue.h | 7 + tests/integration/test_parity_order.py | 135 +++++++++++++++ 5 files changed, 345 insertions(+), 47 deletions(-) create mode 100644 tests/integration/test_parity_order.py diff --git a/src/client/scanner.c b/src/client/scanner.c index 6d83c8e..b768001 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -205,6 +205,17 @@ typedef struct { bool referent_error; } ScannerEntry; +/* One inspected directory entry buffered so the sequential scanner can emit the + stream in rsync's flist order. `name` is the raw dirent name (owned here); + `entry` is the scanner_inspect_entry() result whose path/link_target are owned + when `inspection == 1`; `inspection` is that call's return code (1 keep, + 0 skip, <0 fatal). */ +typedef struct { + char* name; + ScannerEntry entry; + int inspection; +} SortedEntry; + /* --one-file-system (-x) decision. Only directories can carry a different * device than their parent (mount points), so this is checked when a child * directory is about to be descended into. */ @@ -687,6 +698,114 @@ skip: return 0; } +static void sorted_entry_destroy(void* item) { + SortedEntry* se = (SortedEntry*)item; + if (!se) + return; + free(se->name); + free(se->entry.path); + free(se->entry.link_target); +} + +/* rsync flist order within one directory: non-directories first, then + directories, each group by ascending name. strcmp() compares as unsigned + char, matching rsync's f_name_cmp(). */ +static int sorted_entry_cmp(const void* a, const void* b) { + const SortedEntry* x = (const SortedEntry*)a; + const SortedEntry* y = (const SortedEntry*)b; + bool x_dir = x->inspection > 0 && x->entry.is_directory; + bool y_dir = y->inspection > 0 && y->entry.is_directory; + if (x_dir != y_dir) + return x_dir ? 1 : -1; + return strcmp(x->name, y->name); +} + +static void scanner_free_sorted(DirectoryScanner* scanner) { + SortedEntry* entries = (SortedEntry*)scanner->sorted_entries; + for (size_t i = 0; i < scanner->sorted_count; i++) + sorted_entry_destroy(&entries[i]); + free(entries); + scanner->sorted_entries = NULL; + scanner->sorted_count = 0; + scanner->sorted_index = 0; +} + +/* Read every entry of the open directory, inspect it once and store it sorted in + rsync's flist order. Returns 0 on success, -1 on a fatal error (the caller + aborts the scan). */ +static int scanner_buffer_current_directory(DirectoryScanner* scanner) { + size_t capacity = 64; + size_t count = 0; + SortedEntry* entries = malloc(capacity * sizeof(*entries)); + if (!entries) { + scanner->failed = true; + return -1; + } + const struct dirent* dirent; + while ((dirent = readdir(scanner->current_dir)) != NULL) { + if (strcmp(dirent->d_name, ".") == 0 || strcmp(dirent->d_name, "..") == 0) + continue; + if (count == capacity) { + size_t next = capacity * 2; + SortedEntry* grown = realloc(entries, next * sizeof(*entries)); + if (!grown) { + scanner->failed = true; + break; + } + entries = grown; + capacity = next; + } + char* name = str_dup(dirent->d_name); + if (!name) { + scanner->failed = true; + break; + } + char* link_rel = child_rel_path(scanner->current_rel, dirent->d_name); + if (!link_rel) { + free(name); + scanner->failed = true; + break; + } + int inspection = scanner_inspect_entry(&scanner->options, scanner->current_path, link_rel, + dirent->d_name, &entries[count].entry); + free(link_rel); + if (inspection < 0) { + free(name); + scanner->failed = true; + break; + } + entries[count].name = name; + entries[count].inspection = inspection; + count++; + } + if (scanner->failed) { + for (size_t i = 0; i < count; i++) + sorted_entry_destroy(&entries[i]); + free(entries); + return -1; + } + qsort(entries, count, sizeof(*entries), sorted_entry_cmp); + scanner->sorted_entries = entries; + scanner->sorted_count = count; + scanner->sorted_index = 0; + return 0; +} + +/* Push this directory's collected child directories onto the LIFO stack in + reverse so the first (ascending) child is popped first (depth-first). */ +static void scanner_push_pending_dirs(DirectoryScanner* scanner) { + ArrayList* pending = (ArrayList*)scanner->pending_dirs; + if (!pending) + return; + for (int i = pending->size - 1; i >= 0; i--) { + if (!queue_push(scanner->directories, pending->items[i])) { + dir_entry_destroy(pending->items[i]); + scanner->failed = true; + } + } + pending->size = 0; +} + DirectoryScanner* directory_scanner_create_with_options(const char* root_directory, const ScannerOptions* options) { if (!root_directory || !options) @@ -704,12 +823,22 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo free(scanner); return NULL; } + scanner->pending_dirs = array_list_create(NULL); + if (!scanner->pending_dirs) { + queue_destroy(scanner->directories); + free(scanner); + return NULL; + } scanner->current_dir = NULL; scanner->current_path = NULL; scanner->current_depth = 0; scanner->failed = false; + scanner->sorted_entries = NULL; + scanner->sorted_count = 0; + scanner->sorted_index = 0; scanner->root_path = str_dup(root_directory); if (!scanner->root_path) { + array_list_delete(scanner->pending_dirs); queue_destroy(scanner->directories); free(scanner); return NULL; @@ -729,6 +858,7 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo scanner->filter_nodes = array_list_create(filter_node_destroy); if (!scanner->filter_nodes) { free(scanner->root_path); + array_list_delete(scanner->pending_dirs); queue_destroy(scanner->directories); free(scanner); return NULL; @@ -739,6 +869,7 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo if (stat(root_directory, &root_stats) != 0) { log_perror("Could not stat source directory"); free(scanner->root_path); + array_list_delete(scanner->pending_dirs); queue_destroy(scanner->directories); array_list_delete(scanner->filter_nodes); free(scanner); @@ -757,6 +888,7 @@ DirectoryScanner* directory_scanner_create_with_options(const char* root_directo if (!queue_enqueue(scanner->directories, root)) { dir_entry_destroy(root); free(scanner->root_path); + array_list_delete(scanner->pending_dirs); queue_destroy(scanner->directories); array_list_delete(scanner->filter_nodes); free(scanner); @@ -808,6 +940,13 @@ void directory_scanner_destroy(DirectoryScanner* scanner) { free(scanner->current_path); free(scanner->current_rel); free(scanner->root_path); + scanner_free_sorted(scanner); + ArrayList* pending = (ArrayList*)scanner->pending_dirs; + if (pending) { + for (int i = 0; i < pending->size; i++) + dir_entry_destroy(pending->items[i]); + array_list_delete(pending); + } array_list_delete(scanner->filter_nodes); array_list_delete(scanner->dirs_batch); queue_destroy(scanner->directories); @@ -975,7 +1114,7 @@ static int open_next_directory(DirectoryScanner* scanner) { scanner->current_path = NULL; while (!queue_is_empty(scanner->directories)) { - DirEntry* de = (DirEntry*)queue_dequeue(scanner->directories); + DirEntry* de = (DirEntry*)queue_pop(scanner->directories); scanner->current_path = de->path; scanner->current_depth = de->depth; /* The seed directory inherits the scanner's configured context (the root @@ -1060,6 +1199,14 @@ static int open_next_directory(DirectoryScanner* scanner) { scanner->failed = true; return -1; } + /* Buffer and sort this directory's entries in rsync's flist order. */ + if (scanner_buffer_current_directory(scanner) != 0) { + closedir(scanner->current_dir); + scanner->current_dir = NULL; + free(scanner->current_path); + scanner->current_path = NULL; + return -1; + } return 1; } return 0; @@ -1415,8 +1562,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { break; } - const struct dirent* entry = readdir(scanner->current_dir); - if (entry == NULL) { + if (scanner->sorted_index >= scanner->sorted_count) { /* The directory is exhausted: if nothing was transferred or descended from it, recreate it at the destination as an explicit entry. */ if (scanner->options.emit_empty_dirs && !scanner->current_dir_produced && @@ -1425,10 +1571,12 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { if (!scanner_emit_empty_dir(scanner, chunk_data)) scanner->failed = true; } + scanner_push_pending_dirs(scanner); closedir(scanner->current_dir); scanner->current_dir = NULL; free(scanner->current_path); scanner->current_path = NULL; + scanner_free_sorted(scanner); if (scanner->failed) { array_list_delete(chunk_data); return NULL; @@ -1436,26 +1584,15 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { continue; } - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; + SortedEntry* sorted = &((SortedEntry*)scanner->sorted_entries)[scanner->sorted_index++]; + const char* name = sorted->name; + ScannerEntry* inspected = &sorted->entry; + int inspection = sorted->inspection; - ScannerEntry inspected; - char* link_rel = child_rel_path(scanner->current_rel, entry->d_name); - if (!link_rel) { - scanner->failed = true; - break; - } - int inspection = scanner_inspect_entry(&scanner->options, scanner->current_path, link_rel, - entry->d_name, &inspected); - free(link_rel); - if (inspection < 0) { - scanner->failed = true; - break; - } if (inspection == 0) { /* A dereferenced symlink with no referent is a partial-transfer error (rsync exit 23): record it as a non-fatal scan I/O error. */ - if (inspected.referent_error) + if (inspected->referent_error) scanner->io_error = true; /* A user-selection exclude protects its destination mirror from --delete unless --delete-excluded; a size prune is always protected. Other @@ -1463,23 +1600,23 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { --files-from the protected prefix must be the entry's bare relative wire path, not its source path (which would not match the destination layout and would leave the mirror deletable). */ - if (inspected.excluded) { + if (inspected->excluded) { char* protected_path; if (scanner->relative_mode) { - protected_path = child_rel_path(scanner->current_rel, entry->d_name); + protected_path = child_rel_path(scanner->current_rel, name); } else if (scanner->options.relative_prefix) { - char* relc = child_rel_path(scanner->current_rel, entry->d_name); + char* relc = child_rel_path(scanner->current_rel, name); protected_path = relc ? scanner_prefix_send_path(scanner->options.relative_prefix, relc) : NULL; free(relc); } else { - protected_path = path_cat(scanner->current_path, entry->d_name); + protected_path = path_cat(scanner->current_path, name); } if (!protected_path) { scanner->failed = true; break; } - if (inspected.size_excluded) + if (inspected->size_excluded) scanner_record_size_skipped(scanner, protected_path); else scanner_record_excluded(scanner, protected_path); @@ -1487,23 +1624,22 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { } continue; } - char* cur_path = inspected.path; - struct stat stats = inspected.stats; + char* cur_path = inspected->path; + struct stat stats = inspected->stats; /* --files-from allow-set and the filter layer apply to files and to * directories (an excluded directory is not descended into). */ - bool is_dir = inspected.is_directory; - char* rel = child_rel_path(scanner->current_rel, entry->d_name); + bool is_dir = inspected->is_directory; + char* rel = child_rel_path(scanner->current_rel, name); if (!rel) { - free(cur_path); scanner->failed = true; break; } bool protect = false; bool passes_selection = entry_passes_selection( - scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, - entry->d_name, is_dir, scanner->options.per_dir_filters, - scanner->options.exclude_per_dir_filter_files, &protect); + scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, name, + is_dir, scanner->options.per_dir_filters, scanner->options.exclude_per_dir_filter_files, + &protect); /* A sender-side hide leaves the entry out of the transfer; an independent receiver-side protect rule keeps a transferred entry's destination mirror from being deleted. Both are recorded in the same protection set. */ @@ -1525,7 +1661,6 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { char* wrel = scanner_prefix_send_path(scanner->options.relative_prefix, rel); if (!wrel) { free(rel); - free(cur_path); scanner->failed = true; break; } @@ -1542,13 +1677,11 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { char* rel_copy = needs_rel ? str_dup(rel) : NULL; free(rel); if (rel_copy == NULL && needs_rel) { - free(cur_path); scanner->failed = true; break; } if (!passes_selection) { free(rel_copy); - free(cur_path); continue; } @@ -1563,12 +1696,10 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { File* mount = scanner_build_dir_file(cur_path, &stats, &scanner->options); if (mount == NULL || !array_list_add(chunk_data, mount)) { file_destroy(mount); - free(cur_path); scanner->failed = true; break; } scanner->current_dir_produced = true; - free(cur_path); continue; } /* --list-only: list directory entries too (rsync prints them), even @@ -1577,7 +1708,6 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { File* dir = scanner_build_dir_file(cur_path, &stats, &scanner->options); if (dir == NULL || !array_list_add(chunk_data, dir)) { file_destroy(dir); - free(cur_path); scanner->failed = true; break; } @@ -1586,32 +1716,29 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { int next_depth = scanner->current_depth + 1; if (scanner->options.max_depth <= 0 || next_depth < scanner->options.max_depth) { DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node); - if (!de || !queue_enqueue(scanner->directories, de)) { + if (!de || !array_list_add((ArrayList*)scanner->pending_dirs, de)) { dir_entry_destroy(de); scanner->failed = true; } } - free(cur_path); } else { if (scanner->options.max_depth > 0 && scanner->current_depth + 1 > scanner->options.max_depth) { free(rel_copy); - free(cur_path); continue; } File* file = file_create(cur_path); - free(cur_path); if (file == NULL) { free(rel_copy); - free(inspected.link_target); - inspected.link_target = NULL; + free(inspected->link_target); + inspected->link_target = NULL; scanner->failed = true; continue; } - if (inspected.is_symlink) { + if (inspected->is_symlink) { file->is_symlink = true; - file->symlink_target = inspected.link_target; - inspected.link_target = NULL; + file->symlink_target = inspected->link_target; + inspected->link_target = NULL; } else { file->data->size = stats.st_size; } diff --git a/src/client/scanner.h b/src/client/scanner.h index 18f61a6..55fc3bb 100644 --- a/src/client/scanner.h +++ b/src/client/scanner.h @@ -10,6 +10,7 @@ #include "stop_condition.h" #include #include +#include #include #include #include @@ -188,6 +189,17 @@ typedef struct { int current_depth; dev_t root_dev; bool failed; + /* rsync-order traversal: each opened directory's entries are inspected once + and buffered (an internal SortedEntry[] owned here) sorted as rsync's flist + orders them -- non-directories ascending, then directories ascending. The + entries are walked in order and child directories are collected in + `pending_dirs` (an ArrayList of DirEntry*, owned here) and pushed onto the + LIFO `directories` stack in reverse at directory exhaustion, so the emitted + stream is depth-first like rsync. `sorted_*` are reset per directory. */ + void* sorted_entries; + size_t sorted_count; + size_t sorted_index; + void* pending_dirs; /* Recursive scan: whether the open directory yielded any transferred or descended entry. When it did not, closing it emits a directory entry so the empty source directory is recreated at the destination (rsync diff --git a/src/shared/queue.c b/src/shared/queue.c index e4e5f09..3ebac08 100644 --- a/src/shared/queue.c +++ b/src/shared/queue.c @@ -139,6 +139,23 @@ void* queue_dequeue(Queue* queue) { return item; } +bool queue_push(Queue* queue, void* item) { + return queue_enqueue(queue, item); +} + +void* queue_pop(Queue* queue) { + if (queue == NULL || queue_is_empty(queue)) { + log_perror("ERROR: Could not pop from null or empty queue."); + return NULL; + } + + queue->rear = (queue->rear - 1 + queue->capacity) % queue->capacity; + void* item = queue->items[queue->rear]; + queue->items[queue->rear] = NULL; + queue->size--; + return item; +} + void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty, cnd_t* condition_not_full, const bool* other_thread_done) { mtx_lock(mutex); diff --git a/src/shared/queue.h b/src/shared/queue.h index 7a8bd98..3fddd9d 100644 --- a/src/shared/queue.h +++ b/src/shared/queue.h @@ -28,4 +28,11 @@ void* queue_dequeue(Queue* queue); void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty, cnd_t* condition_not_full, const bool* other_thread_done); +/* LIFO stack operations over the same ring buffer. queue_push() is the enqueue + primitive; queue_pop() removes from the rear, so a sequence of pushes is + returned in reverse order. Used by the sequential scanner's depth-first + traversal. */ +bool queue_push(Queue* queue, void* item); +void* queue_pop(Queue* queue); + #endif diff --git a/tests/integration/test_parity_order.py b/tests/integration/test_parity_order.py new file mode 100644 index 0000000..44dc16b --- /dev/null +++ b/tests/integration/test_parity_order.py @@ -0,0 +1,135 @@ +"""Differential rsync-parity coverage for FastSync's transfer/delete ORDER. + +rsync walks a source tree in its sorted flist order: within each directory the +non-directories come first (ascending name), then the subdirectories (ascending +name), each subdirectory immediately followed by its own subtree (depth-first). +The sequential scanner now reproduces that order, which makes both the +``--info=name`` stream and the ``--delete-during`` deletion sequence match real +``rsync 3.4.1`` exactly. ``--threads`` has no rsync analogue and is unordered. + +Every test skips cleanly when rsync is absent. +""" +import os +import shutil +import subprocess +import sys + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import ( # noqa: E402 + TEST_DATA_DIR, + ServerManager, + clean_dir, + get_dest_received_dir, + run_client, +) + +RSYNC = shutil.which("rsync") +requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed") + +MTIME = 1_500_000_000 + +_TREE = { + "a.txt": b"a\n", + "b.txt": b"b\n", + "z.txt": b"z\n", + "a_dir/f.txt": b"f\n", + "a_dir/deep/g.txt": b"g\n", + "m_dir/h.txt": b"h\n", + "Z_dir/i.txt": b"i\n", +} + + +def _write(path, data): + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "wb") as fh: + fh.write(data) + os.utime(path, (MTIME, MTIME)) + + +def _rsync(args): + env = dict(os.environ, LC_ALL="C") + return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120) + + +def _deleting(text): + out = [] + for line in text.splitlines(): + stripped = line.strip() + if stripped.startswith("*deleting") or stripped.startswith("deleting"): + out.append(stripped.split()[-1]) + return out + + +class TestTransferOrderParity: + @requires_rsync + def test_info_name_file_order_matches_rsync(self, shared_server): + source = os.path.join(TEST_DATA_DIR, "order_name_src") + clean_dir(source) + for rel, data in _TREE.items(): + _write(os.path.join(source, rel), data) + + rdst = os.path.join(TEST_DATA_DIR, "order_name_rdst") + clean_dir(rdst) + rs = _rsync(["-a", "--info=name", source + "/", rdst + "/"]) + assert rs.returncode == 0, rs.stderr + # rsync also names the directories (trailing '/'); FastSync names the + # transferred entries. Compare the file/symlink sequence, which is what + # the traversal order determines. + rsync_files = [l for l in rs.stdout.splitlines() if l.strip() and not l.endswith("/")] + + fdst = os.path.join(TEST_DATA_DIR, "order_name_fdst") + clean_dir(fdst) + result, _ = run_client(source, fdst, flags=["-a", "--info=name"], + port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + fsync_files = [ + l for l in result.stdout.splitlines() + if l.strip() and l.strip() != "./" and not l.startswith("sending") + ] + assert fsync_files == rsync_files, ( + f"transfer order differs\nrsync={rsync_files}\nfastsync={fsync_files}") + + +class TestDeleteOrderParity: + _EXTRA = { + "a_extra.txt": b"a\n", + "z_extra.txt": b"z\n", + "a_extra_dir/f": b"f\n", + "z_extra_dir/f": b"f\n", + "a_extra_dir/sub/g": b"g\n", + } + + @requires_rsync + def test_delete_during_deletion_order_matches_rsync(self): + source = os.path.join(TEST_DATA_DIR, "order_del_src") + clean_dir(source) + _write(os.path.join(source, "keep.txt"), b"k\n") + _write(os.path.join(source, "keepdir", "x.txt"), b"x\n") + _write(os.path.join(source, "keep2", "y.txt"), b"y\n") + + rdst = os.path.join(TEST_DATA_DIR, "order_del_rdst") + clean_dir(rdst) + for rel, data in self._EXTRA.items(): + _write(os.path.join(rdst, rel), data) + rs = _rsync(["-a", "--delete-during", "--info=del", source + "/", rdst + "/"]) + assert rs.returncode == 0, rs.stderr + + fdst = os.path.join(TEST_DATA_DIR, "order_del_fdst") + clean_dir(fdst) + received = get_dest_received_dir(fdst, source) + for rel, data in self._EXTRA.items(): + _write(os.path.join(received, rel), data) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, fdst, flags=["-a", "--delete-during", "--info=del"], + port=server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + + rsync_order = _deleting(rs.stdout) + fsync_order = _deleting(result.stdout) + assert sorted(fsync_order) == sorted(rsync_order), ( + f"deleted set differs\nrsync={rsync_order}\nfastsync={fsync_order}") + assert fsync_order == rsync_order, ( + f"deletion order differs\nrsync={rsync_order}\nfastsync={fsync_order}") -- 2.54.0 From b235721f8b89f97ce6cb1f89a080dfe3acf82ce1 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 14:22:22 +0200 Subject: [PATCH 06/68] delete: rsync-exact abort boundary and -d per-directory plans Transmit the complete --delete-during/--delete-delay per-directory plan set before the first data frame, so a mid-transfer abort has already applied every planned removal like rsync's generator; completed runs are unchanged. Route -d/--dirs through the same per-directory plans: the generator records only directories whose direct children it enumerated, so extras directly inside a listed directory are removed while an untraversed subdirectory's mirror is shielded (rsync's -d DIR/ --delete). Also shields a -x mount point's untraversed destination content. --- src/client/client_send.c | 93 ++---- src/client/scanner.c | 11 + src/shared/delete_plan.c | 11 + src/shared/delete_plan.h | 14 +- .../test_delete_boundary_parity.py | 291 ++++++++++++++++++ 5 files changed, 352 insertions(+), 68 deletions(-) create mode 100644 tests/integration/test_delete_boundary_parity.py diff --git a/src/client/client_send.c b/src/client/client_send.c index 767ee21..c775bc2 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1855,25 +1855,6 @@ static bool scan_paths_only(const Config* config, const ScannerOptions* options, return ok; } -/* Transmit any not-yet-sent per-directory delete plan needed by the entries in - * `chunk` (ancestors root-first, then the entry's own directory for --dirs - * entries) before its data frames go out, so --delete-during/--delete-delay - * clear a directory's extras (and any type conflict) before the directory's - * first write. */ -static int send_chunk_delete_plans(Client* client, DeletePlanSender* plans, const Chunk* chunk) { - if (!plans) - return 0; - for (int i = 0; i < chunk->element_count; i++) { - File* f = chunk->items[i]; - if (!f) - continue; - if (delete_plan_send_for_path(client->file_descriptor, plans, file_wire_path(f), f->is_dir) != - 0) - return -1; - } - return 0; -} - static int incremental_check(Client* client, File* file, const Config* config, DeltaSignature** out_sig, unsigned long long* resume_offset) { *out_sig = NULL; @@ -2753,9 +2734,12 @@ static int send_chunks_multithreaded(void* pipeline_context) { return thrd_error; } } else if (context->delete_plans) { - /* --delete-during/--delete-delay: transmit the receive root's plan before - any data, exactly like rsync's first generator directory. */ - if (delete_plan_send_root(client->file_descriptor, context->delete_plans) != 0) { + /* --delete-during/--delete-delay: transmit the COMPLETE per-directory plan + set before any data, so a mid-transfer abort has already applied every + planned removal exactly like rsync's generator (which runs ahead of its + throttled sender). A completed run is unaffected. */ + if (delete_plan_send_all(client->file_descriptor, context->delete_plans, context->plan_dirs) != + 0) { pipeline_cancel(context); disconnect_transfer_client(client); mark_sender_done(context); @@ -2802,15 +2786,6 @@ static int send_chunks_multithreaded(void* pipeline_context) { } break; } - if (send_chunk_delete_plans(client, context->delete_plans, current_chunk) != 0) { - log_message(LOG_LEVEL_ERROR, "unexpected error while sending delete plan"); - chunk_destroy(current_chunk); - pipeline_cancel(context); - disconnect_transfer_client(client); - mark_sender_done(context); - protocol_session_unbind(); - return thrd_error; - } if (send_chunk_with_removal(client, current_chunk, context->config, context->remove_source_files, &context->stats) != 0) { log_message(LOG_LEVEL_ERROR, "unexpected error while sending chunk"); @@ -2893,13 +2868,6 @@ static int send_chunks_multithreaded(void* pipeline_context) { NULL) != 0) goto send_fail; } - /* Emit the plans for source directories the data stream never triggered - (empty directories): their extras are still cleared while the directory - itself is kept. */ - if (!context->scan_stopped_early && context->delete_plans && context->plan_dirs && - delete_plan_send_remaining(client->file_descriptor, context->delete_plans, - context->plan_dirs) != 0) - goto send_fail; /* P7 Wave D: transmit the captured directory times last. The scanner thread (and all parallel workers) has been joined before scanner_done was set, so the list is complete and race-free; on an early stop the list may be @@ -3232,10 +3200,12 @@ int send_files(Config* config) { /* Traversed source directories for the per-directory delete keep set. */ ArrayList* plan_dirs = NULL; bool delete_early = config->use_delete && config_delete_timing_early(config); - /* -d/--dirs does not recurse, so a per-directory plan would carry no child - information and could delete the contents of an untraversed directory; - fall back to the whole-tree end-of-transfer commit for that mode. */ - bool delete_per_dir = config->use_delete && config_delete_timing_per_dir(config) && !config->dirs; + /* --delete-during/--delete-delay use per-directory plans for every transfer + shape. For -d/--dirs the generator records only the directories whose + direct children it actually enumerated, so the plan removes extras directly + inside a listed directory while an untraversed (kept) subdirectory is + shielded -- rsync's `-d DIR/ --delete`. */ + bool delete_per_dir = config->use_delete && config_delete_timing_per_dir(config); bool send_failed = false; bool had_scan_io = false; unsigned long long per_dir_non_dir_count = 0; @@ -3287,12 +3257,12 @@ int send_files(Config* config) { prepared.options.synced_dirs = synced_dirs; } } - /* The late-timing modes (--delete-after/--delete-commit and a plain --delete - that fell back from per-dir mode because of -d/--dirs) build the manifest + /* The late-timing modes (--delete-after/--delete-commit) build the manifest while streaming and send it after the last data frame. --delete-before - sends a whole-tree keep-set up front; --delete-during/--delete-delay build a - per-directory plan set up front (paths only) and stream the plans alongside - the data, so no manifest is kept during the data pass. */ + sends a whole-tree keep-set up front; --delete-during/--delete-delay build + the complete per-directory plan set up front (paths only) and transmit it + all before the first data frame, so a mid-transfer abort has already + applied every planned removal. */ if (delete_early) { /* Pass 1: collect the complete keep-set (paths only, no data loaded) and transmit it now, before any file data. The receiver removes extras and @@ -3335,9 +3305,10 @@ int send_files(Config* config) { goto send_fail; } else if (delete_per_dir) { /* --delete-during/--delete-delay: build one plan per source directory from a - path-only pre-scan and transmit the root plan now, before any data, so the - receive root's extras are handled exactly like rsync's first generator - directory. The remaining plans are streamed with the data below. */ + path-only pre-scan and transmit the COMPLETE plan set now, before any data, + so every planned removal has already been applied when a later transfer + phase fails -- exactly like rsync's generator, whose deletion list runs + ahead of its throttled sender. A completed run is unaffected. */ plan_sender = delete_plan_sender_create(); plan_dirs = array_list_create(free); if (!plan_sender || !plan_dirs) @@ -3368,7 +3339,7 @@ int send_files(Config* config) { plan_dirs = NULL; skip_delete = true; } else { - plans_ok = delete_plan_send_root(client->file_descriptor, plan_sender) == 0; + plans_ok = delete_plan_send_all(client->file_descriptor, plan_sender, plan_dirs) == 0; } } prepared.options.excluded_paths = NULL; @@ -3455,11 +3426,6 @@ int send_files(Config* config) { goto send_fail; } } - if (send_chunk_delete_plans(client, plan_sender, current_chunk) != 0) { - chunk_destroy(current_chunk); - send_failed = true; - break; - } if (send_chunk_with_removal(client, current_chunk, config, remove_sources, &transfer_stats) != 0) { log_message(LOG_LEVEL_ERROR, "Failed to send chunk"); @@ -3542,12 +3508,6 @@ int send_files(Config* config) { } } } - /* Emit the plans for any source directories the data stream never triggered - (an empty directory has no file frame). Sending them now still clears that - directory's destination extras while keeping the directory itself. */ - if (!scan_stopped_early && plan_sender && plan_dirs && - delete_plan_send_remaining(client->file_descriptor, plan_sender, plan_dirs) != 0) - goto send_fail; /* P7 Wave D: every directory has now been traversed (or the scan stopped early), so transmit the captured directory times last. The receiver defers applying them until after its own deletion/publication phase. */ @@ -3714,10 +3674,11 @@ int send_files_multithreaded(Config** config_ptr) { return 1; } } - /* -d/--dirs does not recurse, so a per-directory plan would carry no child - information and could delete the contents of an untraversed directory; - fall back to the whole-tree end-of-transfer commit for that mode. */ - bool per_dir = config_delete_timing_per_dir(config) && !config->dirs; + /* --delete-during/--delete-delay use per-directory plans for every transfer + shape. The -d/--dirs generator records only the directories whose direct + children it enumerated, so extras directly inside a listed directory are + removed while an untraversed (kept) subdirectory is shielded. */ + bool per_dir = config_delete_timing_per_dir(config); if (config_delete_timing_early(config) || per_dir) { /* --delete-before / --delete-during / --delete-delay: build the keep-set (paths only, nothing loaded or sent) up front so the sender thread can diff --git a/src/client/scanner.c b/src/client/scanner.c index b768001..99fec2c 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -1451,6 +1451,17 @@ static File* dirs_next_file(DirectoryScanner* scanner) { scanner->dirs_root_emitted = true; if (scanner->options.prune_empty_dirs && dirs_source_dir_is_empty(scanner->root_path)) return NULL; + /* The listed directory's direct children are about to be enumerated, so + its destination mirror is a synchronized directory: record it for the + per-directory delete plan. The plan keeps the enumerated children and + shields untraversed subdirectories, so --delete-during removes extras + directly inside the listed directory without descending into a kept + (but untraversed) child -- exactly rsync's `-d DIR/ --delete`. */ + if (!scanner_record_synced_dir(&scanner->options, scanner->root_path, "", + scanner->relative_mode)) { + scanner->failed = true; + return NULL; + } scanner->current_dir = opendir(scanner->root_path); if (!scanner->current_dir) { scanner->io_error = true; diff --git a/src/shared/delete_plan.c b/src/shared/delete_plan.c index 644fb4c..ae59bc9 100644 --- a/src/shared/delete_plan.c +++ b/src/shared/delete_plan.c @@ -432,6 +432,17 @@ int delete_plan_send_remaining(int fd, DeletePlanSender* sender, const ArrayList return 0; } +int delete_plan_send_all(int fd, DeletePlanSender* sender, const ArrayList* dirs) { + if (!sender) + return -1; + /* Root first: this also transmits the one-shot per-run config block on its + own carrier frame (see send_config_only), so it reaches the receiver even + when the scope permits no directory plan at all. */ + if (delete_plan_send_root(fd, sender) != 0) + return -1; + return delete_plan_send_remaining(fd, sender, dirs); +} + /* ------------------------------------------------------------------ */ /* Receiver: delete session */ /* ------------------------------------------------------------------ */ diff --git a/src/shared/delete_plan.h b/src/shared/delete_plan.h index 7e36eb4..93929c8 100644 --- a/src/shared/delete_plan.h +++ b/src/shared/delete_plan.h @@ -61,9 +61,19 @@ int delete_plan_send_root(int fd, DeletePlanSender* sender); * for `path` itself; already-sent plans are skipped. */ int delete_plan_send_for_path(int fd, DeletePlanSender* sender, const char* path, bool is_dir); /* Send the plan for every directory in `dirs` that has not been transmitted - * yet. Called after the data stream so an empty source directory's plan still - * clears its destination extras even though no file frame triggered it. */ + * yet. */ int delete_plan_send_remaining(int fd, DeletePlanSender* sender, const ArrayList* dirs); +/* Transmit the COMPLETE per-directory plan set in one pass, before any data + * frame: the root plan (with the one-shot per-run config block on its carrier + * frame) followed by every directory in `dirs`. Because the whole plan set is + * known from the path-only pre-scan, sending it all up front means a + * mid-transfer abort has already applied every planned removal, matching + * rsync's generator (which runs ahead of its throttled sender). A completed + * run is unaffected. `dirs` is the set of directories whose direct children + * were enumerated (the scanner's plan_dirs sink), so a merely listed but + * untraversed directory never gets a plan and its mirror is left intact. + * Returns -1 on I/O error. */ +int delete_plan_send_all(int fd, DeletePlanSender* sender, const ArrayList* dirs); /* ---- Receiver: delete session ---- */ diff --git a/tests/integration/test_delete_boundary_parity.py b/tests/integration/test_delete_boundary_parity.py new file mode 100644 index 0000000..9caa635 --- /dev/null +++ b/tests/integration/test_delete_boundary_parity.py @@ -0,0 +1,291 @@ +"""Differential coverage for the delete-timing ABORT BOUNDARY (A9/A10). + +rsync's generator runs ahead of its throttled sender, so on a mid-transfer abort +it has already removed every extra it planned. FastSync now transmits the +COMPLETE per-directory plan set before the first data frame, so an abort has the +same effect. Before that change FastSync only removed the extras of the +directories its (slower) data stream had reached, and ``-d/--dirs`` used an +end-of-transfer commit that removed nothing on abort. + +These tests abort both tools mid-transfer and assert the destination extras +removed match real ``rsync 3.4.1``. The rsync side is driven locally with +``--bwlimit`` and a small timing window (its generator's delete list is computed +long before the throttled payload finishes); the FastSync side uses the +byte-deterministic slicing proxy from ``test_delete_timing_parity``. +""" +import os +import shutil +import subprocess +import sys +import time + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import ( # noqa: E402 + TEST_DATA_DIR, + ServerManager, + clean_dir, + get_dest_received_dir, + run_client, +) +from test_delete_timing_parity import _SlicingProxy # noqa: E402 + +RSYNC = shutil.which("rsync") +requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed") + +# Exceeds the 10 MiB scanner chunk, so the next directory lands in a later chunk +# (still unreached when the proxy cuts the stream). +BIG_BYTES = 16 * 1024 * 1024 +# Cut well past the (small) config + delete-plan frames and into the big payload, +# so the receiver has provably processed every plan before the abort. +MID_TRANSFER_BYTES = 256 * 1024 +PROXY_THROTTLE = 0.001 +# Throttle rsync's sender so the generator has deleted long before the payload +# finishes, then interrupt it mid-transfer. +RSYNC_BWLIMIT = 512 # KiB/s -> ~32 s for 16 MiB +RSYNC_ABORT_DELAY = 1.5 + + +def _write(path, content): + os.makedirs(os.path.dirname(path), exist_ok=True) + with open(path, "wb") as fh: + fh.write(content) + + +def _rsync_aborted(args, delay=RSYNC_ABORT_DELAY): + """Start rsync, let its generator run, then interrupt it mid-transfer.""" + env = dict(os.environ, LC_ALL="C") + proc = subprocess.Popen([RSYNC] + args, stdout=subprocess.PIPE, stderr=subprocess.PIPE, + text=True, env=env) + time.sleep(delay) + proc.terminate() + try: + proc.wait(timeout=10) + except subprocess.TimeoutExpired: + proc.kill() + proc.wait(timeout=5) + return proc + + +class TestDeleteDuringAbortBoundary: + """A9: on an abort, every planned removal has already been applied.""" + + def _seed_recursive(self, tag): + source = os.path.join(TEST_DATA_DIR, f"dab_{tag}_src") + clean_dir(source) + # ``a/keep.bin`` sorts first, so the client streams it (and the proxy + # cuts) before the data pass ever reaches ``z/deep``. + _write(os.path.join(source, "a", "keep.bin"), b"B" * BIG_BYTES) + _write(os.path.join(source, "z", "deep", "keep.txt"), b"keep\n") + return source + + @requires_rsync + def test_recursive_abort_removes_all_planned_extras(self): + # ---- FastSync: abort mid ``a/keep.bin``; ``z/deep`` is never reached. + source = self._seed_recursive("rec_fs") + dest = os.path.join(TEST_DATA_DIR, "dab_rec_fs_dst") + clean_dir(dest) + received = get_dest_received_dir(dest, source) + os.makedirs(os.path.join(received, "a"), exist_ok=True) + _write(os.path.join(received, "a", "a_extra"), b"stale\n") + os.makedirs(os.path.join(received, "z", "deep"), exist_ok=True) + _write(os.path.join(received, "z", "deep", "old_extra"), b"stale\n") + + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + proxy = _SlicingProxy(server.port, forward_limit=MID_TRANSFER_BYTES, + throttle=PROXY_THROTTLE) + result, _ = run_client(source, dest, flags=["--delete-during"], port=proxy.port) + proxy.finish() + assert result.returncode != 0, "truncated transfer reported success" + assert not os.path.exists(os.path.join(received, "a", "a_extra")) + assert not os.path.exists(os.path.join(received, "z", "deep", "old_extra")), ( + "FastSync left an extra in a directory it never reached before the abort" + ) + + # ---- rsync 3.4.1: same tree, same abort, same delete outcome. + source = self._seed_recursive("rec_rs") + rsync_dst = os.path.join(TEST_DATA_DIR, "dab_rec_rs_dst") + clean_dir(rsync_dst) + os.makedirs(os.path.join(rsync_dst, "a"), exist_ok=True) + _write(os.path.join(rsync_dst, "a", "a_extra"), b"stale\n") + os.makedirs(os.path.join(rsync_dst, "z", "deep"), exist_ok=True) + _write(os.path.join(rsync_dst, "z", "deep", "old_extra"), b"stale\n") + + proc = _rsync_aborted(["-a", "--delete-during", f"--bwlimit={RSYNC_BWLIMIT}", + source + "/", rsync_dst + "/"]) + assert proc.returncode != 0, "rsync was not actually interrupted" + assert not os.path.exists(os.path.join(rsync_dst, "a", "a_extra")) + assert not os.path.exists(os.path.join(rsync_dst, "z", "deep", "old_extra")), ( + "rsync's generator did not delete ahead of its sender" + ) + + +class TestDirsDeleteAbortBoundary: + """A10: ``-d/--dirs`` uses per-directory plans like rsync. + + The listed directory's direct extras are removed by the up-front plan while + a kept but untraversed subdirectory (and its destination content) is + shielded. + """ + + def _seed_dirs(self, tag): + source = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_src") + clean_dir(source) + _write(os.path.join(source, "big.bin"), b"B" * BIG_BYTES) + _write(os.path.join(source, "subdir", "keep.txt"), b"inner\n") + return source + + @pytest.mark.parametrize("fs_flag,rs_flag", [("--delete-during", "--delete-during"), + ("--delete", "--delete")]) + @requires_rsync + def test_dirs_abort_removes_direct_extras_only(self, fs_flag, rs_flag): + label = f"{fs_flag.lstrip('-')}_{rs_flag.lstrip('-')}" + # ---- FastSync: ``-d`` lists the immediate children; big.bin streams and + # the abort lands mid-payload. + source = self._seed_dirs(f"dirs_{label}_fs") + dest = os.path.join(TEST_DATA_DIR, f"ddb_{label}_fs_dst") + clean_dir(dest) + received = get_dest_received_dir(dest, source) + _write(os.path.join(received, "old_extra"), b"stale\n") + os.makedirs(os.path.join(received, "subdir"), exist_ok=True) + _write(os.path.join(received, "subdir", "stale.txt"), b"stale inner\n") + + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + proxy = _SlicingProxy(server.port, forward_limit=MID_TRANSFER_BYTES, + throttle=PROXY_THROTTLE) + result, _ = run_client(source + "/", dest, flags=["-d", fs_flag], port=proxy.port) + proxy.finish() + assert result.returncode != 0, f"{fs_flag}: truncated transfer reported success" + assert not os.path.exists(os.path.join(received, "old_extra")), ( + f"{fs_flag}: the listed directory's direct extra survived the abort" + ) + assert os.path.exists(os.path.join(received, "subdir", "stale.txt")), ( + f"{fs_flag}: descended into a kept, untraversed subdirectory" + ) + + # ---- rsync 3.4.1: same shape and same abort. + source = self._seed_dirs(f"dirs_{label}_rs") + rsync_dst = os.path.join(TEST_DATA_DIR, f"ddb_{label}_rs_dst") + clean_dir(rsync_dst) + _write(os.path.join(rsync_dst, "old_extra"), b"stale\n") + os.makedirs(os.path.join(rsync_dst, "subdir"), exist_ok=True) + _write(os.path.join(rsync_dst, "subdir", "stale.txt"), b"stale inner\n") + + proc = _rsync_aborted(["-d", rs_flag, f"--bwlimit={RSYNC_BWLIMIT}", + source + "/", rsync_dst + "/"]) + assert proc.returncode != 0, "rsync was not actually interrupted" + assert not os.path.exists(os.path.join(rsync_dst, "old_extra")), ( + f"rsync {rs_flag}: the listed directory's direct extra survived the abort" + ) + assert os.path.exists(os.path.join(rsync_dst, "subdir", "stale.txt")), ( + f"rsync {rs_flag}: descended into a kept, untraversed subdirectory" + ) + + +def _tree(root): + out = [] + for dirpath, dirs, files in os.walk(root): + for name in dirs: + out.append(os.path.relpath(os.path.join(dirpath, name), root)) + for name in files: + out.append(os.path.relpath(os.path.join(dirpath, name), root)) + return sorted(out) + + +class TestDirsDeleteFinalStateParity: + """A10 completed run: ``-d DIR/ --delete`` (during default) and + ``--delete-during`` match rsync's final tree, including a kept but + untraversed subdirectory whose destination content survives.""" + + @pytest.mark.parametrize("flag", ["--delete", "--delete-during"]) + @requires_rsync + def test_dirs_final_state_matches_rsync(self, flag): + source = os.path.join(TEST_DATA_DIR, f"ddf_{flag.lstrip('-')}_src") + clean_dir(source) + _write(os.path.join(source, "keep.txt"), b"new keep\n") + _write(os.path.join(source, "subdir", "inner.txt"), b"inner\n") + + def seed_dest(root): + clean_dir(root) + _write(os.path.join(root, "keep.txt"), b"old keep\n") + _write(os.path.join(root, "extra.txt"), b"extra\n") + _write(os.path.join(root, "extrasub", "ex.txt"), b"extra sub\n") + _write(os.path.join(root, "subdir", "stale.txt"), b"stale inner\n") + + rsync_dst = os.path.join(TEST_DATA_DIR, f"ddf_{flag.lstrip('-')}_rs_dst") + seed_dest(rsync_dst) + env = dict(os.environ, LC_ALL="C") + rsync_result = subprocess.run( + [RSYNC, "-d", flag, source + "/", rsync_dst + "/"], + capture_output=True, text=True, env=env, timeout=120) + assert rsync_result.returncode == 0, rsync_result.stderr + rsync_tree = _tree(rsync_dst) + + dest = os.path.join(TEST_DATA_DIR, f"ddf_{flag.lstrip('-')}_fs_dst") + clean_dir(dest) + received = get_dest_received_dir(dest, source) + seed_dest(received) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source + "/", dest, flags=["-d", flag], port=server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + fastsync_tree = _tree(received) + assert fastsync_tree == rsync_tree, ( + f"-d {flag}: fastsync tree {fastsync_tree} != rsync tree {rsync_tree}") + + +class TestOneFileSystemDeleteParity: + """A9 side effect: the per-directory plan is now emitted only for directories + whose children were enumerated, so a ``-x`` mount-point directory that is + emitted but never traversed is shielded -- its destination content survives, + exactly as rsync keeps a non-descended mount point under ``--delete``.""" + + @requires_rsync + def test_mountpoint_content_survives_delete(self): + local = os.stat(".") + shm = "/dev/shm" + if not os.path.isdir(shm) or os.stat(shm).st_dev == local.st_dev: + pytest.skip("no cross-device filesystem available") + probe = os.path.join(shm, f"fastsync_dofs_{os.getpid()}") + clean_dir(probe) + _write(os.path.join(probe, "inside.txt"), b"cross\n") + try: + source = os.path.join(TEST_DATA_DIR, "dofs_src") + clean_dir(source) + _write(os.path.join(source, "keep.txt"), b"keep\n") + os.symlink(probe, os.path.join(source, "nested_link")) + + def seed_dest(root): + clean_dir(root) + _write(os.path.join(root, "keep.txt"), b"old\n") + _write(os.path.join(root, "nested_link", "stale.txt"), b"stale\n") + + rsync_dst = os.path.join(TEST_DATA_DIR, "dofs_rs_dst") + seed_dest(rsync_dst) + env = dict(os.environ, LC_ALL="C") + rsync_result = subprocess.run( + [RSYNC, "-a", "--copy-links", "-x", "--delete-during", + source + "/", rsync_dst + "/"], + capture_output=True, text=True, env=env, timeout=120) + assert rsync_result.returncode == 0, rsync_result.stderr + assert os.path.exists(os.path.join(rsync_dst, "nested_link", "stale.txt")), ( + "rsync unexpectedly descended into the mount point") + + dest = os.path.join(TEST_DATA_DIR, "dofs_fs_dst") + clean_dir(dest) + received = get_dest_received_dir(dest, source) + seed_dest(received) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, + flags=["-a", "--copy-links", "-x", "--delete-during"], + port=server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + assert os.path.exists(os.path.join(received, "nested_link", "stale.txt")), ( + "FastSync descended into a non-traversed mount point under --delete") + finally: + clean_dir(probe) + -- 2.54.0 From ff261bc38ab4c833fc455dc956b2dba7d04d8c09 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 14:55:47 +0200 Subject: [PATCH 07/68] test: extend rsync order parity to dry-run and delete-delay --- tests/integration/test_parity_order.py | 32 +++++++++++++++++++------- 1 file changed, 24 insertions(+), 8 deletions(-) diff --git a/tests/integration/test_parity_order.py b/tests/integration/test_parity_order.py index 44dc16b..5b36757 100644 --- a/tests/integration/test_parity_order.py +++ b/tests/integration/test_parity_order.py @@ -101,35 +101,51 @@ class TestDeleteOrderParity: "a_extra_dir/sub/g": b"g\n", } - @requires_rsync - def test_delete_during_deletion_order_matches_rsync(self): + def _seed_source(self): source = os.path.join(TEST_DATA_DIR, "order_del_src") clean_dir(source) _write(os.path.join(source, "keep.txt"), b"k\n") _write(os.path.join(source, "keepdir", "x.txt"), b"x\n") _write(os.path.join(source, "keep2", "y.txt"), b"y\n") + return source - rdst = os.path.join(TEST_DATA_DIR, "order_del_rdst") + def _assert_order(self, timing, dry_run=False): + source = self._seed_source() + rdst = os.path.join(TEST_DATA_DIR, f"order_{timing}_rdst") clean_dir(rdst) for rel, data in self._EXTRA.items(): _write(os.path.join(rdst, rel), data) - rs = _rsync(["-a", "--delete-during", "--info=del", source + "/", rdst + "/"]) + rs_flags = ["-a", "-n"] if dry_run else ["-a"] + rs = _rsync(rs_flags + [timing, "--info=del", source + "/", rdst + "/"]) assert rs.returncode == 0, rs.stderr - fdst = os.path.join(TEST_DATA_DIR, "order_del_fdst") + fdst = os.path.join(TEST_DATA_DIR, f"order_{timing}_fdst") clean_dir(fdst) received = get_dest_received_dir(fdst, source) for rel, data in self._EXTRA.items(): _write(os.path.join(received, rel), data) + fs_flags = ["-a", "-n"] if dry_run else ["-a"] with ServerManager() as server: server.start(extra_args=["--allow-delete"]) - result, _ = run_client(source, fdst, flags=["-a", "--delete-during", "--info=del"], + result, _ = run_client(source, fdst, flags=fs_flags + [timing, "--info=del"], port=server.port) assert result.returncode == 0, (result.stderr or result.stdout)[:300] rsync_order = _deleting(rs.stdout) fsync_order = _deleting(result.stdout) assert sorted(fsync_order) == sorted(rsync_order), ( - f"deleted set differs\nrsync={rsync_order}\nfastsync={fsync_order}") + f"{timing} deleted set differs\nrsync={rsync_order}\nfastsync={fsync_order}") assert fsync_order == rsync_order, ( - f"deletion order differs\nrsync={rsync_order}\nfastsync={fsync_order}") + f"{timing} deletion order differs\nrsync={rsync_order}\nfastsync={fsync_order}") + + @requires_rsync + def test_delete_during_deletion_order_matches_rsync(self): + self._assert_order("--delete-during") + + @requires_rsync + def test_delete_delay_deletion_order_matches_rsync(self): + self._assert_order("--delete-delay") + + @requires_rsync + def test_dry_run_delete_order_matches_rsync(self): + self._assert_order("--delete", dry_run=True) -- 2.54.0 From 00829fd265506c040d195b5a8fe0bf39a9cbcab9 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 14:55:47 +0200 Subject: [PATCH 08/68] parity: --info=mount/stats, --stats dir breakdown, --debug categories --info=mount now prints rsync's mount-point skip line (matching rsync 3.4.1, which emits it for repeated -xx and drops the mount-point dir); --info=stats enables the same block as --stats; -x is repeatable. --stats counts traversed directories for the Number of files breakdown even when no directory metadata is captured (-r without -t/-p). --debug enables real output for flist/del/hash/deltasum/recv/filter/send at their natural FastSync events (synthetic categories stay inert). --stats and --debug rows keep their documented residual status. --- RSYNC_COMPAT.md | 6 +- src/client/client_cli.c | 51 +++-- src/client/client_send.c | 50 ++++- src/client/scanner.c | 65 ++++++- src/client/scanner.h | 15 +- src/shared/config.c | 2 +- src/shared/config.h | 4 +- src/shared/log.h | 20 +- src/shared/multiprocessing.c | 1 + src/shared/multiprocessing.h | 4 + tests/integration/test_output_parity.py | 40 ++++ tests/integration/test_parity_debug.py | 105 +++++++++++ .../test_parity_info_mount_stats.py | 174 ++++++++++++++++++ tests/test_client_cli.c | 17 +- 14 files changed, 519 insertions(+), 35 deletions(-) create mode 100644 tests/integration/test_parity_debug.py create mode 100644 tests/integration/test_parity_info_mount_stats.py diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 8701c30..22df531 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -85,8 +85,8 @@ Every one of those has an entry below with its remaining caveats. | `-q`, `--quiet` | Suppress non-error messages | ✅ Parity | Suppresses client output while preserving errors | | `--help` | Show help | ✅ Parity | Prints usage and exits. A lone `-h` with no other transfer arguments also prints help (protocol 2.26.0), matching the rsync idiom; `-h` alongside a transfer keeps its rsync meaning of `--human-readable` (see that row) | | `-V`, `--version` | Print version | ✅ Parity | | -| `--info=FLAGS` | Fine-grained info verbosity | ⚠️ Caveat | Accepts rsync 3.4.1's full `--info` vocabulary — `backup`, `copy`, `del`, `flist`, `misc`, `mount`, `name`, `nonreg`, `progress`, `remove`, `skip`, `stats`, `symsafe`, `all`, `none` — with optional level suffixes (`--info=stats2`), so a valid rsync invocation is never rejected up front. Protocol 2.27.0 wires the categories that map to a real FastSync event, matching rsync's line format: `name` prints the updated entry names (with the ` -> target` link suffix), `flist` prints `sending incremental file list`, `del` prints `deleting PATH` (or `*deleting PATH` under `-i`/`--out-format`) for both dry-run would-delete and real deletions (real runs carry the removed paths over the new `report_deletes` wire bool), `remove` prints `sender removed PATH`, `nonreg` prints `skipping non-regular file "NAME"`, `progress` drives the per-file progress output, and `copy`/`misc`/`skip`/`stats` keep their existing channels. `none` suppresses info output, explicit flags override `--verbose`, and a genuinely unknown name is still rejected by name (matching rsync). **Fixed (no-wire):** `--info=name2` (and higher) also prints rsync's `NAME is uptodate` lines for entries the receiver already has, and `--info=name` emits the leading transfer-root `./` name line before the first transferred entry (the marker rides in the existing `info_level` bitset; differential tests vs rsync 3.4.1). **Caveat:** the root `./` line is emitted before the first transferred name rather than keyed off rsync's root-attribute-change decision, so a pre-existing root that rsync leaves untouched can differ; the categories with no client-observable event stay accepted-but-silent — `symsafe`, `mount`, and `backup` (the backup happens on the receiver, which FastSync's protocol does not echo back); and `skip` maps to FastSync's sender-side skip logging rather than rsync's receiver-side "not creating new file" lines | -| `--debug=FLAGS` | Fine-grained debug verbosity | ⚠️ Caveat | Protocol 2.26.0 accepts rsync 3.4.1's full `--debug` vocabulary with optional level suffixes. FastSync emits for its own channels (`io`, `proto`, `pack`, `util`, plus the aliases `hl`/`owner`); the rsync-only categories (`acl`, `filter`, `send`, ...) are accepted silently. `--debug=help` lists the flags; a genuinely unknown name is rejected by name. **Caveat:** most accepted rsync categories produce no output (e.g. `acl`, `filter`, `send`, `flist`, `del`, `deltasum`, `hash`, `recv`, `time`), so they are accepted for CLI compatibility only | +| `--info=FLAGS` | Fine-grained info verbosity | ⚠️ Caveat | Accepts rsync 3.4.1's full `--info` vocabulary — `backup`, `copy`, `del`, `flist`, `misc`, `mount`, `name`, `nonreg`, `progress`, `remove`, `skip`, `stats`, `symsafe`, `all`, `none` — with optional level suffixes (`--info=stats2`), so a valid rsync invocation is never rejected up front. Protocol 2.27.0 wires the categories that map to a real FastSync event, matching rsync's line format: `name` prints the updated entry names (with the ` -> target` link suffix), `flist` prints `sending incremental file list`, `del` prints `deleting PATH` (or `*deleting PATH` under `-i`/`--out-format`) for both dry-run would-delete and real deletions (real runs carry the removed paths over the new `report_deletes` wire bool), `remove` prints `sender removed PATH`, `nonreg` prints `skipping non-regular file "NAME"`, `progress` drives the per-file progress output, `copy`/`misc`/`skip` keep their existing channels, `stats` enables the same transfer-statistics block as `--stats`, and `mount` prints rsync's `[sender] skipping mount-point dir NAME` when `-xx` drops a mount-point directory (plain `-x` keeps the empty directory entry and stays silent, matching rsync; both differential-tested). `none` suppresses info output, explicit flags override `--verbose`, and a genuinely unknown name is still rejected by name (matching rsync). **Fixed (no-wire):** `--info=name2` (and higher) also prints rsync's `NAME is uptodate` lines for entries the receiver already has, and `--info=name` emits the leading transfer-root `./` name line before the first transferred entry (the marker rides in the existing `info_level` bitset; differential tests vs rsync 3.4.1). **Caveat:** the root `./` line is emitted before the first transferred name rather than keyed off rsync's root-attribute-change decision, so a pre-existing root that rsync leaves untouched can differ; the categories with no client-observable event stay accepted-but-silent — `symsafe` and `backup` (the backup happens on the receiver, which FastSync's protocol does not echo back); and `skip` maps to FastSync's sender-side skip logging rather than rsync's receiver-side "not creating new file" lines | +| `--debug=FLAGS` | Fine-grained debug verbosity | ⚠️ Caveat | Protocol 2.26.0 accepts rsync 3.4.1's full `--debug` vocabulary with optional level suffixes. FastSync emits for its own channels (`io`, `proto`, `pack`, `util`, plus the aliases `hl`/`owner`) and maps the remaining categories that have a natural FastSync event onto real debug output: `flist` (per-directory scan progress), `del` (receiver-removed paths, riding the existing `report_deletes` wire bool), `hash`/`deltasum` (whole-file hashing and delta-sum generation), `recv` (receiver verdicts/signatures), `filter` (selection/exclusion decisions) and `send` (files handed to the sender). A normal run prints none of it; `--debug=help` lists the flags and a genuinely unknown name is rejected by name. **Caveat:** the output is FastSync's own timestamped debug format (it does not reproduce rsync's exact per-category lines), and the synthetic/rsync-internal categories (`acl`, `backup`, `bind`, `time`, ...) stay accepted-but-silent, so the row remains ⚠️ | | `--stderr=MODE` | Change stderr output mode | ❌ Divergent | `errors` (default) and `all` are supported; `client` is rejected with a clear error (`--stderr=client is not supported`) because FastSync has no rsync client-message channel — the rejection itself is the documented behavior (Phase 7 Wave B decision). The modes that exist work; the missing rsync channel cannot be emulated without a wire change | | `--msgs2stderr`, `--no-msgs2stderr` | Deprecated `--stderr` aliases | ⚠️ Caveat | `--msgs2stderr` maps to `--stderr=all` (supported, matching rsync). `--no-msgs2stderr` is rsync's spelling of `--stderr=client`, which FastSync has no client-message channel for, so it maps to the errors-only default instead of reproducing rsync's client mode. See `--stderr=MODE` | | `--no-motd` | Suppress daemon MOTD | ✅ Parity | Client-only display switch (Wave C): the daemon still sends the configured `motd file` on a `host::module/path` connection; the client reads and discards the frame without showing it. Without the flag the MOTD is printed to stdout after the config/auth handshake and escaped so control bytes cannot inject terminal sequences | @@ -98,7 +98,7 @@ Every one of those has an entry below with its remaining caveats. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| -| `--stats` | Give transfer stats | ⚠️ Caveat | Prints transfer statistics. Protocol 2.25.0 populates the receiver-only counters the sender cannot observe (`Matched data`, `Number of deleted files`) from the receiver's `STATUS_STATS` report; the sender tracks the scanned file list per type so `Number of files` carries rsync's `(reg: X, dir: Y, link: Z, special: W)` breakdown (directories come from the scanner's captured directory list, present for `-a`/`-t`/`-p`), `Number of regular files transferred` excludes symlinks/specials and up-to-date files, `Total file size` includes symlink target lengths, and `Total transferred file size` counts only transferred files. **Protocol 2.28.0 extends `STATUS_STATS`** with receiver-observed `literal_bytes` and the four `created_*` counters: `Number of created files` now carries rsync's `(reg/dir/link/special)` breakdown (the receiver reports which destination entries it newly created, including implicitly-created parent directories below the transfer root) and `Literal data` is exact for a delta transfer (the receiver counts the literal fragments it stored, not the whole source size) — all differential-tested in the sequential and `--threads` paths against rsync 3.4.1 for fresh-create, update and delta shapes. **Remaining divergences:** a recursive scan that preserves no directory attribute (`-r` without `-t`/`-p`) captures no directory entries, so the `dir:` category is then omitted from `Number of files`; rsync's per-type breakdown on `Number of deleted files` is not reproduced; and `Total bytes sent`/`received` are FastSync wire bytes framed differently from rsync's, so they are not numerically comparable | +| `--stats` | Give transfer stats | ⚠️ Caveat | Prints transfer statistics. Protocol 2.25.0 populates the receiver-only counters the sender cannot observe (`Matched data`, `Number of deleted files`) from the receiver's `STATUS_STATS` report; the sender tracks the scanned file list per type so `Number of files` carries rsync's `(reg: X, dir: Y, link: Z, special: W)` breakdown (directories come from the scanner's captured directory list for `-a`/`-t`/`-p`, or from a lightweight traversed-directory counter on a plain `-r` run so the `dir:` category is present there too), `Number of regular files transferred` excludes symlinks/specials and up-to-date files, `Total file size` includes symlink target lengths, and `Total transferred file size` counts only transferred files. **Protocol 2.28.0 extends `STATUS_STATS`** with receiver-observed `literal_bytes` and the four `created_*` counters: `Number of created files` now carries rsync's `(reg/dir/link/special)` breakdown (the receiver reports which destination entries it newly created, including implicitly-created parent directories below the transfer root) and `Literal data` is exact for a delta transfer (the receiver counts the literal fragments it stored, not the whole source size) — all differential-tested in the sequential and `--threads` paths against rsync 3.4.1 for fresh-create, update and delta shapes. **Remaining divergences:** rsync's per-type breakdown on `Number of deleted files` is not reproduced; and `Total bytes sent`/`received` are FastSync wire bytes framed differently from rsync's, so they are not numerically comparable | | `-h`, `--human-readable` | Human-readable numbers | ✅ Parity | Formats transfer byte and rate counts using rsync's **decimal** (base-1000) units, matching rsync `-h` (e.g. `1.23M`), not binary units. **A lone `-h` with no transfer arguments prints help instead** (protocol 2.26.0), matching the rsync idiom; `-h` alongside a transfer remains human-readable | | `-i`, `--itemize-changes` | Per-file change summary | ✅ Parity | Prints rsync-style `>f+++++++++` lines to stdout only for files actually sent (also under `-j`/`--threads`); unchanged files print nothing, matching single-`-i` behavior | | `--progress` | Show progress | ⚠️ Caveat | Protocol 2.25.0 prints rsync-style per-file progress blocks (percent, transferred/total bytes, rate, elapsed, `(xfr#N, to-chk=M/T)`) fed by the receiver's `STATUS_STATS`, in both the sequential and `--threads` send paths. FastSync also prints rsync's leading `./` transfer-root line and, when progress is requested (`--progress`/`-P`/`--info=progress`) and not `--quiet`, runs a **paths-only metadata pre-scan** (no file reads, no hashing) that supplies rsync's file-list total `T` for the `to-chk` denominator and the directory names; `--delete-during`/`--delete-delay` reuse their existing keep-set pre-scan instead of walking twice, and non-progress runs are untouched. Per-directory name lines are emitted (trailing `/`), and symlink (` -> target`) and special entries are named too, so a **fresh multi-directory tree's name set and `to-chk` denominator match rsync 3.4.1** (differential test, sequential and `--threads`) and a **single-file transfer's name lines and deterministic frames remain byte-identical** to rsync. **Remaining divergences:** rsync emits entries in sorted depth-first order while FastSync streams them in the scanner's readdir/BFS order, so the interleaving and the `to-chk` numerator differ (the denominator matches); the leading `./` root line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an ancestor directory line is emitted whenever a child transfers (rsync suppresses it when the directory itself is unchanged); on a re-run, entries without a quick-check (symlinks, empty directories) are still named where rsync stays silent; and the rate/ETA are wall-clock dependent | diff --git a/src/client/client_cli.c b/src/client/client_cli.c index c830ca9..8004d2d 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -504,9 +504,8 @@ static bool split_flag_level(const char* token, char* name, size_t name_size, in * of rsync's `symsafe`, `hlink`, and `own`. */ static bool is_accepted_debug_category(const char* name) { static const char* const categories[] = { - "acl", "backup", "bind", "chdir", "cmd", "connect", "del", "deltasum", - "dup", "exit", "filter", "flist", "fuzzy", "genr", "hash", "hl", - "hlink", "iconv", "nstr", "own", "owner", "recv", "send", "time", + "acl", "backup", "bind", "chdir", "cmd", "connect", "dup", "exit", "fuzzy", + "genr", "hl", "hlink", "iconv", "nstr", "own", "owner", "time", }; for (size_t i = 0; i < sizeof(categories) / sizeof(categories[0]); i++) { if (strcmp(name, categories[i]) == 0) @@ -518,7 +517,6 @@ static bool is_accepted_debug_category(const char* name) { static bool is_accepted_info_category(const char* name) { static const char* const categories[] = { "backup", - "mount", "syms", "symsafe", }; @@ -571,6 +569,18 @@ static int parse_debug_flags(const char* value, Config* config) { flag = LOG_DEBUG_PACK; } else if (strcmp(name, "util") == 0) { flag = LOG_DEBUG_UTIL; + } else if (strcmp(name, "flist") == 0) { + flag = LOG_DEBUG_FLIST; + } else if (strcmp(name, "del") == 0) { + flag = LOG_DEBUG_DEL; + } else if (strcmp(name, "hash") == 0 || strcmp(name, "deltasum") == 0) { + flag = LOG_DEBUG_HASH; + } else if (strcmp(name, "recv") == 0) { + flag = LOG_DEBUG_RECV; + } else if (strcmp(name, "filter") == 0) { + flag = LOG_DEBUG_FILTER; + } else if (strcmp(name, "send") == 0) { + flag = LOG_DEBUG_SEND; } else if (is_accepted_debug_category(name)) { continue; } else { @@ -645,9 +655,12 @@ static int parse_info_flags(const char* value, Config* config) { flag = LOG_INFO_MISC; else if (strcmp(name, "skip") == 0) flag = LOG_INFO_SKIP; - else if (strcmp(name, "stats") == 0) + else if (strcmp(name, "stats") == 0) { flag = LOG_INFO_STATS; - else if (strcmp(name, "del") == 0) + /* `--info=stats` requests the same transfer-statistics block as + `--stats`; `--info=stats0` turns it back off. */ + config->stats = level > 0; + } else if (strcmp(name, "del") == 0) flag = LOG_INFO_DEL; else if (strcmp(name, "remove") == 0) flag = LOG_INFO_REMOVE; @@ -655,6 +668,8 @@ static int parse_info_flags(const char* value, Config* config) { flag = LOG_INFO_FLIST; else if (strcmp(name, "nonreg") == 0) flag = LOG_INFO_NONREG; + else if (strcmp(name, "mount") == 0) + flag = LOG_INFO_MOUNT; else if (strcmp(name, "progress") == 0) flag = LOG_INFO_PROGRESS; else if (is_accepted_info_category(name)) @@ -1213,7 +1228,13 @@ static int apply_table_option(Config* config, const OptionEntry* entry, const ch void* field = (char*)config + entry->offset; switch (entry->kind) { case OPT_FLAG: - *(bool*)field = true; + /* -x/--one-file-system is repeatable in rsync: `-xx` increments the level so + the scanner drops mount-point directories instead of recreating them + empty. Everything else is a plain boolean. */ + if (entry->offset == offsetof(Config, one_file_system)) + (*(int*)field)++; + else + *(bool*)field = true; return 0; case OPT_NOOP: return 0; @@ -2574,7 +2595,11 @@ static bool cli_handle_outbuf_option(CliParseCtx* ctx) { * load --files-from once every argument has been seen. Returns 0 on success, * -1 on error. */ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool no_incremental) { - set_log_level(config->quiet ? LOG_LEVEL_ERROR : (verbose ? LOG_LEVEL_DEBUG : LOG_LEVEL_WARNING)); + /* An explicit --debug=FLAGS enables the debug log level by itself (rsync + behaviour); -v enables every other INFO-level message. */ + bool debug_enabled = verbose || config->debug_level != 0; + set_log_level(config->quiet ? LOG_LEVEL_ERROR + : (debug_enabled ? LOG_LEVEL_DEBUG : LOG_LEVEL_WARNING)); /* rsync's plain --delete defaults to delete-during (--del): each directory's extras are removed as that directory is processed, so space is freed progressively and a tight destination never has to hold the whole old+new @@ -2764,10 +2789,12 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool } /* --info=del on a real --delete run asks the receiver to report the paths it actually removed; the report rides the STATUS_STATS path list, so the wire - stats frame must be negotiated too. */ - config->report_deletes = config->use_delete && !config->dry_run && - ((config->info_level & LOG_INFO_DEL) != 0 || config->itemize_changes || - config->out_format != NULL); + stats frame must be negotiated too. --debug=del needs the same paths, so + it opts into the existing report (no new wire field). */ + config->report_deletes = + config->use_delete && !config->dry_run && + ((config->info_level & LOG_INFO_DEL) != 0 || config->itemize_changes || + config->out_format != NULL || (config->debug_level & LOG_DEBUG_DEL) != 0); config->report_stats = config->stats || config->show_progress || (config->info_level & LOG_INFO_PROGRESS) || format_needs_wire || config->report_deletes || (config->dry_run && config->use_delete); diff --git a/src/client/client_send.c b/src/client/client_send.c index c775bc2..535cb5d 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -129,6 +129,21 @@ static void stats_type_breakdown(const TransferStats* stats, char* out, size_t o out_size); } +/* rsync's `Number of files` counts every directory. A recursive scan that + preserves a directory attribute captures them in `dir_entries`; a `-r` scan + (no -t/-p) captures nothing, so fall back to the scanner's shared counter of + traversed directories that are not already represented by an inline + directory entry. The -d generator counts its explicit directory entries + inline and does not traverse, so it is excluded here. */ +static unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries, + atomic_ullong* counter) { + if (config == NULL || config->dirs || config->list_only) + return 0; + if (dir_metadata_should_capture(config)) + return dir_entries != NULL ? (unsigned long long)dir_entries->size : 0; + return counter != NULL ? (unsigned long long)atomic_load(counter) : 0; +} + /* Print the rsync `--stats` block on stdout. The source-side flist and transferred counters come from `stats` (filled while scanning/sending), the receiver-only counters from the STATUS_STATS frame, and the wire byte totals @@ -387,6 +402,15 @@ static bool info_flag_enabled(const Config* config, LogInfoFlag flag) { static void print_delete_reports(const Config* config, const ArrayList* paths) { if (!config || !paths || config->quiet) return; + /* --debug=del is independent of the --info=del/itemize/out-format display: + emit the debug trace even when no deletion line would be printed. */ + if (log_debug_enabled(LOG_DEBUG_DEL)) { + for (int i = 0; i < paths->size; i++) { + const char* raw = (const char*)paths->items[i]; + const char* path = delete_display_path(config, raw); + log_debug_message(LOG_DEBUG_DEL, "del: %s", path ? path : raw); + } + } if (!(config->itemize_changes || config->out_format != NULL || info_flag_enabled(config, LOG_INFO_DEL))) return; @@ -665,6 +689,7 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann options->ignore_io_errors = config->ignore_errors; options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args; options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet; + options->note_mount = (config->info_level & LOG_INFO_MOUNT) != 0 && !config->quiet; options->send_directory = config->send_directory; options->eight_bit_output = config->eight_bit_output; options->excluded_paths = NULL; @@ -672,6 +697,9 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann options->size_skipped_paths = NULL; options->synced_dirs = NULL; options->hardlinks = NULL; + /* Set by the real send paths; NULL for the metadata-only scans (progress + pre-count, batch) that must not perturb the sender's --stats counter. */ + options->dir_count = NULL; /* P7 Wave D: capture source directory metadata when a directory attribute is requested (-p for modes, -t for times unless -O omits them). Whether they are APPLIED is decided receiver-side. */ @@ -730,6 +758,8 @@ static bool progress_precount_scan(const Config* config, ProgressPrecount* out) ScannerOptions local = prepared.options; local.list_dirs = true; local.note_nonreg = false; + local.note_mount = false; + local.dir_count = NULL; local.use_metadata = false; local.preserve_xattrs = false; local.preserve_acls = false; @@ -1336,6 +1366,7 @@ static bool finalize_transfer(Client* client, const Config* config, ArrayList* r return false; if (status == STATUS_STATS) { ReceiverStats scratch; + log_debug_message(LOG_DEBUG_RECV, "recv: receiver stats"); /* A real --info=del run carries the actually-removed paths in the stats frame's path list; collect and print them in rsync's format. */ ArrayList* deleted = config->report_deletes ? array_list_create(free) : NULL; @@ -1885,6 +1916,8 @@ static int incremental_check(Client* client, File* file, const Config* config, if (!file_checksum(file, (ChecksumAlgo)config->checksum_algo, config->checksum_seed, digest, sizeof(digest), &digest_len)) return -1; + log_debug_message(LOG_DEBUG_HASH, "hash: %s (algo %d)", file_wire_path(file), + config->checksum_algo); uint8_t wire_len = (uint8_t)digest_len; if (!send_n_data(client->file_descriptor, &wire_len, sizeof(wire_len)) || !send_n_data(client->file_descriptor, digest, wire_len)) @@ -1919,6 +1952,7 @@ static int incremental_check(Client* client, File* file, const Config* config, log_server_rejection("Server reported error for file"); return -1; } + log_debug_message(LOG_DEBUG_RECV, "recv: check reply for %s", file_wire_path(file)); if (s == STATUS_OK) return 1; if (s == STATUS_DELTA_SIGNATURE) { @@ -1933,6 +1967,8 @@ static int incremental_check(Client* client, File* file, const Config* config, send_status(client->file_descriptor, STATUS_ERROR); return -1; } + log_debug_message(LOG_DEBUG_RECV, "recv: delta signature for %s (%d blocks)", + file_wire_path(file), sig->block_count); *out_sig = sig; return 2; } @@ -1973,6 +2009,7 @@ static int incremental_check(Client* client, File* file, const Config* config, static int send_delta(Client* client, File* file, DeltaSignature* sig, Config* config) { Delta* delta = delta_compute_seeded(file->data->data, file->data->size, sig, config->delta_block_size, (uint32_t)config->checksum_seed); + log_debug_message(LOG_DEBUG_HASH, "deltasum: %s", file_wire_path(file)); /* The receiver is blocked after sending the signature. Every local fallback therefore needs the explicit NEXT response before full data. */ if (!delta) @@ -2446,6 +2483,7 @@ static int send_single_file(Client* client, File* file, Config* config, bool use bool use_sendfile) { int compression_level = config->use_compression ? config->compression_level : 0; log_info_message(LOG_INFO_COPY, "Transferring %s", file->path); + log_debug_message(LOG_DEBUG_SEND, "send: %s", file_wire_path(file)); if (!use_incremental) { if (use_sendfile) { @@ -2886,8 +2924,8 @@ static int send_chunks_multithreaded(void* pipeline_context) { "server reported a deletion failure (--delete); see the server log for the reason"); if (ok) remove_transferred_sources(context->config, context->remove_source_files); - if (context->dir_entries) - context->stats.flist_dir += (unsigned long long)context->dir_entries->size; + context->stats.flist_dir += + dir_count_for_stats(context->config, context->dir_entries, &context->dir_count); report_transfer_stats(context->config, &context->stats, start, &recv_stats); log_info_message(LOG_INFO_STATS, "Transfer summary: %llu files, %.1f MB", context->stats.transferred_regular, @@ -2924,6 +2962,7 @@ static int scan_directory_multithreaded(void* pipeline_context) { parallel workers append under the context's dedicated mutex. */ prepared.options.dir_entries = context->dir_entries; prepared.options.dir_entries_mutex = &context->dir_entries_mutex; + prepared.options.dir_count = context->config->stats ? &context->dir_count : NULL; if (!append_implied_dir_times(context->config, context->dir_entries)) { pipeline_cancel(context); protocol_session_unbind(); @@ -3192,6 +3231,9 @@ int send_files(Config* config) { /* P7 Wave D: captured source directory times, transmitted in trailing STATUS_DIR_TIMES frame(s) (only when metadata rides the wire). */ ArrayList* dir_entries = NULL; + /* --stats directory accounting for the no-metadata (-r) case. */ + atomic_ullong dir_count; + atomic_init(&dir_count, 0); /* Protected excluded prefixes (delete-excluded default protection). */ ArrayList* excluded = NULL; /* Size-pruned prefixes (always protected) and synchronized directories. */ @@ -3373,6 +3415,7 @@ int send_files(Config* config) { the directory-time list (otherwise every directory would be captured twice). */ prepared.options.dir_entries = dir_entries; + prepared.options.dir_count = config->stats ? &dir_count : NULL; scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); if (!scanner) goto send_fail; @@ -3526,8 +3569,7 @@ int send_files(Config* config) { from the scanner's captured directory list (present whenever a directory attribute is preserved, e.g. -a/-t/-p). The -d generator counts its explicit directory entries inline instead. */ - if (dir_entries) - transfer_stats.flist_dir += (unsigned long long)dir_entries->size; + transfer_stats.flist_dir += dir_count_for_stats(config, dir_entries, &dir_count); report_transfer_stats(config, &transfer_stats, start, &recv_stats); log_info_message(LOG_INFO_STATS, "Transfer summary: %llu files, %.1f MB", transfer_stats.transferred_regular, diff --git a/src/client/scanner.c b/src/client/scanner.c index 99fec2c..1adbab8 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -219,8 +219,8 @@ typedef struct { /* --one-file-system (-x) decision. Only directories can carry a different * device than their parent (mount points), so this is checked when a child * directory is about to be descended into. */ -bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entry_device) { - return !one_file_system || entry_device == root_device; +bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device) { + return one_file_system <= 0 || entry_device == root_device; } /* Build a payload-less directory File carrying the captured metadata (when @@ -456,6 +456,40 @@ static void scanner_note_nonreg(const ScannerOptions* options, const char* fs_pa fflush(stdout); } +/* rsync 3.4.1's `--info=mount` line, emitted when `-xx` drops a mount-point + * directory: `[sender] skipping mount-point dir NAME` (the client is the + * sender). Plain `-x` keeps the empty directory and prints nothing, matching + * rsync. */ +static void scanner_note_mount(const ScannerOptions* options, const char* fs_path) { + if (!options || !options->note_mount || !fs_path) + return; + const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); + char* escaped = output_escape(rel, options->eight_bit_output); + printf("[sender] skipping mount-point dir %s\n", escaped ? escaped : rel); + free(escaped); + fflush(stdout); +} + +/* --debug=filter: a selection/filter decision dropped an entry. */ +static void scanner_note_filter(const ScannerOptions* options, const char* name) { + if (!options || !log_debug_enabled(LOG_DEBUG_FILTER) || !name) + return; + log_debug_message(LOG_DEBUG_FILTER, "filter: excluded %s", name); +} + +/* Account for a directory that will not be represented by an inline directory + * entry. Paired with scanner_dir_count_uncount for empty directories that are + * emitted inline, so every traversed directory is counted exactly once. */ +static void scanner_dir_count_count(const ScannerOptions* options) { + if (options && options->dir_count) + atomic_fetch_add(options->dir_count, 1); +} + +static void scanner_dir_count_uncount(const ScannerOptions* options) { + if (options && options->dir_count) + atomic_fetch_sub(options->dir_count, 1); +} + /* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */ static void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) { scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths); @@ -1096,6 +1130,9 @@ static bool scanner_emit_empty_dir(DirectoryScanner* scanner, ArrayList* chunk_d file_destroy(dir); return false; } + /* The directory was counted when it was opened; this inline entry represents + it, so drop the counter to avoid counting it twice in --stats. */ + scanner_dir_count_uncount(&scanner->options); return true; } @@ -1185,6 +1222,8 @@ static int open_next_directory(DirectoryScanner* scanner) { scanner->failed = true; return -1; } + scanner_dir_count_count(&scanner->options); + log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", scanner->current_path); if (scanner->options.capture_dir_times && !scanner_capture_dir_time( scanner->options.dir_entries, scanner->options.dir_entries_mutex, scanner->root_path, @@ -1692,6 +1731,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { break; } if (!passes_selection) { + scanner_note_filter(&scanner->options, name); free(rel_copy); continue; } @@ -1700,6 +1740,13 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { free(rel_copy); if (!scanner_same_filesystem(scanner->options.one_file_system, scanner->root_dev, stats.st_dev)) { + if (scanner->options.one_file_system > 1) { + /* rsync's -xx drops the mount-point directory entirely (the plain -x + path below keeps it as an empty directory) and prints the + --info=mount line when that category is enabled. */ + scanner_note_mount(&scanner->options, cur_path); + continue; + } /* rsync's -x/--one-file-system emits the mount-point directory entry itself (so the destination gets an empty directory) but does NOT descend into it. Build a payload-less directory File and hand it to @@ -2113,6 +2160,7 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo free(prefixed); } if (!passes) { + scanner_note_filter(options, entry->d_name); free(rel); free(cur_path); return; @@ -2120,6 +2168,14 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo } if (is_dir) { if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { + if (options->one_file_system > 1) { + /* -xx: drop the mount-point directory entirely (rsync) and print the + --info=mount line when enabled. */ + scanner_note_mount(options, cur_path); + free(rel); + free(cur_path); + return; + } /* -x/--one-file-system: emit the mount-point directory entry (empty) but do not descend into it (see the sequential scanner for the same rule). */ File* mount = file_create(cur_path); @@ -2257,6 +2313,7 @@ static bool scan_root_directory(ParallelScanner* ps, const char* root_directory, ps->failed = true; return false; } + log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", root_directory); const struct dirent* entry; while ((entry = readdir(dir)) != NULL) { if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) @@ -2421,6 +2478,10 @@ ParallelScanner* parallel_scanner_create_with_options(const char* root_directory parallel_scanner_destroy(ps); return NULL; } + /* The root itself is a traversed directory (rsync counts it in + `Number of files`); the worker DirectoryScanners account for every + subdirectory below it. */ + scanner_dir_count_count(options); /* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the transfer root itself (it hands the root's immediate subdirectories to workers), so capture the root's directory time here. */ diff --git a/src/client/scanner.h b/src/client/scanner.h index 55fc3bb..a9fa3d0 100644 --- a/src/client/scanner.h +++ b/src/client/scanner.h @@ -51,7 +51,7 @@ typedef struct { bool copy_dirlinks; bool munge_links; bool checksum; - bool one_file_system; + int one_file_system; /* Phase 4 special/devices: whether device nodes (--devices) and special files * (--specials) are preserved via recreation, and whether --copy-devices * copies a device's content as an ordinary regular file. */ @@ -130,6 +130,17 @@ typedef struct { /* --info=nonreg: print rsync's `skipping non-regular file "NAME"` line for a * non-regular entry that is not being preserved. Client-only. */ bool note_nonreg; + /* --info=mount: print rsync's `[sender] skipping mount-point dir NAME` when + * -xx drops a mount-point directory. Client-only. */ + bool note_mount; + /* --stats directory accounting for a `-r` run (no -t/-p): a shared counter of + * traversed directories that are NOT otherwise represented by an inline + * directory entry (rsync still counts every directory in `Number of files`). + * Incremented when a directory is opened and decremented when an empty + * directory is emitted inline (so it is counted exactly once). Atomic + * because the parallel scanner's workers share it; NULL disables the + * accounting. Client-only. */ + atomic_ullong* dir_count; /* Source root and 8-bit-output policy used to render a `--info=nonreg` name * relative to the transfer root. Borrowed read-only. */ const char* send_directory; @@ -266,7 +277,7 @@ void directory_scanner_destroy(DirectoryScanner* scanner); /* --one-file-system (-x) decision: a directory entry may be descended into * only when the option is disabled or the entry lives on the same device as * the transfer root. Exposed so tests can exercise the rule directly. */ -bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entry_device); +bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device); /* Relative path of an on-disk path below `root` ("" == the root itself, NULL * when `fs_path` is not under `root`). Handles trailing slashes and a root of diff --git a/src/shared/config.c b/src/shared/config.c index d0e25a8..2a2941b 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -79,7 +79,7 @@ static void config_set_defaults(Config* config) { config->cvs_exclude = false; config->per_dir_filter = false; config->per_dir_filter_count = 0; - config->one_file_system = false; + config->one_file_system = 0; config->no_implied_dirs = false; config->dirs = false; config->rsh_command = NULL; diff --git a/src/shared/config.h b/src/shared/config.h index 2eb6d3b..c4d7130 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -463,7 +463,9 @@ typedef struct Config { * /.rsync-filter' (the .rsync-filter files themselves are transferred); a * repeated -F adds --filter='- .rsync-filter' so they are excluded too. */ int per_dir_filter_count; - bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */ + int one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries. + Repeated -x (rsync's -xx) drops the mount-point + directory entirely instead of recreating it empty. */ /* --no-implied-dirs: client-only. With -R, do not transfer the source * metadata of the parent directories implied by a listed path; an unlisted * implied parent is still created (with default attributes) so the listed diff --git a/src/shared/log.h b/src/shared/log.h index 52cf38f..b3a258c 100644 --- a/src/shared/log.h +++ b/src/shared/log.h @@ -13,7 +13,19 @@ typedef enum { LOG_DEBUG_PROTO = 1u << 1, LOG_DEBUG_PACK = 1u << 2, LOG_DEBUG_UTIL = 1u << 3, - LOG_DEBUG_ALL = (1u << 4) - 1, + /* rsync --debug categories that now map to a natural FastSync event: + * flist (file-list scan progress), del (deletions), hash/deltasum + * (whole-file hashing and delta-sum generation), recv (receiver + * responses/signatures), filter (selection/exclusion decisions) and send + * (files handed to the sender). Only emitted when the category is + * explicitly enabled; a normal run stays silent. */ + LOG_DEBUG_FLIST = 1u << 4, + LOG_DEBUG_DEL = 1u << 5, + LOG_DEBUG_HASH = 1u << 6, + LOG_DEBUG_RECV = 1u << 7, + LOG_DEBUG_FILTER = 1u << 8, + LOG_DEBUG_SEND = 1u << 9, + LOG_DEBUG_ALL = (1u << 10) - 1, } LogDebugFlag; typedef enum { @@ -38,9 +50,13 @@ typedef enum { the info_level bitset (there is no separate Config field) and is never set by --info=all (which selects level 1). */ LOG_INFO_NAME_UPTODATE = 1u << 10, + /* --info=mount: print rsync's `[sender] skipping mount-point dir NAME` when + * -xx/--one-file-system drops a mount-point directory (FastSync's client is + * the sender). */ + LOG_INFO_MOUNT = 1u << 11, LOG_INFO_ALL = LOG_INFO_COPY | LOG_INFO_MISC | LOG_INFO_SKIP | LOG_INFO_STATS | LOG_INFO_DEL | LOG_INFO_REMOVE | LOG_INFO_NAME | LOG_INFO_FLIST | LOG_INFO_NONREG | - LOG_INFO_PROGRESS, + LOG_INFO_PROGRESS | LOG_INFO_MOUNT, } LogInfoFlag; void log_message(LogLevel log_level, const char* message, ...); diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index c45109a..307c736 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -50,6 +50,7 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que protocol_session_set_max_alloc(&context->allocation_session, config->max_alloc); context->dir_entries = NULL; context->dir_entries_mutex_init = false; + atomic_init(&context->dir_count, 0); context->delete_limit = false; int init = 0; if (config->use_metadata) { diff --git a/src/shared/multiprocessing.h b/src/shared/multiprocessing.h index a6fda8c..e53c542 100644 --- a/src/shared/multiprocessing.h +++ b/src/shared/multiprocessing.h @@ -119,6 +119,10 @@ typedef struct { ArrayList* dir_entries; mtx_t dir_entries_mutex; bool dir_entries_mutex_init; + /* --stats directory accounting for a `-r` scan (no directory metadata): + shared by the parallel scanner workers, read by the sender thread once the + scanner is done. See ScannerOptions.dir_count. */ + atomic_ullong dir_count; /* Set by the sender thread when the receiver reported a --max-delete-capped deletion (STATUS_DELETE_LIMIT): the transfer succeeded and the process must exit 25 like rsync. Read by the caller after the sender thread is joined. */ diff --git a/tests/integration/test_output_parity.py b/tests/integration/test_output_parity.py index 8cae816..0710d78 100644 --- a/tests/integration/test_output_parity.py +++ b/tests/integration/test_output_parity.py @@ -662,6 +662,46 @@ class TestWireStatsParity: assert re.match(r"Number of created files: 1 \(reg: 1\)$", r_created), r_created assert f_created == r_created, (r_created, f_created) + @requires_rsync + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_stats_r_directory_breakdown_matches_rsync(self, shared_server, mt): + """A recursive `-r` scan (no -t/-p) exposes no directory metadata, but + rsync still counts every directory in `Number of files`; the sender's + lightweight directory counter must reproduce the `dir: N` category.""" + source = os.path.join(TEST_DATA_DIR, "wire_stdir_src") + dest = os.path.join(TEST_DATA_DIR, "wire_stdir_dst") + rdst = os.path.join(TEST_DATA_DIR, "wire_stdir_rdst") + clean_dir(source) + clean_dir(dest) + clean_dir(rdst) + os.makedirs(os.path.join(source, "sub", "deep")) + os.makedirs(os.path.join(source, "empty")) + for rel in ("a.txt", os.path.join("sub", "b.txt"), os.path.join("sub", "deep", "c.txt")): + with open(os.path.join(source, rel), "wb") as fh: + fh.write(b"x\n") + os.makedirs(get_dest_received_dir(dest, source), exist_ok=True) + + rsync_result = _rsync(["-r", "--stats", source + "/", rdst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + flags = ["-r", "--stats"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, result.stderr[:300] + + def stats_line(text, key): + for line in text.splitlines(): + if line.startswith(key + ":"): + return line + return None + + r_files = stats_line(rsync_result.stdout, "Number of files") + f_files = stats_line(result.stdout, "Number of files") + # 3 regular files, 4 directories (root, sub, sub/deep, empty). + assert re.match(r"Number of files: 7 \(reg: 3, dir: 4\)$", r_files), r_files + assert f_files == r_files, (r_files, f_files) + assert (stats_line(result.stdout, "Number of regular files transferred") == + stats_line(rsync_result.stdout, "Number of regular files transferred")) + @requires_rsync @pytest.mark.ci @pytest.mark.parametrize("mt", [False, True]) diff --git a/tests/integration/test_parity_debug.py b/tests/integration/test_parity_debug.py new file mode 100644 index 0000000..b094887 --- /dev/null +++ b/tests/integration/test_parity_debug.py @@ -0,0 +1,105 @@ +"""`--debug=FLAGS` natural-event categories (no-wire). + +FastSync maps the rsync `--debug` categories that correspond to a real event it +already performs (``flist``, ``del``, ``hash``/``deltasum``, ``recv``, +``filter`` and ``send``) onto debug output. A normal run prints none of it. +""" +import os +import sys + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import TEST_DATA_DIR, run_client, clean_dir, get_dest_received_dir, ServerManager + + +def _make_tree(root): + clean_dir(root) + os.makedirs(os.path.join(root, "sub")) + with open(os.path.join(root, "a.txt"), "wb") as fh: + fh.write(b"alpha\n") + with open(os.path.join(root, "keep.log"), "wb") as fh: + fh.write(b"log\n") + with open(os.path.join(root, "sub", "b.txt"), "wb") as fh: + fh.write(b"beta\n") + + +@pytest.mark.ci +def test_debug_flist_and_send_emit_output(shared_server): + """`--debug=flist,send` produces category-tagged debug output.""" + source = os.path.join(TEST_DATA_DIR, "dbg_src") + dest = os.path.join(TEST_DATA_DIR, "dbg_dst") + _make_tree(source) + clean_dir(dest) + result, _ = run_client(source, dest, flags=["-a", "--debug=flist,send"], + port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + assert "flist: scanning" in result.stdout, result.stdout + assert "send: " in result.stdout, result.stdout + + +@pytest.mark.ci +def test_debug_filter_emits_excluded_entry(shared_server): + source = os.path.join(TEST_DATA_DIR, "dbg_filter_src") + dest = os.path.join(TEST_DATA_DIR, "dbg_filter_dst") + _make_tree(source) + clean_dir(dest) + result, _ = run_client(source, dest, + flags=["-a", "--debug=filter", "--exclude=*.log"], + port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + assert "filter: excluded keep.log" in result.stdout, result.stdout + + +@pytest.mark.ci +def test_debug_hash_and_recv_emit_on_incremental(shared_server): + source = os.path.join(TEST_DATA_DIR, "dbg_hash_src") + dest = os.path.join(TEST_DATA_DIR, "dbg_hash_dst") + _make_tree(source) + clean_dir(dest) + result, _ = run_client(source, dest, + flags=["-a", "--incremental", "--checksum", + "--debug=hash,recv"], + port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + assert "hash: " in result.stdout, result.stdout + assert "recv: " in result.stdout, result.stdout + + +@pytest.mark.ci +def test_debug_del_emits_deleted_path(): + """`--debug=del` reports the paths the receiver actually removed. + + A deletion-capable server is required (the shared fixture refuses + client-requested deletion).""" + source = os.path.join(TEST_DATA_DIR, "dbg_del_src") + dest = os.path.join(TEST_DATA_DIR, "dbg_del_dst") + _make_tree(source) + clean_dir(dest) + seeded = get_dest_received_dir(dest, source) + os.makedirs(seeded) + with open(os.path.join(seeded, "extra.tmp"), "wb") as fh: + fh.write(b"stale\n") + server = ServerManager() + server.start(extra_args=["--allow-super", "--allow-delete"]) + try: + result, _ = run_client(source, dest, flags=["-a", "--delete", "--debug=del"], + port=server.port) + finally: + server.stop() + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + assert "del: " in result.stdout and "extra.tmp" in result.stdout, result.stdout + assert not os.path.exists(os.path.join(seeded, "extra.tmp")) + + +@pytest.mark.ci +def test_normal_run_has_no_debug_output(shared_server): + source = os.path.join(TEST_DATA_DIR, "dbg_quiet_src") + dest = os.path.join(TEST_DATA_DIR, "dbg_quiet_dst") + _make_tree(source) + clean_dir(dest) + result, _ = run_client(source, dest, flags=["-a"], port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + assert "[DEBUG]" not in result.stdout + assert "flist: scanning" not in result.stdout + assert "send: " not in result.stdout diff --git a/tests/integration/test_parity_info_mount_stats.py b/tests/integration/test_parity_info_mount_stats.py new file mode 100644 index 0000000..fa9e50e --- /dev/null +++ b/tests/integration/test_parity_info_mount_stats.py @@ -0,0 +1,174 @@ +"""Differential parity for `--info=mount` and `--info=stats` (no-wire). + +Both behaviours are compared against real rsync 3.4.1: + +* `--info=mount` prints rsync's ``[sender] skipping mount-point dir NAME`` line + when ``-xx`` drops a mount-point directory. Plain ``-x`` keeps the empty + directory and stays silent, exactly like rsync. +* `--info=stats` requests the same transfer-statistics block as `--stats` + (rsync spells the full block ``--info=stats2``/``--stats``). + +The tests are skipped when rsync is unavailable. +""" +import os +import re +import shutil +import subprocess +import sys + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import TEST_DATA_DIR, run_client, clean_dir, get_dest_received_dir + +RSYNC = shutil.which("rsync") +requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed") + + +def _rsync(args): + env = dict(os.environ, LC_ALL="C") + return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120) + + +def _cross_device_mount_tree(source): + """Build a source whose ``nested_link`` is a symlink onto a tmpfs directory. + + ``--copy-links`` dereferences it so ``-x`` sees a mount-point directory on a + different device. Returns the probe path to remove, or skips the test when + no cross-device filesystem is available. + """ + local = os.stat(".") + shm = "/dev/shm" + try: + shm_stat = os.stat(shm) + except OSError: + pytest.skip("/dev/shm not available") + if shm_stat.st_dev == local.st_dev: + pytest.skip("no cross-device filesystem available") + + clean_dir(source) + with open(os.path.join(source, "keep.txt"), "wb") as fh: + fh.write(b"keep\n") + probe = os.path.join(shm, f"fastsync_info_mount_{os.getpid()}") + shutil.rmtree(probe, ignore_errors=True) + os.makedirs(probe) + with open(os.path.join(probe, "inside.txt"), "wb") as fh: + fh.write(b"cross\n") + try: + os.symlink(probe, os.path.join(source, "nested_link")) + except OSError: + shutil.rmtree(probe, ignore_errors=True) + pytest.skip("cannot create symlink") + return probe + + +@requires_rsync +@pytest.mark.ci +def test_info_mount_xx_matches_rsync(shared_server): + """`-xx --info=mount` drops the mount-point dir and prints rsync's line.""" + source = os.path.join(TEST_DATA_DIR, "info_mount_src") + dest = os.path.join(TEST_DATA_DIR, "info_mount_dst") + rdst = os.path.join(TEST_DATA_DIR, "info_mount_rdst") + probe = _cross_device_mount_tree(source) + clean_dir(dest) + clean_dir(rdst) + flags = ["-a", "--copy-links", "-xx", "--info=mount"] + try: + rsync_result = _rsync(flags + [source + "/", rdst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + + expected = "[sender] skipping mount-point dir nested_link" + assert expected in rsync_result.stdout, rsync_result.stdout + assert expected in result.stdout, (result.stdout, result.stderr) + + received = get_dest_received_dir(dest, source) + assert os.path.exists(os.path.join(received, "keep.txt")) + # -xx omits the mount-point directory entirely. + assert not os.path.exists(os.path.join(received, "nested_link")) + assert not os.path.exists(os.path.join(rdst, "nested_link")) + finally: + shutil.rmtree(probe, ignore_errors=True) + + +@requires_rsync +@pytest.mark.ci +def test_info_mount_single_x_is_silent(shared_server): + """Plain `-x` keeps the empty mount-point directory and prints no line.""" + source = os.path.join(TEST_DATA_DIR, "info_mount1_src") + dest = os.path.join(TEST_DATA_DIR, "info_mount1_dst") + rdst = os.path.join(TEST_DATA_DIR, "info_mount1_rdst") + probe = _cross_device_mount_tree(source) + clean_dir(dest) + clean_dir(rdst) + flags = ["-a", "--copy-links", "-x", "--info=mount"] + try: + rsync_result = _rsync(flags + [source + "/", rdst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, (result.stderr or result.stdout)[:300] + + assert "skipping mount-point dir" not in rsync_result.stdout + assert "skipping mount-point dir" not in result.stdout + + received = get_dest_received_dir(dest, source) + assert os.path.isdir(os.path.join(received, "nested_link")) + assert not os.path.exists(os.path.join(received, "nested_link", "inside.txt")) + assert os.path.isdir(os.path.join(rdst, "nested_link")) + assert not os.path.exists(os.path.join(rdst, "nested_link", "inside.txt")) + finally: + shutil.rmtree(probe, ignore_errors=True) + + +def _make_stats_tree(root): + clean_dir(root) + os.makedirs(os.path.join(root, "sub")) + with open(os.path.join(root, "a.txt"), "wb") as fh: + fh.write(b"alpha\n") + with open(os.path.join(root, "sub", "b.txt"), "wb") as fh: + fh.write(b"beta\n") + + +def _pick_stats(text): + keys = ("Number of files", "Number of regular files transferred", "Total file size", + "Total transferred file size", "Literal data", "Matched data") + out = {} + for line in text.splitlines(): + for key in keys: + if line.startswith(key + ":"): + out[key] = line + return out + + +@requires_rsync +@pytest.mark.ci +def test_info_stats_emits_full_stats_block(shared_server): + """`--info=stats` is the same full block as `--stats` and matches rsync.""" + source = os.path.join(TEST_DATA_DIR, "info_stats_src") + dest = os.path.join(TEST_DATA_DIR, "info_stats_dst") + rdst = os.path.join(TEST_DATA_DIR, "info_stats_rdst") + dest2 = os.path.join(TEST_DATA_DIR, "info_stats_dst2") + _make_stats_tree(source) + for path in (dest, rdst, dest2): + clean_dir(path) + os.makedirs(get_dest_received_dir(path, source), exist_ok=True) + + rsync_result = _rsync(["-a", "--stats", source + "/", rdst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + + info_result, _ = run_client(source, dest, flags=["-a", "--info=stats"], + port=shared_server.port) + assert info_result.returncode == 0, (info_result.stderr or info_result.stdout)[:300] + stats_result, _ = run_client(source, dest2, flags=["-a", "--stats"], + port=shared_server.port) + assert stats_result.returncode == 0, (stats_result.stderr or stats_result.stdout)[:300] + + # --info=stats must print the same block as --stats... + assert _pick_stats(info_result.stdout) == _pick_stats(stats_result.stdout), ( + f"info={info_result.stdout} stats={stats_result.stdout}") + # ...and the protocol-independent counters must match real rsync. + assert _pick_stats(info_result.stdout) == _pick_stats(rsync_result.stdout), ( + f"rsync={_pick_stats(rsync_result.stdout)} fastsync={_pick_stats(info_result.stdout)}") + assert re.search(r"^Number of files: \d+ \(reg: 2, dir: 2\)$", info_result.stdout, + re.MULTILINE), info_result.stdout diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index 468987b..b511dd5 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -762,8 +762,9 @@ static void test_parse_args_debug_flags() { int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); - EXPECT_EQ_INT(cfg->debug_level, LOG_DEBUG_ALL); - EXPECT_EQ_INT(get_log_debug_flags(), LOG_DEBUG_ALL); + EXPECT_EQ_INT(cfg->debug_level, LOG_DEBUG_IO | LOG_DEBUG_PROTO | LOG_DEBUG_PACK | LOG_DEBUG_UTIL); + EXPECT_EQ_INT(get_log_debug_flags(), + LOG_DEBUG_IO | LOG_DEBUG_PROTO | LOG_DEBUG_PACK | LOG_DEBUG_UTIL); config_delete(cfg); } @@ -1421,10 +1422,9 @@ static void test_parse_args_info_name_and_help() { config_delete(cfg); } -/* rsync 3.4.1's full --info/--debug vocabulary parses. The info categories - * with a FastSync event set their flag; the remaining rsync-only categories - * (backup/mount/symsafe/syms) parse but stay silent. Every --debug category - * listed here is FastSync-silent, so debug_level stays 0. */ +/* rsync 3.4.1's full --info/--debug vocabulary parses. The categories with a + * FastSync event set their flag; the remaining rsync-only categories + * (backup/symsafe/syms, acl/bind/chdir/...) parse but stay silent. */ static void test_parse_args_rsync_flag_vocabulary_accepted() { Config* cfg = config_create(); char* argv[] = {"fastsync", "--info=backup,del,flist,mount,nonreg,progress,remove,symsafe,syms", @@ -1436,9 +1436,10 @@ static void test_parse_args_rsync_flag_vocabulary_accepted() { int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); - EXPECT_EQ_INT(cfg->info_level, LOG_INFO_DEL | LOG_INFO_FLIST | LOG_INFO_NONREG | + EXPECT_EQ_INT(cfg->info_level, LOG_INFO_DEL | LOG_INFO_FLIST | LOG_INFO_MOUNT | LOG_INFO_NONREG | LOG_INFO_PROGRESS | LOG_INFO_REMOVE); - EXPECT_EQ_INT(cfg->debug_level, 0); + EXPECT_EQ_INT(cfg->debug_level, LOG_DEBUG_DEL | LOG_DEBUG_FLIST | LOG_DEBUG_HASH | + LOG_DEBUG_RECV | LOG_DEBUG_FILTER | LOG_DEBUG_SEND); config_delete(cfg); } -- 2.54.0 From d119f3506646bd24f8a249083ac0bdec08ac47fe Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 15:02:44 +0200 Subject: [PATCH 09/68] docs: record parity cycle 2.29 (120/10/27) and deferred residuals --- CHANGELOG.md | 43 +++++++++++++++++++++++++++++++++++++++++++ HANDOFF.md | 31 ++++++++++++++++--------------- RSYNC_COMPAT.md | 25 ++++++++++++++++--------- 3 files changed, 75 insertions(+), 24 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 7195a19..658e241 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -4,6 +4,49 @@ All notable changes to FastSync are documented here. Versions match `PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must run the same version because the handshake is strict. +## [Unreleased] + +The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0). +`RSYNC_COMPAT.md` moves from **116 ✅ / 14 ⚠️ / 27 ❌** to +**120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows. + +### Changed + +- **rsync-exact traversal order.** The sequential scanner now walks each + directory's entries in rsync 3.4.1's flist order (non-directories ascending, + then directories ascending, depth-first), so `--info=name`, the + `--delete-during`/`--delete-delay`/`-n` would-delete order and the partial + `--max-delete` survivor set match rsync byte-for-byte. `--threads` has no + rsync analogue and stays unordered. +- **Delete timing.** The complete `--delete-during`/`--delete-delay` + per-directory plan set is transmitted before the first data frame, so a + mid-transfer abort has already removed every planned extra like rsync's + generator; `-d/--dirs` uses per-directory plans (shielded untraversed + subdirectories) instead of the end-of-transfer commit. `-n`, `--delete`, + `--del`/`--delete-during` and `--delete-delay` are now ✅ Parity. +- **Basis directories.** A relative `--compare-dest`/`--copy-dest`/`--link-dest` + DIR resolves against the destination directory with the transfer-relative + name appended, exactly like rsync 3.4.1. +- **`-y`/`--fuzzy`.** The candidate search no longer inherits the ordinary delta + engine's 16 KiB minimum or 10× size-ratio bound, so an oversized or + sub-16-KiB sibling is reused exactly as rsync reuses it. +- `--info=mount` prints rsync's mount-point skip line (repeated `-xx` drops the + mount-point directory); `--info=stats` enables the `--stats` block; `-x` is + repeatable. `--stats` counts traversed directories for the `Number of files` + breakdown under a plain `-r` scan. `--debug` emits real output for + `flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send`. + +### Known residuals + +- `--progress` and `--info` still need a receiver→sender event channel for the + root `./` line, ancestor-directory suppression, receiver-side `skip`/`backup` + wording, and symlink/empty-directory quick-checks. +- `--delete-before`'s phase-0 late-file divergence remains (rsync's pre-scan + fixes the file list before the data pass). +- A single file larger than 256 MiB cannot be streamed in the default path + (a general whole-file limit, not basis-specific). +- `--stats` byte totals and `--msgs2stderr` stay documented divergences. + ## [2.28.0] - 2026-09-20 The rsync-parity cycle. `PROTOCOL_VERSION` moves `2.26.0 → 2.27.0 → 2.28.0`; diff --git a/HANDOFF.md b/HANDOFF.md index c479f73..73dba0d 100644 --- a/HANDOFF.md +++ b/HANDOFF.md @@ -1,21 +1,22 @@ -# FastSync — Session Handoff (2026-09-19) +# FastSync — Session Handoff (2026-09-20) ## Current status -- **rsync-parity tracks 1-6 landed on `dev`** via **PR #303** (`10159dc`, - "feat(parity): rsync parity tracks 1-6 (protocol 2.28.0)"). Dev push CI run - **581** fully green: lint, build-and-test, parity-full, ASan, UBSan, - fuzz-build, coverage, valgrind. +- **Release `v2.28.0`** is tagged and merged to `main` (PR #304, `b4d54504`). + `dev` is at `558782d` (the incremental-check flake fix). - **`PROTOCOL_VERSION` = `"2.28.0"`** (`src/shared/config.h`); CMake - `project(FastFileTransfer VERSION 2.28.0)`. The cycle batched all wire - changes (stats counters, filter-rule block, `--verify-basis`) under the one - bump. -- **Release `v2.28.0` tagged and merged to `main`** via PR #304 - (`b4d54504`); tag `v2.28.0`. Main push CI run **585** fully green (lint, - build-and-test, parity-full, ASan, UBSan, fuzz-build, coverage, valgrind). - Gitea release `v2.28.0` published. `dev` and `main` are at the release - content. -- Parity matrix: **116 ✅ / 14 ⚠️ / 27 ❌ = 157** (was 111/13/33 at cycle start). -- Working tree clean; feature branch deleted; no scratch trees or worktrees. + `project(FastFileTransfer VERSION 2.28.0)`. +- **Parity cycle 2.29 on branch `feat/parity-2.29`** (from `dev` @ `558782d`), + no wire change. It closes the scanner-order, delete-timing, relative-basis and + fuzzy-eligibility residuals and improves the `--info`/`--stats`/`--debug` + partials. Parity matrix: **120 ✅ / 10 ⚠️ / 27 ❌ = 157** (was 116/14/27). + Remaining ⚠️ rows: `--info`, `--debug`, `--msgs2stderr`, `--stats`, + `--progress`, `--delete-before`, `--compare-dest`/`--copy-dest`/`--link-dest` + (over-256-MiB basis MISS), `-y`/`--fuzzy` (256 MiB buffer cap). +- **Deferred (needs a wire bump):** the `--progress`/`--info` receiver→sender + event channel (root `./` line, ancestor suppression, `skip`/`backup` echo, + symlink/empty-dir quick-check); `--delete-before` phase-0 keep-set; and the + general >256 MiB single-file streaming limit (B4). +- Feature branch `feat/parity-2.29`; integration PR to `dev` pending. ## What landed this session diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 22df531..11aa01b 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -55,6 +55,13 @@ matrix is **111 ✅ / 13 ⚠️ / 33 ❌ = 157**. **Lockstep track 6 (delete default; `PROTOCOL_VERSION` stays 2.28.0).** Plain `--delete` with no explicit timing flag now defaults to rsync's delete-during (`--del`) timing: the client normalizes it onto the existing `delete_during` wire bool in `cli_finalize_config`, so no config-frame field was added, and a tight destination no longer has to hold the whole old+new tree at once (the old atomic commit could hit `ENOSPC`). The old late whole-tree commit is opt-in via `--delete-after` or the FastSync-only long spelling `--delete-commit`, which selects the identical `delete_after` timing (documented equivalence). Precedence is unchanged and order-independent: each timing flag implies `--delete`, at most one timing flag may be given, and a timing flag with `--no-delete` is rejected. `-d/--dirs` still falls back to the end commit; `--delay-updates` still deletes genuine extras before publication (the per-directory skip list protects the staging dir); `--files-from`/`-R` scope is unchanged. The per-directory `STATUS_DELETE_PLAN` frame gained a one-int `apply` flag (still 2.28.0): the one-shot per-run config block (protected prefixes, size-pruned mirrors, `--delete-missing-args` exact paths) is now always sent first on a config-only carrier with `apply=false`, fixing a latent bug where a `--delete-missing-args` run whose `--files-from` list synchronized no directory never transmitted its exact deletions. Differential evidence: `delete` (plain, vs rsync's default), `delete_commit` (FastSync `--delete-commit` vs rsync `--delete-after`), and `filter_protect_after` (whole-tree protect) cases; `TestDeleteTimingFinalStateParity` compares plain `--delete`/`--delete-commit` against rsync on completed runs, and `TestDeleteTimingFailure` proves plain `--delete` removes reached extras on a mid-transfer abort while `--delete-commit` removes nothing. The matrix is unchanged at **116 ✅ / 14 ⚠️ / 27 ❌ = 157** (the `--delete`/`--delete-during` rows stay ⚠️ for the abort boundary; `--delete-after` stays ✅). +**Parity cycle 2.29 (on `feat/parity-2.29`; `PROTOCOL_VERSION` stays 2.28.0 — no wire change was needed).** Five independent residuals were closed and four rows moved to ✅: +- **Scanner order.** The sequential scanner now buffers and sorts each directory's inspected entries (non-directories ascending, then directories ascending) and walks them depth-first, reproducing rsync 3.4.1's flist order. This makes the `--info=name` transfer order, the `--delete-during`/`--delete-delay`/`-n` would-delete order, and the partial-`--max-delete` survivor set byte-identical to rsync (`test_parity_order.py`). `--threads` has no rsync analogue and stays unordered. +- **Delete timing.** The complete per-directory plan set is transmitted before the first data frame, so a mid-transfer abort has already removed every planned extra like rsync's generator; `-d/--dirs` uses the same per-directory plans (shielded untraversed subdirectories) instead of the end-of-transfer commit (`test_delete_boundary_parity.py`). `-n`/`--delete`/`--del`/`--delete-delay` move ⚠️ → ✅. +- **Basis relative-DIR.** A relative `--compare-dest`/`--copy-dest`/`--link-dest` DIR resolves against the destination directory with the transfer-relative name appended, exactly like rsync (`test_parity_basis_fuzzy.py`); the >256 MiB basis-MISS limit remains (a general whole-file limit, not basis-specific). +- **Fuzzy eligibility.** The `-y/--fuzzy` candidate search no longer inherits the ordinary delta engine's 16 KiB minimum or 10× ratio bound, so an oversized or sub-16-KiB sibling is reused as rsync reuses it (`test_parity_basis_fuzzy.py`). +- **Output partials.** `--info=mount`/`--info=stats`, the `--stats` `dir:` breakdown under `-r`, and real `--debug` output for `flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send` were added (`test_parity_info_mount_stats.py`, `test_output_parity.py`, `test_parity_debug.py`); those rows stay ⚠️ for their remaining documented residuals. `--delete-before`'s phase-0 late-file divergence and the `--progress` root/ancestor/symlink feedback remain open (they need a receiver→sender event channel), and the >256 MiB single-file streaming limit (B4) was not addressed. The matrix is now **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. + **Parity completion wave (protocol 2.23.0 → 2.26.0).** This wave closed the remaining gaps the rsync-parity wave left open (delete timing, wire counters and output, codec breadth, general `-R`/`-d`, the full filter grammar, receiver-side @@ -101,7 +108,7 @@ Every one of those has an entry below with its remaining caveats. | `--stats` | Give transfer stats | ⚠️ Caveat | Prints transfer statistics. Protocol 2.25.0 populates the receiver-only counters the sender cannot observe (`Matched data`, `Number of deleted files`) from the receiver's `STATUS_STATS` report; the sender tracks the scanned file list per type so `Number of files` carries rsync's `(reg: X, dir: Y, link: Z, special: W)` breakdown (directories come from the scanner's captured directory list for `-a`/`-t`/`-p`, or from a lightweight traversed-directory counter on a plain `-r` run so the `dir:` category is present there too), `Number of regular files transferred` excludes symlinks/specials and up-to-date files, `Total file size` includes symlink target lengths, and `Total transferred file size` counts only transferred files. **Protocol 2.28.0 extends `STATUS_STATS`** with receiver-observed `literal_bytes` and the four `created_*` counters: `Number of created files` now carries rsync's `(reg/dir/link/special)` breakdown (the receiver reports which destination entries it newly created, including implicitly-created parent directories below the transfer root) and `Literal data` is exact for a delta transfer (the receiver counts the literal fragments it stored, not the whole source size) — all differential-tested in the sequential and `--threads` paths against rsync 3.4.1 for fresh-create, update and delta shapes. **Remaining divergences:** rsync's per-type breakdown on `Number of deleted files` is not reproduced; and `Total bytes sent`/`received` are FastSync wire bytes framed differently from rsync's, so they are not numerically comparable | | `-h`, `--human-readable` | Human-readable numbers | ✅ Parity | Formats transfer byte and rate counts using rsync's **decimal** (base-1000) units, matching rsync `-h` (e.g. `1.23M`), not binary units. **A lone `-h` with no transfer arguments prints help instead** (protocol 2.26.0), matching the rsync idiom; `-h` alongside a transfer remains human-readable | | `-i`, `--itemize-changes` | Per-file change summary | ✅ Parity | Prints rsync-style `>f+++++++++` lines to stdout only for files actually sent (also under `-j`/`--threads`); unchanged files print nothing, matching single-`-i` behavior | -| `--progress` | Show progress | ⚠️ Caveat | Protocol 2.25.0 prints rsync-style per-file progress blocks (percent, transferred/total bytes, rate, elapsed, `(xfr#N, to-chk=M/T)`) fed by the receiver's `STATUS_STATS`, in both the sequential and `--threads` send paths. FastSync also prints rsync's leading `./` transfer-root line and, when progress is requested (`--progress`/`-P`/`--info=progress`) and not `--quiet`, runs a **paths-only metadata pre-scan** (no file reads, no hashing) that supplies rsync's file-list total `T` for the `to-chk` denominator and the directory names; `--delete-during`/`--delete-delay` reuse their existing keep-set pre-scan instead of walking twice, and non-progress runs are untouched. Per-directory name lines are emitted (trailing `/`), and symlink (` -> target`) and special entries are named too, so a **fresh multi-directory tree's name set and `to-chk` denominator match rsync 3.4.1** (differential test, sequential and `--threads`) and a **single-file transfer's name lines and deterministic frames remain byte-identical** to rsync. **Remaining divergences:** rsync emits entries in sorted depth-first order while FastSync streams them in the scanner's readdir/BFS order, so the interleaving and the `to-chk` numerator differ (the denominator matches); the leading `./` root line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an ancestor directory line is emitted whenever a child transfers (rsync suppresses it when the directory itself is unchanged); on a re-run, entries without a quick-check (symlinks, empty directories) are still named where rsync stays silent; and the rate/ETA are wall-clock dependent | +| `--progress` | Show progress | ⚠️ Caveat | Protocol 2.25.0 prints rsync-style per-file progress blocks (percent, transferred/total bytes, rate, elapsed, `(xfr#N, to-chk=M/T)`) fed by the receiver's `STATUS_STATS`, in both the sequential and `--threads` send paths. FastSync also prints rsync's leading `./` transfer-root line and, when progress is requested (`--progress`/`-P`/`--info=progress`) and not `--quiet`, runs a **paths-only metadata pre-scan** (no file reads, no hashing) that supplies rsync's file-list total `T` for the `to-chk` denominator and the directory names; `--delete-during`/`--delete-delay` reuse their existing keep-set pre-scan instead of walking twice, and non-progress runs are untouched. Per-directory name lines are emitted (trailing `/`), and symlink (` -> target`) and special entries are named too, so a **fresh multi-directory tree's name set and `to-chk` denominator match rsync 3.4.1** (differential test, sequential and `--threads`) and a **single-file transfer's name lines and deterministic frames remain byte-identical** to rsync. **Order parity (parity-2.29):** the sequential scanner now emits entries in rsync's sorted depth-first flist order (non-directories ascending, then directories ascending), so the interleaving and the `to-chk` numerator match rsync for the default single-threaded transfer (differential `test_parity_order.py`; `--threads` has no rsync analogue and stays unordered). **Remaining divergences:** the leading `./` root line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an ancestor directory line is emitted whenever a child transfers (rsync suppresses it when the directory itself is unchanged); on a re-run, entries without a quick-check (symlinks, empty directories) are still named where rsync stays silent; and the rate/ETA are wall-clock dependent | | `-P` | Same as --partial --progress | ✅ Parity | Parses to `--partial` + `--progress`. The independent `--partial` retention semantics are rsync parity: an interrupted write retains the already-written temp at the destination (best-effort) so a later `--append`/`--append-verify` can resume. Progress presentation is owned by the `--progress` row; there is no separate `-P` divergence | | `--out-format=FORMAT` | Custom output format | ❌ Divergent | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%c` `%C` `%i` `%M` `%o` `%U` `%G` `%t` `%%`. `%C` now uses the negotiated transfer algorithm (`--checksum-choice`, default `xxh128`, seed 0) and renders every algorithm exactly like rsync — xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, `none` a blank 2-char column — differential-tested across all algorithms. `%f`/`%n`/`%l`/`%i`/`%M`/`%U`/`%G`/`%B` also match. **Reclassified because `%b`/`%c` are protocol-specific and cannot match:** a differential against rsync 3.4.1 shows whole-file `%c = 16` for both, but rsync whole-file `%b = filesize + 27 + transfer-digest-bytes` (39 for a 0-byte file; 43/35/47 for xxh128/xxh64/sha1 on a 12-byte file) while FastSync `%b` counts its own framing; in delta mode rsync `%c = 16 + 6·ceil(filesize/block_size)` (verified at block sizes 512/700/1024/2048) while FastSync counts its own signature handshake, and rsync `%b` is its token stream. FastSync's wire bytes are a different quantity, so exact `%b`/delta-`%c` equality is impossible | | `--log-file=FILE` | Log to file | ✅ Parity | `log_file` config field | @@ -156,7 +163,7 @@ Every one of those has an entry below with its remaining caveats. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| -| `-n`, `--dry-run` | Trial run with no changes | ⚠️ Caveat | Server-contacting since protocol 2.21.0. The routing predicate `dry_run_targets_server()` selects the server-contacting path for any target a real run would reach over the wire (SSH, daemon `host::module`, explicit `--server-host`/`--server-port`, TLS, source-bind `--address`); the client handshakes with the receiver, which runs the normal read-only per-file check and answers `STATUS_DRY_RUN_TRANSFER`/`STATUS_OK` without mutating anything. Protocol 2.25.0 also reports would-delete lines: with `--delete` the receiver's `STATUS_STATS` carries the extras it would have removed and the client prints rsync-style `*deleting` lines (sequential and `--threads`; control bytes escaped). Dry-run never deletes. **Fixed (no-wire):** the dry-run keep-set manifest now carries the same filter-excluded and size-pruned protected prefixes and synchronized-directory scope a real run sends, and the would-be-transferred entries are in the keep-set, so `-n --delete` lists exactly rsync's extras for source-derived protections — the file merely being updated is kept and an excluded/pruned source entry is protected (`test_dry_run_delete_lines_match_rsync`, differential vs rsync 3.4.1). **Track 4a (protocol 2.28.0):** the received receiver-side `protect`/`risk` rules are also applied to the would-delete enumeration, so a destination-only entry matching a `P` rule is no longer reported (or removed in a real run) — matching rsync (`TestFilterProtect::test_protect_dest_only_dry_run_enumeration`). **Remaining caveat:** the would-delete line ordering follows the destination readdir order rather than rsync's reverse-sorted walk (the differential compares the sorted set) | +| `-n`, `--dry-run` | Trial run with no changes | ✅ Parity | Server-contacting since protocol 2.21.0. The routing predicate `dry_run_targets_server()` selects the server-contacting path for any target a real run would reach over the wire (SSH, daemon `host::module`, explicit `--server-host`/`--server-port`, TLS, source-bind `--address`); the client handshakes with the receiver, which runs the normal read-only per-file check and answers `STATUS_DRY_RUN_TRANSFER`/`STATUS_OK` without mutating anything. Protocol 2.25.0 also reports would-delete lines: with `--delete` the receiver's `STATUS_STATS` carries the extras it would have removed and the client prints rsync-style `*deleting` lines (sequential and `--threads`; control bytes escaped). Dry-run never deletes. **Fixed (no-wire):** the dry-run keep-set manifest now carries the same filter-excluded and size-pruned protected prefixes and synchronized-directory scope a real run sends, and the would-be-transferred entries are in the keep-set, so `-n --delete` lists exactly rsync's extras for source-derived protections — the file merely being updated is kept and an excluded/pruned source entry is protected (`test_dry_run_delete_lines_match_rsync`, differential vs rsync 3.4.1). **Track 4a (protocol 2.28.0):** the received receiver-side `protect`/`risk` rules are also applied to the would-delete enumeration, so a destination-only entry matching a `P` rule is no longer reported (or removed in a real run) — matching rsync (`TestFilterProtect::test_protect_dest_only_dry_run_enumeration`). **Order parity (parity-2.29):** the sequential scanner now walks the tree in rsync's sorted depth-first flist order (non-directories ascending, then directories ascending) and the delete walkers sort each directory the same way, so the would-delete lines are emitted in rsync's exact reverse-sorted order — differential `test_parity_order.py::test_dry_run_delete_order_matches_rsync` compares the ordered sequence, not a sorted set | | `-b`, `--backup` | Make backups of overwritten files | ✅ Parity | Backup before overwrite | | `--backup-dir=DIR` | Backup directory hierarchy | ✅ Parity | `backup_dir` config field | | `--suffix=SUFFIX` | Backup suffix (default ~) | ✅ Parity | `suffix` config field | @@ -169,10 +176,10 @@ Every one of those has an entry below with its remaining caveats. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| -| `--delete` | Delete extraneous files from dest | ⚠️ Caveat | `use_delete` config field. Deletion is always derived from the keep-set the sender actually transmitted (the per-directory `STATUS_DELETE_PLAN` set by default, or the whole-tree manifest for the late timings — never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. **Lockstep track 6 (protocol 2.28.0): plain `--delete` with no explicit timing flag now defaults to `--delete-during`**, exactly like rsync's `--del` (the client normalizes it to the existing `delete_during` wire bool; no new wire field). This frees destination space progressively during the transfer and avoids the whole-old+new-tree peak that could `ENOSPC` a tight destination. The old late whole-tree commit is opt-in via `--delete-after` or the FastSync-only long spelling `--delete-commit`. **Caveat (shared with `--delete-during`):** the exact mid-transfer abort boundary can differ from rsync's generator (rsync removes all extras ahead of its throttled sender; FastSync removes only the directories it has reached), `-d/--dirs` falls back to the end-of-transfer commit, and the surviving-set ordering under a partial `--max-delete` can differ — on a completed run the trees agree. By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). Deletion is scoped to the **synchronized directories** sent on the wire (protocol 2.23.0), so a `--files-from` subset no longer deletes untransmitted paths outside the listed directory subtrees. The walk is bounded: a client `--max-delete=NUM` (or the 100000-entry server bound) makes it **partial** — entries up to the bound are removed, the rest are skipped, and the client exits **25** (`RERR_PARTIAL`), matching rsync, rather than failing the transfer. Extraneous destination symlinks are unlinked by name (never followed); a directory still holding a kept/protected entry is left behind rather than failing | +| `--delete` | Delete extraneous files from dest | ✅ Parity | `use_delete` config field. Deletion is always derived from the keep-set the sender actually transmitted (the per-directory `STATUS_DELETE_PLAN` set by default, or the whole-tree manifest for the late timings — never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. **Lockstep track 6 (protocol 2.28.0): plain `--delete` with no explicit timing flag now defaults to `--delete-during`**, exactly like rsync's `--del` (the client normalizes it to the existing `delete_during` wire bool; no new wire field). This frees destination space progressively during the transfer and avoids the whole-old+new-tree peak that could `ENOSPC` a tight destination. The old late whole-tree commit is opt-in via `--delete-after` or the FastSync-only long spelling `--delete-commit`. **Abort/ordering parity (parity-2.29):** the complete per-directory plan set is transmitted before the first data frame, so a mid-transfer abort has already applied every planned removal exactly like rsync's generator (which runs ahead of its throttled sender); `-d/--dirs` uses the same per-directory plans (the generator records only the directories whose direct children it enumerated, so an untraversed subdirectory's mirror is shielded); and the sorted depth-first traversal makes the removal order — and therefore the survivor set under a partial `--max-delete` — match rsync exactly (`test_delete_boundary_parity.py`, `test_parity_order.py`). By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). Deletion is scoped to the **synchronized directories** sent on the wire (protocol 2.23.0), so a `--files-from` subset no longer deletes untransmitted paths outside the listed directory subtrees. The walk is bounded: a client `--max-delete=NUM` (or the 100000-entry server bound) makes it **partial** — entries up to the bound are removed, the rest are skipped, and the client exits **25** (`RERR_PARTIAL`), matching rsync, rather than failing the transfer. Extraneous destination symlinks are unlinked by name (never followed); a directory still holding a kept/protected entry is left behind rather than failing | | `--delete-before` | Delete before transfer | ⚠️ Caveat | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. **Phase-0 divergence (sharpened):** rsync builds the full file list first, so a source file created after that scan is NOT transferred and its destination extra is deleted; FastSync's single-threaded data pass re-scans the source, so the late file IS transferred (a safe superset), while FastSync `--threads` pipelines the scan and matches rsync | -| `--del`, `--delete-during` | Delete during transfer | ⚠️ Caveat | Both spellings accepted; imply `--delete`, and since lockstep track 6 this is also the default timing of a plain `--delete`. **Protocol 2.24.0 implements per-directory delete plans:** as the sender reaches each source directory it streams a `STATUS_DELETE_PLAN` for that directory and the receiver removes that directory's extras (verified with a byte-slicing proxy). The one-shot per-run config block (protected prefixes, size-pruned mirrors, `--delete-missing-args` exact paths) rides a dedicated config-only carrier frame with an `apply=false` flag, so it reaches the receiver even when the scope allows no directory plan at all (a `--files-from` list of bare files synchronizes no directory). **Phase-0 divergence (sharpened):** on a mid-transfer abort rsync's generator (which runs ahead of its throttled sender) has already removed ALL the extras it planned, whereas FastSync has removed only the directories it actually reached; on a completed run both agree. Also `-d`/`--dirs` (no descent) falls back to the end-of-transfer whole-tree commit, and the surviving-set ordering under a partial `--max-delete` can differ. `-R` plans are scoped to the transferred prefix subtree | -| `--delete-delay` | Find deletions during, delete after | ⚠️ Caveat | Implies `--delete`. **Protocol 2.24.0 implements rsync's delete-delay timing:** the sender records each directory's delete plan while scanning and the receiver commits those removals only after the whole transfer succeeds (per plan), so an extra created in the destination after its directory's plan survives while `--delete-after` re-scans and removes it, and a failed transfer removes nothing. The **reported** deleted count advances only on an actual removal. **Fixed (no-wire):** the `--max-delete` budget is now charged on ACTUAL removals (an unlink/rmdir that succeeded), not at plan/snapshot time, and a queued directory is re-scanned at commit and removed recursively (content created after the plan included), matching rsync: a snapshotted entry that fails or is skipped consumes no budget, so a later extra rsync would delete is still deleted. The deferred snapshot list keeps an independent hard cap (`DELETE_PLAN_SERVER_LIMIT`) so it cannot grow without bound now that the budget is no longer charged while scanning. A `--max-delete=2` partial delete reports exactly 2 and exits 25 in both tools, and the refilled-directory differential (late content removed, directory removed, budget shared) now matches rsync 3.4.1 on both sides (`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`). Unit tests cover recursive removal, actual-removal charging, and the bounded deferred list. **Caveat:** the exact ordering of which extras are removed first under a partial `--max-delete` can still differ from rsync's generator (the survivor set is compared by count, not identity) | +| `--del`, `--delete-during` | Delete during transfer | ✅ Parity | Both spellings accepted; imply `--delete`, and since lockstep track 6 this is also the default timing of a plain `--delete`. **Protocol 2.24.0 implements per-directory delete plans:** as the sender reaches each source directory it streams a `STATUS_DELETE_PLAN` for that directory and the receiver removes that directory's extras (verified with a byte-slicing proxy). The one-shot per-run config block (protected prefixes, size-pruned mirrors, `--delete-missing-args` exact paths) rides a dedicated config-only carrier frame with an `apply=false` flag, so it reaches the receiver even when the scope allows no directory plan at all (a `--files-from` list of bare files synchronizes no directory). **Abort/ordering parity (parity-2.29):** the complete plan set is transmitted before the first data frame, so on a mid-transfer abort every planned extra has already been removed exactly like rsync's generator (which runs ahead of its throttled sender); `-d/--dirs` no longer falls back to the end-of-transfer commit but records only the directories whose direct children it enumerated; and the sorted depth-first traversal makes the removal order — and the partial-`--max-delete` survivor set — identical to rsync (`test_delete_boundary_parity.py`, `test_parity_order.py`). `-R` plans are scoped to the transferred prefix subtree | +| `--delete-delay` | Find deletions during, delete after | ✅ Parity | Implies `--delete`. **Protocol 2.24.0 implements rsync's delete-delay timing:** the sender records each directory's delete plan while scanning and the receiver commits those removals only after the whole transfer succeeds (per plan), so an extra created in the destination after its directory's plan survives while `--delete-after` re-scans and removes it, and a failed transfer removes nothing. The **reported** deleted count advances only on an actual removal. **Fixed (no-wire):** the `--max-delete` budget is now charged on ACTUAL removals (an unlink/rmdir that succeeded), not at plan/snapshot time, and a queued directory is re-scanned at commit and removed recursively (content created after the plan included), matching rsync: a snapshotted entry that fails or is skipped consumes no budget, so a later extra rsync would delete is still deleted. The deferred snapshot list keeps an independent hard cap (`DELETE_PLAN_SERVER_LIMIT`) so it cannot grow without bound now that the budget is no longer charged while scanning. A `--max-delete=2` partial delete reports exactly 2 and exits 25 in both tools, and the refilled-directory differential (late content removed, directory removed, budget shared) now matches rsync 3.4.1 on both sides (`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`). Unit tests cover recursive removal, actual-removal charging, and the bounded deferred list. **Ordering parity (parity-2.29):** the sorted depth-first traversal plus the up-front plan set make the order in which extras are removed — and therefore the survivor set under a partial `--max-delete` — match rsync exactly (differential `test_parity_order.py::test_delete_delay_deletion_order_matches_rsync` and `::test_partial_max_delete_survivor_order_matches_rsync`) | | `--delete-after` | Delete after transfer | ✅ Parity | Implies `--delete`. Selects the late whole-tree commit: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing. Since lockstep track 6 a plain `--delete` defaults to delete-during (rsync's `--del`); `--delete-after` — or the FastSync-only `--delete-commit` spelling, which selects the identical timing — is the explicit way to keep the old commit-style behavior | | `--delete-excluded` | Also delete excluded files | ✅ Parity | `delete_excluded` config field. Under `--delete` FastSync protects (rsync's default) the destination mirror of paths the sender's source scan pruned by the user-selection rules — the `--filter`/`-F`/`-C` layer and the legacy `--exclude`/`--include` layer. The sender transmits those concrete pruned paths as **protected prefixes** in the delete-manifest frame (see the Phase-3 notes below); the walker never descends into or removes them. `--delete-excluded` opts back in: the sender sends an empty protected list, so the excluded destination mirrors become ordinary extras and are removed. **`--max-size`/`--min-size` pruned mirrors are a separate, always-on protection** (protocol 2.23.0, rsync parity): size-pruned source mirrors survive `--delete` even with `--delete-excluded`. Track 4a (protocol 2.28.0) additionally re-applies the received `protect`/`risk` rules on the receiver, so a destination-only entry matching an exclude rule is protected (or left at risk) exactly like rsync; the remaining sender-derived `--delete-excluded` behavior (an unqualified rule becomes sender-only, so its source mirror and matching destination-only extras are deleted) is unchanged | | `--max-delete=NUM` | Max files to delete | ✅ Parity | `max_delete` config field (default -1 = no client limit; 0 = delete nothing). **Protocol 2.23.0 matches rsync's partial semantics:** the receiver deletes up to NUM entries (regular files, symlinks and empty directories; each directory removal counts as one) and then **stops deleting, skips the rest, and reports the run as partial**. The client prints a "deletions stopped due to `--max-delete` limit" message and exits **25** (rsync's `RERR_PARTIAL`), not a hard failure — the transfer itself succeeded. NUM only applies together with `--delete` (it is inert otherwise, matching rsync). A client NUM below the server hard bound `MAX_SERVER_DELETE_COUNT` (100000) replaces it; a NUM above it never raises that cap. Deleting an entire destination with no limit is still bounded by the server's 100000-entry ceiling. `--delete-missing-args` exact-path deletions and the ordinary extras walk draw from the same budget, matching rsync | @@ -667,10 +674,10 @@ targets verbatim, matching rsync. |------|-------------------|-----------------|-------| | `--checksum` | Skip based on checksum | ✅ Parity | `-c`/`--checksum` compares per-file whole-file content digests to skip unchanged files. **As of protocol 2.23.0 the short `-c` implies the checksum quick-check**, so a plain `-c` run verifies content rather than only affecting the `--incremental` handshake. The digest algorithm is `xxh128` by default (protocol 2.26.0's negotiated default) and is selectable via `--checksum-choice`/`--cc` (`xxh128`/`xxh3`/`xxh64`/`xxhash`/`md5`/`md4`/`sha1`/`none`/`auto`, plus rsync's two-name form) and `--checksum-seed=NUM` (see those rows) | | `--checksum-choice=STR`, `--cc=STR` | Choose checksum algorithm | ✅ Parity | Real algorithm selection for the per-file whole-file digest used by the `--incremental`/`--checksum` handshake and basis-dir verification. **Protocol 2.26.0 accepts rsync 3.4.1's full set** — `xxh128` (the negotiated default), `xxh3`, `xxh64`, `xxhash`, `md5`, `md4`, `sha1`, `none`, `auto`, and the two-name `transfer,pre-transfer` form — with rsync's exit-4 rejection of an unknown name and of `none` on the transfer side when `--checksum` is on. `--cc=ALG` and space forms both parse. The algorithm id and seed cross the wire; the receiver hashes its old file with the same algorithm+seed and the per-file `STATUS_CHECK` handshake carries a bounded digest pinned to the negotiated length. `checksum_digest_file` now streams **every** supported algorithm (md4 via the self-contained RFC 1320 code, sha1/md5 via EVP, none as an empty digest), so the streaming path matches its contract, and `--out-format %C` uses the selected **transfer** half of a two-name choice and renders each algorithm byte-for-byte like rsync (xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, none a blank 2-char column) — differential-tested across all algorithms. **Track-3b finding — the block-checksum residual is not observable.** rsync applies the choice to the block checksum on its wire too, while FastSync selects only the whole-file comparison digest and keeps the delta BLOCK strong checksum fixed at xxHash32 (`DeltaBlockSig`, `delta_signature_create_seeded`). Because FastSync does not interoperate with rsync on the wire, only the compared surface matters, and a pre-seeded delta differential against rsync 3.4.1 (`--no-whole-file -B8192 --stats --out-format=%c|%C %n` vs `--incremental --delta`) shows the choice does not move it: across `xxh64`, `xxh128`, `xxh3`, `md5`, `md4`, `sha1` and both two-name orders the destination tree is byte-identical, `Matched data`/`Literal data`/`Total transferred file size` are unchanged (and equal to rsync's with the block size pinned), the `%c` block-checksum token is invariant (rsync `16 + 6·ceil(size/block)`, already documented under `--out-format`; FastSync its own basis-read counter), and the exit code is 0. The negotiated algorithm is visible only in `%C`, which applies it to the whole-file transfer digest and matches rsync byte-for-byte. A false block match would require the 4-byte adler32 AND the 4-byte xxHash32 to collide; at the 256 MiB maximum with the 1 KiB minimum block size the expected false matches are ≤2⁻¹⁸, and the choice cannot change this because FastSync's block strong sum is fixed. `auto` now consults `RSYNC_CHECKSUM_LIST` (rsync's whitespace-separated preference list; unknown names skipped, first supported wins, all-unknown is exit 4) before the compiled-in order; because both peers run the identical build this deterministic resolution needs no rsync peer probe, and an explicit `--cc` still wins. The list is differential-tested through `--out-format %C` (byte-identical digests to rsync for md5/sha1/xxh3) | -| `--compare-dest=DIR` | Compare dest files relative to DIR | ⚠️ Caveat | DIR is a receiver-side basis; protocol 2.26.0 uses an absolute path verbatim (rsync semantics) and resolves a relative path below the destination root (`..` components are rejected, `//` collapsed and trailing `/` dropped) — note rsync resolves a relative DIR against the destination directory while FastSync resolves it below the receive root and appends the mirrored source path, so the same relative spelling addresses a different tree (use an absolute DIR for exact parity). On the receiver's per-file check (implies `--incremental`) an exact match is rsync's metadata quick-check: same size and mtime (unless `--size-only`; `-I` disables matching), with NO content digest required by default (track 5a). A match suppresses the data transfer. The FastSync-only `--verify-basis` restores the stricter whole-file content equality. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Differential-tested against rsync 3.4.1 (`compare_dest`, and `test_verify_basis_restores_strict_content`). Residual: a basis MISS above the 256 MiB whole-file payload bound is refused up front (FastSync's general whole-file limit, not basis-specific); rsync applies basis dirs to arbitrary sizes. Wire: a basis-count field plus the `verify_basis` bool are present on the config frame (protocol 2.9.0/2.28.0) | +| `--compare-dest=DIR` | Compare dest files relative to DIR | ⚠️ Caveat | DIR is a receiver-side basis; protocol 2.26.0 uses an absolute path verbatim (rsync semantics) and resolves a relative path below the destination root (`..` components are rejected, `//` collapsed and trailing `/` dropped) — note rsync resolves a relative DIR against the destination directory while FastSync resolves it below the receive root and appends the mirrored source path, so the same relative spelling addresses a different tree (use an absolute DIR for exact parity). On the receiver's per-file check (implies `--incremental`) an exact match is rsync's metadata quick-check: same size and mtime (unless `--size-only`; `-I` disables matching), with NO content digest required by default (track 5a). A match suppresses the data transfer. The FastSync-only `--verify-basis` restores the stricter whole-file content equality. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Differential-tested against rsync 3.4.1 (`compare_dest`, and `test_verify_basis_restores_strict_content`). **Relative-DIR parity (parity-2.29):** a relative DIR now resolves against the destination directory with the file's transfer-relative name appended, exactly like rsync 3.4.1, instead of FastSync's source-mirrored wire path (the historical spelling stays as a fallback; differential `test_parity_basis_fuzzy.py`). Residual: a basis MISS above the 256 MiB whole-file payload bound is refused up front (FastSync's general whole-file limit, not basis-specific); rsync applies basis dirs to arbitrary sizes. Wire: a basis-count field plus the `verify_basis` bool are present on the config frame (protocol 2.9.0/2.28.0) | | `--copy-dest=DIR` | Include copies of unchanged files | ⚠️ Caveat | Same basis rules as `--compare-dest`, but an exact match materializes a **local copy** of the DIR file into the destination (via the atomic temp+rename store path, so `--existing`/`--ignore-existing`/`--update`/`--backup`/`--delay-updates` all still apply) instead of transferring data. Track 5a re-applies the SOURCE attributes on the copy (rsync's "copy then fix attributes"): the sender transmits the source metadata with the basis check frame, so the copy's mode/uid/gid/mtime match the source rather than the basis inode (differential `copy_dest` compares modes). The copy streams the basis file through a bounded buffer, so a basis larger than the whole-file payload bound still materializes. Repeatable; command-line order = priority. Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 | -| `--link-dest=DIR` | Hardlink to files when unchanged | ⚠️ Caveat | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy (streamed from the basis, so an over-limit basis still works), never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Differential-tested against rsync 3.4.1 (`link_dest`). Inherent shared-inode semantics (identical to rsync): a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); protocol 2.26.0 re-links an already up-to-date destination file to the basis; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. Residual: a basis MISS above the 256 MiB whole-file payload bound is refused (FastSync's general whole-file limit). Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 | -| `-y`, `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ⚠️ Caveat | `-y/--fuzzy` is a pure bandwidth optimization on the existing receiver-driven delta path: when a file must be transferred and the destination holds no usable content at the exact path (file absent, or the destination file is outside the delta engine's size bounds), the receiver searches the SAME destination directory for an existing regular file whose basename is similar to the incoming name and uses it as the delta basis, so the sender transmits only the differences instead of the whole file. The output is always byte-exact regardless of which (or whether any) basis is chosen. Decision location: the receiver performs the candidate search inside `receive_incremental_check` and sends the normal `STATUS_DELTA_SIGNATURE`; the sender never learns the basis was a different file, so no new frame type or sender logic was needed — only the config frame grew a `fuzzy` boolean, so `PROTOCOL_VERSION` was bumped **2.8.0 → 2.9.0** (peers must match). Similarity heuristic (a deterministic port of rsync 3.4.1's matcher — `util1.c` `fuzzy_distance`/`find_filename_suffix` plus `generator.c find_fuzzy`'s exact size+mtime pass — documented precisely): candidates are the target's sibling entries in its destination directory, opened `O_NOFOLLOW`/`AT_SYMLINK_NOFOLLOW` under the confined root (symlinks never followed; nothing outside the destination root is ever read or hashed); dotfiles, directories, the target's own name, and the `.fastsync-stage`/temp scratch names are excluded; like the ordinary delta path, the block signature the receiver transmits is derived from on-disk content it may not otherwise send, so a negotiated `--fuzzy` run exposes the destination's sibling files (at block granularity) to the sender as a known-plaintext oracle — the same information class as the normal delta handshake over the file being replaced; the size gate is the delta engine's own bounds (both files ≥ 16 KiB, ≤ `--delta-max`, ratio ≤ 10×); rsync's fuzzy matcher is not tied to a delta size bound and empirically reuses a basis well outside FastSync's window (a 64 KiB source against a repeated-content sibling from 0.25× to 10000×, and files as small as 300 B), so candidate ELIGIBILITY — and hence the chosen basis — can differ even though the name heuristic is the same; protocol 2.26.0 uses rsync's weighted-Levenshtein name/suffix distance plus an exact size+mtime pass and reads a single best candidate; the tie-break (smallest size gap, then lexical name) is deterministic where rsync leaves equal distances to its file-list order; the directory scan is capped at 4096 entries so a pathological directory cannot stall a transfer. When fuzzy applies: only to files the receiver would otherwise send whole — the destination's own file is always preferred as the delta basis when it exists and fits the delta size bounds, so fuzzy does NOT replace an existing-but-different destination basis; FastSync's 10× delta size-ratio bound means an existing destination file that is too far away in size still lets the fuzzy search run. When no similar candidate exists the transfer falls back to the normal whole-file transfer. rsync-divergence note: the name matching is rsync's own rule; the residual is eligibility bounded by FastSync's delta engine, so no name-matcher port can widen it. Because FastSync's delta machinery is off by default (rsync's is on), `--fuzzy` implies `--incremental` + `--delta` (unless `--whole-file`/`-W` or an explicit `--no-delta` switched delta off, in which case fuzzy is inert — matching rsync where `--whole-file` makes fuzzy irrelevant). Unlike the basis-dir options, `--fuzzy` honors an explicit `--no-incremental` (it does not force the handshake back on); an explicit `--no-incremental` also suppresses the delta implication so no invalid `--delta requires --incremental` config results. `--no-fuzzy` negates it. All surrounding semantics are untouched: a fuzzy-reconstructed file is stored as a normal file, so `--remove-source-files`, itemize/`-i`, `--stats`, `--backup`, `--delay-updates`, `--existing`/`--ignore-existing`/`--update` behave exactly as for a whole-file transfer (the fuzzy delta does not skip the file). **Reclassified Caveat (track 5b):** the output is always byte-exact regardless of the basis, and the name rule is rsync's, so the residual is candidate ELIGIBILITY: FastSync's 10× ratio / 16 KiB delta gates make its size window strictly narrower than rsync's, and when eligibility differs the chosen basis — and therefore the `--stats` `Matched data`/`Literal data`/`Total transferred file size` counters — can differ even though the tree cannot. Where the two tools' choices coincide and the block size is pinned, both the tree and the counters match rsync (differential `fuzzy_basis`); `TestFuzzy` pins the window boundary on both sides (a >10× sibling and a <16 KiB sibling are declined, with a byte-exact whole-file fallback). A wider window would require loosening the delta engine's bounds, not changing the name heuristic. | +| `--link-dest=DIR` | Hardlink to files when unchanged | ⚠️ Caveat | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy (streamed from the basis, so an over-limit basis still works), never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Differential-tested against rsync 3.4.1 (`link_dest`). Inherent shared-inode semantics (identical to rsync): a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); protocol 2.26.0 re-links an already up-to-date destination file to the basis; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. **Relative-DIR parity (parity-2.29):** a relative DIR resolves against the destination directory with the transfer-relative name appended, exactly like rsync 3.4.1 (differential `test_parity_basis_fuzzy.py`). Residual: a basis MISS above the 256 MiB whole-file payload bound is refused (FastSync's general whole-file limit). Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 | +| `-y`, `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ⚠️ Caveat | `-y/--fuzzy` is a pure bandwidth optimization on the existing receiver-driven delta path: when a file must be transferred and the destination holds no usable content at the exact path (file absent, or the destination file is outside the delta engine's size bounds), the receiver searches the SAME destination directory for an existing regular file whose basename is similar to the incoming name and uses it as the delta basis, so the sender transmits only the differences instead of the whole file. The output is always byte-exact regardless of which (or whether any) basis is chosen. Decision location: the receiver performs the candidate search inside `receive_incremental_check` and sends the normal `STATUS_DELTA_SIGNATURE`; the sender never learns the basis was a different file, so no new frame type or sender logic was needed — only the config frame grew a `fuzzy` boolean, so `PROTOCOL_VERSION` was bumped **2.8.0 → 2.9.0** (peers must match). Similarity heuristic (a deterministic port of rsync 3.4.1's matcher — `util1.c` `fuzzy_distance`/`find_filename_suffix` plus `generator.c find_fuzzy`'s exact size+mtime pass — documented precisely): candidates are the target's sibling entries in its destination directory, opened `O_NOFOLLOW`/`AT_SYMLINK_NOFOLLOW` under the confined root (symlinks never followed; nothing outside the destination root is ever read or hashed); dotfiles, directories, the target's own name, and the `.fastsync-stage`/temp scratch names are excluded; like the ordinary delta path, the block signature the receiver transmits is derived from on-disk content it may not otherwise send, so a negotiated `--fuzzy` run exposes the destination's sibling files (at block granularity) to the sender as a known-plaintext oracle — the same information class as the normal delta handshake over the file being replaced; the size gate is the delta engine's own bounds (both files ≥ 16 KiB, ≤ `--delta-max`, ratio ≤ 10×); rsync's fuzzy matcher is not tied to a delta size bound and empirically reuses a basis well outside FastSync's window (a 64 KiB source against a repeated-content sibling from 0.25× to 10000×, and files as small as 300 B), so candidate ELIGIBILITY — and hence the chosen basis — can differ even though the name heuristic is the same; protocol 2.26.0 uses rsync's weighted-Levenshtein name/suffix distance plus an exact size+mtime pass and reads a single best candidate; the tie-break (smallest size gap, then lexical name) is deterministic where rsync leaves equal distances to its file-list order; the directory scan is capped at 4096 entries so a pathological directory cannot stall a transfer. When fuzzy applies: only to files the receiver would otherwise send whole — the destination's own file is always preferred as the delta basis when it exists and fits the delta size bounds, so fuzzy does NOT replace an existing-but-different destination basis; FastSync's 10× delta size-ratio bound means an existing destination file that is too far away in size still lets the fuzzy search run. When no similar candidate exists the transfer falls back to the normal whole-file transfer. rsync-divergence note: the name matching is rsync's own rule; the residual is eligibility bounded by FastSync's delta engine, so no name-matcher port can widen it. Because FastSync's delta machinery is off by default (rsync's is on), `--fuzzy` implies `--incremental` + `--delta` (unless `--whole-file`/`-W` or an explicit `--no-delta` switched delta off, in which case fuzzy is inert — matching rsync where `--whole-file` makes fuzzy irrelevant). Unlike the basis-dir options, `--fuzzy` honors an explicit `--no-incremental` (it does not force the handshake back on); an explicit `--no-incremental` also suppresses the delta implication so no invalid `--delta requires --incremental` config results. `--no-fuzzy` negates it. All surrounding semantics are untouched: a fuzzy-reconstructed file is stored as a normal file, so `--remove-source-files`, itemize/`-i`, `--stats`, `--backup`, `--delay-updates`, `--existing`/`--ignore-existing`/`--update` behave exactly as for a whole-file transfer (the fuzzy delta does not skip the file). **Reclassified Caveat (track 5b):** the output is always byte-exact regardless of the basis, and the name rule is rsync's, **Eligibility parity (parity-2.29):** the fuzzy candidate search no longer inherits the ordinary delta engine's 16 KiB minimum or 10× size-ratio bound, so an oversized or sub-16-KiB sibling is now reused exactly as rsync reuses it — the chosen basis (and therefore `Matched data`/`Literal data`/`Total transferred file size`) matches rsync where the block size is pinned (differential `test_parity_basis_fuzzy.py`; the old boundary tests were flipped to assert both tools reuse the basis). Residual: the whole-basis buffer cap (`MAX_RECEIVE_WHOLE_FILE_SIZE`, 256 MiB) and the deterministic tie-break where rsync leaves equal distances to its file-list order. | ## 12. Compression @@ -930,7 +937,7 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass.** ✅ Parity 120 / ⚠️ Caveat 10 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `--delete-before`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that receiver filter engine, flipping `--filter=RULE` back to ✅ (see above). The fs pass flips `-d/--dirs` and `--iconv` to ✅ — recursive transfers now recreate empty source directories (and replace a blocking destination non-directory with an incoming directory); `-R --no-implied-dirs --files-from` places a listed file under a missing implied parent with default attributes instead of refusing; and `--iconv` now reproduces rsync's push direction (destination charset = the spec's REMOTE half) — and reclassified six rows to ❌ after reproducing their exact residual with differential tests: `--temp-dir` (the receiver confines the scratch dir to the receive root, so an absolute temp dir is deliberately rejected although standalone rsync follows it), the three basis-dir options (FastSync xxHash-verifies a basis hit while rsync's `--size-only` quick check installs the wrong basis content), `--delay-updates` (fixed staging name wipes an unrelated destination entry of that name), and `--dry-run` (would-delete report over-reports). `--fuzzy` was also reclassified to ❌ (deterministic heuristic with a 10× size window, not rsync's matcher), but its residual is the candidate-selection heuristic itself: the final tree is byte-exact by design, so no destination differential can expose it and the row is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync differential. (Track 5b later showed the name heuristic is in fact rsync's own and moved the row ❌ → ⚠️, leaving only the narrower delta size window as the residual; see the track 5b paragraph above.) The remaining ⚠️ rows are the ones with a documented residual (see the row notes and the **Parity Completion Wave (protocol 2.26.0)** section below). -- 2.54.0 From 9b059723756f71cd5cf30203951536f15f208be9 Mon Sep 17 00:00:00 2001 From: TapTap Date: Sun, 20 Sep 2026 15:02:51 +0200 Subject: [PATCH 10/68] test: assert partial --max-delete survivor order matches rsync --- tests/integration/test_parity_order.py | 34 ++++++++++++++++++++++++++ 1 file changed, 34 insertions(+) diff --git a/tests/integration/test_parity_order.py b/tests/integration/test_parity_order.py index 5b36757..02b00ab 100644 --- a/tests/integration/test_parity_order.py +++ b/tests/integration/test_parity_order.py @@ -149,3 +149,37 @@ class TestDeleteOrderParity: @requires_rsync def test_dry_run_delete_order_matches_rsync(self): self._assert_order("--delete", dry_run=True) + + @requires_rsync + def test_partial_max_delete_survivor_order_matches_rsync(self): + """With the exact removal order matching rsync, a --max-delete cap stops + after the same entries, so the survivor set is identical too.""" + source = os.path.join(TEST_DATA_DIR, "order_maxdel_src") + clean_dir(source) + _write(os.path.join(source, "keep.txt"), b"k\n") + extra = {f"e{i}.txt": b"x\n" for i in range(6)} + extra["ed/f"] = b"f\n" + extra["ed/g"] = b"g\n" + + rdst = os.path.join(TEST_DATA_DIR, "order_maxdel_rdst") + clean_dir(rdst) + for rel, data in extra.items(): + _write(os.path.join(rdst, rel), data) + rs = _rsync(["-a", "--delete-during", "--max-delete=3", "--info=del", + source + "/", rdst + "/"]) + assert rs.returncode in (0, 25), (rs.returncode, rs.stderr) + + fdst = os.path.join(TEST_DATA_DIR, "order_maxdel_fdst") + clean_dir(fdst) + received = get_dest_received_dir(fdst, source) + for rel, data in extra.items(): + _write(os.path.join(received, rel), data) + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, fdst, + flags=["-a", "--delete-during", "--max-delete=3", "--info=del"], + port=server.port) + assert result.returncode in (0, 25), (result.returncode, result.stderr[:300]) + assert _deleting(result.stdout) == _deleting(rs.stdout), ( + f"partial --max-delete survivor order differs\n" + f"rsync={_deleting(rs.stdout)}\nfastsync={_deleting(result.stdout)}") -- 2.54.0 From bc1e1191afe19813f06729fa121519a4570d3d01 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 18:30:14 +0200 Subject: [PATCH 11/68] fix(io): pace sendfile with --bwlimit, retry poll EINTR, clamp SSL_read --- src/shared/file_send.c | 3 +++ src/shared/protocol.c | 22 ++++++++++++++++---- src/shared/protocol.h | 6 ++++++ tests/test_protocol.c | 47 ++++++++++++++++++++++++++++++++++++++++++ 4 files changed, 74 insertions(+), 4 deletions(-) diff --git a/src/shared/file_send.c b/src/shared/file_send.c index f064f93..b5224bd 100644 --- a/src/shared/file_send.c +++ b/src/shared/file_send.c @@ -166,6 +166,8 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta } struct pollfd pfd = {.fd = file_descriptor, .events = POLLOUT}; int polled = poll(&pfd, 1, timeout); + if (polled < 0 && errno == EINTR) + continue; if (polled <= 0 || (pfd.revents & (POLLERR | POLLHUP | POLLNVAL))) { close(fd); return false; @@ -183,6 +185,7 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta return false; } protocol_note_bytes_written((unsigned long long)sent); + protocol_throttle_bytes((size_t)sent); } close(fd); diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 2a3f844..a78fa0a 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -301,6 +301,14 @@ static ProtocolSession* legacy_session(int read_fd, int write_fd) { return &legacy_io_session; } +/* Pace an out-of-band write that bypassed protocol_send_n_data (the plaintext + * sendfile fast path). The bound/legacy session is resolved exactly as + * send_n_data resolves it, so the same token-bucket state is throttled and the + * TLS and plaintext transports share identical --bwlimit semantics. */ +void protocol_throttle_bytes(size_t bytes) { + bw_throttle_session(legacy_session(-1, -1), bytes); +} + bool send_n_data(int file_descriptor, const void* data, size_t data_size) { return protocol_send_n_data(legacy_session(-1, file_descriptor), data, data_size); } @@ -439,12 +447,18 @@ static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, } ssize_t bytes_received; - if (session->ssl) - bytes_received = SSL_read(session->ssl, (char*)data + total_bytes_received, - data_size - total_bytes_received); - else + if (session->ssl) { + /* SSL_read takes an int length; clamp a >INT_MAX request into chunks + * (mirrors the send path) so the size_t downcast can never truncate into + * a negative/partial read. */ + size_t ssl_chunk = data_size - total_bytes_received > (size_t)INT_MAX + ? (size_t)INT_MAX + : data_size - total_bytes_received; + bytes_received = SSL_read(session->ssl, (char*)data + total_bytes_received, (int)ssl_chunk); + } else { bytes_received = read(fd, (char*)data + total_bytes_received, data_size - total_bytes_received); + } if (bytes_received <= 0) { if (session->ssl) { int ssl_err = SSL_get_error(session->ssl, (int)bytes_received); diff --git a/src/shared/protocol.h b/src/shared/protocol.h index 4348f09..1ecdab3 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -220,6 +220,12 @@ SSL* io_get_ssl(void); unsigned long long protocol_bytes_written(void); unsigned long long protocol_bytes_read(void); void protocol_note_bytes_written(unsigned long long bytes); +/* Apply --bwlimit pacing to bytes written outside protocol_send_n_data (the + * plaintext zero-copy sendfile fast path). Resolves the bound/legacy session + * exactly as send_n_data does and runs the same token-bucket throttle, so the + * sendfile transport is paced identically to the buffered/TLS paths. A no-op + * when the effective session has no bandwidth limit. */ +void protocol_throttle_bytes(size_t bytes); void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd); /* Transitional bridge for helpers whose signatures still carry only an fd. */ diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 53e59d7..f1b3c64 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -2,6 +2,7 @@ #include "test_utils.h" #include #include +#include #include #include @@ -669,6 +670,50 @@ static void test_receive_status_keepalive_emits() { close(to_peer[1]); } +/* protocol_throttle_bytes() must apply the same token-bucket pacing as the + * buffered protocol send path, so the plaintext sendfile fast path honors + * --bwlimit exactly like the TLS path. With bwlimit=1 MB/s the initial burst + * is 100 KB (bwlimit/10); pacing 150 KB therefore owes ~50 KB of debt, i.e. a + * ~50 ms sleep. */ +static void test_protocol_throttle_bytes_paces() { + ProtocolSession session; + protocol_session_init(&session, -1, -1); + protocol_session_bind(&session); + protocol_session_set_bwlimit(&session, 1000000ULL); + + struct timespec start; + clock_gettime(CLOCK_MONOTONIC, &start); + protocol_throttle_bytes(150000); + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long elapsed_ms = + (now.tv_sec - start.tv_sec) * 1000LL + (now.tv_nsec - start.tv_nsec) / 1000000LL; + /* Allow for scheduler slack but require the bulk of the expected 50 ms. */ + EXPECT_TRUE(elapsed_ms >= 40); + + protocol_session_unbind(); +} + +/* With no bandwidth limit the primitive must not sleep, however many bytes it + * is handed. */ +static void test_protocol_throttle_bytes_unlimited() { + ProtocolSession session; + protocol_session_init(&session, -1, -1); + protocol_session_bind(&session); + protocol_session_set_bwlimit(&session, 0); + + struct timespec start; + clock_gettime(CLOCK_MONOTONIC, &start); + protocol_throttle_bytes(100000000ULL); + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long elapsed_ms = + (now.tv_sec - start.tv_sec) * 1000LL + (now.tv_nsec - start.tv_nsec) / 1000000LL; + EXPECT_TRUE(elapsed_ms < 50); + + protocol_session_unbind(); +} + void test_protocol() { test_send_receive_n_data(); test_send_receive_n_data_zero(); @@ -698,4 +743,6 @@ void test_protocol() { test_protocol_accounting_release_does_not_underflow(); test_receive_data_charge_follows_owning_session(); test_data_create_starts_uncharged_and_unowned(); + test_protocol_throttle_bytes_paces(); + test_protocol_throttle_bytes_unlimited(); } -- 2.54.0 From 10a61c6101f6cb191bf8e4af6399b52158bf13dc Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 18:31:25 +0200 Subject: [PATCH 12/68] fix(compression): raise decompression ceiling to the protocol whole-file limit MAX_DECOMPRESSED_SIZE was 100 MiB while the receiver advertises and the sender compresses whole files up to MAX_RECEIVE_WHOLE_FILE_SIZE (256 MiB), so -z on a 100-256 MiB regular file failed with 'Declared decompressed size exceeds 104857600 bytes'. Define the internal bomb-guard ceiling in terms of the protocol constant so the two bounds cannot drift, and add unit coverage for a 130 MiB payload (accepted) and an over-ceiling declared size (still rejected). --- src/shared/compression.c | 9 +++++- tests/test_compression.c | 65 ++++++++++++++++++++++++++++++++++++++++ 2 files changed, 73 insertions(+), 1 deletion(-) diff --git a/src/shared/compression.c b/src/shared/compression.c index c1fb5c9..c7bba22 100644 --- a/src/shared/compression.c +++ b/src/shared/compression.c @@ -16,7 +16,14 @@ #include #define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024) -#define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */ + +/* Hard ceiling for a single decompression. The sender compresses whole files + * up to the protocol's whole-file receive bound, so the decompressor must + * accept payloads that large; referencing the protocol constant keeps the two + * bounds from drifting apart (they previously did: a 100 MB ceiling rejected + * 100-256 MB files). This remains a real bomb guard -- every allocation in the + * paths below is clamped to it -- so it must not exceed the protocol bound. */ +#define MAX_DECOMPRESSED_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE /* rsync 3.4.1's built-in skip-compress suffix list (the `--skip-compress` * defaults, in the man page's order). rsync stores it as space-separated diff --git a/tests/test_compression.c b/tests/test_compression.c index 1e35293..f21d87a 100644 --- a/tests/test_compression.c +++ b/tests/test_compression.c @@ -3,7 +3,9 @@ #include "compression.h" #include "data.h" #include "file.h" +#include "protocol.h" #include "utils.h" +#include #include #include #include @@ -216,6 +218,67 @@ static void test_data_decompress_unknown_size_frame() { data_destroy(frame); } +/* The receiver advertises MAX_RECEIVE_WHOLE_FILE_SIZE (256 MiB) and the sender + * compresses whole files, so the decompressor's internal ceiling must match that + * protocol bound. A 130 MiB payload -- above the old 100 MiB ceiling but below + * the protocol bound -- must round-trip through + * data_decompress_limited(..., MAX_RECEIVE_WHOLE_FILE_SIZE). The payload is all + * zeros so it compresses to a tiny frame while still declaring its full size. */ +static void test_data_decompress_limited_whole_file_ceiling() { + const size_t size = 130ULL * 1024 * 1024; + Data* input = data_create_empty(size); + EXPECT_NOT_NULL(input); + memset(input->data, 0, size); + input->size = size; + + /* Threaded zstd stores the content size in the frame header, as the sender + * does for whole files, so the decompressor sees the exact declared size. */ + Data* compressed = data_compress_codec(input, COMPRESSION_ALGO_ZSTD, 1, 2); + EXPECT_NOT_NULL(compressed); + /* Premise: the declared size sits between the old 100 MiB cap and the + * protocol whole-file bound -- exactly the range that used to be rejected. */ + unsigned long long declared = + ZSTD_getFrameContentSize((uint8_t*)compressed->data + 1, compressed->size - 1); + EXPECT_TRUE(declared > (100ULL * 1024 * 1024)); + EXPECT_TRUE(declared <= MAX_RECEIVE_WHOLE_FILE_SIZE); + + Data* out = data_decompress_limited(compressed, MAX_RECEIVE_WHOLE_FILE_SIZE); + EXPECT_NOT_NULL(out); + EXPECT_EQ_INT((int)out->size, (int)size); + EXPECT_EQ_INT(memcmp(out->data, input->data, size), 0); + + data_destroy(out); + data_destroy(compressed); + data_destroy(input); +} + +/* A frame declaring an uncompressed size above the hard ceiling is still + * rejected before any allocation, even when the caller passes a limit higher + * than the protocol bound. The 13-byte header is a valid zstd frame header with + * an 8-byte content size and no blocks; rejection happens at the size check. */ +static void test_data_decompress_limited_rejects_over_ceiling() { + const uint64_t declared = MAX_RECEIVE_WHOLE_FILE_SIZE + 1; + Data* frame = data_create_empty(1 + 13); + EXPECT_NOT_NULL(frame); + uint8_t* p = (uint8_t*)frame->data; + p[0] = (uint8_t)COMPRESSION_ALGO_ZSTD; + p[1] = 0x28; /* zstd magic number, little-endian */ + p[2] = 0xB5; + p[3] = 0x2F; + p[4] = 0xFD; + p[5] = 0xE0; /* Frame_Header_Descriptor: 8-byte content size, single segment */ + for (int i = 0; i < 8; i++) + p[6 + i] = (uint8_t)((declared >> (8 * i)) & 0xff); + frame->size = 1 + 13; + /* Guard the premise: zstd reads back exactly the declared over-ceiling size. */ + EXPECT_EQ_INT((int)ZSTD_getFrameContentSize(p + 1, frame->size - 1), (int)declared); + + EXPECT_NULL(data_decompress_limited(frame, MAX_RECEIVE_WHOLE_FILE_SIZE * 2)); + EXPECT_NULL(data_decompress_limited(frame, MAX_RECEIVE_WHOLE_FILE_SIZE)); + + data_destroy(frame); +} + typedef struct { int id; int iterations; @@ -467,6 +530,8 @@ void test_compression() { test_data_compress_decompress_roundtrip(); test_data_compress_decompress_large(); test_data_decompress_unknown_size_frame(); + test_data_decompress_limited_whole_file_ceiling(); + test_data_decompress_limited_rejects_over_ceiling(); test_data_decompress_truncated_frame_fails(); test_skip_compress_suffix_matching(); test_data_compress_with_threads_roundtrip(); -- 2.54.0 From bc18ae205ba89a21e3d043e39a8f71737c507189 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 18:37:39 +0200 Subject: [PATCH 13/68] fix(cli): --partial-dir implies --partial (rsync parity) rsync 3.4.1 resolves --partial-dir after option parsing and sets keep_partial, so --partial-dir=DIR alone retains an interrupted transfer's partial file. FastSync only used the partial dir when --partial was also given, silently discarding it otherwise. Set Config->partial in cli_finalize_config whenever partial_dir is set. Following rsync, an explicit --no-partial does NOT win (verified on rsync 3.4.1 in either option order); --inplace is guarded because it writes the destination in place with no partial staging. Tests: CLI unit coverage for the implication/precedence/inplace guard, and a deterministic integration case that blocks the final install (non-empty directory at the destination) and asserts the staged partial survives under --partial-dir alone. --- src/client/client_cli.c | 9 +++++ src/client/usage.c | 2 +- tests/integration/test_features.py | 30 ++++++++++++++++ tests/test_client_cli.c | 56 ++++++++++++++++++++++++++++++ 4 files changed, 96 insertions(+), 1 deletion(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 8004d2d..e8d660f 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -2612,6 +2612,15 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool if (config->use_delete && !config->delete_before && !config->delete_during && !config->delete_delay && !config->delete_after) config->delete_during = true; + /* rsync parity: --partial-dir=DIR chooses where an interrupted transfer's + partial file is kept, so it implies --partial. rsync applies the + implication after option parsing, so it wins over an explicit --no-partial + regardless of the order the two options appear in (verified on rsync + 3.4.1). --inplace is the exception: the destination file is written in + place with no partial/temp staging, so the partial machinery is bypassed + and the implication is skipped to leave --inplace behavior untouched. */ + if (config->partial_dir && !config->inplace) + config->partial = true; if (config->compress_choice) { int algo = compression_algo_from_name(config->compress_choice); if (algo >= 0) { diff --git a/src/client/usage.c b/src/client/usage.c index c6f6e96..a24c2f8 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -288,7 +288,7 @@ void print_usage(void) { printf(" --log-file , --log-file= Write log messages to file\n"); printf(" --stderr=MODE Route logging to stderr: errors or all\n"); printf(" --partial Keep partial files on interrupted transfer\n"); - printf(" --partial-dir Directory for partial files\n"); + printf(" --partial-dir Directory for partial files (implies --partial)\n"); printf(" -T, --temp-dir Scratch dir for temp files before atomic install.\n"); printf(" Confined to the receive root: a relative dir resolves below\n"); printf(" it and an absolute/traversal dir is rejected. The dir must\n"); diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index 1a2c05c..f649ab5 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -2275,6 +2275,36 @@ class TestPartialDir: partial = os.path.join(dest, ".partial", os.path.relpath(source_file, os.path.sep)) assert not os.path.exists(partial) + def test_partial_dir_alone_implies_partial(self, shared_server): + """--partial-dir=DIR with no --partial implies --partial, like rsync. + + rsync 3.4.1 retains the staged partial when --partial-dir is given by + itself; before the implication was added FastSync discarded it. The + transfer is made to fail deterministically by placing a non-empty + directory at the destination path, so the final partial-dir -> + destination rename fails and whatever was staged under the partial dir + stays on disk.""" + source = os.path.join(TEST_DATA_DIR, "partial_dir_implied_src") + dest = os.path.join(TEST_DATA_DIR, "partial_dir_implied_dst") + clean_dir(source) + clean_dir(dest) + source_file = os.path.join(source, "f.bin") + with open(source_file, "wb") as f: + f.write(b"partial payload") + + received = get_dest_received_dir(dest, source) + os.makedirs(os.path.join(received, "f.bin")) + with open(os.path.join(received, "f.bin", "keep"), "wb") as f: + f.write(b"keep") + + result, _ = run_client(source, dest, flags=["--partial-dir=.partial"], + port=shared_server.port) + assert result.returncode != 0, "expected the blocked install to fail" + + partial = os.path.join(dest, ".partial", os.path.relpath(source_file, os.path.sep)) + assert os.path.exists(partial), \ + "--partial-dir alone must imply --partial and retain the partial file" + class TestLargeFile: def test_transfer_100mb_file(self, shared_server): diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index b511dd5..d8eb597 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -511,6 +511,61 @@ static void test_parse_args_ignore_existing() { config_delete(cfg); } +/* --partial-dir=DIR implies --partial, matching rsync 3.4.1. rsync resolves + * this after option parsing, so the implication wins over an explicit + * --no-partial in either order. It is skipped under --inplace, where partial + * staging is bypassed and the destination is written in place. */ +static void test_parse_args_partial_dir_implies_partial() { + { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--partial-dir=.partial", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->partial); + config_delete(cfg); + } + { + /* Explicit --no-partial before --partial-dir: --partial-dir still wins. */ + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--no-partial", "--partial-dir=.partial", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->partial); + config_delete(cfg); + } + { + /* Reversed order must not change the precedence. */ + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--partial-dir=.partial", "--no-partial", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->partial); + config_delete(cfg); + } + { + /* --inplace bypasses partial staging: the implication must not fire. */ + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--inplace", "--partial-dir=.partial", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->partial); + config_delete(cfg); + } + { + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--no-partial", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); + EXPECT_FALSE(cfg->partial); + config_delete(cfg); + } +} + static void test_parse_args_executability() { Config* cfg = config_create(); char* argv[] = {"fastsync", "-E", "/src", "/dst"}; @@ -4784,6 +4839,7 @@ void test_client_cli() { test_parse_args_valid_port(); test_parse_args_size_only(); test_parse_args_ignore_existing(); + test_parse_args_partial_dir_implies_partial(); test_parse_args_executability(); test_parse_args_chmod(); test_parse_args_numeric_chmod(); -- 2.54.0 From 423a62e69171ea262ec7261467a5af25a8cbb160 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 18:45:29 +0200 Subject: [PATCH 14/68] fix(receiver): confine --temp-dir scratch dir and gate setuid bits Three receiver security fixes from the audit: 1. --temp-dir symlink escape (High): file_open_temp_dir() opened the client-controlled scratch dir with a bare open(), so a symlink planted under the receive root let a peer redirect receiver scratch files outside the authorized root. The opened dir is now judged by the REAL path of its fd (via /proc/self/fd), and any target outside the authorized receive root is refused with a logged error (EACCES). An in-root symlink (the EXDEV cross-filesystem fallback case) still works, and the no-root local batch path is unchanged. 2. setuid/setgid/sticky under SUPER_MODE_OFF (High): the special bits were applied under --perms (and via --chmod) even when the connection forbade super-user activities. FileAttrPolicy gains super_permitted, set by file_attr_policy_from_config() from privilege_super_mode_permitted(); metadata_mode_for_policy(), the symlink path, the special-node creation path, and the deferred directory-mode apply now strip the special bits when it is false. Exact rsync semantics are preserved when permitted. 3. daemon umask (Low): daemonize() forced umask(0), so implied parent directories created without -p were world-writable 0777. Set the conventional daemon umask 022 instead (rsync never forces 0); -p/-a mode preservation is unaffected because it restores modes via fchmod. Tests: new unit tests for file_open_temp_dir confinement and the masked/unmasked special-bit policy (incl. the --chmod path), a daemon world-writable-dir regression test, an integration escape test, and a root-only integration test asserting special bits are masked without --allow-super. The old cross-filesystem test encoded the vulnerable behavior (symlink target outside the root) and is replaced by the escape test; the EXDEV fallback code is retained for in-root links. --- src/server/server.c | 14 +++-- src/shared/file.c | 52 +++++++++++++--- src/shared/file.h | 8 ++- src/shared/file_attr.h | 6 ++ src/shared/file_receive.c | 12 +++- src/shared/metadata.c | 25 ++++++-- tests/integration/test_daemon.py | 15 +++++ tests/integration/test_features.py | 74 ++++++++++++++-------- tests/test_file.c | 97 ++++++++++++++++++++++++----- tests/test_metadata.c | 98 ++++++++++++++++++++---------- tests/test_xattr.c | 6 +- 11 files changed, 309 insertions(+), 98 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index aebf384..fa19876 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -1178,13 +1178,19 @@ static bool daemonize(void) { close(devnull); } /* Do not pin the launch CWD (module-relative 'path' entries would resolve - * against an unstable working directory) and drop the restrictive host umask - * so modules can create files/dirs with the modes the config requests. */ + * against an unstable working directory). Set a conservative daemon umask + * of 022 (the conventional service default): rsync never forces umask 0 -- + * it reads and restores the inherited umask and creates new entries as + * 0777 & ~umask / source & ~umask without -p. Forcing 0 here made every + * implied parent directory world-writable (0777) whenever -p metadata was not + * applied. 022 gives 0755 directories and source&~022 files, matching rsync + * under a normal daemon umask; -p/-a still restore the exact source mode via + * fchmod, which is unaffected by the umask. */ if (chdir("/") != 0) log_message(LOG_LEVEL_WARNING, "daemon: chdir to / failed: %s", strerror(errno)); - umask(0); + umask(022); /* Refresh the cached umask: main() captured the launch umask before this - * (single-threaded) umask(0), and file_mode_base() must see the daemon's + * (single-threaded) umask(022), and file_mode_base() must see the daemon's * actual umask. */ file_umask_capture(); return true; diff --git a/src/shared/file.c b/src/shared/file.c index 7e44023..4fdf3fe 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -1136,17 +1136,51 @@ int file_open_private_dir(const char* dir_path) { return fd; } -/* Open a --temp-dir scratch directory exactly as rsync does: the directory must - * already exist and is used as given (an absolute path is used verbatim, a - * relative one was already resolved against the destination root by the - * caller). Unlike file_open_private_dir this neither creates it nor confines - * it below the receive root, because rsync accepts any temp dir -- including - * one outside the destination tree or on another filesystem. Returns an - * O_DIRECTORY|O_CLOEXEC fd, or -1 on error. */ +/* Open a --temp-dir scratch directory. The directory must already exist (rsync + * never creates it); a relative path was already resolved against the + * destination root by the caller. Unlike file_open_private_dir this neither + * creates it nor requires it to be a direct child of the receive root, because + * rsync permits a scratch dir that (via a symlink) lands on another filesystem + * -- but it MUST resolve inside the authorized receive root. The directory is + * opened following symlinks and then judged by the REAL path of the opened fd + * (through /proc/self/fd), so a client-planted symlink under the receive root + * can never redirect receiver scratch files outside the sandbox while an + * in-root link to another filesystem (the EXDEV fallback case) still works. + * Returns an O_DIRECTORY|O_CLOEXEC fd, or -1 on error (errno set; an escaping + * target is reported as EACCES with a logged reason). */ int file_open_temp_dir(const char* dir_path) { if (!dir_path) return -1; - return open(dir_path, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + int fd = open(dir_path, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + if (fd < 0) + return -1; + const char* root = utils_get_authorized_root_path(); + if (!root) { + /* No authorized root (e.g. a local batch apply): nothing to confine + against, so preserve the historical open-as-given behavior. */ + return fd; + } + char fd_path[64]; + int fd_path_length = snprintf(fd_path, sizeof(fd_path), "/proc/self/fd/%d", fd); + char resolved[PATH_MAX]; + if (fd_path_length < 0 || (size_t)fd_path_length >= sizeof(fd_path) || + !realpath(fd_path, resolved)) { + int saved_errno = errno; + close(fd); + errno = saved_errno; + return -1; + } + if (!path_is_within_root(root, resolved)) { + char* escaped = output_escape(dir_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, + "--temp-dir '%s' resolves outside the authorized receive root; refusing", + escaped ? escaped : ""); + free(escaped); + close(fd); + errno = EACCES; + return -1; + } + return fd; } /* After the content and mode/times are restored on the just-written file, apply @@ -1859,6 +1893,6 @@ bool file_write_to_disk(const char* path, const void* data, unsigned long long d bool inplace, bool sparse) { if (!path || (!data && data_size != 0) || has_path_traversal(path)) return false; - FileAttrPolicy policy = {false, false, false, false}; + FileAttrPolicy policy = {0}; return file_to_disk_secure(path, data, data_size, inplace, sparse, false, NULL, policy, NULL); } diff --git a/src/shared/file.h b/src/shared/file.h index 612c50b..12654ee 100644 --- a/src/shared/file.h +++ b/src/shared/file.h @@ -103,8 +103,12 @@ bool file_remove_tree_secure(const char* path); the authorized root. Used for the --delay-updates staging directory. */ int file_open_private_dir(const char* dir_path); -/* Open an existing --temp-dir scratch directory as-is (absolute or relative; - no creation, no root confinement), matching rsync's --temp-dir handling. */ +/* Open an existing --temp-dir scratch directory (relative or absolute; no + creation). When an authorized receive root is configured the directory's + REAL path (symlinks resolved) must lie within it, so a client-planted + symlink cannot redirect receiver scratch files outside the sandbox; an + in-root symlink to another filesystem is still allowed for rsync's EXDEV + fallback. */ int file_open_temp_dir(const char* dir_path); /* The file_to_disk_secure* variants write a temporary copy in the destination diff --git a/src/shared/file_attr.h b/src/shared/file_attr.h index d965773..3beb8ab 100644 --- a/src/shared/file_attr.h +++ b/src/shared/file_attr.h @@ -29,6 +29,12 @@ typedef struct FileAttrPolicy { bool times; /* config->preserve_times: apply the source mtime */ bool atimes; /* config->preserve_atimes (-U): apply the source atime */ bool executability; /* config->use_executability (-E): exec-bits-only mode */ + /* privilege_super_mode_permitted(): when false (SUPER_MODE_OFF / --no-super, + or a daemon that did not grant `client owner = yes`), the setuid/setgid/ + sticky bits are stripped from every applied mode (source mode and any + --chmod result) even under --perms. When true, rsync's exact semantics are + preserved: -p copies the special bits and the kernel decides. */ + bool super_permitted; } FileAttrPolicy; /* Build the per-attribute policy from a connection's Config. A NULL config diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index d41ca1d..3abbda1 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -474,9 +474,13 @@ static FileSaveResult file_save_special_to_disk(const char* root_directory, cons /* Under -p/--perms rsync copies the source's permission and special bits; a * kernel that denies setuid/setgid/sticky reports the failure rather than * having them masked here. Without -p the node is created like any other new - * entry: source_mode & 0777 & ~umask. */ + * entry: source_mode & 0777 & ~umask. When super-user activities are + * forbidden, the special bits are stripped even under -p (they are + * super-user activities just like device-node creation). */ mode_t perms = config->preserve_perms ? (mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777)) : (mode & 0777 & ~(mode_t)file_process_umask()); + if (!privilege_super_mode_permitted(config->super_mode)) + perms &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); int rc = is_fifo ? mkfifoat(parent_fd, leaf, perms) : mknodat(parent_fd, leaf, create_mode | perms, rdev); @@ -2949,8 +2953,12 @@ void dir_metadata_list_apply(const DirTimeList* list, const char* root_directory } if (mode_ready) { /* rsync -p copies the source directory mode exactly, including - * group/other write and the setgid/sticky bits. */ + * group/other write and the setgid/sticky bits. Setuid/setgid/sticky + * are super-user activities: when the connection forbade them + * (SUPER_MODE_OFF / --no-super), strip them even under -p. */ mode_t safe_mode = dir_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777); + if (!privilege_super_mode_permitted(config->super_mode)) + safe_mode &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); if (dir_fd < 0) { char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); log_message(LOG_LEVEL_WARNING, "Failed to open directory %s to set its mode: %s", diff --git a/src/shared/metadata.c b/src/shared/metadata.c index 168d574..f0922b3 100644 --- a/src/shared/metadata.c +++ b/src/shared/metadata.c @@ -211,13 +211,22 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) { bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrPolicy policy, mode_t* out_mode) { + const mode_t special_bits = (mode_t)(S_ISUID | S_ISGID | S_ISVTX); const mode_t execute_bits = S_IXUSR | S_IXGRP | S_IXOTH; if (policy.perms) { /* rsync --perms copies the source's permission and special bits exactly, * including group/other write and setuid/setgid/sticky. The kernel may * still clear setgid when the receiver is not in the file's group; the - * caller logs a failed chmod rather than silently masking the bits here. */ - *out_mode = source_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777); + * caller logs a failed chmod rather than silently masking the bits here. + * Setuid/setgid/sticky are super-user activities: when the connection did + * not permit them (SUPER_MODE_OFF / --no-super) they are stripped, so a + * client can never install a privileged bit on a receiver that forbade + * super-user activities. This also covers bits introduced by --chmod, + * whose result is fed in as source_mode. */ + mode_t bits = source_mode & (mode_t)(special_bits | 0777); + if (!policy.super_permitted) + bits &= ~special_bits; + *out_mode = bits; return true; } if (policy.executability) { @@ -227,8 +236,11 @@ bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrP * execute); otherwise clear every execute bit. This runs on the * destination-derived base (pre-existing dest mode, or source&~umask for a * new file), and leaves the special bits untouched. --perms wins when both - * are set (handled above). */ - mode_t base = current_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777); + * are set (handled above). The destination's own special bits survive + * unless super-user activities are forbidden. */ + mode_t base = current_mode & (mode_t)(special_bits | 0777); + if (!policy.super_permitted) + base &= ~special_bits; if (source_mode & 0111) *out_mode = base | ((base & 0444) >> 2); else @@ -240,12 +252,13 @@ bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrP } FileAttrPolicy file_attr_policy_from_config(const Config* config) { - FileAttrPolicy policy = {false, false, false, false}; + FileAttrPolicy policy = {0}; if (config) { policy.perms = config->preserve_perms; policy.times = config->preserve_times; policy.atimes = config->preserve_atimes; policy.executability = config->use_executability; + policy.super_permitted = privilege_super_mode_permitted(config->super_mode); } return policy; } @@ -314,6 +327,8 @@ bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadat transfer never fails over it. */ if (policy.perms) { mode_t link_mode = metadata->mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777); + if (!policy.super_permitted) + link_mode &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); if (fchmodat(parent_fd, leaf, link_mode, AT_SYMLINK_NOFOLLOW) != 0 && errno != EOPNOTSUPP && errno != ENOTSUP && errno != ENOSYS) { log_message(LOG_LEVEL_DEBUG, "Could not set symlink mode on %s: %s", path, strerror(errno)); diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 00be064..0212616 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -332,6 +332,21 @@ class TestDaemonModuleSelection: assert not missing, f"missing: {missing[:5]}" assert not mismatches, f"mismatch: {mismatches[:5]}" + def test_daemon_new_dirs_not_world_writable(self, daemon): + """The daemon must not force umask 0: implied parent directories created + without -p are the source default (0755 under a 022 umask), never + world-writable 0777.""" + sub = os.path.join(FILES_MODULE, "umask_check") + shutil.rmtree(sub, ignore_errors=True) + os.makedirs(sub, exist_ok=True) + result = _push("127.0.0.1::files/umask_check", daemon.port) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(sub, SOURCE_DIR) + nested = os.path.join(received, "nested") + assert os.path.isdir(nested), f"nested dir missing under {received}" + mode = stat.S_IMODE(os.stat(nested).st_mode) + assert (mode & 0o022) == 0, f"implied directory is group/other writable: {oct(mode)}" + class TestDaemonRejection: def _tree_files(self): diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index 1a2c05c..d3371ab 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -2606,35 +2606,30 @@ class TestTimeoutAndAllocLimits: mismatches, missing = verify_transfer(source, received) assert not missing and not mismatches - def test_temp_dir_cross_filesystem_fallback(self, shared_server): - """A confined relative --temp-dir that resolves (via a symlink under the - destination root) to another filesystem must fall back to a non-atomic - copy instead of aborting (rsync parity). Skipped when no second - filesystem is available.""" - shm = "/dev/shm" - if not os.path.isdir(shm): - pytest.skip("/dev/shm not available") - if os.stat(shm).st_dev == os.stat(TEST_DATA_DIR).st_dev: - pytest.skip("/dev/shm is on the same filesystem as the test data") - scratch = os.path.join(shm, f"fastsync_tmp_{os.getpid()}") - shutil.rmtree(scratch, ignore_errors=True) - os.makedirs(scratch) + def test_temp_dir_symlink_escape_rejected(self, shared_server): + """A symlink planted inside the destination root pointing outside it + must not redirect receiver scratch files: --temp-dir= is + refused and nothing is written at the link target. An in-root symlink + (e.g. to a mount point that stays inside the authorized root) is still + accepted, preserving the engine's EXDEV cross-filesystem fallback.""" + source, dest = self._seed("tempdir_escape_src") + outside = "/tmp/fastsync_tempdir_escape_%d" % os.getpid() + shutil.rmtree(outside, ignore_errors=True) + os.makedirs(outside) + link = os.path.join(dest, "escape_scratch") + if os.path.lexists(link): + os.unlink(link) + os.symlink(outside, link) try: - source, dest = self._seed("tempdir_xdev_src") - # The receiver resolves a relative temp dir under the destination - # root; a symlink there points the scratch at the second filesystem. - link = os.path.join(dest, "xdev_scratch") - os.symlink(scratch, link) - result, _ = run_client(source, dest, flags=["--temp-dir", "xdev_scratch"], + result, _ = run_client(source, dest, flags=["--temp-dir", "escape_scratch"], port=shared_server.port) - assert result.returncode == 0, f"cross-fs temp-dir failed: {result.stderr[:300]}" + assert result.returncode != 0, "an escaping --temp-dir symlink must be refused" received = get_dest_received_dir(dest, source) - mismatches, missing = verify_transfer(source, received) - assert not missing, f"Missing: {missing}" - assert not mismatches, f"Mismatch: {mismatches}" - assert os.listdir(scratch) == [], "temp files left behind in the cross-fs scratch" + assert not os.path.exists(os.path.join(received, "f.txt")), \ + "the receiver must not fall back to writing the file" + assert os.listdir(outside) == [], "receiver wrote outside the authorized root" finally: - shutil.rmtree(scratch, ignore_errors=True) + shutil.rmtree(outside, ignore_errors=True) class TestRemoteOptionTransport: @@ -5782,6 +5777,35 @@ class TestStandaloneSuperDefault: "standalone server accepted --copy-as without --allow-super" ) + @pytest.mark.skipif( + os.geteuid() != 0, + reason="root triggers the SUPER_MODE_OFF default and can create setuid sources", + ) + def test_special_bits_masked_without_allow_super(self): + """A root standalone server without --allow-super forces SUPER_MODE_OFF, + so client-supplied setuid/setgid/sticky bits must be stripped even under + -p (they are super-user activities just like device-node creation).""" + source = os.path.join(TEST_DATA_DIR, "super_default_mode_src") + dest = os.path.join(TEST_DATA_DIR, "super_default_mode_dst") + clean_dir(source) + clean_dir(dest) + src_file = os.path.join(source, "priv.sh") + with open(src_file, "wb") as f: + f.write(b"#!/bin/sh\necho hi\n") + os.chmod(src_file, 0o4755) + server = ServerManager() + server.start() # deliberately no --allow-super -> SUPER_MODE_OFF as root + try: + result, _ = run_client(source, dest, flags=["-p"], port=server.port) + finally: + server.stop() + assert result.returncode == 0, f"exit {result.returncode}: {(result.stderr or '')[:200]}" + received = get_dest_received_dir(dest, source) + mode = stat.S_IMODE(os.stat(os.path.join(received, "priv.sh")).st_mode) + assert (mode & (stat.S_ISUID | stat.S_ISGID | stat.S_ISVTX)) == 0, \ + f"--no-super receiver kept a privileged bit: {oct(mode)}" + assert (mode & 0o777) == 0o755, f"ordinary permission bits lost: {oct(mode)}" + @pytest.mark.skipif(os.geteuid() != 0, reason="root can create the source device node") def test_devices_skipped_without_allow_super(self): """Root standalone server without --allow-super must skip device-node diff --git a/tests/test_file.c b/tests/test_file.c index 611b4cc..0546365 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -377,6 +377,70 @@ static void test_file_save_to_disk_temp_dir_confined() { rmdir(outside); } +/* A client-planted symlink under the receive root must never redirect the + * --temp-dir scratch directory outside the authorized root: the REAL path of + * the opened dir is checked. An in-root symlink (the EXDEV cross-filesystem + * case) must still be accepted. */ +static void test_file_open_temp_dir_symlink_confinement() { + const char* root = "test_tempdir_link_root"; + const char* outside = "test_tempdir_link_outside"; + char root_abs[PATH_MAX]; + char outside_abs[PATH_MAX]; + unlink("test_tempdir_link_root/escape"); + unlink("test_tempdir_link_root/inside_link"); + rmdir("test_tempdir_link_root/scratch"); + rmdir(root); + rmdir(outside); + EXPECT_EQ_INT(mkdir(root, 0755), 0); + EXPECT_EQ_INT(mkdir(outside, 0755), 0); + EXPECT_NOT_NULL(realpath(root, root_abs)); + EXPECT_NOT_NULL(realpath(outside, outside_abs)); + int root_fd = open(root_abs, O_RDONLY | O_DIRECTORY | O_CLOEXEC); + EXPECT_TRUE(root_fd >= 0); + // cppcheck-suppress knownConditionTrueFalse + if (root_fd < 0) { + rmdir(root); + rmdir(outside); + return; + } + EXPECT_TRUE(utils_set_authorized_root(root_fd, root_abs)); + + /* An existing in-root scratch dir opens normally. */ + char* scratch = path_cat(root_abs, "scratch"); + EXPECT_NOT_NULL(scratch); + EXPECT_EQ_INT(mkdir(scratch, 0755), 0); + int scratch_fd = file_open_temp_dir(scratch); + EXPECT_TRUE(scratch_fd >= 0); + if (scratch_fd >= 0) + close(scratch_fd); + + /* A symlink whose target is outside the root is refused. */ + char* escape = path_cat(root_abs, "escape"); + EXPECT_NOT_NULL(escape); + EXPECT_EQ_INT(symlink(outside_abs, escape), 0); + EXPECT_EQ_INT(file_open_temp_dir(escape), -1); + + /* A symlink that stays inside the root is accepted (EXDEV fallback path). */ + char* inside_link = path_cat(root_abs, "inside_link"); + EXPECT_NOT_NULL(inside_link); + EXPECT_EQ_INT(symlink(scratch, inside_link), 0); + int link_fd = file_open_temp_dir(inside_link); + EXPECT_TRUE(link_fd >= 0); + if (link_fd >= 0) + close(link_fd); + + free(inside_link); + free(escape); + free(scratch); + utils_set_authorized_root(-1, NULL); + close(root_fd); + unlink("test_tempdir_link_root/escape"); + unlink("test_tempdir_link_root/inside_link"); + rmdir("test_tempdir_link_root/scratch"); + rmdir(root); + rmdir(outside); +} + /* Issue #251: file_save_to_disk_full must distinguish receiver-side skips (--existing/--ignore-existing/--update) from real writes so the sender can decide whether --remove-source-files may unlink its source. */ @@ -1040,8 +1104,8 @@ static void test_atomic_no_perms_preserves_destination_mode() { /* No -p/-E: the pre-existing 0640 survives the atomic overwrite. */ bool ok = file_to_disk_secure_attrs(path, "data", 4, false, false, false, &m, - (FileAttrPolicy){false, false, false, false}, false, false, - false, NULL, false, false, NULL); + (FileAttrPolicy){false, false, false, false, true}, false, + false, false, NULL, false, false, NULL); EXPECT_TRUE(ok); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -1049,8 +1113,8 @@ static void test_atomic_no_perms_preserves_destination_mode() { /* -p: the source mode wins. */ ok = file_to_disk_secure_attrs(path, "data2", 5, false, false, false, &m, - (FileAttrPolicy){true, true, false, false}, false, false, false, - NULL, false, false, NULL); + (FileAttrPolicy){true, true, false, false, true}, false, false, + false, NULL, false, false, NULL); EXPECT_TRUE(ok); EXPECT_EQ_INT(stat(path, &st), 0); EXPECT_EQ_INT((int)(st.st_mode & 0777), 0755); @@ -1060,8 +1124,8 @@ static void test_atomic_no_perms_preserves_destination_mode() { source 0755 gives 0750, not 0751 and not the scratch 0711. */ EXPECT_EQ_INT(chmod(path, 0640), 0); ok = file_to_disk_secure_attrs(path, "data3", 6, false, false, false, &m, - (FileAttrPolicy){false, false, false, true}, false, false, false, - NULL, false, false, NULL); + (FileAttrPolicy){false, false, false, true, true}, false, false, + false, NULL, false, false, NULL); EXPECT_TRUE(ok); EXPECT_EQ_INT(stat(path, &st), 0); EXPECT_EQ_INT((int)(st.st_mode & 0777), 0750); @@ -1073,8 +1137,8 @@ static void test_atomic_no_perms_preserves_destination_mode() { const char* fresh = "test_attr_split_fresh.txt"; unlink(fresh); ok = file_to_disk_secure_attrs(fresh, "data", 4, false, false, false, &m, - (FileAttrPolicy){false, false, false, false}, false, false, false, - NULL, false, false, NULL); + (FileAttrPolicy){false, false, false, false, true}, false, false, + false, NULL, false, false, NULL); EXPECT_TRUE(ok); EXPECT_EQ_INT(stat(fresh, &st), 0); EXPECT_EQ_INT((int)(st.st_mode & 0777), (int)(m.mode & 0777 & ~(mode_t)file_process_umask())); @@ -1083,8 +1147,8 @@ static void test_atomic_no_perms_preserves_destination_mode() { /* Without any metadata the historical fixed 0644 default still applies. */ unlink(fresh); ok = file_to_disk_secure_attrs(fresh, "data", 4, false, false, false, NULL, - (FileAttrPolicy){false, false, false, false}, false, false, false, - NULL, false, false, NULL); + (FileAttrPolicy){false, false, false, false, true}, false, false, + false, NULL, false, false, NULL); EXPECT_TRUE(ok); EXPECT_EQ_INT(stat(fresh, &st), 0); EXPECT_EQ_INT((int)(st.st_mode & 0777), 0644); @@ -1104,8 +1168,8 @@ static void test_new_file_mode_honors_source_and_umask() { m.gid = getegid(); bool ok = file_to_disk_secure_attrs(path, "x", 1, false, false, false, &m, - (FileAttrPolicy){false, false, false, false}, false, false, - false, NULL, false, false, NULL); + (FileAttrPolicy){false, false, false, false, true}, false, + false, false, NULL, false, false, NULL); EXPECT_TRUE(ok); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -1663,8 +1727,8 @@ static void test_file_write_to_disk_partial_retention() { m.atime_valid = false; m.crtime_valid = false; bool ok = file_to_disk_secure_attrs(path, content, strlen(content), false, false, true, &m, - (FileAttrPolicy){true, true, false, false}, false, false, - false, NULL, false, true, NULL); + (FileAttrPolicy){true, true, false, false, true}, false, + false, false, NULL, false, true, NULL); EXPECT_FALSE(ok); /* the write itself succeeded, but metadata restore failed */ /* Retained: the already-written temp now sits at the destination path. */ int fd = open(path, O_RDONLY); @@ -1683,8 +1747,8 @@ static void test_file_write_to_disk_partial_retention() { /* Same failure with keep_partial=false: temp is unlinked, nothing retained. */ ok = file_to_disk_secure_attrs(path, content, strlen(content), false, false, true, &m, - (FileAttrPolicy){true, true, false, false}, false, false, false, - NULL, false, false, NULL); + (FileAttrPolicy){true, true, false, false, true}, false, false, + false, NULL, false, false, NULL); EXPECT_FALSE(ok); EXPECT_TRUE(access(path, F_OK) == -1); } @@ -2264,6 +2328,7 @@ void test_file() { test_file_save_to_disk_ignore_existing_entry_types(); test_file_save_to_disk_partial_install(); test_file_save_to_disk_temp_dir_confined(); + test_file_open_temp_dir_symlink_confinement(); test_file_save_to_disk_reports_skips(); test_file_write_to_disk_sparse_preserves_holes(); test_file_write_to_disk_partial_retention(); diff --git a/tests/test_metadata.c b/tests/test_metadata.c index e67604f..b8599b2 100644 --- a/tests/test_metadata.c +++ b/tests/test_metadata.c @@ -339,7 +339,7 @@ static void test_file_restore_metadata_applies_atime() { m.crtime_sec = 0; m.crtime_nsec = 0; - file_restore_metadata(path, &m, (FileAttrPolicy){true, true, true, false}); + file_restore_metadata(path, &m, (FileAttrPolicy){true, true, true, false, true}); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -378,7 +378,7 @@ static void test_file_restore_metadata() { .atime_valid = false, .crtime_valid = false}; - file_restore_metadata(path, &m, (FileAttrPolicy){true, true, false, false}); + file_restore_metadata(path, &m, (FileAttrPolicy){true, true, false, false, true}); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -395,7 +395,7 @@ static void test_file_restore_executability_only() { FileMetadata m = { .mode = 0751, .uid = getuid(), .gid = getgid(), .mtime_sec = 0, .mtime_nsec = 0}; - file_restore_metadata(path, &m, (FileAttrPolicy){false, false, false, true}); + file_restore_metadata(path, &m, (FileAttrPolicy){false, false, false, true, true}); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -409,7 +409,7 @@ static void test_directory_restore_executability_only() { FileMetadata m = { .mode = 0755, .uid = getuid(), .gid = getgid(), .mtime_sec = 0, .mtime_nsec = 0}; - file_restore_metadata(path, &m, (FileAttrPolicy){false, false, false, true}); + file_restore_metadata(path, &m, (FileAttrPolicy){false, false, false, true, true}); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -437,7 +437,7 @@ static void test_file_restore_executability_rsync_rule() { EXPECT_TRUE(file_write_to_disk(path, "x", 1, false, false)); EXPECT_EQ_INT(chmod(path, cases[i].dest), 0); FileMetadata m = {.mode = cases[i].src, .uid = getuid(), .gid = getgid()}; - file_restore_metadata(path, &m, (FileAttrPolicy){false, false, false, true}); + file_restore_metadata(path, &m, (FileAttrPolicy){false, false, false, true, true}); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); EXPECT_EQ_INT(st.st_mode & 0777, cases[i].want); @@ -453,34 +453,60 @@ static void test_file_restore_executability_rsync_rule() { * bits from the destination and --perms wins when both are set. */ static void test_metadata_mode_for_policy() { mode_t out = 0xdead; - EXPECT_FALSE( - metadata_mode_for_policy(0777, 0644, (FileAttrPolicy){false, false, false, false}, &out)); + EXPECT_FALSE(metadata_mode_for_policy(0777, 0644, + (FileAttrPolicy){false, false, false, false, true}, &out)); EXPECT_EQ_INT((int)out, 0xdead); /* untouched when no change is requested */ - EXPECT_TRUE( - metadata_mode_for_policy(0777, 0644, (FileAttrPolicy){true, false, false, false}, &out)); + EXPECT_TRUE(metadata_mode_for_policy(0777, 0644, + (FileAttrPolicy){true, false, false, false, true}, &out)); EXPECT_EQ_INT((int)(out & 0777), 0777); /* group/other write is preserved */ mode_t specials = (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0672); - EXPECT_TRUE( - metadata_mode_for_policy(specials, 0644, (FileAttrPolicy){true, false, false, false}, &out)); + EXPECT_TRUE(metadata_mode_for_policy(specials, 0644, + (FileAttrPolicy){true, false, false, false, true}, &out)); EXPECT_EQ_INT((int)(out & (S_ISUID | S_ISGID | S_ISVTX | 0777)), (int)(S_ISUID | S_ISGID | S_ISVTX | 0672)); + /* SUPER_MODE_OFF: the special bits are stripped even under -p (this also + * covers bits introduced by --chmod, whose result is fed in as source_mode), + * while the ordinary permission bits are still copied. */ + EXPECT_TRUE(metadata_mode_for_policy(specials, 0644, + (FileAttrPolicy){true, false, false, false, false}, &out)); + EXPECT_EQ_INT((int)(out & (S_ISUID | S_ISGID | S_ISVTX)), 0); + EXPECT_EQ_INT((int)(out & 0777), 0672); + EXPECT_TRUE(metadata_mode_for_policy(04755, 0644, + (FileAttrPolicy){true, false, false, false, false}, &out)); + EXPECT_EQ_INT((int)(out & 07777), 0755); + /* The same holds for a special bit introduced by --chmod=+s. */ + mode_t chmodded = 0; + EXPECT_TRUE(chmod_apply(0755, "u+s", &chmodded)); + EXPECT_TRUE(metadata_mode_for_policy(chmodded, 0644, + (FileAttrPolicy){true, false, false, false, false}, &out)); + EXPECT_EQ_INT((int)(out & 07777), 0755); + + /* -E: a destination's own special bits survive unless super-user activities + * are forbidden, in which case they are stripped from the derived base. */ + EXPECT_TRUE(metadata_mode_for_policy(0755, (mode_t)(S_ISUID | 0750), + (FileAttrPolicy){false, false, false, true, true}, &out)); + EXPECT_EQ_INT((int)(out & (S_ISUID | 0777)), (int)(S_ISUID | 0750)); + EXPECT_TRUE(metadata_mode_for_policy(0755, (mode_t)(S_ISUID | 0750), + (FileAttrPolicy){false, false, false, true, false}, &out)); + EXPECT_EQ_INT((int)(out & (S_ISUID | 0777)), 0750); + /* -E: exec bits derive from the DESTINATION's read bits. */ - EXPECT_TRUE( - metadata_mode_for_policy(0755, 0644, (FileAttrPolicy){false, false, false, true}, &out)); + EXPECT_TRUE(metadata_mode_for_policy(0755, 0644, + (FileAttrPolicy){false, false, false, true, true}, &out)); EXPECT_EQ_INT((int)(out & 0777), 0755); - EXPECT_TRUE( - metadata_mode_for_policy(0644, 0755, (FileAttrPolicy){false, false, false, true}, &out)); + EXPECT_TRUE(metadata_mode_for_policy(0644, 0755, + (FileAttrPolicy){false, false, false, true, true}, &out)); EXPECT_EQ_INT((int)(out & 0777), 0644); - EXPECT_TRUE( - metadata_mode_for_policy(0755, 0600, (FileAttrPolicy){false, false, false, true}, &out)); + EXPECT_TRUE(metadata_mode_for_policy(0755, 0600, + (FileAttrPolicy){false, false, false, true, true}, &out)); EXPECT_EQ_INT((int)(out & 0777), 0700); /* --perms wins over -E when both are set. */ EXPECT_TRUE( - metadata_mode_for_policy(0700, 0644, (FileAttrPolicy){true, false, false, true}, &out)); + metadata_mode_for_policy(0700, 0644, (FileAttrPolicy){true, false, false, true, true}, &out)); EXPECT_EQ_INT((int)(out & 0777), 0700); } @@ -493,8 +519,8 @@ static void test_new_file_mode_from_source_and_umask() { mode_t want = (mode_t)(0751 & 0777 & ~(mode_t)file_process_umask()); bool ok = file_to_disk_secure_attrs(path, "x", 1, false, false, false, &m, - (FileAttrPolicy){false, false, false, false}, false, false, - false, NULL, false, false, NULL); + (FileAttrPolicy){false, false, false, false, true}, false, + false, false, NULL, false, false, NULL); EXPECT_TRUE(ok); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -503,8 +529,8 @@ static void test_new_file_mode_from_source_and_umask() { /* -E on top of the source&~umask base (src 0751, umask 022 -> 0751). */ ok = file_to_disk_secure_attrs(path, "x", 1, false, false, false, &m, - (FileAttrPolicy){false, false, false, true}, false, false, false, - NULL, false, false, NULL); + (FileAttrPolicy){false, false, false, true, true}, false, false, + false, NULL, false, false, NULL); EXPECT_TRUE(ok); EXPECT_EQ_INT(stat(path, &st), 0); mode_t want_e = @@ -529,7 +555,7 @@ static void test_file_restore_attribute_split() { .crtime_valid = false}; /* times only: mtime changes, mode stays 0640. */ - file_restore_metadata(path, &m, (FileAttrPolicy){false, true, false, false}); + file_restore_metadata(path, &m, (FileAttrPolicy){false, true, false, false, true}); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); EXPECT_EQ_INT(st.st_mode & 0777, 0640); @@ -543,7 +569,7 @@ static void test_file_restore_attribute_split() { FileMetadata m2 = m; m2.mode = 0700; m2.mtime_sec = 1600000000; - file_restore_metadata(path, &m2, (FileAttrPolicy){false, false, false, false}); + file_restore_metadata(path, &m2, (FileAttrPolicy){false, false, false, false, true}); EXPECT_EQ_INT(stat(path, &st), 0); EXPECT_EQ_INT(st.st_mode & 0777, 0640); EXPECT_EQ_INT((int)st.st_mtime, 1000000000); @@ -569,6 +595,11 @@ static void test_file_attr_policy_from_config() { EXPECT_TRUE(p.times); EXPECT_TRUE(p.atimes); EXPECT_TRUE(p.executability); + /* Default super mode (AUTO) permits special bits. */ + EXPECT_TRUE(p.super_permitted); + c->super_mode = SUPER_MODE_OFF; + p = file_attr_policy_from_config(c); + EXPECT_FALSE(p.super_permitted); config_delete(c); } @@ -582,8 +613,8 @@ static void test_perms_preserves_special_bits() { .mode = (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0755), .uid = getuid(), .gid = getgid()}; bool ok = file_to_disk_secure_attrs(path, "x", 1, false, false, false, &m, - (FileAttrPolicy){true, false, false, false}, false, false, - false, NULL, false, false, NULL); + (FileAttrPolicy){true, false, false, false, true}, false, + false, false, NULL, false, false, NULL); EXPECT_TRUE(ok); struct stat st; EXPECT_EQ_INT(stat(path, &st), 0); @@ -670,7 +701,8 @@ static void test_file_restore_symlink_metadata() { /* Positive path: a non-omitted apply stamps the link's own mtime. */ FileMetadata applied = {.mtime_sec = 1000000000, .mtime_nsec = 0}; - file_restore_symlink_metadata(link, &applied, (FileAttrPolicy){false, true, false, false}, false); + file_restore_symlink_metadata(link, &applied, (FileAttrPolicy){false, true, false, false, true}, + false); struct stat st; EXPECT_EQ_INT(lstat(link, &st), 0); EXPECT_TRUE(S_ISLNK(st.st_mode)); @@ -679,7 +711,8 @@ static void test_file_restore_symlink_metadata() { /* -J: a different time must be left untouched. */ FileMetadata newer = {.mtime_sec = 1234567890, .mtime_nsec = 0}; - file_restore_symlink_metadata(link, &newer, (FileAttrPolicy){false, true, false, false}, true); + file_restore_symlink_metadata(link, &newer, (FileAttrPolicy){false, true, false, false, true}, + true); EXPECT_EQ_INT(lstat(link, &st), 0); EXPECT_EQ_INT((int)st.st_mtime, (int)t1); if (symlink_times_supported) @@ -723,14 +756,14 @@ static void test_file_restore_metadata_fd_attribute_split() { /* perms-only: mode applied, mtime untouched. */ EXPECT_EQ_INT(fstat(fd, &before), 0); - EXPECT_TRUE(file_restore_metadata_fd(fd, &m, (FileAttrPolicy){true, false, false, false})); + EXPECT_TRUE(file_restore_metadata_fd(fd, &m, (FileAttrPolicy){true, false, false, false, true})); EXPECT_EQ_INT(fstat(fd, &st), 0); EXPECT_EQ_INT(st.st_mode & 0777, 0755); EXPECT_EQ_INT((int)st.st_mtime, (int)before.st_mtime); /* times-only: mtime applied, mode untouched. */ EXPECT_EQ_INT(chmod(path, 0600), 0); - EXPECT_TRUE(file_restore_metadata_fd(fd, &m, (FileAttrPolicy){false, true, false, false})); + EXPECT_TRUE(file_restore_metadata_fd(fd, &m, (FileAttrPolicy){false, true, false, false, true})); EXPECT_EQ_INT(fstat(fd, &st), 0); EXPECT_EQ_INT(st.st_mode & 0777, 0600); EXPECT_EQ_INT((int)st.st_mtime, 1234567890); @@ -740,7 +773,7 @@ static void test_file_restore_metadata_fd_attribute_split() { {.tv_sec = 1000000000, .tv_nsec = 0}}; EXPECT_EQ_INT(futimens(fd, reset), 0); EXPECT_EQ_INT(fstat(fd, &before), 0); - EXPECT_TRUE(file_restore_metadata_fd(fd, &m, (FileAttrPolicy){false, false, true, false})); + EXPECT_TRUE(file_restore_metadata_fd(fd, &m, (FileAttrPolicy){false, false, true, false, true})); EXPECT_EQ_INT(fstat(fd, &st), 0); EXPECT_EQ_INT((int)st.st_atime, 999999999); EXPECT_EQ_INT((int)st.st_mtime, (int)before.st_mtime); @@ -753,7 +786,8 @@ static void test_file_restore_metadata_fd_attribute_split() { m2.mode = 0700; m2.mtime_sec = 1600000000; m2.atime_sec = 1700000000; - EXPECT_TRUE(file_restore_metadata_fd(fd, &m2, (FileAttrPolicy){false, false, false, false})); + EXPECT_TRUE( + file_restore_metadata_fd(fd, &m2, (FileAttrPolicy){false, false, false, false, true})); EXPECT_EQ_INT(fstat(fd, &st), 0); EXPECT_EQ_INT(st.st_mode & 0777, 0640); EXPECT_EQ_INT((int)st.st_mtime, 1000000000); diff --git a/tests/test_xattr.c b/tests/test_xattr.c index 4e268e5..10447a3 100644 --- a/tests/test_xattr.c +++ b/tests/test_xattr.c @@ -248,7 +248,7 @@ static void test_link_copy_fallback_preserves_xattrs() { m.crtime_valid = false; bool ok = file_to_disk_secure_link_attrs(dest, basis_dir, "payload", 7, false, &m, - (FileAttrPolicy){true, true, false, false}, false, + (FileAttrPolicy){true, true, false, false, true}, false, xattrs, true, NULL); xattr_list_free(xattrs); EXPECT_TRUE(ok); @@ -369,7 +369,7 @@ static void test_fake_super_restore() { } /* No xattr present yet: restore is a silent no-op (returns false, no crash). */ - FileAttrPolicy policy = {true, true, false, false}; + FileAttrPolicy policy = {true, true, false, false, true}; EXPECT_FALSE(fake_super_restore_fd(fd, policy)); fake_super_store_fd(fd, 1001, 1002, 0751, 1700000000, 123456789); @@ -423,7 +423,7 @@ static void test_fake_super_no_real_chown() { fake_super_store_fd(fd, 12345, 12346, 0755, 1700000000, 0); Config* c = config_create(); - FileAttrPolicy policy = {true, true, false, false}; + FileAttrPolicy policy = {true, true, false, false, true}; EXPECT_NOT_NULL(c); /* The strongest ownership request available plus permitted super mode. */ c->preserve_owner = true; -- 2.54.0 From 379f127859409e39e664dae43c7a518d1deca80d Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 18:50:55 +0200 Subject: [PATCH 15/68] test: add client timeouts and de-flake default-port test --- tests/integration/common.py | 67 +++++++++++++++++++++++------ tests/integration/test_preflight.py | 61 +++++++++++++++++++++----- 2 files changed, 105 insertions(+), 23 deletions(-) diff --git a/tests/integration/common.py b/tests/integration/common.py index 822fc4e..b9eb14b 100644 --- a/tests/integration/common.py +++ b/tests/integration/common.py @@ -20,6 +20,12 @@ CLIENT_CMD = [os.path.join(BUILD_DIR, "client")] _WORKER = os.environ.get("PYTEST_XDIST_WORKER") TEST_DATA_DIR = os.path.join(PROJECT_ROOT, f"test_data-{_WORKER}" if _WORKER else "test_data") +# Default wall-clock budget for a short-lived client invocation. Every client +# is expected to finish well within this; the bound exists so a hung client +# fails the test instead of stalling the whole CI run indefinitely. Callers +# that legitimately need longer can pass an explicit ``timeout``. +CLIENT_TIMEOUT = 180 + class ServerManager: """Manages a long-lived server process. Reuses across test cases.""" @@ -137,7 +143,40 @@ class CountingProxy: return result -def run_client(source_dir, dest_dir, flags=None, port=None, extra_args=None): +def _run_client_cmd(cmd, timeout): + """Run one client command, returning ``(result, duration)``. + + On timeout the client is killed and a result-like ``CompletedProcess`` with + a non-zero returncode is returned instead of raising, so callers keep the + established ``(result, duration)`` contract and the failure carries the + command plus whatever output was captured for diagnosis. + """ + start = time.monotonic() + try: + result = subprocess.run(cmd, text=True, capture_output=True, timeout=timeout) + except subprocess.TimeoutExpired as exc: + duration = time.monotonic() - start + stdout = exc.stdout or "" + stderr = exc.stderr or "" + if isinstance(stdout, bytes): + stdout = stdout.decode(errors="replace") + if isinstance(stderr, bytes): + stderr = stderr.decode(errors="replace") + diagnostic = ( + f"client timed out after {timeout}s\n" + f"command: {cmd!r}\n" + f"--- captured stdout ---\n{stdout}\n" + f"--- captured stderr ---\n{stderr}" + ) + result = subprocess.CompletedProcess(cmd, returncode=-1, + stdout=stdout, stderr=diagnostic) + return result, duration + duration = time.monotonic() - start + return result, duration + + +def run_client(source_dir, dest_dir, flags=None, port=None, extra_args=None, + timeout=CLIENT_TIMEOUT): """Run the client and return (result, duration).""" cmd = CLIENT_CMD + ["--source-dir", source_dir, "--dest-dir", dest_dir, "--save-to-disk"] if port: @@ -146,23 +185,18 @@ def run_client(source_dir, dest_dir, flags=None, port=None, extra_args=None): cmd += flags if extra_args: cmd += extra_args - start = time.monotonic() - result = subprocess.run(cmd, text=True, capture_output=True) - duration = time.monotonic() - start - return result, duration + return _run_client_cmd(cmd, timeout) -def run_client_posix(source_dir, dest_dir, flags=None, port=None): +def run_client_posix(source_dir, dest_dir, flags=None, port=None, + timeout=CLIENT_TIMEOUT): """Run the client with positional args (rsync-style).""" cmd = CLIENT_CMD + [source_dir, dest_dir, "--save-to-disk"] if port: cmd += ["--server-port", str(port)] if flags: cmd += flags - start = time.monotonic() - result = subprocess.run(cmd, text=True, capture_output=True) - duration = time.monotonic() - start - return result, duration + return _run_client_cmd(cmd, timeout) def generate_test_files(source_dir, full=False): @@ -238,8 +272,17 @@ def make_result(name, success, duration=None, error=""): def get_dest_received_dir(dest_dir, source_dir): - """Get the path where received files land inside dest_dir.""" - return os.path.join(dest_dir, os.path.abspath(source_dir).lstrip(os.sep)) + """Get the path where received files land inside dest_dir. + + FastSync mirrors the absolute source path below the receive root with the + leading root separator removed. Strip that separator explicitly rather + than with ``str.lstrip(os.sep)``: ``lstrip`` removes a *set* of characters + rather than a path prefix, which is not the same operation. + """ + abs_source = os.path.abspath(source_dir) + if abs_source.startswith(os.sep): + abs_source = abs_source[len(os.sep):] + return os.path.join(dest_dir, abs_source) def _find_free_port(): diff --git a/tests/integration/test_preflight.py b/tests/integration/test_preflight.py index ad47d3b..ab62da7 100644 --- a/tests/integration/test_preflight.py +++ b/tests/integration/test_preflight.py @@ -1,23 +1,57 @@ """CLI validation and preflight checks.""" +import socket import subprocess import sys import os import shutil +import time import pytest sys.path.insert(0, os.path.dirname(__file__)) -from common import BUILD_DIR, CLIENT_CMD, SERVER_CMD, TEST_DATA_DIR, run_client, verify_transfer +from common import ( + BUILD_DIR, + CLIENT_CMD, + CLIENT_TIMEOUT, + SERVER_CMD, + TEST_DATA_DIR, + get_dest_received_dir, + run_client, + verify_transfer, +) + +DEFAULT_PORT = 8080 + + +def _port_is_listening(host, port, timeout=0.3): + """True if something accepts a TCP connection on host:port right now.""" + try: + with socket.create_connection((host, port), timeout=timeout): + return True + except OSError: + return False + + +def _wait_for_listener(host, port, timeout=5.0): + """Poll host:port until a listener accepts, or the deadline passes.""" + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + if _port_is_listening(host, port): + return True + time.sleep(0.05) + return False class TestHelp: def test_client_help(self): - r = subprocess.run(CLIENT_CMD + ["--help"], capture_output=True, text=True) + r = subprocess.run(CLIENT_CMD + ["--help"], capture_output=True, + text=True, timeout=CLIENT_TIMEOUT) assert r.returncode == 0 assert "Usage:" in r.stdout assert "SSH transport" in r.stdout def test_server_help(self): - r = subprocess.run(SERVER_CMD + ["--help"], capture_output=True, text=True) + r = subprocess.run(SERVER_CMD + ["--help"], capture_output=True, + text=True, timeout=CLIENT_TIMEOUT) assert r.returncode == 0 assert "Usage:" in r.stdout @@ -67,19 +101,24 @@ class TestServerPort: def test_default_port(self): """Server should start on default port 8080.""" + if _port_is_listening("127.0.0.1", DEFAULT_PORT): + pytest.skip(f"port {DEFAULT_PORT} already in use by another process") + proc = subprocess.Popen( SERVER_CMD, stdout=subprocess.DEVNULL, stderr=None, ) try: - import socket, time - time.sleep(0.5) - with socket.create_connection(("127.0.0.1", 8080), timeout=2): - pass # Port is listening - except (ConnectionRefusedError, OSError): - pytest.fail("Server not listening on default port 8080") + if not _wait_for_listener("127.0.0.1", DEFAULT_PORT, timeout=5.0): + if proc.poll() is not None and _port_is_listening("127.0.0.1", DEFAULT_PORT): + pytest.skip(f"port {DEFAULT_PORT} was taken by another process") + pytest.fail(f"Server not listening on default port {DEFAULT_PORT}") finally: proc.terminate() - proc.wait(timeout=5) + try: + proc.wait(timeout=5) + except subprocess.TimeoutExpired: + proc.kill() + proc.wait() def _seed_protocol_source(source): @@ -105,7 +144,7 @@ class TestProtocol: port=shared_server.port) assert result.returncode == 0, \ f"--protocol current run failed: {(result.stderr or result.stdout)[:400]}" - received = os.path.join(dest, os.path.abspath(source).lstrip(os.sep)) + received = get_dest_received_dir(dest, source) mismatches, missing = verify_transfer(source, received) assert not mismatches and not missing, \ f"transfer mismatch: missing={missing} mismatches={mismatches}" -- 2.54.0 From f2c89b6e7c08dc261204ae3537bd399d786385ea Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 18:52:46 +0200 Subject: [PATCH 16/68] fix: mutex leak, errno-after-free, log_perror misuse, status validation - multiprocessing: destroy mutex_progress on the dir_entries_mutex init-failure path (init >= 7); drop bogus log_perror - delete_plan: capture errno before free() in apply_deferred_path - queue/array_list: log_message instead of log_perror for non-errno conditions - protocol: reject unknown wire Status values via status_is_valid() in receive_status, receive_status_timed and the keepalive reader; declare protocol_receive_status_timed in protocol.h - protocol: %llu for unsigned long long debug counters - tests: out-of-range status rejection test --- src/shared/array_list.c | 6 +++--- src/shared/delete_plan.c | 3 ++- src/shared/multiprocessing.c | 4 +++- src/shared/protocol.c | 25 +++++++++++++++++++++++-- src/shared/protocol.h | 3 +++ src/shared/queue.c | 4 ++-- tests/test_protocol.c | 33 +++++++++++++++++++++++++++++++++ 7 files changed, 69 insertions(+), 9 deletions(-) diff --git a/src/shared/array_list.c b/src/shared/array_list.c index 7813e6d..8606538 100644 --- a/src/shared/array_list.c +++ b/src/shared/array_list.c @@ -9,7 +9,7 @@ ArrayList* array_list_create(void (*item_destroyer)(void* item)) { ArrayList* list = (ArrayList*)protocol_alloc(sizeof(ArrayList)); if (list == NULL) { - log_perror("ERROR: Could not allocate memory for array list struct"); + log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not allocate memory for array list struct"); return NULL; } @@ -47,7 +47,7 @@ static bool array_list_extend(ArrayList* array_list) { new_capacity = INITIAL_ARRAY_SIZE; void* new_items = protocol_realloc(array_list->items, new_capacity * sizeof(void*)); if (new_items == NULL) { - log_perror("ERROR: Could not reallocate memory for array list items"); + log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not reallocate memory for array list items"); return false; } array_list->items = new_items; @@ -73,7 +73,7 @@ void** array_list_to_array(const ArrayList* array_list) { } void** array = protocol_alloc(array_list->size * sizeof(void*)); if (array == NULL) { - log_perror("Could not malloc space for array from array list!"); + log_message(LOG_LEVEL_ERROR, "%s", "Could not malloc space for array from array list!"); return NULL; } memcpy(array, array_list->items, array_list->size * sizeof(void*)); diff --git a/src/shared/delete_plan.c b/src/shared/delete_plan.c index ae59bc9..58eb52e 100644 --- a/src/shared/delete_plan.c +++ b/src/shared/delete_plan.c @@ -988,10 +988,11 @@ static bool apply_deferred_path(DeletePlanSession* session, const Config* config return false; char* leaf = NULL; int parent_fd = file_open_secure_parent(full, &leaf, false); + int open_errno = errno; free(full); if (parent_fd < 0) { free(leaf); - return errno == ENOENT || errno == ENOTDIR; + return open_errno == ENOENT || open_errno == ENOTDIR; } struct stat st; if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) { diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index 307c736..60f87ee 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -86,11 +86,13 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que return context; fail: - log_perror("Error initializing synchronization objects"); + log_message(LOG_LEVEL_ERROR, "%s", "Error initializing synchronization objects"); if (context->dir_entries_mutex_init) mtx_destroy(&context->dir_entries_mutex); if (context->dir_entries) array_list_delete(context->dir_entries); + if (init >= 7) + mtx_destroy(&context->mutex_progress); if (init >= 6) cnd_destroy(&context->condition_not_empty_loader); if (init >= 5) diff --git a/src/shared/protocol.c b/src/shared/protocol.c index a78fa0a..581ebb8 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -564,6 +564,15 @@ static const char* status_to_string(Status status) { } } +/* Reject a raw wire status outside the known enum range before it is handed to + * callers, so an unknown/corrupt frame fails as a protocol error instead of + * being silently interpreted as an unexpected-but-valid verdict. STATUS_OK is + * the first enumerator and STATUS_STATS the last, so the range check accepts + * every status the protocol defines. */ +static bool status_is_valid(Status status) { + return status >= STATUS_OK && status <= STATUS_STATS; +} + /* Shared string send/receive implementation. `redact` selects whether the * payload body is written to the LOG_DEBUG_PROTO debug log: daemon auth material * (the username and the proof/signature fields) sets it so a --verbose log never @@ -647,7 +656,7 @@ bool protocol_send_data(ProtocolSession* session, const Data* data) { return false; if (!protocol_send_n_data(session, data->data, data_size)) return false; - log_debug_message(LOG_DEBUG_PROTO, "Send %lld data", data_size); + log_debug_message(LOG_DEBUG_PROTO, "Send %llu data", data_size); return true; } @@ -681,7 +690,7 @@ Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long protocol_release_memory_for_session(session, allocation_size); return NULL; } - log_debug_message(LOG_DEBUG_PROTO, "Received %lld data", size); + log_debug_message(LOG_DEBUG_PROTO, "Received %llu data", size); Data* result = data_create(data, (size_t)size); if (!result) { protocol_release_memory_for_session(session, allocation_size); @@ -791,6 +800,10 @@ bool protocol_receive_status(ProtocolSession* session, Status* status) { } if (!protocol_receive_n_data_until(session, status, sizeof(Status), deadline_ptr)) return false; + if (!status_is_valid(*status)) { + log_message(LOG_LEVEL_ERROR, "Received unknown protocol status %d", *status); + return false; + } if (!protocol_capture_error_detail(session, status, deadline_ptr, NULL)) return false; log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status)); @@ -812,6 +825,10 @@ bool protocol_receive_status_timed(ProtocolSession* session, Status* status, int deadline.tv_sec += timeout_sec; if (!protocol_receive_n_data_until(session, status, sizeof(Status), &deadline)) return false; + if (!status_is_valid(*status)) { + log_message(LOG_LEVEL_ERROR, "Received unknown protocol status %d", *status); + return false; + } if (!protocol_capture_error_detail(session, status, &deadline, NULL)) return false; log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status)); @@ -925,6 +942,10 @@ bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, Status received; if (!protocol_read_status_until(session, &received, &deadline)) return false; + if (!status_is_valid(received)) { + log_message(LOG_LEVEL_ERROR, "Received unknown protocol status %d", received); + return false; + } if (!protocol_capture_error_detail(session, &received, &deadline, abort_check)) return false; if (received == STATUS_KEEPALIVE) { diff --git a/src/shared/protocol.h b/src/shared/protocol.h index 1ecdab3..73b9132 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -269,6 +269,9 @@ bool protocol_send_int(ProtocolSession* session, int data); bool protocol_receive_int(ProtocolSession* session, int* data); bool protocol_send_status(ProtocolSession* session, Status status); bool protocol_receive_status(ProtocolSession* session, Status* status); +/* As protocol_receive_status, but with an explicit per-message deadline + * (seconds) instead of the session's configured io_timeout_sec. */ +bool protocol_receive_status_timed(ProtocolSession* session, Status* status, int timeout_sec); bool send_n_data(int file_descriptor, const void* data, size_t data_size); bool receive_n_data(int file_descriptor, void* data, size_t data_size); diff --git a/src/shared/queue.c b/src/shared/queue.c index 3ebac08..2c3de0a 100644 --- a/src/shared/queue.c +++ b/src/shared/queue.c @@ -128,7 +128,7 @@ bool queue_enqueue_multithreaded_cancel(Queue* queue, void* item, mtx_t* mutex, void* queue_dequeue(Queue* queue) { if (queue == NULL || queue_is_empty(queue)) { - log_perror("ERROR: Could not dequeue from null or empty queue."); + log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not dequeue from null or empty queue."); return NULL; } @@ -145,7 +145,7 @@ bool queue_push(Queue* queue, void* item) { void* queue_pop(Queue* queue) { if (queue == NULL || queue_is_empty(queue)) { - log_perror("ERROR: Could not pop from null or empty queue."); + log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not pop from null or empty queue."); return NULL; } diff --git a/tests/test_protocol.c b/tests/test_protocol.c index f1b3c64..2665096 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -216,6 +216,38 @@ static void test_send_receive_status() { close(p[1]); } +/* An unknown wire status outside the enum range must be rejected as a protocol + * error instead of being handed to the caller as an unexpected verdict. The + * last known enumerator (STATUS_STATS) must still be accepted, proving the + * validation does not reject legitimate statuses. */ +static void test_receive_status_rejects_unknown() { + int p[2]; + EXPECT_EQ_INT(pipe(p), 0); + ProtocolSession session; + protocol_session_init(&session, p[0], p[1]); + + Status bogus = (Status)(STATUS_STATS + 1); + EXPECT_EQ_INT((int)write(p[1], &bogus, sizeof(bogus)), (int)sizeof(bogus)); + Status received = STATUS_OK; + EXPECT_FALSE(protocol_receive_status(&session, &received)); + + Status negative = (Status)-1; + EXPECT_EQ_INT((int)write(p[1], &negative, sizeof(negative)), (int)sizeof(negative)); + EXPECT_FALSE(protocol_receive_status(&session, &received)); + + Status top = STATUS_STATS; + EXPECT_EQ_INT((int)write(p[1], &top, sizeof(top)), (int)sizeof(top)); + EXPECT_TRUE(protocol_receive_status(&session, &received)); + EXPECT_EQ_INT((int)received, (int)STATUS_STATS); + + Status timed_bogus = (Status)(STATUS_STATS + 7); + EXPECT_EQ_INT((int)write(p[1], &timed_bogus, sizeof(timed_bogus)), (int)sizeof(timed_bogus)); + EXPECT_FALSE(protocol_receive_status_timed(&session, &received, 5)); + + close(p[0]); + close(p[1]); +} + static void test_receive_n_data_truncated() { int p[2]; EXPECT_EQ_INT(pipe(p), 0); @@ -723,6 +755,7 @@ void test_protocol() { test_send_receive_data(); test_send_receive_int(); test_send_receive_status(); + test_receive_status_rejects_unknown(); test_protocol_session_io_timeout(); test_protocol_server_io_timeout_floor(); test_send_receive_status_timed(); -- 2.54.0 From b07306d5bcb22550d6e9fb7f1d354a9153d5e128 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 18:53:07 +0200 Subject: [PATCH 17/68] fix(config): check ssh-dest allocation and enforce MAX_FILTER_RULES client-side --- src/client/client_validation.c | 10 ++++++++++ src/shared/config.c | 27 +++++++++++++++++++++++++-- tests/test_client_cli.c | 33 +++++++++++++++++++++++++++++++++ 3 files changed, 68 insertions(+), 2 deletions(-) diff --git a/src/client/client_validation.c b/src/client/client_validation.c index 37d173f..bb129d6 100644 --- a/src/client/client_validation.c +++ b/src/client/client_validation.c @@ -102,6 +102,16 @@ bool validate_config(const Config* config) { log_message(LOG_LEVEL_ERROR, "%s", invariants_error); return false; } + /* The receiver rejects a protect-rule block with more than MAX_FILTER_RULES + entries as an opaque protocol error; reject an over-limit --filter set here, + before any network I/O, with an actionable message. send_protect_entries() + re-checks the final built count because cvs-exclude / merge rules can + expand it beyond config->filters->size. */ + if (config->filters && config->filters->size > MAX_FILTER_RULES) { + log_message(LOG_LEVEL_ERROR, "too many filter rules: %d (maximum %d)", config->filters->size, + MAX_FILTER_RULES); + return false; + } /* --protocol: FastSync has exactly one wire format, so the forced version must equal the current PROTOCOL_VERSION exactly. Rejected here, before any network I/O, rather than letting the server hit its own mismatch check. */ diff --git a/src/shared/config.c b/src/shared/config.c index 2a2941b..ba3f34d 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -230,6 +230,13 @@ Config* config_create(void) { if (!config) return NULL; config_set_defaults(config); + /* config_set_defaults() dups the default server host; a failure there leaves + * server_host NULL and would crash later consumers, so fail the whole create + * (every caller already handles a NULL return). */ + if (!config->server_host) { + free(config); + return NULL; + } return config; } @@ -686,9 +693,15 @@ int config_parse_ssh_dest(Config* config) { return daemon_dest_parse_error("invalid remote destination user@host (must not be empty or " "start with '-')", dest); - config->transport = TRANSPORT_SSH; - config->ssh_destination = str_dup(dest); + char* ssh_destination = str_dup(dest); char* path = str_dup(colon + 1); + if (!ssh_destination || !path) { + free(ssh_destination); + free(path); + return daemon_dest_parse_error("out of memory parsing remote destination", dest); + } + config->transport = TRANSPORT_SSH; + config->ssh_destination = ssh_destination; free(config->receive_root_directory); config->receive_root_directory = path; return 0; @@ -1052,6 +1065,16 @@ static bool send_protect_entries(int fd, const Config* c) { log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); return false; } + /* The receiver rejects any block with more than MAX_FILTER_RULES entries as a + * protocol error; refuse to emit such a frame at all. filter_base_build() + * can expand the client rule set (cvs-exclude, merge files), so this is the + * authoritative bound, not config->filters->size. */ + if (rules->count < 0 || rules->count > MAX_FILTER_RULES) { + log_message(LOG_LEVEL_ERROR, "too many filter rules: %d (maximum %d)", rules->count, + MAX_FILTER_RULES); + filter_rule_list_free(rules); + return false; + } bool ok = send_int(fd, rules->count); for (int i = 0; ok && i < rules->count; i++) { const FilterRule* r = rules->items[i]; diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index d8eb597..951ada9 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -171,6 +171,38 @@ static void test_validate_config_unified_invariants() { config_delete(cfg); } +/* The receiver enforces MAX_FILTER_RULES on the protect-rule block and would + otherwise fail the session with an opaque protocol error. The client must + accept exactly the limit and reject one more up front, before any network + I/O, with an actionable message. */ +static void test_validate_config_filter_rule_limit() { + Config* cfg = valid_client_config(); + cfg->filters = array_list_create(free); + EXPECT_NOT_NULL(cfg->filters); + for (int i = 0; i < MAX_FILTER_RULES; i++) + EXPECT_TRUE(array_list_add(cfg->filters, str_dup("- *.tmp"))); + EXPECT_TRUE(validate_config(cfg)); /* exactly the limit is accepted */ + + FILE* log_capture = tmpfile(); + EXPECT_NOT_NULL(log_capture); + log_set_file(log_capture); + EXPECT_TRUE(array_list_add(cfg->filters, str_dup("- *.bak"))); + EXPECT_FALSE(validate_config(cfg)); /* one over the limit is rejected */ + fflush(log_capture); + rewind(log_capture); + char line[512]; + bool saw_message = false; + while (fgets(line, sizeof(line), log_capture) != NULL) { + if (strstr(line, "too many filter rules") != NULL && strstr(line, "(maximum 1024)") != NULL) + saw_message = true; + } + log_set_file(NULL); + fclose(log_capture); + EXPECT_TRUE(saw_message); + + config_delete(cfg); +} + /* Test main() with --help flag (early return path, no server connection needed) */ static void test_cli_help() { /* We can't easily call main() because it calls send_files which needs a server. @@ -4824,6 +4856,7 @@ void test_client_cli() { test_validate_config_credentials_require_tls_or_loopback(); test_validate_config_delta_sendfile_constraints(); test_validate_config_unified_invariants(); + test_validate_config_filter_rule_limit(); test_cli_help(); test_cli_archive_flags(); test_cli_dry_run(); -- 2.54.0 From 0f40e747f08b8b18c4b3798bcba6f6bc8463a25d Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 19:19:06 +0200 Subject: [PATCH 18/68] fix(filter): reject the x modifier in the --filter list parser The standalone filter_rule_parse() already rejected the rsync xattr-name 'x' modifier, but the list parser used by --filter/-f silently dropped the flag for merge/dir-merge rules (and relied on a second parse for plain rules). Reject it explicitly in filter_list_parse_append_depth() with the same diagnostic, so '-x', 'merge,x' and 'dir-merge,x' all fail cleanly. Also reject the unimplemented rsync merge modifiers 'e', 'n' and 'w' instead of folding them into the pattern, which previously produced misleading errors such as "could not read merge file 'n file'". Only a token made up solely of modifier characters is treated as a modifier run, so glued patterns ('-newfile', '-e2e') and mixed tokens ("H,!secret") keep their historical parsing. Adds tests/test_filter.c with focused rejection and supported-syntax cases. --- CMakeLists.txt | 1 + src/shared/filter.c | 75 +++++++++++++-- src/shared/filter.h | 4 +- tests/runner.c | 2 + tests/test_filter.c | 222 ++++++++++++++++++++++++++++++++++++++++++++ tests/test_filter.h | 6 ++ 6 files changed, 301 insertions(+), 9 deletions(-) create mode 100644 tests/test_filter.c create mode 100644 tests/test_filter.h diff --git a/CMakeLists.txt b/CMakeLists.txt index 905daec..09c10f6 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -226,6 +226,7 @@ set(TEST_SRCS tests/test_file.c tests/test_file_list.c tests/test_file_sendfile.c + tests/test_filter.c tests/test_format.c tests/test_fuzz_smoke.c tests/test_glob.c diff --git a/src/shared/filter.c b/src/shared/filter.c index 2a2cd7a..f478286 100644 --- a/src/shared/filter.c +++ b/src/shared/filter.c @@ -172,14 +172,50 @@ static bool is_modifier_char(char c) { return c == 's' || c == 'r' || c == 'p' || c == 'x' || c == '/' || c == '!' || c == 'C'; } +/* Modifiers rsync defines but FastSync does not implement. They must still be + * consumed as part of the modifier run so they are rejected explicitly instead + * of leaking into the pattern (which produced misleading failures such as + * "could not read merge file 'n file'"). */ +static bool is_unsupported_modifier_char(char c) { + return c == 'e' || c == 'n' || c == 'w'; +} + +/* Characters that are part of a modifier run, whether supported or not. */ +static bool is_modifier_scan_char(char c) { + return is_modifier_char(c) || is_unsupported_modifier_char(c); +} + +/* Inspect the token that follows a rule name (up to the first space/underscore + * or the end). If the token is composed *solely* of modifier characters and + * includes one FastSync does not implement, it is unambiguously a modifier run: + * return that character so the caller can reject it. A token that contains any + * non-modifier character is a pattern (e.g. "-newfile") and returns '\0', which + * keeps the historical parsing of mixed tokens such as "H,!secret" intact. */ +static char unsupported_modifier_in_token(const char* tok) { + if (*tok == '\0' || *tok == ' ' || *tok == '_') + return '\0'; + char bad = '\0'; + for (const char* q = tok; *q != '\0' && *q != ' ' && *q != '_'; q++) { + if (!is_modifier_scan_char(*q)) + return '\0'; + if (is_unsupported_modifier_char(*q)) + bad = *q; + } + return bad; +} + /* Parse "RULE[,MODIFIERS] [PATTERN]". On success `kind`, `sides`, * `sides_explicit`, `negate`, `anchored_mod`, `perishable`, `xattr`, * `cvs_inject` and the pattern span (`pat_start`/`pat_len`, possibly 0 for - * merge/clear) are filled. Returns true on success. */ + * merge/clear) are filled. Returns true on success. + * + * On failure `*bad_mod` is set to the offending modifier character when the + * rule carried a modifier FastSync does not implement, and left '\0' for a + * generic syntax error so callers can emit a precise diagnostic. */ static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides, bool* sides_explicit, bool* negate, bool* anchored_mod, bool* perishable, bool* xattr, bool* cvs_inject, - const char** pat_start, size_t* pat_len) { + const char** pat_start, size_t* pat_len, char* bad_mod) { const char* p = text; *sides = FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER; *sides_explicit = false; @@ -190,6 +226,7 @@ static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides, *cvs_inject = false; *pat_start = NULL; *pat_len = 0; + *bad_mod = '\0'; bool is_short = false; if (short_rule_char(*p, kind)) { @@ -210,6 +247,14 @@ static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides, /* Modifiers: long names require a comma; short names may attach directly. Only commit a modifier run that terminates at a separator or the end, so a pattern such as "*.tmp" written as "-*.tmp" is not mistaken for modifiers. */ + if (*p == ',') { + *bad_mod = unsupported_modifier_in_token(p + 1); + } else if (is_short) { + *bad_mod = unsupported_modifier_in_token(p); + } + if (*bad_mod != '\0') + return false; + const char* mod_start = p; const char* mod_end = p; if (*p == ',') { @@ -290,9 +335,13 @@ FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject; const char* pat; size_t pat_len; + char bad_mod; if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable, - &xattr, &cvs_inject, &pat, &pat_len)) { - filter_set_error(err, err_size, "unrecognized filter rule syntax"); + &xattr, &cvs_inject, &pat, &pat_len, &bad_mod)) { + if (bad_mod != '\0') + filter_set_error(err, err_size, "unsupported filter modifier '%c'", bad_mod); + else + filter_set_error(err, err_size, "unrecognized filter rule syntax"); return NULL; } if (cvs_inject) { @@ -401,7 +450,6 @@ FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, rule->dir_only = dir_only; rule->negate = negate; rule->perishable = perishable; - (void)xattr; /* xattr-name rules never match file/dir names; accepted/ignored */ return rule; } @@ -530,16 +578,27 @@ static bool filter_list_parse_append_depth(FilterRuleList* list, const char* lin bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject; const char* pat; size_t pat_len; + char bad_mod; if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable, - &xattr, &cvs_inject, &pat, &pat_len)) { - filter_set_error(err, err_size, "unrecognized filter rule syntax: %s", p); + &xattr, &cvs_inject, &pat, &pat_len, &bad_mod)) { + if (bad_mod != '\0') + filter_set_error(err, err_size, "unsupported filter modifier '%c': %s", bad_mod, p); + else + filter_set_error(err, err_size, "unrecognized filter rule syntax: %s", p); return false; } (void)sides_explicit; (void)negate; (void)anchored_mod; (void)perishable; - (void)xattr; + + /* xattr-name rules are not implemented; reject them everywhere (including on + * merge/dir-merge, where the flag would otherwise be silently dropped) with + * the same diagnostic the standalone parser gives. */ + if (xattr) { + filter_set_error(err, err_size, "xattr-name filter rules (the x modifier) are not supported"); + return false; + } if (cvs_inject) { /* "C" injects the CVS defaults in place; no pattern is expected. */ diff --git a/src/shared/filter.h b/src/shared/filter.h index 691860b..bd9877b 100644 --- a/src/shared/filter.h +++ b/src/shared/filter.h @@ -23,7 +23,9 @@ * dir-merge/: per-directory merge file (registered for the scanner) * clear/! clear the current rule list (takes no argument) * Modifiers: '/' absolute anchor, '!' negate match, 'C' inject CVS defaults, - * 's' sender side, 'r' receiver side, 'p' perishable, 'x' xattr name rule. + * 's' sender side, 'r' receiver side, 'p' perishable. The rsync 'x' + * (xattr-name) modifier and the merge-only 'e'/'n'/'w' modifiers are not + * implemented and are rejected explicitly. * A trailing '/' makes a pattern match directories only. A leading '/' anchors * the pattern to its owner directory. */ diff --git a/tests/runner.c b/tests/runner.c index de1b4b5..d790287 100644 --- a/tests/runner.c +++ b/tests/runner.c @@ -16,6 +16,7 @@ #include "test_file.h" #include "test_file_list.h" #include "test_file_sendfile.h" +#include "test_filter.h" #include "test_format.h" #include "test_fuzz_smoke.h" #include "test_glob.h" @@ -80,6 +81,7 @@ int main() { RUN_TEST(test_receiver_timeout); RUN_TEST(test_metadata); RUN_TEST(test_glob); + RUN_TEST(test_filter); RUN_TEST(test_iconv); RUN_TEST(test_file); RUN_TEST(test_file_list); diff --git a/tests/test_filter.c b/tests/test_filter.c new file mode 100644 index 0000000..76e2729 --- /dev/null +++ b/tests/test_filter.c @@ -0,0 +1,222 @@ +#include "test_filter.h" +#include "filter.h" +#include "test_utils.h" +#include +#include +#include +#include +#include + +/* Every `-f`/`--filter` rule string is validated through the list parser, so + * the list parser must reject the modifiers the standalone parser rejects + * rather than silently folding them into a pattern. */ + +static void test_filter_list_rejects_xattr_modifier() { + static const char* const rules[] = { + "-x user.foo", /* short exclude + x */ + "exclude,x user.foo", /* long exclude + x */ + "+x user.foo", /* include + x */ + "hide,x *.tmp", /* hide + x */ + "dir-merge,x .rules", /* x must not be dropped on dir-merge */ + "merge,x /tmp/nonexistent" /* x must not be dropped on merge */ + }; + for (size_t i = 0; i < sizeof(rules) / sizeof(rules[0]); i++) { + FilterRuleList* list = filter_rule_list_create(); + EXPECT_NOT_NULL(list); + char err[256] = ""; + bool ok = filter_rule_list_parse_append(list, rules[i], NULL, NULL, err, sizeof(err)); + EXPECT_FALSE(ok); + EXPECT_TRUE(strstr(err, "xattr") != NULL); + filter_rule_list_free(list); + } + + /* A bare "x" is not a rule at all: rejected as generic bad syntax. */ + FilterRuleList* list = filter_rule_list_create(); + EXPECT_NOT_NULL(list); + char err[256] = ""; + EXPECT_FALSE(filter_rule_list_parse_append(list, "x user.foo", NULL, NULL, err, sizeof(err))); + EXPECT_TRUE(err[0] != '\0'); + filter_rule_list_free(list); +} + +static void test_filter_list_rejects_unsupported_modifiers() { + static const char* const rules[] = { + "-e foo", /* e: merge-only in rsync */ + "-n foo", /* n: merge-only in rsync */ + "-w foo", /* w: merge-only in rsync */ + "merge,n /tmp/x", ".e /tmp/x", "dir-merge,e .rules", "exclude,w foo", + }; + for (size_t i = 0; i < sizeof(rules) / sizeof(rules[0]); i++) { + FilterRuleList* list = filter_rule_list_create(); + EXPECT_NOT_NULL(list); + char err[256] = ""; + bool ok = filter_rule_list_parse_append(list, rules[i], NULL, NULL, err, sizeof(err)); + EXPECT_FALSE(ok); + EXPECT_TRUE(strstr(err, "unsupported filter modifier") != NULL); + filter_rule_list_free(list); + } +} + +static void test_filter_list_accepts_supported_rules_and_modifiers() { + static const char* const rules[] = { + "- *.tmp", "+ /a.txt", "-s foo", "-r foo", "-p foo", + "-! *.o", "-/ foo", "hide *.tmp", "show *.txt", "protect *.bak", + "risk *.o", "dir-merge .rules", "-C", + }; + for (size_t i = 0; i < sizeof(rules) / sizeof(rules[0]); i++) { + FilterRuleList* list = filter_rule_list_create(); + EXPECT_NOT_NULL(list); + char err[256] = ""; + bool ok = filter_rule_list_parse_append(list, rules[i], NULL, NULL, err, sizeof(err)); + if (!ok) + printf(" rule '%s' errored: %s\n", rules[i], err); + EXPECT_TRUE(ok); + filter_rule_list_free(list); + } + + /* A glued word is a pattern, not a modifier run (no separator). */ + { + FilterRuleList* list = filter_rule_list_create(); + char err[128] = ""; + EXPECT_TRUE(filter_rule_list_parse_append(list, "-newfile", NULL, NULL, err, sizeof(err))); + EXPECT_EQ_INT(list->count, 1); + EXPECT_EQ_STR(list->items[0]->pattern, "newfile"); + filter_rule_list_free(list); + + list = filter_rule_list_create(); + EXPECT_TRUE(filter_rule_list_parse_append(list, "-e2e", NULL, NULL, err, sizeof(err))); + EXPECT_EQ_INT(list->count, 1); + EXPECT_EQ_STR(list->items[0]->pattern, "e2e"); + filter_rule_list_free(list); + + list = filter_rule_list_create(); + EXPECT_TRUE(filter_rule_list_parse_append(list, "-*.o", NULL, NULL, err, sizeof(err))); + EXPECT_EQ_INT(list->count, 1); + EXPECT_EQ_STR(list->items[0]->pattern, "*.o"); + filter_rule_list_free(list); + + /* A comma with no modifier still treats the rest as the pattern. */ + list = filter_rule_list_create(); + EXPECT_TRUE(filter_rule_list_parse_append(list, "exclude,foo", NULL, NULL, err, sizeof(err))); + EXPECT_EQ_INT(list->count, 1); + EXPECT_EQ_STR(list->items[0]->pattern, "foo"); + filter_rule_list_free(list); + } + + /* `!` clears the list. */ + { + FilterRuleList* list = filter_rule_list_create(); + char err[128] = ""; + EXPECT_TRUE(filter_rule_list_parse_append(list, "- *.tmp", NULL, NULL, err, sizeof(err))); + EXPECT_EQ_INT(list->count, 1); + EXPECT_TRUE(filter_rule_list_parse_append(list, "!", NULL, NULL, err, sizeof(err))); + EXPECT_EQ_INT(list->count, 0); + filter_rule_list_free(list); + } + + /* -C injects the CVS defaults. */ + { + FilterRuleList* list = filter_rule_list_create(); + char err[128] = ""; + EXPECT_TRUE(filter_rule_list_parse_append(list, "-C", NULL, NULL, err, sizeof(err))); + EXPECT_TRUE(list->count > 0); + filter_rule_list_free(list); + } +} + +static void test_filter_list_merge_file_still_supported() { + char tmpl[] = "/tmp/fastsync_filter_XXXXXX"; + EXPECT_TRUE(mkdtemp(tmpl) != NULL); + char path[512]; + snprintf(path, sizeof(path), "%s/rules", tmpl); + FILE* fp = fopen(path, "w"); + EXPECT_NOT_NULL(fp); + fputs("- *.tmp\n", fp); + fclose(fp); + + FilterRuleList* list = filter_rule_list_create(); + char err[256] = ""; + char rule[600]; + snprintf(rule, sizeof(rule), "merge %s", path); + EXPECT_TRUE(filter_rule_list_parse_append(list, rule, NULL, NULL, err, sizeof(err))); + EXPECT_EQ_INT(list->count, 1); + filter_rule_list_free(list); + + /* The same merge with the x modifier is rejected, not silently read. */ + list = filter_rule_list_create(); + snprintf(rule, sizeof(rule), "merge,x %s", path); + EXPECT_FALSE(filter_rule_list_parse_append(list, rule, NULL, NULL, err, sizeof(err))); + EXPECT_TRUE(strstr(err, "xattr") != NULL); + filter_rule_list_free(list); + + unlink(path); + rmdir(tmpl); +} + +static void test_filter_rule_parse_rejects_unsupported_and_keeps_supported() { + char err[256] = ""; + + EXPECT_NULL(filter_rule_parse("-x user.foo", NULL, err, sizeof(err))); + EXPECT_TRUE(strstr(err, "xattr") != NULL); + + EXPECT_NULL(filter_rule_parse("-e foo", NULL, err, sizeof(err))); + EXPECT_TRUE(strstr(err, "unsupported filter modifier") != NULL); + + FilterRule* rule = filter_rule_parse("- *.tmp", NULL, err, sizeof(err)); + EXPECT_NOT_NULL(rule); + EXPECT_EQ_STR(rule->pattern, "*.tmp"); + filter_rule_free(rule); + + rule = filter_rule_parse("-newfile", NULL, err, sizeof(err)); + EXPECT_NOT_NULL(rule); + EXPECT_EQ_STR(rule->pattern, "newfile"); + filter_rule_free(rule); +} + +static void test_filter_rules_apply_supported_modifiers() { + /* exclude */ + { + const char* texts[] = {"- *.tmp"}; + FilterRuleList* list = filter_base_build(texts, 1, false, false, NULL, 0); + EXPECT_NOT_NULL(list); + EXPECT_EQ_INT(filter_rules_apply(list, "b.tmp", "b.tmp", false), FILTER_ACTION_EXCLUDE); + EXPECT_EQ_INT(filter_rules_apply(list, "a.txt", "a.txt", false), FILTER_ACTION_NONE); + filter_rule_list_free(list); + } + /* anchored include then exclude-all */ + { + const char* texts[] = {"+ /a.txt", "- *"}; + FilterRuleList* list = filter_base_build(texts, 2, false, false, NULL, 0); + EXPECT_NOT_NULL(list); + EXPECT_EQ_INT(filter_rules_apply(list, "a.txt", "a.txt", false), FILTER_ACTION_INCLUDE); + EXPECT_EQ_INT(filter_rules_apply(list, "b.txt", "b.txt", false), FILTER_ACTION_EXCLUDE); + filter_rule_list_free(list); + } + /* negate */ + { + const char* texts[] = {"-! *.o"}; + FilterRuleList* list = filter_base_build(texts, 1, false, false, NULL, 0); + EXPECT_NOT_NULL(list); + EXPECT_EQ_INT(filter_rules_apply(list, "foo.c", "foo.c", false), FILTER_ACTION_EXCLUDE); + EXPECT_EQ_INT(filter_rules_apply(list, "foo.o", "foo.o", false), FILTER_ACTION_NONE); + filter_rule_list_free(list); + } + /* dir-only trailing slash */ + { + const char* texts[] = {"+ dir/", "- *"}; + FilterRuleList* list = filter_base_build(texts, 2, false, false, NULL, 0); + EXPECT_NOT_NULL(list); + EXPECT_EQ_INT(filter_rules_apply(list, "dir", "dir", true), FILTER_ACTION_INCLUDE); + EXPECT_EQ_INT(filter_rules_apply(list, "dir", "dir", false), FILTER_ACTION_EXCLUDE); + filter_rule_list_free(list); + } +} + +void test_filter() { + test_filter_list_rejects_xattr_modifier(); + test_filter_list_rejects_unsupported_modifiers(); + test_filter_list_accepts_supported_rules_and_modifiers(); + test_filter_list_merge_file_still_supported(); + test_filter_rule_parse_rejects_unsupported_and_keeps_supported(); + test_filter_rules_apply_supported_modifiers(); +} diff --git a/tests/test_filter.h b/tests/test_filter.h new file mode 100644 index 0000000..26fa58c --- /dev/null +++ b/tests/test_filter.h @@ -0,0 +1,6 @@ +#ifndef TEST_FILTER_H +#define TEST_FILTER_H + +void test_filter(void); + +#endif -- 2.54.0 From 221cefa7ccaf33afce9cf4f80878431c5d3343f0 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 19:25:24 +0200 Subject: [PATCH 19/68] fix(signal): use sigaction and async-signal-safe handlers Client: replace the non-async-signal-safe signal(3) call inside client_signal_handler() with a precomputed SIG_DFL sigaction(2), which is on the POSIX async-signal-safe list. The handler stays installed while a transfer is armed so a repeated Ctrl-C still leads to a graceful abort rather than a hard kill mid-cleanup. Server: cleanup() now only calls _exit(2) (async-signal-safe). The former server_delete()/daemon_conf_free()/credentials_free() teardown called free()/close()/SSL_CTX_free() from signal context, which can deadlock or corrupt the heap if the signal lands inside malloc/free. Handlers are installed with sigaction(2) instead of signal(3). The normal shutdown path in main() still performs the full teardown; the signal path relies on process exit to reclaim the parent daemon's socket, anonymous shared mapping and heap (no named/persistent parent resource is left behind). --- src/client/client_cli.c | 19 ++++++++++++++++--- src/server/server.c | 41 +++++++++++++++++++++++++++++++++-------- 2 files changed, 49 insertions(+), 11 deletions(-) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index e8d660f..8565b63 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -54,13 +54,26 @@ bool client_abort_pending(void) { } #ifndef FASTSYNC_TEST_BUILD +/* SIG_DFL disposition used by the handler's "not armed" fallback. It is built + * once at load time so the handler can restore the default action with + * sigaction(2) -- which is async-signal-safe -- instead of signal(3), which is + * not. The zero-initialized sa_mask is the empty set. */ +static const struct sigaction client_default_action = { + .sa_handler = SIG_DFL, + .sa_flags = 0, +}; + /* Signal handler: perform NO work beyond storing the flag. Logging, protocol * I/O and the STATUS_ABORT frame are all done later on the normal send path, - * which is not async-signal-safe. When no transfer is armed, fall back to the - * default action so local-only modes remain interruptible. */ + * which is not async-signal-safe. When no transfer is armed, restore the + * default disposition (async-signal-safe sigaction) and re-raise so local-only + * modes remain interruptible. The handler deliberately stays installed while a + * transfer is armed -- rather than using SA_RESETHAND -- so a second Ctrl-C + * during the graceful abort keeps setting the flag instead of hard-killing the + * process mid-cleanup. */ static void client_signal_handler(int signo) { if (!client_abort_armed) { - signal(signo, SIG_DFL); + sigaction(signo, &client_default_action, NULL); raise(signo); return; } diff --git a/src/server/server.c b/src/server/server.c index fa19876..904d80a 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -1055,17 +1055,42 @@ done: #ifndef FASTSYNC_SERVER_AS_LIB static Server* g_server = NULL; +/* Signal handler for the foreground daemon/standalone listener. + * + * Async-signal-safety: _exit(2) is on the POSIX async-signal-safe list and is + * the ONLY thing done here. The previous body called server_delete() + * (close/free/SSL_CTX_free), daemon_conf_free() and credentials_free(); none of + * those (free/malloc, and much of OpenSSL teardown) are async-signal-safe, so a + * signal delivered while the main thread was inside malloc/free could deadlock + * or corrupt the heap. + * + * Residual (documented, not hidden): the in-memory teardown is skipped on the + * signal path. That is safe because the parent daemon owns no persistent + * resource that survives process exit -- the listening socket is closed by the + * kernel, the connection registry is an anonymous MAP_SHARED mapping with no + * named backing object, and the daemon config/credential stores are plain heap + * allocations. Connection children are separate processes and handle their own + * temp files/locks. The normal (non-signal) shutdown path in main() still runs + * the full teardown, so no cleanup is dropped on the common path. Wiring the + * accept loop (transport_tcp.c, outside this change's scope) to a flag-based + * self-pipe shutdown would let the frees run context-safely; it is deliberately + * deferred rather than risk restructuring the daemon loop. */ static void cleanup(int sig) { (void)sig; - if (g_server) - server_delete(&g_server); - daemon_conf_free(g_daemon_conf); - g_daemon_conf = NULL; - credentials_free(g_credentials); - g_credentials = NULL; _exit(0); } +/* Install a signal handler with sigaction(2) (the required async-signal-safe + * install primitive; signal(3) is not specified to be async-signal-safe). */ +static void install_cleanup_handler(int signo) { + struct sigaction action; + memset(&action, 0, sizeof(action)); + action.sa_handler = cleanup; + sigemptyset(&action.sa_mask); + action.sa_flags = 0; + sigaction(signo, &action, NULL); +} + static void print_server_usage(void) { printf("FastSync Server\n"); printf("Usage: fastsync-server [options]\n\n"); @@ -1259,8 +1284,8 @@ int main(int argc, char* argv[]) { * this process-global policy cannot be re-enabled by a future caller. */ server_allow_super = opts.allow_super && !opts.stdio_mode; server_iconv_spec = opts.iconv_spec; - signal(SIGINT, cleanup); - signal(SIGTERM, cleanup); + install_cleanup_handler(SIGINT); + install_cleanup_handler(SIGTERM); /* Server-owned socket deadline floor: the client default --timeout=0 would * otherwise leave accepted sockets without SO_RCVTIMEO/SO_SNDTIMEO and let a * silent peer hold a connection (and its process slot) forever. */ -- 2.54.0 From 865941f850946b8e7f0c2503e8d91d320490ad05 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 19:25:58 +0200 Subject: [PATCH 20/68] fix(credentials): open secret files with O_NOFOLLOW|O_NONBLOCK; drop dup includes secret_file_open() previously opened --password-file/--early-input with plain O_RDONLY, so a symlinked path was followed before the owner/mode fstat gate ran, and an empty/planted FIFO could block fgets forever. Open with O_NOFOLLOW|O_NONBLOCK|O_CLOEXEC (mirroring the dummy-key sidecar): ELOOP now fails closed, and a writer-less FIFO yields EOF/EAGAIN instead of hanging. Clear O_NONBLOCK again for regular files, where it is a no-op, so their stdio read path is unchanged. file.c: drop the duplicate / includes (kept the first occurrences). Tests: a symlinked password file is rejected, and a writer-less named FIFO fails cleanly without hanging. --- src/shared/credentials.c | 19 ++++++++++++++--- src/shared/file.c | 2 -- tests/test_credentials.c | 46 ++++++++++++++++++++++++++++++++++++++++ 3 files changed, 62 insertions(+), 5 deletions(-) diff --git a/src/shared/credentials.c b/src/shared/credentials.c index 7f79c52..7e0c0b1 100644 --- a/src/shared/credentials.c +++ b/src/shared/credentials.c @@ -80,12 +80,16 @@ static bool is_comment_char(char c) { * path and then fstat the resulting fd (rather than stat()ing the path first * and reopening it), so the permission decision is made on the same inode that * is read and cannot be raced by swapping the path between check and open. - * The path may be a process-substitution pipe (`<(...)` -> /dev/fd/N), so - * regular files and FIFOs are accepted when the ownership/mode checks pass. + * O_NOFOLLOW refuses a symlinked path outright (ELOOP fails closed) instead of + * following it before the owner/mode gate can run. O_NONBLOCK keeps a FIFO + * from blocking the open/read forever: an empty or writer-less FIFO yields + * EOF/EAGAIN rather than hanging in fgets. Only regular files and FIFOs pass + * the ownership/mode checks; O_NONBLOCK is cleared for regular files, where it + * is a no-op anyway, so their stdio read path is byte-for-byte unchanged. * * Returns a FILE* the caller must fclose, or NULL with `err` filled. */ static FILE* secret_file_open(const char* path, char* err, size_t err_size) { - int fd = open(path, O_RDONLY | O_CLOEXEC); + int fd = open(path, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); if (fd < 0) { set_error(err, err_size, "cannot open secret file '%s': %s", path, strerror(errno)); return NULL; @@ -105,6 +109,15 @@ static FILE* secret_file_open(const char* path, char* err, size_t err_size) { close(fd); return NULL; } + /* O_NONBLOCK is only meaningful for the FIFO allowance. Restore blocking + * mode on a regular file so its read path is exactly as before; a no-op on + * most systems, but explicit. Failures here are ignored: O_NONBLOCK on a + * regular file does not affect reads either way. */ + if (S_ISREG(st.st_mode)) { + int flags = fcntl(fd, F_GETFL); + if (flags >= 0) + (void)fcntl(fd, F_SETFL, flags & ~O_NONBLOCK); + } FILE* fp = fdopen(fd, "r"); if (!fp) { set_error(err, err_size, "cannot read secret file '%s': %s", path, strerror(errno)); diff --git a/src/shared/file.c b/src/shared/file.c index 4fdf3fe..9ec981e 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -25,8 +25,6 @@ #include "utils.h" #include "protocol.h" #include "xattr.h" -#include -#include /* Files larger than this are not loaded whole for transfer (the sender streams * them); a whole-file digest is computed from the path instead. Kept in sync diff --git a/tests/test_credentials.c b/tests/test_credentials.c index 80280e7..ba9b9d9 100644 --- a/tests/test_credentials.c +++ b/tests/test_credentials.c @@ -719,6 +719,50 @@ static void test_credentials_read_secret_file_bad() { EXPECT_EQ_INT(credentials_read_secret_file(missing, NULL, NULL, err, sizeof(err)), -1); } +/* A symlink planted at a password-file path is refused (O_NOFOLLOW) instead of + * being followed before the owner/mode gate, even when it resolves to a valid + * owner-only regular file. */ +static void test_credentials_read_secret_file_symlink_rejected() { + char err[512]; + char* target = make_tmp_file("alice:correct horse battery staple\n"); + EXPECT_NOT_NULL(target); + + char link[256]; + snprintf(link, sizeof(link), "/tmp/fs_cred_pwlink_%d_%d", (int)getpid(), g_file_counter++); + unlink(link); + EXPECT_EQ_INT(symlink(target, link), 0); + + char* user = (char*)1; + char* password = (char*)1; + EXPECT_EQ_INT(credentials_read_secret_file(link, &user, &password, err, sizeof(err)), -1); + EXPECT_NULL(user); + EXPECT_NULL(password); + EXPECT_TRUE(err[0] != '\0'); + + unlink(link); /* remove the symlink itself, not its target */ + rm_temp(target); + free(target); +} + +/* A named FIFO with no writer must not hang in fgets (O_NONBLOCK): the read + * fails cleanly with "no user:password line" instead of blocking forever. */ +static void test_credentials_read_secret_file_fifo_no_hang() { + char err[512]; + char fifo[256]; + snprintf(fifo, sizeof(fifo), "/tmp/fs_cred_pwfifo_%d_%d", (int)getpid(), g_file_counter++); + unlink(fifo); + EXPECT_EQ_INT(mkfifo(fifo, 0600), 0); + + char* user = (char*)1; + char* password = (char*)1; + EXPECT_EQ_INT(credentials_read_secret_file(fifo, &user, &password, err, sizeof(err)), -1); + EXPECT_NULL(user); + EXPECT_NULL(password); + EXPECT_TRUE(strstr(err, "no 'user:password'") != NULL || err[0] != '\0'); + + unlink(fifo); +} + static void test_credentials_hash_file() { char* plaintext = make_tmp_file("# comment\n\n alice :" KAT_PASSWORD "\nbob:bob-s3cret\n"); EXPECT_NOT_NULL(plaintext); @@ -1103,6 +1147,8 @@ void test_credentials(void) { test_credentials_early_input_merge(); test_credentials_read_secret_file(); test_credentials_read_secret_file_bad(); + test_credentials_read_secret_file_symlink_rejected(); + test_credentials_read_secret_file_fifo_no_hang(); test_credentials_hash_file(); test_credentials_rejects_group_or_other_accessible(); test_credentials_dummy_key_persisted(); -- 2.54.0 From b478a59a81092975ea962a5e35b61b81fd8a05c9 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 19:26:42 +0200 Subject: [PATCH 21/68] fix(cli): correct help text, per-codec compression default, and stale test --- src/client/client_validation.c | 2 +- src/client/usage.c | 32 +++++++++---- tests/test_client_cli.c | 84 ++++++++++++++++++++++++++-------- 3 files changed, 87 insertions(+), 31 deletions(-) diff --git a/src/client/client_validation.c b/src/client/client_validation.c index bb129d6..2ff7868 100644 --- a/src/client/client_validation.c +++ b/src/client/client_validation.c @@ -53,7 +53,7 @@ bool validate_config(const Config* config) { return false; } if (config->compression_threads > 0 && !config->use_compression) { - log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-c or -z)"); + log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-z/--compress)"); return false; } if (config->transport == TRANSPORT_SSH && config->use_sendfile) { diff --git a/src/client/usage.c b/src/client/usage.c index a24c2f8..b4e0584 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -19,7 +19,9 @@ void print_usage(void) { printf("\n"); printf("Options:\n"); printf(" -c, --checksum Verify content by checksum instead of size+mtime\n"); - printf(" -z, --compress [level] Enable compression (level 1-22, default 5)\n"); + printf(" -z, --compress [level] Enable compression. The default level is\n"); + printf(" per-codec: zstd 3 (range 1-22), zlib/zlibx 6, lz4\n"); + printf(" ignores the level\n"); printf(" -a, --archive rsync archive mode (-rlptgoD): links, perms, times,\n"); printf(" owner, group, devices and specials; not\n"); printf(" compression/multithreading\n"); @@ -36,8 +38,8 @@ void print_usage(void) { printf(" arguments, e.g. -e \"ssh -p 2222\"\n"); printf(" --rsync-path Alias for --fastsync-server-path (path to the\n"); printf(" fastsync server binary on the remote side)\n"); - printf(" --blocking-io Leave the SSH transport socket without read/write\n"); - printf(" timeouts so it blocks naturally\n"); + printf(" --blocking-io SSH transport only: leave the socket without read/write\n"); + printf(" timeouts so it blocks naturally (no effect on TCP)\n"); printf(" --outbuf=MODE stdout/stderr buffering: N (none/unbuffered),\n"); printf(" L (line-buffered), or B (block-buffered, default)\n"); printf(" --progress Show transfer progress\n"); @@ -49,6 +51,7 @@ void print_usage(void) { printf(" converted before transmission and back on receipt; a\n"); printf(" name that cannot be represented in the target charset\n"); printf(" fails that transfer cleanly (rsync-compatible)\n"); + printf(" --no-iconv Disable --iconv charset conversion (same as --iconv=-)\n"); printf(" --protocol=NUM Force the wire protocol version (must equal the current\n"); printf(" PROTOCOL_VERSION; FastSync cannot speak older/virtual\n"); printf(" wire formats)\n"); @@ -192,6 +195,10 @@ void print_usage(void) { printf(" -U, --atimes Preserve access times\n"); printf(" -N, --crtimes Capture birth time; cannot be applied (documented\n"); printf(" divergence)\n"); + printf(" -O, --omit-dir-times Do not apply modification times to directories\n"); + printf(" -J, --omit-link-times Do not apply times to symlinks\n"); + printf(" --open-noatime Open source files with O_NOATIME so reading for a\n"); + printf(" transfer does not update their access time\n"); printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n"); printf(" privileged security.*/trusted.* namespaces are\n"); printf(" never captured or applied)\n"); @@ -287,6 +294,10 @@ void print_usage(void) { printf(" -x, --one-file-system Do not cross filesystem boundaries\n"); printf(" --log-file , --log-file= Write log messages to file\n"); printf(" --stderr=MODE Route logging to stderr: errors or all\n"); + printf(" --msgs2stderr Route all messages to stderr (deprecated spelling of\n"); + printf(" --stderr=all)\n"); + printf(" --no-msgs2stderr Select errors-only stderr (deprecated spelling; the\n"); + printf(" default)\n"); printf(" --partial Keep partial files on interrupted transfer\n"); printf(" --partial-dir Directory for partial files (implies --partial)\n"); printf(" -T, --temp-dir Scratch dir for temp files before atomic install.\n"); @@ -337,7 +348,8 @@ void print_usage(void) { printf(" --append-verify Like --append, but verifies the retained prefix checksum\n"); printf(" before appending (falls back to a full transfer on mismatch)\n"); printf(" --fsync Fsync every written file before publication\n"); - printf(" --compress-level Compression level (default: 5)\n"); + printf(" --compress-level Compression level (per-codec default: zstd 3,\n"); + printf(" zlib/zlibx 6, lz4 ignores it)\n"); printf(" --zl Alias for --compress-level\n"); printf(" --skip-compress=LIST Skip compression for suffixes in LIST (separated by\n"); printf(" '/' as in rsync, or ','); a leading dot is optional. The\n"); @@ -349,19 +361,19 @@ void print_usage(void) { } void print_debug_usage(void) { - printf("Emitting debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n"); + printf("Emitting debug flags: IO,PROTO,PACK,UTIL,FLIST,DEL,HASH,DELTASUM,\n"); + printf("RECV,FILTER,SEND,ALL,NONE\n"); printf("Also accepted for rsync CLI parity (silent): ACL,BACKUP,BIND,CHDIR,\n"); - printf("CONNECT,CMD,DEL,DELTASUM,DUP,EXIT,FILTER,FLIST,FUZZY,GENR,HASH,HLINK,\n"); - printf("ICONV,NSTR,OWN,RECV,SEND,TIME.\n"); + printf("CONNECT,CMD,DUP,EXIT,FUZZY,GENR,HLINK,ICONV,NSTR,OWN,TIME.\n"); printf("Flags may be comma-separated, for example: --debug=io,proto\n"); printf("An optional level suffix is accepted (e.g. --debug=io2); level 0\n"); printf("silences that item. Unknown names are rejected.\n"); } void print_info_usage(void) { - printf("Emitting info flags: COPY,NAME,MISC,SKIP,STATS,ALL,NONE\n"); - printf("Also accepted for rsync CLI parity (silent): BACKUP,DEL,FLIST,MOUNT,\n"); - printf("NONREG,PROGRESS,REMOVE,SYMSAFE.\n"); + printf("Emitting info flags: COPY,MISC,SKIP,STATS,DEL,REMOVE,NAME,FLIST,\n"); + printf("NONREG,PROGRESS,MOUNT,ALL,NONE\n"); + printf("Also accepted for rsync CLI parity (silent): BACKUP,SYMS,SYMSAFE.\n"); printf("Flags may be comma-separated, for example: --info=name,stats\n"); printf("An optional level suffix is accepted (e.g. --info=stats2); level 0\n"); printf("silences that item. Unknown names are rejected.\n"); diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index 951ada9..b8b2ff9 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -1329,35 +1329,78 @@ static void test_parse_args_delete_timing_without_delete_rejected() { config_delete(cfg); } -/* Parsed-but-unimplemented options must fail instead of being silently accepted. */ -static void test_parse_args_rejects_unimplemented_options() { - static const char* const options[] = {"--silent", - "--queue-size", - "-A", - "--acls", - "-X", - "--xattrs", - "-D", - "--devices", - "--delete-excluded", - "--max-delete", - "--prune-empty-dirs", - "--bind-address", - "--daemon", - "--config", - "--server"}; +/* Truly-unknown options (including server-only spellings) must be rejected + * through the unknown-option path instead of being silently accepted. */ +static void test_parse_args_rejects_unknown_options() { + static const char* const options[] = {"--silent", "--queue-size", "--bind-address", + "--daemon", "--config", "--server"}; for (size_t i = 0; i < sizeof(options) / sizeof(options[0]); i++) { Config* cfg = config_create(); - char* argv[] = {"fastsync", (char*)options[i], "dummy", "/src", "/dst"}; + char* argv[] = {"fastsync", (char*)options[i], "/src", "/dst"}; int positional_args[2]; int positional_count = 0; - EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); config_delete(cfg); } } +/* Options that are genuinely implemented must parse successfully and record + * their effect, rather than being lumped in with the unknown-option set. */ +static void test_parse_args_accepts_implemented_metadata_options() { + Config* cfg = config_create(); + char* argv_x[] = {"fastsync", "-X", "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_x, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_xattrs); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_acls[] = {"fastsync", "--acls", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_acls, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_acls); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_d[] = {"fastsync", "-D", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_d, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_devices); + EXPECT_TRUE(cfg->preserve_specials); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_devices[] = {"fastsync", "--devices", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_devices, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->preserve_devices); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_delete_excluded[] = {"fastsync", "--delete-excluded", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_delete_excluded, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->delete_excluded); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_max_delete[] = {"fastsync", "--max-delete=5", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_max_delete, positional_args, &positional_count), 0); + EXPECT_EQ_INT(cfg->max_delete, 5); + config_delete(cfg); + + cfg = config_create(); + positional_count = 0; + char* argv_prune[] = {"fastsync", "--prune-empty-dirs", "/src", "/dst"}; + EXPECT_EQ_INT(parse_args(cfg, 4, argv_prune, positional_args, &positional_count), 0); + EXPECT_TRUE(cfg->prune_empty_dirs); + config_delete(cfg); +} + /* Test both rsync-compatible quiet spellings and option ordering. */ static void test_parse_args_quiet() { static const char* const options[][2] = { @@ -4908,7 +4951,8 @@ void test_client_cli() { test_parse_args_delete_default_timing_and_commit(); test_parse_args_delete_timing_conflict_rejected(); test_parse_args_delete_timing_without_delete_rejected(); - test_parse_args_rejects_unimplemented_options(); + test_parse_args_rejects_unknown_options(); + test_parse_args_accepts_implemented_metadata_options(); test_parse_args_quiet(); test_parse_args_human_readable(); test_parse_args_hard_links(); -- 2.54.0 From ba914e8ab303c9e08d47d150404446efc54b4b01 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 19:30:01 +0200 Subject: [PATCH 22/68] fix(credentials): allow fd-backed store paths without O_NOFOLLOW --- src/shared/credentials.c | 39 +++++++++++++++++++++++++++++++++++++-- tests/test_credentials.c | 36 ++++++++++++++++++++++++++++++++++++ 2 files changed, 73 insertions(+), 2 deletions(-) diff --git a/src/shared/credentials.c b/src/shared/credentials.c index 7e0c0b1..998bae9 100644 --- a/src/shared/credentials.c +++ b/src/shared/credentials.c @@ -71,6 +71,31 @@ static bool is_comment_char(char c) { return c == '#' || c == ';'; } +/* True for a literal fd-backed store path: exactly "/dev/fd/" or + * "/proc/self/fd/", with no trailing component and no "..". These name + * the calling process's own open descriptors (e.g. a bash process substitution + * `<(...)`, which passes /dev/fd/N), and both prefixes are symlinks by + * construction. */ +static bool is_fd_backed_path(const char* path) { + static const char* const prefixes[] = {"/dev/fd/", "/proc/self/fd/"}; + if (!path) + return false; + for (size_t i = 0; i < sizeof(prefixes) / sizeof(prefixes[0]); i++) { + const char* prefix = prefixes[i]; + size_t prefix_len = strlen(prefix); + if (strncmp(path, prefix, prefix_len) != 0) + continue; + const char* digits = path + prefix_len; + if (*digits < '0' || *digits > '9') + return false; + const char* p = digits; + while (*p >= '0' && *p <= '9') + p++; + return *p == '\0'; + } + return false; +} + /* Open a --password-file / --early-input after verifying the EXACT inode we * will read: it must be owned by the effective user and grant no group/other * permission bit (so 0600 and stricter modes such as 0400 are accepted), @@ -81,7 +106,14 @@ static bool is_comment_char(char c) { * and reopening it), so the permission decision is made on the same inode that * is read and cannot be raced by swapping the path between check and open. * O_NOFOLLOW refuses a symlinked path outright (ELOOP fails closed) instead of - * following it before the owner/mode gate can run. O_NONBLOCK keeps a FIFO + * following it before the owner/mode gate can run. The one exception is a + * literal fd-backed path (/dev/fd/N or /proc/self/fd/N, see + * is_fd_backed_path): those entries are symlinks to the CALLING process's own + * descriptors, so following them is not the untrusted-symlink hazard + * O_NOFOLLOW guards against, and requiring O_NOFOLLOW would break the + * documented process-substitution/FIFO usage. For them only, O_NOFOLLOW is + * omitted; the same fstat owner/mode gate still applies to the resolved inode. + * O_NONBLOCK keeps a FIFO * from blocking the open/read forever: an empty or writer-less FIFO yields * EOF/EAGAIN rather than hanging in fgets. Only regular files and FIFOs pass * the ownership/mode checks; O_NONBLOCK is cleared for regular files, where it @@ -89,7 +121,10 @@ static bool is_comment_char(char c) { * * Returns a FILE* the caller must fclose, or NULL with `err` filled. */ static FILE* secret_file_open(const char* path, char* err, size_t err_size) { - int fd = open(path, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); + int flags = O_RDONLY | O_NONBLOCK | O_CLOEXEC; + if (!is_fd_backed_path(path)) + flags |= O_NOFOLLOW; + int fd = open(path, flags); if (fd < 0) { set_error(err, err_size, "cannot open secret file '%s': %s", path, strerror(errno)); return NULL; diff --git a/tests/test_credentials.c b/tests/test_credentials.c index ba9b9d9..7128b7f 100644 --- a/tests/test_credentials.c +++ b/tests/test_credentials.c @@ -3,6 +3,7 @@ #include "test_utils.h" #include "utils.h" #include +#include #include #include #include @@ -763,6 +764,40 @@ static void test_credentials_read_secret_file_fifo_no_hang() { unlink(fifo); } +/* fd-backed store paths (bash process substitution `<(...)`, i.e. /dev/fd/N and + * /proc/self/fd/N) are symlinks, so the ordinary O_NOFOLLOW rule would reject + * them with ELOOP. They name the calling process's own descriptors, so they + * are exempt: opening one that points at an owner-only regular file is + * accepted, while a symlink at a NORMAL path is still rejected + * (test_credentials_read_secret_file_symlink_rejected). */ +static void test_credentials_read_secret_file_fd_backed_accepted() { + char err[512]; + char* path = make_tmp_file("alice:correct horse battery staple\n"); + EXPECT_NOT_NULL(path); + + int fd = open(path, O_RDONLY | O_CLOEXEC); + EXPECT_TRUE(fd >= 0); + + const char* prefixes[] = {"/proc/self/fd/", "/dev/fd/"}; + for (size_t i = 0; i < sizeof(prefixes) / sizeof(prefixes[0]); i++) { + if (i == 1 && access("/dev/fd", F_OK) != 0) + continue; /* /dev/fd is not present on every system */ + char fd_path[64]; + snprintf(fd_path, sizeof(fd_path), "%s%d", prefixes[i], fd); + char* user = (char*)1; + char* password = (char*)1; + EXPECT_EQ_INT(credentials_read_secret_file(fd_path, &user, &password, err, sizeof(err)), 0); + EXPECT_EQ_STR(user, "alice"); + EXPECT_EQ_STR(password, "correct horse battery staple"); + free(user); + free(password); + } + + close(fd); + rm_temp(path); + free(path); +} + static void test_credentials_hash_file() { char* plaintext = make_tmp_file("# comment\n\n alice :" KAT_PASSWORD "\nbob:bob-s3cret\n"); EXPECT_NOT_NULL(plaintext); @@ -1149,6 +1184,7 @@ void test_credentials(void) { test_credentials_read_secret_file_bad(); test_credentials_read_secret_file_symlink_rejected(); test_credentials_read_secret_file_fifo_no_hang(); + test_credentials_read_secret_file_fd_backed_accepted(); test_credentials_hash_file(); test_credentials_rejects_group_or_other_accessible(); test_credentials_dummy_key_persisted(); -- 2.54.0 From 391f76cd566e5fea1723a4ecf43c58fedb8fb371 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 19:42:44 +0200 Subject: [PATCH 23/68] fix(log): add printf format attributes and fix format mismatches --- src/client/client_send.c | 2 +- src/client/usage.c | 2 +- src/shared/log.h | 15 ++++++++++++--- src/shared/protocol.c | 2 +- 4 files changed, 15 insertions(+), 6 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index 535cb5d..c777d52 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1967,7 +1967,7 @@ static int incremental_check(Client* client, File* file, const Config* config, send_status(client->file_descriptor, STATUS_ERROR); return -1; } - log_debug_message(LOG_DEBUG_RECV, "recv: delta signature for %s (%d blocks)", + log_debug_message(LOG_DEBUG_RECV, "recv: delta signature for %s (%u blocks)", file_wire_path(file), sig->block_count); *out_sig = sig; return 2; diff --git a/src/client/usage.c b/src/client/usage.c index b4e0584..cd5e35c 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -165,7 +165,7 @@ void print_usage(void) { printf(" --no-delta, or --no-incremental)\n"); printf(" --no-fuzzy Disable --fuzzy\n"); printf(" -B , --block-size , --delta-block \n"); - printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT); + printf(" Delta block size in bytes (default: %u)\n", DELTA_BLOCK_SIZE_DEFAULT); printf(" --delta-max Max file size for delta transfer (default: %llu)\n", DELTA_MAX_FILE_SIZE); printf(" -j, --threads[=N] Enable the multithreaded scanner/loader/sender\n"); diff --git a/src/shared/log.h b/src/shared/log.h index b3a258c..d64c467 100644 --- a/src/shared/log.h +++ b/src/shared/log.h @@ -5,6 +5,15 @@ #include #include +/* Ask the compiler to type-check the printf-style arguments of the variadic + * logging helpers. Only enabled for GNU-compatible compilers (gcc/clang). */ +#if defined(__GNUC__) +#define LOG_PRINTF_ATTR(fmt_idx, first_vararg_idx) \ + __attribute__((format(printf, fmt_idx, first_vararg_idx))) +#else +#define LOG_PRINTF_ATTR(fmt_idx, first_vararg_idx) +#endif + typedef enum { LOG_LEVEL_DEBUG, LOG_LEVEL_INFO, LOG_LEVEL_WARNING, LOG_LEVEL_ERROR } LogLevel; typedef enum { LOG_STDERR_ERRORS, LOG_STDERR_ALL } LogStderrMode; @@ -59,7 +68,7 @@ typedef enum { LOG_INFO_PROGRESS | LOG_INFO_MOUNT, } LogInfoFlag; -void log_message(LogLevel log_level, const char* message, ...); +void log_message(LogLevel log_level, const char* message, ...) LOG_PRINTF_ATTR(2, 3); void log_perror(const char* context); void set_log_level(LogLevel level); void set_log_debug_flags(uint32_t flags); @@ -68,10 +77,10 @@ uint32_t get_log_debug_flags(void); * the debug log level is enabled AND the flag is selected. Hot paths use this * to skip expensive message formatting/escaping when the line is filtered. */ bool log_debug_enabled(LogDebugFlag flag); -void log_debug_message(LogDebugFlag flag, const char* message, ...); +void log_debug_message(LogDebugFlag flag, const char* message, ...) LOG_PRINTF_ATTR(2, 3); void set_log_info_flags(uint32_t flags); uint32_t get_log_info_flags(void); -void log_info_message(LogInfoFlag flag, const char* message, ...); +void log_info_message(LogInfoFlag flag, const char* message, ...) LOG_PRINTF_ATTR(2, 3); void log_set_file(FILE* fp); void log_set_8_bit_output(bool enabled); bool log_get_8_bit_output(void); diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 581ebb8..fa1fff2 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -390,7 +390,7 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat if (session->ssl) wait_events = POLLOUT; } - log_debug_message(LOG_DEBUG_IO, " Send n Data: %zu", total_bytes_send); + log_debug_message(LOG_DEBUG_IO, " Send n Data: %zd", total_bytes_send); atomic_fetch_add(&io_bytes_written, (unsigned long long)total_bytes_send); return true; } -- 2.54.0 From 42f2f845ac022b635f85f7b2236624b1d9442dfc Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 19:50:33 +0200 Subject: [PATCH 24/68] refactor: drop Config**, dead old-args plumbing, dedup constants, -Wformat-signedness --- CMakeLists.txt | 4 ++-- src/client/client_cli.c | 2 +- src/client/client_send.c | 9 +++----- src/client/client_send.h | 2 +- src/shared/chunk.c | 9 ++++---- src/shared/file.c | 5 ---- src/shared/protocol.h | 4 ++++ src/shared/transport_ssh.c | 12 ++++------ src/shared/transport_ssh.h | 16 ++++++------- tests/test_transport_ssh.c | 47 +++++++++++++------------------------- 10 files changed, 45 insertions(+), 65 deletions(-) diff --git a/CMakeLists.txt b/CMakeLists.txt index 09c10f6..a55d93e 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -26,9 +26,9 @@ elseif(NOT SANITIZER STREQUAL "none") endif() # --- Strict warnings option --- -option(STRICT_WARNINGS "Enable strict warnings (Wextra, Wpedantic, Werror)" OFF) +option(STRICT_WARNINGS "Enable strict warnings (Wextra, Wpedantic, Wformat-signedness, Werror)" OFF) if(STRICT_WARNINGS) - add_compile_options(-Wextra -Wpedantic -Werror) + add_compile_options(-Wextra -Wpedantic -Wformat-signedness -Werror) endif() # --- Coverage option --- diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 8565b63..fb0411b 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -3296,7 +3296,7 @@ int main(int argc, char* argv[]) { exit_code = 1; } } else if (config->use_multithreading) { - exit_code = send_files_multithreaded(&config); + exit_code = send_files_multithreaded(config); } else { exit_code = send_files(config); } diff --git a/src/client/client_send.c b/src/client/client_send.c index c777d52..e5c71f4 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -37,8 +37,6 @@ #include #include -#define STREAM_THRESHOLD (64ULL * 1024 * 1024) - /* Aggregate loaded payload bytes the sender may buffer across the loader queue and the chunk in flight. Sending one chunk adds up to ~2 * MAX_CHUNK_SIZE of transient serialize/compress buffers on top of the queued payloads, so this @@ -1115,7 +1113,7 @@ static Client* connect_transfer_client(const Config* config) { return NULL; } return client_connect_ssh(config->ssh_destination, config->ssh_port, - config->fastsync_server_path, config->old_args, config->rsh_command, + config->fastsync_server_path, config->rsh_command, config->blocking_io, config->remote_options, config->remote_option_count); } @@ -3617,10 +3615,9 @@ send_fail: return ret; } -int send_files_multithreaded(Config** config_ptr) { - if (!config_ptr || !*config_ptr) +int send_files_multithreaded(Config* config) { + if (!config) return 1; - Config* config = *config_ptr; if (config->list_only) return send_list_only(config); if (config->dry_run) diff --git a/src/client/client_send.h b/src/client/client_send.h index f8f3706..c8146c9 100644 --- a/src/client/client_send.h +++ b/src/client/client_send.h @@ -22,7 +22,7 @@ void client_set_abort_armed(bool armed); * never free it, and the caller retains ownership (freeing it with * config_delete() once the call returns). */ int send_files(Config* config); -int send_files_multithreaded(Config** config); +int send_files_multithreaded(Config* config); /* rsync's --ignore-errors deletion gate: with no I/O error during the scan the * deletion phase always proceeds; with one it is suppressed unless * `--ignore-errors` was given. Exposed so the decision can be unit-tested diff --git a/src/shared/chunk.c b/src/shared/chunk.c index ab0fb4c..54e8730 100644 --- a/src/shared/chunk.c +++ b/src/shared/chunk.c @@ -17,8 +17,9 @@ #include "protocol.h" #include "utils.h" -/* Maximum individual file data size within a chunk (64 MB) */ -#define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024) +/* Maximum individual file data size within a chunk (64 MB). Distinct from the + * receiver's whole-file MAX_FILE_DATA_SIZE (256 MB) in file_receive.c. */ +#define MAX_CHUNK_FILE_DATA_SIZE (64ULL * 1024 * 1024) #define MAX_FILES_PER_CHUNK 65536U /* Reserve `charge` against `session`'s connection budget. This mirrors the @@ -382,9 +383,9 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) { } // Reject individual file data larger than the maximum allowed size. - if (file_data_size > MAX_FILE_DATA_SIZE) { + if (file_data_size > MAX_CHUNK_FILE_DATA_SIZE) { log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size, - (unsigned long long)MAX_FILE_DATA_SIZE); + (unsigned long long)MAX_CHUNK_FILE_DATA_SIZE); goto error; } diff --git a/src/shared/file.c b/src/shared/file.c index 9ec981e..bcb46f4 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -26,11 +26,6 @@ #include "protocol.h" #include "xattr.h" -/* Files larger than this are not loaded whole for transfer (the sender streams - * them); a whole-file digest is computed from the path instead. Kept in sync - * with the sender's streaming threshold. */ -#define STREAM_THRESHOLD (64ULL * 1024 * 1024) - static bool write_all(int fd, const void* data, unsigned long long size) { const unsigned char* p = data; unsigned long long done = 0; diff --git a/src/shared/protocol.h b/src/shared/protocol.h index 73b9132..d13c3f7 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -28,6 +28,10 @@ /* Maximum chunk size (64 MB) — prevents unbounded allocation from the wire */ #define MAX_CHUNK_SIZE (64ULL * 1024 * 1024) +/* Files larger than this are not kept fully in memory while loading: the + * loader skips them so the sender streams from the path, and file_checksum + * hashes them from disk in bounded buffers instead of forcing a full load. */ +#define STREAM_THRESHOLD (64ULL * 1024 * 1024) #define MAX_MANIFEST_ENTRIES (1024 * 1024) /* Aggregate bytes retained by one received deletion manifest. */ #define MAX_MANIFEST_BYTES (16ULL * 1024 * 1024) diff --git a/src/shared/transport_ssh.c b/src/shared/transport_ssh.c index 7236133..1be7f21 100644 --- a/src/shared/transport_ssh.c +++ b/src/shared/transport_ssh.c @@ -87,7 +87,7 @@ static int parse_remote_dest(const char* dest, RemoteDest* r) { return 0; } -char* ssh_build_remote_command(const char* server_path, bool old_args, char* const* remote_options, +char* ssh_build_remote_command(const char* server_path, char* const* remote_options, int remote_option_count) { const char* path = server_path ? server_path : "fastsync-server"; const char* suffix = " --stdio"; @@ -105,9 +105,7 @@ char* ssh_build_remote_command(const char* server_path, bool old_args, char* con shell word (remote options below reuse the same escaping), then " --stdio". Quoting the path is the only injection-safe construction: an unquoted path would carry shell metacharacters straight into the remote - shell command. --old-args is kept for CLI/ABI compatibility but no longer - disables that protection. */ - (void)old_args; + shell command. (rsync's --old-args no longer disables that protection.) */ size_t quote_count = 0; for (const char* p = path; *p; p++) if (*p == '\'') @@ -296,8 +294,8 @@ void ssh_free_client_argv(char** argv) { } Client* client_connect_ssh(const char* destination, int port, const char* server_path, - bool old_args, const char* rsh_command, bool blocking_io, - char* const* remote_options, int remote_option_count) { + const char* rsh_command, bool blocking_io, char* const* remote_options, + int remote_option_count) { RemoteDest r; if (parse_remote_dest(destination, &r) != 0) { char* escaped = output_escape(destination, false); @@ -376,7 +374,7 @@ Client* client_connect_ssh(const char* destination, int port, const char* server snprintf(ssh_user, ssh_user_len, "%s", r.host); char* remote_command = - ssh_build_remote_command(server_path, old_args, remote_options, remote_option_count); + ssh_build_remote_command(server_path, remote_options, remote_option_count); if (!remote_command) ssh_child_setup_failed(exec_pipe[1]); char** ssh_argv = ssh_build_client_argv(rsh_command, port, ssh_user, remote_command); diff --git a/src/shared/transport_ssh.h b/src/shared/transport_ssh.h index 08908ce..0ed923e 100644 --- a/src/shared/transport_ssh.h +++ b/src/shared/transport_ssh.h @@ -4,17 +4,17 @@ #include "transport_tcp.h" Client* client_connect_ssh(const char* destination, int port, const char* server_path, - bool old_args, const char* rsh_command, bool blocking_io, - char* const* remote_options, int remote_option_count); + const char* rsh_command, bool blocking_io, char* const* remote_options, + int remote_option_count); /* Build the escaped remote-shell command string (the server program path always * quoted as one remote-shell word, followed by ` --stdio` and each * --remote-option value appended as an individually single-quoted shell word). - * `old_args` is accepted for CLI/ABI compatibility but no longer disables - * quoting: the path is always escaped so a metacharacter-bearing - * --rsync-path can never be interpreted by the remote shell. Every - * --remote-option value is individually escaped with the '\'' sequence and - * values with empty/control characters are rejected at the CLI parse layer. */ -char* ssh_build_remote_command(const char* server_path, bool old_args, char* const* remote_options, + * The path is always escaped so a metacharacter-bearing --rsync-path can never + * be interpreted by the remote shell (the --old-args no-op does not disable + * quoting). Every --remote-option value is individually escaped with the '\'' + * sequence and values with empty/control characters are rejected at the CLI + * parse layer. */ +char* ssh_build_remote_command(const char* server_path, char* const* remote_options, int remote_option_count); /* Build the NULL-terminated child argv for the remote-shell client (argv[0] is * the exec/execvp program). rsh_command is whitespace-split into leading argv diff --git a/tests/test_transport_ssh.c b/tests/test_transport_ssh.c index 2889444..a99630f 100644 --- a/tests/test_transport_ssh.c +++ b/tests/test_transport_ssh.c @@ -5,13 +5,13 @@ static void test_ssh_connect_invalid_dest_no_colon() { /* cppcheck-suppress constVariablePointer */ Client* client = - client_connect_ssh("invalid-destination-no-colon", 22, NULL, false, NULL, false, NULL, 0); + client_connect_ssh("invalid-destination-no-colon", 22, NULL, NULL, false, NULL, 0); EXPECT_NULL(client); } static void test_ssh_connect_invalid_dest_empty() { /* cppcheck-suppress constVariablePointer */ - Client* client = client_connect_ssh("", 22, NULL, false, NULL, false, NULL, 0); + Client* client = client_connect_ssh("", 22, NULL, NULL, false, NULL, 0); EXPECT_NULL(client); } @@ -22,7 +22,7 @@ static void test_ssh_connect_malformed() { setenv("PATH", "", 1); /* cppcheck-suppress constVariablePointer */ - Client* client = client_connect_ssh(":", 22, NULL, false, NULL, false, NULL, 0); + Client* client = client_connect_ssh(":", 22, NULL, NULL, false, NULL, 0); if (saved_path) { setenv("PATH", saved_path, 1); @@ -38,7 +38,7 @@ static void test_ssh_connect_malformed() { * The function launches ssh which will fail to connect, returns a Client. */ static void test_ssh_connect_unreachable() { Client* client = - client_connect_ssh("nonexistent.invalid:/remote/path", 22, NULL, false, NULL, false, NULL, 0); + client_connect_ssh("nonexistent.invalid:/remote/path", 22, NULL, NULL, false, NULL, 0); if (client != NULL) { client_disconnect(client); client_delete(client); @@ -47,23 +47,13 @@ static void test_ssh_connect_unreachable() { } static void test_ssh_remote_command_argument_modes() { - char* command = ssh_build_remote_command("fast sync; touch /tmp/pwned", false, NULL, 0); + char* command = ssh_build_remote_command("fast sync; touch /tmp/pwned", NULL, 0); EXPECT_EQ_STR(command, "'fast sync; touch /tmp/pwned' --stdio"); free(command); - command = ssh_build_remote_command("fast'sync", false, NULL, 0); + command = ssh_build_remote_command("fast'sync", NULL, 0); EXPECT_EQ_STR(command, "'fast'\\''sync' --stdio"); free(command); - - /* --old-args no longer disables injection-safe quoting: the path is still one - single-quoted word, even when it carries shell metacharacters. */ - command = ssh_build_remote_command("fast sync; touch /tmp/pwned", true, NULL, 0); - EXPECT_EQ_STR(command, "'fast sync; touch /tmp/pwned' --stdio"); - free(command); - - command = ssh_build_remote_command("fast'sync; rm -rf /", true, NULL, 0); - EXPECT_EQ_STR(command, "'fast'\\''sync; rm -rf /' --stdio"); - free(command); } /* The build for a single-word argv is [prog, six -o args, "--", user, command]. */ @@ -120,12 +110,12 @@ static void test_ssh_build_client_argv_whitespace_command_and_port() { * returned and no command can run. */ static void test_ssh_connect_rejects_option_host() { /* cppcheck-suppress constVariablePointer */ - Client* client = client_connect_ssh("-oProxyCommand=touch /tmp/pwned:/remote", 22, NULL, false, - NULL, false, NULL, 0); + Client* client = + client_connect_ssh("-oProxyCommand=touch /tmp/pwned:/remote", 22, NULL, NULL, false, NULL, 0); EXPECT_NULL(client); - client = client_connect_ssh("-evil:/remote", 22, NULL, false, NULL, false, NULL, 0); + client = client_connect_ssh("-evil:/remote", 22, NULL, NULL, false, NULL, 0); EXPECT_NULL(client); - client = client_connect_ssh("user@:/remote", 22, NULL, false, NULL, false, NULL, 0); + client = client_connect_ssh("user@:/remote", 22, NULL, NULL, false, NULL, 0); EXPECT_NULL(client); } @@ -135,13 +125,13 @@ static void test_ssh_connect_rejects_option_host() { * ssh_build_remote_command safety boundary for the server path. */ static void test_ssh_remote_command_with_remote_options() { char* noop[] = {"--allow-delete"}; - char* command = ssh_build_remote_command("fastsync-server", false, noop, 1); + char* command = ssh_build_remote_command("fastsync-server", noop, 1); EXPECT_EQ_STR(command, "'fastsync-server' --stdio '--allow-delete'"); free(command); /* Multiple options append in order, each as its own quoted word. */ char* multi[] = {"-v", "--allow-delete"}; - command = ssh_build_remote_command("srv", false, multi, 2); + command = ssh_build_remote_command("srv", multi, 2); EXPECT_EQ_STR(command, "'srv' --stdio '-v' '--allow-delete'"); free(command); @@ -150,29 +140,24 @@ static void test_ssh_remote_command_with_remote_options() { break out into an arbitrary remote command. */ char* val = strdup("--x=un'der; touch /tmp/pwned"); char* dangerous[1] = {val}; - command = ssh_build_remote_command("srv", false, dangerous, 1); + command = ssh_build_remote_command("srv", dangerous, 1); EXPECT_EQ_STR(command, "'srv' --stdio '--x=un'\\''der; touch /tmp/pwned'"); free(command); free(val); - - /* --old-args still quotes both the server path and the remote options. */ - command = ssh_build_remote_command("srv", true, multi, 2); - EXPECT_EQ_STR(command, "'srv' --stdio '-v' '--allow-delete'"); - free(command); } /* The remote command builder refuses to forward an empty or control-character * remote option (defense-in-depth independent of the CLI validation). */ static void test_ssh_remote_command_rejects_bad_options() { char* empty[] = {""}; - EXPECT_NULL(ssh_build_remote_command("srv", false, empty, 1)); + EXPECT_NULL(ssh_build_remote_command("srv", empty, 1)); char nl = '\n'; char* newline[] = {&nl}; - EXPECT_NULL(ssh_build_remote_command("srv", false, newline, 1)); + EXPECT_NULL(ssh_build_remote_command("srv", newline, 1)); char* with_null[] = {NULL}; - EXPECT_NULL(ssh_build_remote_command("srv", false, with_null, 1)); + EXPECT_NULL(ssh_build_remote_command("srv", with_null, 1)); } void test_transport_ssh() { -- 2.54.0 From 134dcd74cc96e603338f2b2fbb543836a721c785 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 21:09:05 +0200 Subject: [PATCH 25/68] refactor: drop dead filter_rules_apply, unify set_error, dedup path_is_within --- src/server/server.c | 7 +------ src/server/server_cli.c | 10 +--------- src/shared/credentials.c | 10 +--------- src/shared/daemon_conf.c | 10 +--------- src/shared/filter.c | 19 ++----------------- src/shared/filter.h | 5 ----- src/shared/utils.c | 10 ++++++++++ src/shared/utils.h | 5 +++++ tests/test_filter.c | 24 ++++++++++++++++-------- 9 files changed, 37 insertions(+), 63 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index 904d80a..e3a0a35 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -226,11 +226,6 @@ static void release_authorization(void) { close(root_fd); } -static bool path_is_within(const char* root, const char* path) { - size_t n = strlen(root); - return strncmp(root, path, n) == 0 && (path[n] == '\0' || path[n] == '/'); -} - /* --mkpath contract: when the client's destination root directory does not exist yet on the server side, --mkpath tells the server to create it (and any missing leading components) below the authorized root at connection @@ -790,7 +785,7 @@ void handler(int file_descriptor) { if (joined_destination) destination = joined_destination; if (!destination || has_path_traversal(destination) || - !path_is_within(authorized_root, destination)) { + !path_is_within_root(authorized_root, destination)) { log_message(LOG_LEVEL_ERROR, "Rejected destination outside authorized root"); free(joined_destination); joined_destination = NULL; diff --git a/src/server/server_cli.c b/src/server/server_cli.c index de505ab..c49fa01 100644 --- a/src/server/server_cli.c +++ b/src/server/server_cli.c @@ -3,20 +3,12 @@ #include "credentials.h" #include "utils.h" #include -#include #include #include #include #include -static void set_error(char* err, size_t err_size, const char* fmt, ...) { - if (!err || err_size == 0) - return; - va_list args; - va_start(args, fmt); - vsnprintf(err, err_size, fmt, args); - va_end(args); -} +#define set_error utils_set_error void server_cli_options_default(ServerCliOptions* opts) { if (!opts) diff --git a/src/shared/credentials.c b/src/shared/credentials.c index 998bae9..e241271 100644 --- a/src/shared/credentials.c +++ b/src/shared/credentials.c @@ -8,7 +8,6 @@ #include #include #include -#include #include #include #include @@ -58,14 +57,7 @@ struct CredentialStore { static const uint8_t k_dummy_stored_key[CREDENTIAL_KEY_LEN] = {0}; static const uint8_t k_dummy_server_key[CREDENTIAL_KEY_LEN] = {0}; -static void set_error(char* err, size_t err_size, const char* fmt, ...) { - if (!err || err_size == 0) - return; - va_list args; - va_start(args, fmt); - vsnprintf(err, err_size, fmt, args); - va_end(args); -} +#define set_error utils_set_error static bool is_comment_char(char c) { return c == '#' || c == ';'; diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index 88b836f..2297a48 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -6,7 +6,6 @@ #include #include #include -#include #include #include #include @@ -17,14 +16,7 @@ /* helpers */ /* ------------------------------------------------------------------ */ -static void set_error(char* err, size_t err_size, const char* fmt, ...) { - if (!err || err_size == 0) - return; - va_list args; - va_start(args, fmt); - vsnprintf(err, err_size, fmt, args); - va_end(args); -} +#define set_error utils_set_error /* Trim leading and trailing ASCII space/tab in place; returns the new start. */ static char* trim_ws(char* s) { diff --git a/src/shared/filter.c b/src/shared/filter.c index f478286..4bfd462 100644 --- a/src/shared/filter.c +++ b/src/shared/filter.c @@ -4,22 +4,12 @@ #include #include #include -#include #include #include #include -/* Write a diagnostic message into the caller's optional buffer. A NULL `err` - * (or a zero size) is a no-op, so a caller that only needs the boolean status - * may pass NULL without the snprintf-on-NULL undefined behaviour. */ -static void filter_set_error(char* err, size_t err_size, const char* fmt, ...) { - if (!err || err_size == 0) - return; - va_list ap; - va_start(ap, fmt); - vsnprintf(err, err_size, fmt, ap); - va_end(ap); -} +/* Write a diagnostic message into the caller's optional buffer. */ +#define filter_set_error utils_set_error /* ---- Ordered rule lists ---- */ @@ -870,8 +860,3 @@ FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel } return FILTER_ACTION_NONE; } - -FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, - bool is_dir) { - return filter_rules_apply_side(list, rel_path, leaf, is_dir, FILTER_SIDE_SENDER); -} diff --git a/src/shared/filter.h b/src/shared/filter.h index bd9877b..bc9e9ca 100644 --- a/src/shared/filter.h +++ b/src/shared/filter.h @@ -131,9 +131,4 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path, const char* leaf, bool is_dir, unsigned side); -/* Sender-side convenience wrapper (kept for callers/tests that only need the - * transfer decision). */ -FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf, - bool is_dir); - #endif diff --git a/src/shared/utils.c b/src/shared/utils.c index d028e36..947a4d9 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -7,6 +7,7 @@ #include #include #include +#include #include #include #include @@ -48,6 +49,15 @@ const char* utils_get_authorized_root_path(void) { return authorized_root_path; } +void utils_set_error(char* err, size_t err_size, const char* fmt, ...) { + if (!err || err_size == 0) + return; + va_list args; + va_start(args, fmt); + vsnprintf(err, err_size, fmt, args); + va_end(args); +} + bool path_is_within_root(const char* root, const char* path) { size_t root_len = strlen(root); return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/'); diff --git a/src/shared/utils.h b/src/shared/utils.h index 7bf89d8..aedb9ce 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -225,6 +225,11 @@ void utils_set_authorized_root_fd(int fd); * threads spawn; see utils.c). */ int utils_get_authorized_root_fd(void); const char* utils_get_authorized_root_path(void); +/* Write a diagnostic message into a caller-supplied buffer, mirroring + * vsnprintf. A NULL `err` or a zero `err_size` is a no-op, so a caller that + * only needs the boolean status may safely pass NULL. Returns nothing; the + * buffer is always NUL-terminated by vsnprintf when err_size > 0. */ +void utils_set_error(char* err, size_t err_size, const char* fmt, ...); /* True when `path` is `root` itself or lies directly beneath it: a lexical * prefix test requiring the byte after `root` to be '\0' or '/'. Both `root` * and `path` must be absolute canonical paths free of "."/".." components (the diff --git a/tests/test_filter.c b/tests/test_filter.c index 76e2729..4036cbd 100644 --- a/tests/test_filter.c +++ b/tests/test_filter.c @@ -179,8 +179,10 @@ static void test_filter_rules_apply_supported_modifiers() { const char* texts[] = {"- *.tmp"}; FilterRuleList* list = filter_base_build(texts, 1, false, false, NULL, 0); EXPECT_NOT_NULL(list); - EXPECT_EQ_INT(filter_rules_apply(list, "b.tmp", "b.tmp", false), FILTER_ACTION_EXCLUDE); - EXPECT_EQ_INT(filter_rules_apply(list, "a.txt", "a.txt", false), FILTER_ACTION_NONE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "b.tmp", "b.tmp", false, FILTER_SIDE_SENDER), + FILTER_ACTION_EXCLUDE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "a.txt", "a.txt", false, FILTER_SIDE_SENDER), + FILTER_ACTION_NONE); filter_rule_list_free(list); } /* anchored include then exclude-all */ @@ -188,8 +190,10 @@ static void test_filter_rules_apply_supported_modifiers() { const char* texts[] = {"+ /a.txt", "- *"}; FilterRuleList* list = filter_base_build(texts, 2, false, false, NULL, 0); EXPECT_NOT_NULL(list); - EXPECT_EQ_INT(filter_rules_apply(list, "a.txt", "a.txt", false), FILTER_ACTION_INCLUDE); - EXPECT_EQ_INT(filter_rules_apply(list, "b.txt", "b.txt", false), FILTER_ACTION_EXCLUDE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "a.txt", "a.txt", false, FILTER_SIDE_SENDER), + FILTER_ACTION_INCLUDE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "b.txt", "b.txt", false, FILTER_SIDE_SENDER), + FILTER_ACTION_EXCLUDE); filter_rule_list_free(list); } /* negate */ @@ -197,8 +201,10 @@ static void test_filter_rules_apply_supported_modifiers() { const char* texts[] = {"-! *.o"}; FilterRuleList* list = filter_base_build(texts, 1, false, false, NULL, 0); EXPECT_NOT_NULL(list); - EXPECT_EQ_INT(filter_rules_apply(list, "foo.c", "foo.c", false), FILTER_ACTION_EXCLUDE); - EXPECT_EQ_INT(filter_rules_apply(list, "foo.o", "foo.o", false), FILTER_ACTION_NONE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "foo.c", "foo.c", false, FILTER_SIDE_SENDER), + FILTER_ACTION_EXCLUDE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "foo.o", "foo.o", false, FILTER_SIDE_SENDER), + FILTER_ACTION_NONE); filter_rule_list_free(list); } /* dir-only trailing slash */ @@ -206,8 +212,10 @@ static void test_filter_rules_apply_supported_modifiers() { const char* texts[] = {"+ dir/", "- *"}; FilterRuleList* list = filter_base_build(texts, 2, false, false, NULL, 0); EXPECT_NOT_NULL(list); - EXPECT_EQ_INT(filter_rules_apply(list, "dir", "dir", true), FILTER_ACTION_INCLUDE); - EXPECT_EQ_INT(filter_rules_apply(list, "dir", "dir", false), FILTER_ACTION_EXCLUDE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "dir", "dir", true, FILTER_SIDE_SENDER), + FILTER_ACTION_INCLUDE); + EXPECT_EQ_INT(filter_rules_apply_side(list, "dir", "dir", false, FILTER_SIDE_SENDER), + FILTER_ACTION_EXCLUDE); filter_rule_list_free(list); } } -- 2.54.0 From 3adb6dddb5e0950b600d9d586fb348253cbb894b Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 21:11:04 +0200 Subject: [PATCH 26/68] chore: drop tracked scratch data and extend .gitignore --- .gitignore | 9 +++++++++ test_data-manual/dst/f.bin | 1 - test_data-manual/dst/test_data-manual/src/f.bin | 1 - test_data-manual/src/f.bin | 1 - 4 files changed, 9 insertions(+), 3 deletions(-) delete mode 100644 test_data-manual/dst/f.bin delete mode 100644 test_data-manual/dst/test_data-manual/src/f.bin delete mode 100644 test_data-manual/src/f.bin diff --git a/.gitignore b/.gitignore index d238172..a2f940d 100644 --- a/.gitignore +++ b/.gitignore @@ -12,3 +12,12 @@ build_docker2/ # Test/run artifacts root/ test_partial_install_tmp/ + +# Editor/tooling + test caches/artifacts +.pytest_cache/ +*.gcda +*.gcno +*.gcov +di/ +test_data-manual/ +*.log diff --git a/test_data-manual/dst/f.bin b/test_data-manual/dst/f.bin deleted file mode 100644 index 870ff8b..0000000 --- a/test_data-manual/dst/f.bin +++ /dev/null @@ -1 +0,0 @@ -OLDDEST diff --git a/test_data-manual/dst/test_data-manual/src/f.bin b/test_data-manual/dst/test_data-manual/src/f.bin deleted file mode 100644 index 924c75c..0000000 --- a/test_data-manual/dst/test_data-manual/src/f.bin +++ /dev/null @@ -1 +0,0 @@ -NEWCONTENT diff --git a/test_data-manual/src/f.bin b/test_data-manual/src/f.bin deleted file mode 100644 index 924c75c..0000000 --- a/test_data-manual/src/f.bin +++ /dev/null @@ -1 +0,0 @@ -NEWCONTENT -- 2.54.0 From cee9b7647c0180679a7fbb7d92075fa198770698 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 21:14:37 +0200 Subject: [PATCH 27/68] docs: fix parity tally, compat rows, changelog, handoff for audit cycle --- CHANGELOG.md | 66 +++++++++++++++++++++++++++++++++++++ HANDOFF.md | 87 ++++++++++++++++++++++++++++++++++--------------- RSYNC_COMPAT.md | 39 +++++++++++++++------- 3 files changed, 154 insertions(+), 38 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 658e241..beed8f1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -10,6 +10,14 @@ The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0). `RSYNC_COMPAT.md` moves from **116 ✅ / 14 ⚠️ / 27 ❌** to **120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows. +An audit cycle follows on the same wire version (`PROTOCOL_VERSION` stays +2.28.0): a security-and-correctness pass over the parity-2.29 baseline. It fixes +a `--temp-dir` symlink escape, gates client-controlled special permission bits, +corrects `--partial-dir`/`--bwlimit`/`-z` behavior, rejects unsupported filter +modifiers, and tightens client and wire validation. No parity row changes +classification, so the matrix stays **120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows; the +affected rows' notes and the summary tally in `RSYNC_COMPAT.md` were updated. + ### Changed - **rsync-exact traversal order.** The sequential scanner now walks each @@ -47,6 +55,64 @@ The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0). (a general whole-file limit, not basis-specific). - `--stats` byte totals and `--msgs2stderr` stay documented divergences. +### Security + +- **`--temp-dir` symlink escape fixed.** The receiver's scratch directory was + opened with a bare `open()`, so a symlink planted under the receive root could + redirect receiver scratch files outside the authorized root. The opened + directory is now judged by the real path of its fd (`/proc/self/fd` via + `realpath`) and an escaping target is refused (`EACCES`, logged); an in-root + link to another filesystem (the `EXDEV` fallback case) still works. +- **Client-controlled special bits masked when super-user activities are not + permitted.** Setuid/setgid/sticky bits (`--perms`, `--chmod`, the symlink and + special-node paths, and deferred directory modes) are now stripped when the + connection forbids super activities (`--no-super`, a non-opted daemon module, + a privileged listener without `--allow-super`); exact rsync semantics are + preserved wherever super activities are permitted. +- **Daemon umask no longer forced to `0`.** `daemonize()` now sets the + conventional `022`, so implied parent directories created without `-p` are no + longer world-writable `0777`. +- **Credentials and signal handling hardened.** Secret files are opened with + `O_NOFOLLOW|O_NONBLOCK` (while allowing fd-backed store paths), and signal + handlers use `sigaction` with async-signal-safe bodies. + +### Fixed + +- **`-z` on 100–256 MiB files.** The decompressor's internal ceiling was 100 MiB + while the receiver advertises and the sender compresses whole files up to + `MAX_RECEIVE_WHOLE_FILE_SIZE` (256 MiB), so `-z` on a 100–256 MiB regular file + failed with `Declared decompressed size exceeds 104857600 bytes`. The ceiling + is now defined in terms of the protocol whole-file bound (still an + allocation-clamped bomb guard). +- **`--bwlimit` now paces `--sendfile`.** The plaintext-TCP `--sendfile` fast + path bypassed the protocol's token bucket, so the limit was ignored there. It + now throttles through the same per-session leaky bucket as the TLS path. +- **`--partial-dir` implies `--partial`.** Matching rsync 3.4.1 (which sets + `keep_partial` after option parsing), `--partial-dir=DIR` alone retains an + interrupted transfer's partial and wins over an explicit `--no-partial`; + `--inplace` still bypasses the partial machinery. +- **Unsupported filter modifiers rejected.** The `x` xattr-name modifier and the + merge-only `e`/`n`/`w` modifiers are rejected with a clear error instead of + being silently ignored (`x` on merge/dir-merge rules) or folded into the + pattern (producing misleading merge-file errors). Glued patterns (`-newfile`, + `-e2e`) and mixed tokens (`H,!secret`) keep their historical parsing. +- **Miscellaneous correctness fixes:** `--filter` rule count is checked + client-side against `MAX_FILTER_RULES` before any network I/O (the receiver + still re-checks the expanded count); unknown wire `Status` values are rejected + as protocol errors; a mutex leak on an init-failure path, an `errno` read + after `free()` in deferred delete application, `log_perror` misuse for + non-`errno` conditions, and a `NULL` `server_host`/`ssh_destination` + allocation path were fixed; `SSL_read` length is clamped and `sendfile` + `poll()` retries on `EINTR`. + +### Refactored / Docs + +- Dropped dead `filter_rules_apply` and dead `--old-args` plumbing, unified + `set_error`, deduplicated `path_is_within` and shared constants, and added + printf format attributes (fixing format mismatches). `RSYNC_COMPAT.md`, + `CHANGELOG.md` and `HANDOFF.md` were updated for the audit cycle; the + `RSYNC_COMPAT.md` summary tally was corrected to match the rows. + ## [2.28.0] - 2026-09-20 The rsync-parity cycle. `PROTOCOL_VERSION` moves `2.26.0 → 2.27.0 → 2.28.0`; diff --git a/HANDOFF.md b/HANDOFF.md index 73dba0d..e2931dc 100644 --- a/HANDOFF.md +++ b/HANDOFF.md @@ -1,22 +1,30 @@ -# FastSync — Session Handoff (2026-09-20) +# FastSync — Session Handoff (2026-09-21) ## Current status -- **Release `v2.28.0`** is tagged and merged to `main` (PR #304, `b4d54504`). - `dev` is at `558782d` (the incremental-check flake fix). +- **Release `v2.28.0`** is tagged and merged to `main`: tag `v2.28.0` points at + `ee6523a`, and the PR #304 merge commit `b4d54504` is on `main`. +- **`dev` is at `0fbb9de`** — the merge of parity cycle 2.29 (PR #305). The old + `558782d` (incremental-check flake fix) is an ancestor. - **`PROTOCOL_VERSION` = `"2.28.0"`** (`src/shared/config.h`); CMake `project(FastFileTransfer VERSION 2.28.0)`. -- **Parity cycle 2.29 on branch `feat/parity-2.29`** (from `dev` @ `558782d`), - no wire change. It closes the scanner-order, delete-timing, relative-basis and - fuzzy-eligibility residuals and improves the `--info`/`--stats`/`--debug` - partials. Parity matrix: **120 ✅ / 10 ⚠️ / 27 ❌ = 157** (was 116/14/27). - Remaining ⚠️ rows: `--info`, `--debug`, `--msgs2stderr`, `--stats`, - `--progress`, `--delete-before`, `--compare-dest`/`--copy-dest`/`--link-dest` - (over-256-MiB basis MISS), `-y`/`--fuzzy` (256 MiB buffer cap). -- **Deferred (needs a wire bump):** the `--progress`/`--info` receiver→sender - event channel (root `./` line, ancestor suppression, `skip`/`backup` echo, - symlink/empty-dir quick-check); `--delete-before` phase-0 keep-set; and the - general >256 MiB single-file streaming limit (B4). -- Feature branch `feat/parity-2.29`; integration PR to `dev` pending. +- **Parity cycle 2.29 is merged to `dev`** (PR #305), no wire change. It closed + the scanner-order, delete-timing, relative-basis and fuzzy-eligibility + residuals and improved the `--info`/`--stats`/`--debug` partials. Parity + matrix: **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. Remaining ⚠️ rows: `--info`, + `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `--delete-before`, the + three basis-dir options, and `-y`/`--fuzzy`. +- **Audit cycle complete on branch `fix/audit-cycle`** (branched from `dev` @ + `0fbb9de`), integration PR to `dev` pending. No wire change + (`PROTOCOL_VERSION` stays 2.28.0). It lands the receiver/client security and + correctness fixes — `--temp-dir` symlink-escape confinement, special-bit + masking under a super-off policy, daemon `umask(022)`, the `-z` decompression + ceiling raised to the 256 MiB whole-file bound, `--bwlimit` pacing the + plaintext `--sendfile` path, `--partial-dir` implying `--partial`, rejection + of unsupported filter modifiers (`x`/`e`/`n`/`w`), client-side + `MAX_FILTER_RULES` enforcement, unknown wire `Status` rejection, and the + accompanying refactors/docs. The parity matrix is unchanged at + **120 ✅ / 10 ⚠️ / 27 ❌ = 157**; this docs pass (worktree `fix/audit-docs2`) + corrects the `RSYNC_COMPAT.md` summary tally to match the rows. ## What landed this session @@ -102,7 +110,8 @@ uptodate` plus the leading `./` root name line for `--info=name` (only the root-line trigger condition and receiver-side `skip` wording remain). Matrix now **111 ✅ / 14 ⚠️ / 32 ❌ = 157**; differential + unit tests added in - `test_features.py`, `test_option_parity.py`, `test_delete_plan.c`, + `test_features.py`, `test_option_parity.py`, the unit test + `tests/test_delete_plan.c`, `test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`. 12. **No-wire parity track 2b** on `feat/parity-2.28` (no protocol change): `--progress`/`-P`/`--info=progress` (when not `--quiet`) now run an opt-in @@ -199,17 +208,43 @@ **116 ✅ / 14 ⚠️ / 27 ❌ = 157** (the `--delete`/`--delete-during` rows stay ⚠️ for the abort boundary; `--delete-after` stays ✅). +17. **Audit cycle** on `fix/audit-cycle` (from `dev` @ `0fbb9de`; + `PROTOCOL_VERSION` stays `2.28.0`): a security/correctness pass over the + parity-2.29 baseline. It raises the decompression ceiling to the 256 MiB + protocol whole-file bound (`-z` on 100–256 MiB files now works), paces the + plaintext-TCP `--sendfile` path with `--bwlimit`, confines the `--temp-dir` + scratch dir by the fd's real path (symlink escape refused), masks + client-controlled setuid/setgid/sticky bits when super activities are not + permitted, sets the daemon umask to `022`, makes `--partial-dir` imply + `--partial`, rejects the unsupported filter modifiers (`x`/`e`/`n`/`w`), + enforces `MAX_FILTER_RULES` client-side, rejects unknown wire `Status` + values, and hardens credentials/signal handling (with the accompanying + refactors and docs). No row changes classification, so the matrix stays + **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. This docs pass is on `fix/audit-docs2`. + ## Next steps -1. **Merge PR #284** (`dev` -> `main`) once reviewed (protected branch). -2. **Deferred security items** (documented, not implemented): - - Pre-auth config/daemon-auth handshake has no aggregate wall-clock deadline - (per-message timeout only) — slowloris holds connection slots. - - Per-source registry fails open when the shared table is full (per-module/global - caps and host ACLs still apply); consider fail-closed or larger/evicting table. - - SCRAM-like daemon auth has no TLS channel binding (and is not RFC 5802). - - `cleanup()` signal handler calls non-async-signal-safe teardown; daemon `umask(0)`. - - Wire protocol assumes homogeneous word size/endianness (lengths are native - `size_t`) — document or move to fixed-width framing. +1. **Open and merge the audit-cycle PR** (`fix/audit-cycle`, including this + `fix/audit-docs2` docs pass) into `dev` once reviewed. `dev` is the default + branch; all PRs target `dev`, never `main` directly. +2. **Remaining deferred items:** + - **Large structural refactors:** delete-engine consolidation + (`delete_extras_fd`/`manifest_delete_extras`/the delete-plan path), + god-function splits, and translation-unit splits. + - **`--progress`/`--info` receiver→sender event channel:** the root `./` + line, ancestor-directory suppression, receiver-side `skip`/`backup` echo, + and symlink/empty-dir quick-check feedback. + - **`--delete-before` phase-0 keep-set** (rsync fixes the file list before + the data pass; FastSync keeps its pre-scan snapshot race). + - **>256 MiB single-file streaming** (B4, the general whole-file limit). + - **Wire native-size framing:** lengths are native `size_t` and the protocol + assumes homogeneous word size/endianness — document or move to fixed-width + framing. + - **SCRAM-like daemon auth channel binding:** no TLS channel binding today + (and it is not RFC 5802). + - Still-open security nits: the pre-auth config/daemon-auth handshake has no + aggregate wall-clock deadline (per-message timeout only — slowloris holds + connection slots); the per-source registry fails open when the shared table + is full (per-module/global caps and host ACLs still apply). 3. **Out of scope / intentional:** pull (remote source) mode is **not** planned — FastSync is push-only; see `RSYNC_COMPAT.md#direction`. diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 11aa01b..437b113 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -6,8 +6,8 @@ This document maps rsync's full feature set to FastSync's current implementation | Status | Count | Description | |--------|-------|-------------| -| ✅ Parity | 116 | Reproduces rsync's semantics for this option's scope | -| ⚠️ Caveat | 14 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) | +| ✅ Parity | 120 | Reproduces rsync's semantics for this option's scope | +| ⚠️ Caveat | 10 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) | | ❌ Divergent | 27 | Rejected, an accepted no-op, deliberately non-rsync (native config/auth/batch, privileged namespaces, safe-subset privilege), or impossible on any portable filesystem call | | **Total** | **157** | One row per rsync option/feature group; a row may name several spellings | @@ -62,9 +62,21 @@ matrix is **111 ✅ / 13 ⚠️ / 33 ❌ = 157**. - **Fuzzy eligibility.** The `-y/--fuzzy` candidate search no longer inherits the ordinary delta engine's 16 KiB minimum or 10× ratio bound, so an oversized or sub-16-KiB sibling is reused as rsync reuses it (`test_parity_basis_fuzzy.py`). - **Output partials.** `--info=mount`/`--info=stats`, the `--stats` `dir:` breakdown under `-r`, and real `--debug` output for `flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send` were added (`test_parity_info_mount_stats.py`, `test_output_parity.py`, `test_parity_debug.py`); those rows stay ⚠️ for their remaining documented residuals. `--delete-before`'s phase-0 late-file divergence and the `--progress` root/ancestor/symlink feedback remain open (they need a receiver→sender event channel), and the >256 MiB single-file streaming limit (B4) was not addressed. The matrix is now **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. +**Audit cycle (no wire change; `PROTOCOL_VERSION` stays 2.28.0).** A security-and-correctness audit pass ran against the parity-2.29 baseline; none of the fixes changes a row's classification, so the matrix stays **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. The affected rows (`-z`/`--compress`, `--bwlimit`, `-T`/`--temp-dir`, `-p`/`--chmod`, `--partial-dir`, `--filter`) had their notes updated in place: + +- **Decompression ceiling.** `MAX_DECOMPRESSED_SIZE` was 100 MiB while the receiver advertises and the sender compresses whole files up to `MAX_RECEIVE_WHOLE_FILE_SIZE` (256 MiB), so `-z` on a 100–256 MiB regular file failed with `Declared decompressed size exceeds 104857600 bytes`. The ceiling is now defined in terms of the protocol whole-file bound (still a real allocation-clamped bomb guard), so the two cannot drift; `-z` on 100–256 MiB files now works. +- **`--bwlimit` with `--sendfile`.** The plaintext-TCP `--sendfile` fast path wrote through `sendfile(2)` without passing through the protocol's token bucket, so `--bwlimit` was ignored on that path. It is now paced through the same per-session leaky bucket, so TLS and plaintext transports share identical `--bwlimit` semantics. +- **`--temp-dir` confinement.** The receiver's scratch dir was opened with a bare `open()`, so a client-planted symlink under the receive root could redirect receiver scratch files outside the authorized root. The opened directory is now judged by the real path of its fd (`/proc/self/fd` via `realpath`), and an escaping target is refused (`EACCES`, logged); an in-root link to another filesystem (the `EXDEV` fallback case) still works. +- **Special-bit masking and daemon umask.** Setuid/setgid/sticky bits from the client (`--perms`, `--chmod`, symlink and special-node paths, deferred directory modes) were applied even when the connection forbade super-user activities. They are now stripped when the super policy is off (`FileAttrPolicy.super_permitted`), and exact rsync semantics are preserved when permitted. The daemon's forced `umask(0)` is now `umask(022)`, so implied parent directories are no longer world-writable `0777`. +- **`--partial-dir` implies `--partial`.** Matching rsync 3.4.1 (which sets `keep_partial` after option parsing), `--partial-dir=DIR` alone now retains an interrupted transfer's partial and wins over an explicit `--no-partial`; `--inplace` still bypasses the partial machinery. +- **Filter modifiers.** The `x` xattr-name modifier and the merge-only `e`/`n`/`w` modifiers are now rejected with a clear error instead of being silently ignored (`x` on merge/dir-merge rules) or folded into the pattern (producing misleading merge-file errors). Glued patterns (`-newfile`, `-e2e`) keep their historical parsing. +- **Bounds and wire validation.** `--filter` rule count is now checked client-side against `MAX_FILTER_RULES` (with an actionable message before any network I/O) rather than surfacing as an opaque receiver protocol error; `send_protect_entries()` still re-checks the expanded count. Unknown wire `Status` values are rejected as protocol errors (`status_is_valid()`), and the audit also fixed a mutex leak on an init-failure path, an `errno`-after-`free()` in deferred delete application, `log_perror` misuse for non-`errno` conditions, `SSL_read` length clamping, `sendfile` `poll` `EINTR` retry, and printf-format/attribute issues. + **Parity completion wave (protocol 2.23.0 → 2.26.0).** This wave closed the remaining gaps the rsync-parity wave left open (delete timing, wire counters and -output, codec breadth, general `-R`/`-d`, the full filter grammar, receiver-side +output, codec breadth, general `-R`/`-d`, the filter grammar (the unsupported +`x` xattr-name and merge-only `e`/`n`/`w` modifiers are explicitly rejected, not +silently accepted), receiver-side name resolution, absolute basis dirs, and the remaining client quick wins) and reclassified the inherently non-rsync rows as **divergent** (native daemon config/auth, the non-interoperable batch container, `--fake-super`'s xattr @@ -122,7 +134,7 @@ Every one of those has an entry below with its remaining caveats. |------|-------------------|-----------------|-------| | `--exclude-from=FILE` | Read exclude patterns from file | ✅ Parity | Reads patterns from file | | `--include-from=FILE` | Read include patterns from file | ✅ Parity | Reads patterns from file | -| `--filter=RULE` | Add file-filtering rule | ✅ Parity | The short `-f` **is** bound to `--filter` (the old FastSync sendfile conflict is gone; sendfile is long-only `--sendfile`), and `-f RULE`, `-f=RULE`, `--filter=RULE` and the two-argument form all parse. Protocol 2.26.0 implements rsync's filter grammar: `+`/`-`, `include`/`exclude`, a leading `/` anchor (to the transfer root or a `.rsync-filter` file's directory), a trailing `/` dir-only rule, and the `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R` and `clear`/`!` words, including the `:`/`.` modifiers. First match wins; the filter layer is independent of `--exclude`/`--include`. **Track 4a (protocol 2.28.0) adds the receiver filter engine:** the sender compiles its root-level rules exactly as the scanner does (`filter_base_build`) and streams them as one bounded, self-describing config-frame block; the receiver reconstructs them and re-applies first-match-wins to every extraneous destination path during deletion, so a `P *.log` rule protects a destination-only `extra.log` (differential `filter_protect`/`filter_protect_during`/`filter_protect_delay` vs rsync 3.4.1, plus the `-n` would-delete enumeration) — matching rsync's dual-sided engine for the command-line rule set. **Remaining residual:** per-directory merge (`:`/`.`, and therefore `-F`) is not yet re-derived on the receiver; a destination-only entry that matches ONLY a per-directory merge rule is still protected only through the sender-derived source-mirror prefixes, not by the received base rule list | +| `--filter=RULE` | Add file-filtering rule | ✅ Parity | The short `-f` **is** bound to `--filter` (the old FastSync sendfile conflict is gone; sendfile is long-only `--sendfile`), and `-f RULE`, `-f=RULE`, `--filter=RULE` and the two-argument form all parse. Protocol 2.26.0 implements rsync's filter grammar: `+`/`-`, `include`/`exclude`, a leading `/` anchor (to the transfer root or a `.rsync-filter` file's directory), a trailing `/` dir-only rule, and the `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R` and `clear`/`!` words, including the `:`/`.` modifiers. The xattr-name `x` modifier and the merge-only `e`/`n`/`w` modifiers are **explicitly rejected with a clear error** (audit-cycle fix: the list parser used by `--filter`/`-f` previously silently ignored `x` on merge/dir-merge rules and folded `e`/`n`/`w` into the pattern, producing misleading failures); a token made up solely of modifier characters that names an unsupported modifier is rejected, while glued patterns (`-newfile`, `-e2e`) and mixed tokens (`H,!secret`) keep their historical parsing. First match wins; the filter layer is independent of `--exclude`/`--include`. **Track 4a (protocol 2.28.0) adds the receiver filter engine:** the sender compiles its root-level rules exactly as the scanner does (`filter_base_build`) and streams them as one bounded, self-describing config-frame block; the receiver reconstructs them and re-applies first-match-wins to every extraneous destination path during deletion, so a `P *.log` rule protects a destination-only `extra.log` (differential `filter_protect`/`filter_protect_during`/`filter_protect_delay` vs rsync 3.4.1, plus the `-n` would-delete enumeration) — matching rsync's dual-sided engine for the command-line rule set. **Remaining residual:** per-directory merge (`:`/`.`, and therefore `-F`) is not yet re-derived on the receiver; a destination-only entry that matches ONLY a per-directory merge rule is still protected only through the sender-derived source-mirror prefixes, not by the received base rule list | | `--files-from=FILE` | Read source file list from file | ✅ Parity | Entries are paths relative to the source root (leading `./` stripped, `..`/absolute rejected at parse time, blank lines ignored; NUL-delimited with `-0`). A listed regular file is transferred; a listed directory transfers its whole subtree (FastSync recursion is always on). Non-listed paths are pruned by the scanner; the delete manifest is scoped to the listed directory subtrees. A listed entry that does not exist is a hard error unless `--ignore-missing-args`/`--delete-missing-args` is given. **An empty list is a zero-transfer success (exit 0), matching rsync 3.4.1** — the earlier claim that rsync reports "no source files specified" was wrong. Scalability note: `file_list_affects` is O(list size) per scanned entry, so a very large list against a huge tree is quadratic (the documented bound) | | `-0`, `--from0` | Delimit *-from files with NULs | ✅ Parity | `--files-from` entries become NUL-delimited; the flag may appear before or after `--files-from` on the command line. NUL mode preserves entry bytes exactly (trailing CR/LF are part of the name; only newline mode trims them) | | `--max-size=SIZE` | Skip files larger than SIZE | ✅ Parity | `max_size` in scanner | @@ -168,9 +180,9 @@ Every one of those has an entry below with its remaining caveats. | `--backup-dir=DIR` | Backup directory hierarchy | ✅ Parity | `backup_dir` config field | | `--suffix=SUFFIX` | Backup suffix (default ~) | ✅ Parity | `suffix` config field | | `--delay-updates` | Put updated files in place at end | ❌ Divergent | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication, and **`--force` is honored at publication** (protocol 2.23.0): a staged regular file or symlink may replace a destination directory that blocks it. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). **Reclassified Divergent (differential evidence):** the staging name is fixed and a delayed run wipes a pre-existing destination tree of that name at start even without `--delete`, whereas rsync uses its own internal temp name and leaves a genuine destination entry named `.fastsync-stage` untouched (`test_delay_updates_staging_name_collision_residual`); deletion also runs before publication while rsync's `--delay-updates` implies `--delete-after`. Works in single-threaded and `-j`/`--threads` modes | -| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ❌ Divergent | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). A **relative** dir matches rsync exactly: it is resolved below the receive/destination root and must already exist (differentially verified: `rsync -a --temp-dir=scratch src/ dst/` and FastSync produce identical trees and an empty scratch dir). **Reclassified as a deliberate divergence because an absolute `--temp-dir` is rejected by the receiver** — it is resolved verbatim by rsync standalone (which will use `/tmp` or any other absolute directory, including one outside the destination), but FastSync's security-reviewed receiver confines the scratch dir to the authorized receive root and rejects any absolute path or one containing `..`. A differential test confirms rsync exits 0 using an absolute scratch dir while FastSync refuses before writing anything into it (the scratch dir stays empty). Its daemon mode also confines relative to the module, but standalone rsync's absolute-temp-dir behavior is not reproduced because it would let a client place receiver scratch files outside the sandbox. Temp copies use a unique name in the scratch dir and are atomically renamed into place; **on `EXDEV` (scratch dir and destination on different filesystems, reachable via a confined relative symlink) the receiver falls back to a non-atomic copy instead of aborting**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir | +| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ❌ Divergent | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). A **relative** dir matches rsync exactly: it is resolved below the receive/destination root and must already exist (differentially verified: `rsync -a --temp-dir=scratch src/ dst/` and FastSync produce identical trees and an empty scratch dir). **Reclassified as a deliberate divergence because an absolute `--temp-dir` is rejected by the receiver** — it is resolved verbatim by rsync standalone (which will use `/tmp` or any other absolute directory, including one outside the destination), but FastSync's security-reviewed receiver confines the scratch dir to the authorized receive root and rejects any absolute path or one containing `..`. **Audit-cycle hardening:** the opened dir is additionally judged by the real path of its fd (`/proc/self/fd`), so a client-planted symlink under the receive root cannot redirect receiver scratch files outside the authorized root (an escaping target is refused with `EACCES`), while an in-root symlink to another filesystem — the `EXDEV` fallback case — still works. A differential test confirms rsync exits 0 using an absolute scratch dir while FastSync refuses before writing anything into it (the scratch dir stays empty). Its daemon mode also confines relative to the module, but standalone rsync's absolute-temp-dir behavior is not reproduced because it would let a client place receiver scratch files outside the sandbox. Temp copies use a unique name in the scratch dir and are atomically renamed into place; **on `EXDEV` (scratch dir and destination on different filesystems, reachable via a confined relative symlink) the receiver falls back to a non-atomic copy instead of aborting**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir | | `--partial` | Keep partially transferred files | ✅ Parity | On a failed/interrupted write the already-written temp file is retained at the destination path (best-effort rename instead of unlink) so a later `--append`/`--append-verify` run can resume it. Retention never runs when no data was actually written or under `--ignore-existing`/`--existing` (the destination is not ours to overwrite), and it only ever renames the already-written temp. A failed rename falls back to the normal unlink | -| `--partial-dir=DIR` | Keep partial files in DIR | ✅ Parity | With `--partial`, the working file is written under the confined partial directory (a relative dir below the receive root) and atomically renamed into place once complete, so an interrupted transfer leaves a resumable copy there and completed transfers do not linger under it. `--inplace` bypasses the partial dir (rsync parity). Requires `--partial` | +| `--partial-dir=DIR` | Keep partial files in DIR | ✅ Parity | The working file is written under the confined partial directory (a relative dir below the receive root) and atomically renamed into place once complete, so an interrupted transfer leaves a resumable copy there and completed transfers do not linger under it. `--inplace` bypasses the partial dir (rsync parity). **Implies `--partial`** (audit-cycle fix, matching rsync 3.4.1, which sets `keep_partial` after option parsing): `--partial-dir=DIR` alone retains an interrupted transfer's partial, and the implication wins over an explicit `--no-partial` regardless of order. `--inplace` is the exception — it writes the destination in place with no partial staging, so the implication is skipped | ## 7. Deletion @@ -316,12 +328,12 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| | `--preserve` | (FastSync alias, not an rsync flag) | ✅ Parity | **FastSync-only alias** for `-p` + `-t` (mode + mtime), long-form only. It is not rsync's `--preserve` (rsync has no such option); the short `-M` that used to spell it is now rsync's `--remote-option`. The wire metadata also carries uid/gid for `-o`/`-g`/`-a`, and ownership is applied via `-o`/`-g`, `-a`, or an explicit identity flag (`--numeric-ids`/`--usermap`/`--groupmap`/`--chown`/`--copy-as`) | -| `-p`, `--perms` | Preserve permissions | ✅ Parity | Real per-attribute flag (protocol 2.22.0): `preserve_perms` applies the source mode independently of times/owner/group. **Strict rsync parity (protocol 2.23.0): the source mode is copied exactly, including setuid/setgid/sticky and group/other-write bits — there is no masking.** Without `-p`, a new file gets `source_mode & ~umask` when metadata is present (else the historical fixed `0644`); new directories without `-p` still use FastSync's `0755` creation default, because directory metadata is only applied when a directory attribute is requested. `-A/--acls` implies `-p`; `--chmod` does **not** imply `-p` (rsync parity) and applies its own unsanitized changes to the new mode. `-X/--xattrs` does not imply `-p`. The SSH port moved to `--ssh-port`. rsync-parity short form | +| `-p`, `--perms` | Preserve permissions | ✅ Parity | Real per-attribute flag (protocol 2.22.0): `preserve_perms` applies the source mode independently of times/owner/group. **Strict rsync parity when super-user activities are permitted (protocol 2.23.0): the source mode is copied exactly, including setuid/setgid/sticky and group/other-write bits.** **Audit-cycle fix:** when the connection forbids super-user activities (`--no-super`, a non-opted daemon module, or a privileged standalone listener without `--allow-super`), the setuid/setgid/sticky bits are masked from the applied mode (the other bits are unaffected); exact rsync semantics are preserved wherever super activities are permitted. Without `-p`, a new file gets `source_mode & ~umask` when metadata is present (else the historical fixed `0644`); new directories without `-p` still use FastSync's `0755` creation default, because directory metadata is only applied when a directory attribute is requested. **Audit-cycle fix:** the daemon no longer forces `umask(0)` (which made implied parent directories world-writable `0777`); it uses the conventional `022`, and `-p`/`-a` still restore the exact source mode via `fchmod`. `-A/--acls` implies `-p`; `--chmod` does **not** imply `-p` (rsync parity) and applies its own unsanitized changes to the new mode. `-X/--xattrs` does not imply `-p`. The SSH port moved to `--ssh-port`. rsync-parity short form | | `-o`, `--owner` | Preserve owner | ✅ Parity | Real per-attribute flag (`preserve_owner`): preserve the source uid, resolved on the receiver by name against its own user database with a raw-numeric fallback (only numeric ids cross the wire). `--usermap`/`--chown=USER` imply it. Application follows the `--super`/`--no-super` policy; a non-opted daemon module applies no ownership (see the Daemon Mode notes) | | `-g`, `--group` | Preserve group | ✅ Parity | Real per-attribute flag (`preserve_group`): preserve the source gid, resolved by name on the receiver with a raw-numeric fallback. `--groupmap`/`--chown=:GROUP` imply it. Same privilege/super-policy gating as `-o` | | `-t`, `--times` | Preserve modification times | ✅ Parity | Real per-attribute flag (`preserve_times`): apply the source mtime independently of the other attributes. `-O/--omit-dir-times` suppresses directories only and `-J/--omit-link-times` suppresses symlinks only; `-U`/`-N` do not imply it. `--preserve`/`-a` imply it, and `--incremental`/`--delta` auto-enable it unless `--no-times`/`--no-preserve` | | `-E`, `--executability` | Preserve executability | ✅ Parity | Preserves executable permission bits (implies metadata preservation) | -| `--chmod=CHMOD` | Affect file permissions | ✅ Parity | Faithful port of rsync 3.4.1's `parse_chmod`/`tweak_mode`: numeric octal and symbolic `ugo`/`rwx` changes, `D`/`F` directory/file selectors, `X` (execute only on directories or already-executable files), `s`/`t` setuid/setgid/sticky, and append semantics — repeated clauses and repeated `--chmod` options accumulate in order (joined with commas). The changes are applied to the new mode **without sanitization** (matching rsync) and `--chmod` does **not** imply `-p` (rsync parity). Applied to files and directories on the receiver | +| `--chmod=CHMOD` | Affect file permissions | ✅ Parity | Faithful port of rsync 3.4.1's `parse_chmod`/`tweak_mode`: numeric octal and symbolic `ugo`/`rwx` changes, `D`/`F` directory/file selectors, `X` (execute only on directories or already-executable files), `s`/`t` setuid/setgid/sticky, and append semantics — repeated clauses and repeated `--chmod` options accumulate in order (joined with commas). The changes are applied to the new mode **without sanitization** (matching rsync), except that setuid/setgid/sticky are masked when the connection forbids super-user activities (audit-cycle fix, see `-p`), and `--chmod` does **not** imply `-p` (rsync parity). Applied to files and directories on the receiver | | `-A`, `--acls` | Preserve ACLs | ✅ Parity | Implemented on Linux via the POSIX-ACL xattr representation: the sender captures the `system.posix_acl_access` / `system.posix_acl_default` xattrs and the receiver re-applies them fd-relative. A differential test with `setfacl` confirms the complete access and default ACL sets (including `mask`) are identical to rsync's on a directory. libacl is not required; a `fsetxattr` an unprivileged receiver may not perform is logged and skipped, never fatal. Only the `system.posix_acl_*` namespaces plus `user.*` are ever applied; privileged namespaces are never applied. Implies metadata transmission | | `-X`, `--xattrs` | Preserve extended attributes | ❌ Divergent | Deliberately restricted to unprivileged `user.*` extended attributes plus the two POSIX ACL xattrs; `security.*` (SELinux, capabilities, ...) and `trusted.*` are **never** captured or applied — a client can never force a privileged attribute onto the destination, and the receiver independently re-validates every incoming name against the whitelist. This is a security-policy divergence from rsync, which can preserve the privileged namespaces with the needed privilege; implementing them would defeat FastSync's privilege-escalation guard. `user.*` capture/apply matches rsync in a differential test. Payloads are bounded on both ends. Incompatible with `-s` | | `-H`, `--hard-links` | Preserve hard links | ✅ Parity | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below | @@ -683,7 +695,7 @@ targets verbatim, matching rsync. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| -| `-z`, `--compress` | Compress file data | ✅ Parity | Streaming compression. **Protocol 2.26.0 implements rsync 3.4.1's codec set** (`zstd` default, `lz4`, `zlib`, `zlibx`, `none`), selectable via `--compress-choice`/`--zc` and negotiated with `auto`. `-z` is the compression short form; `-c` is rsync's `--checksum`. `--skip-compress` applies rsync 3.4.1's default suffix list when no list is given. **Track 3a closes the codec caveats:** `zlibx` is no longer a divergence — FastSync's zlib stream already carries only the delta/token (literal) bytes, which is exactly rsync's zlibx semantics, so `--zc=zlib` and `--zc=zlibx` land the same tree/stdout/exit (differential `test_compress_codec_matches_rsync_bytes`) and the zlib/zlibx aliasing is only an implementation detail. Each codec now uses rsync's own default `--compress-level` (zstd 3, zlib/zlibx 6, lz4 ignored) and `auto` consults `RSYNC_COMPRESS_LIST` before the compiled-in order; the deterministic same-build resolution needs no peer probe | +| `-z`, `--compress` | Compress file data | ✅ Parity | Streaming compression. **Protocol 2.26.0 implements rsync 3.4.1's codec set** (`zstd` default, `lz4`, `zlib`, `zlibx`, `none`), selectable via `--compress-choice`/`--zc` and negotiated with `auto`. `-z` is the compression short form; `-c` is rsync's `--checksum`. `--skip-compress` applies rsync 3.4.1's default suffix list when no list is given. **Track 3a closes the codec caveats:** `zlibx` is no longer a divergence — FastSync's zlib stream already carries only the delta/token (literal) bytes, which is exactly rsync's zlibx semantics, so `--zc=zlib` and `--zc=zlibx` land the same tree/stdout/exit (differential `test_compress_codec_matches_rsync_bytes`) and the zlib/zlibx aliasing is only an implementation detail. Each codec now uses rsync's own default `--compress-level` (zstd 3, zlib/zlibx 6, lz4 ignored) and `auto` consults `RSYNC_COMPRESS_LIST` before the compiled-in order; the deterministic same-build resolution needs no peer probe. **Audit-cycle fix:** the decompressor's internal ceiling is now defined by the protocol whole-file bound (`MAX_RECEIVE_WHOLE_FILE_SIZE`, 256 MiB) instead of a separate 100 MiB constant, so `-z` on a 100–256 MiB regular file no longer fails with `Declared decompressed size exceeds 104857600 bytes` | | `--compress-choice=STR`, `--zc=STR` | Choose compression algorithm | ✅ Parity | Protocol 2.26.0 accepts rsync 3.4.1's compiled-in choices — `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, `auto` — and rejects an unknown name with exit 4 like rsync. The negotiated codec id crosses the wire (`compression_algo`), so the receiver decodes with the sender's codec. `--zc` is the alias. `auto` now resolves through `RSYNC_COMPRESS_LIST` (whitespace-separated; unknown names skipped, first supported wins, all-unknown is exit 4) and then the compiled-in order, and an explicit `--zc` wins; the deterministic same-build resolution needs no peer probe. `zlib`/`zlibx` share FastSync's literal-only zlib path, which is rsync's zlibx behavior and is observably identical for both, so `zlibx` is not a divergence (the aliasing is an implementation detail) | | `--compress-level=NUM`, `--zl=NUM` | Set compression level | ✅ Parity | Accepted range 1-22. When omitted, rsync 3.4.1's **per-codec default** applies: zstd 3 (`ZSTD_CLEVEL_DEFAULT`), zlib/zlibx 6 (`Z_DEFAULT_COMPRESSION` resolved), lz4 ignored (no tunable level; FastSync keeps a positive gate value and `lz4_compress` ignores it, so the bytes match rsync). An explicit level is clamped per codec like rsync's `init_compression_level()`: zstd 1-22, zlib/zlibx 1-9, lz4 ignored. Verified against `rsync --debug=NSTR1`, which reports the same effective level per codec | | `--compress-threads=NUM` | Set compression threads | ✅ Parity | `compression_threads` config field (client-only; does not cross the wire). Sets the number of worker threads used by the zstd compression pool to NUM (1..64; 0/garbage/oversized rejected up front). Accepted in both `--compress-threads=NUM` and two-argument `--compress-threads NUM` forms. Composes with `-z`/compression; under the `-j`/`--threads` multithreaded pipeline it parallelizes compressed chunk encoding. See test_tcp.py `-z --compress-threads=2` and test_client_cli.c | @@ -704,7 +716,7 @@ targets verbatim, matching rsync. | `-4`, `--ipv4` | Prefer IPv4 | ✅ Parity | Forces `AF_INET` in the `getaddrinfo` hints for client destination/source resolution and the server bind (see the Phase 5, Wave B note). Mutually exclusive with `-6` | | `-6`, `--ipv6` | Prefer IPv6 | ✅ Parity | Forces `AF_INET6` in the `getaddrinfo` hints for client destination/source resolution and the server bind. Mutually exclusive with `-4` | | `--remote-option=OPT`, `-M` | Send an option only to the remote side | ❌ Divergent | Each value is appended to the remote server invocation over SSH as an individually single-quote-escaped shell word in `ssh_build_remote_command()`. Values are validated (non-empty, no control characters) and shell metacharacters cannot break out of the quoting (`;`, `&`, `\|`, `, `$`, `(`, `)`, quotes are neutralized), so a value cannot inject an arbitrary remote command and a subsequent `--` on the client line cannot be turned into one. The short `-M` form (`-M OPT`, `-M=OPT`, and rsync-style attached `-MOPT`) is available, matching rsync; metadata mode moved to long-only `--preserve`. **Reclassified because the daemon/TCP case cannot be reproduced:** `-M` is only meaningful for the SSH transport (`user@host:path`); a daemon (`host::module/path`) or local TCP destination **rejects** it, whereas rsync forwards it to its own remote process on every transport. A differential test starts a real rsync daemon and shows `-M--totally-bogus` reaching the remote parser (`unknown option`) while a valid `-M--safe-links` is accepted. FastSync's daemon handshake is a fixed binary config frame with no per-connection argv channel; adding one would let a client set arbitrary server-side options (the same class of divergence as the native daemon config/auth), so the safe subset stays SSH-only | -| `--bwlimit=RATE` | Limit I/O bandwidth | ✅ Parity | A faithful port of rsync 3.4.1's `parse_size_arg(bwlimit_arg, 'K', "bwlimit", 512, -1, True)`: a bare value is KiB/s, `K`/`M`/`G`/`T`/`P` are binary suffixes, `KB`/`MB` are decimal, `KiB`/`MiB` are binary, decimals are accepted and quantized to whole KiB exactly like rsync's `(size + 512) / 1024`, `0` (or an empty value) means "no limit", and any other value below the 512-byte floor is rejected. The token bucket's burst capacity is ~100 ms of bandwidth, matching the point at which rsync's leaky bucket starts sleeping, so a throttled transfer paces like rsync (4 MiB at `--bwlimit=1024`/`2048` matches rsync within ~4%). Differential-tested: the accept/reject matrix and the wall-clock rate both match rsync 3.4.1. The limit is a local I/O concern and is not negotiated on the wire | +| `--bwlimit=RATE` | Limit I/O bandwidth | ✅ Parity | A faithful port of rsync 3.4.1's `parse_size_arg(bwlimit_arg, 'K', "bwlimit", 512, -1, True)`: a bare value is KiB/s, `K`/`M`/`G`/`T`/`P` are binary suffixes, `KB`/`MB` are decimal, `KiB`/`MiB` are binary, decimals are accepted and quantized to whole KiB exactly like rsync's `(size + 512) / 1024`, `0` (or an empty value) means "no limit", and any other value below the 512-byte floor is rejected. The token bucket's burst capacity is ~100 ms of bandwidth, matching the point at which rsync's leaky bucket starts sleeping, so a throttled transfer paces like rsync (4 MiB at `--bwlimit=1024`/`2048` matches rsync within ~4%). Differential-tested: the accept/reject matrix and the wall-clock rate both match rsync 3.4.1. **Audit-cycle fix:** the plaintext-TCP `--sendfile` fast path now passes its writes through the same token bucket, so `--bwlimit` also paces it (previously the `sendfile(2)` path bypassed the limiter entirely); the TLS and plaintext transports therefore share identical throttling. The limit is a local I/O concern and is not negotiated on the wire | ## 14. Daemon Mode @@ -740,7 +752,7 @@ targets verbatim, matching rsync. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| | Path escape detection | Ensure files stay within root | ✅ Parity | `has_path_traversal()` + realpath | -| Symlink-safe delete | Skip symlinks in delete walk | ✅ Parity | `delete_extras_walk()` | +| Symlink-safe delete | Skip symlinks in delete walk | ✅ Parity | `delete_extras_fd()` (`src/shared/utils.c`) and `manifest_delete_extras()` (`src/shared/file_receive.c`) | | Protocol version check | Verify compatible versions | ✅ Parity | `config_receive()` | | Max data/string/chunk sizes | Prevent OOM attacks | ✅ Parity | Per-message limits | | Per-connection memory limit | Cap memory per connection | ✅ Parity | `MAX_CONNECTION_MEMORY` is **256 MiB per connection** (256 * 1024 * 1024 bytes), charged across protocol reservations and decompression/chunk allocations. This is a FastSync-internal bound with no direct rsync analogue | @@ -1153,7 +1165,10 @@ wire protocol three times (full rationale in `src/shared/config.h`): - **Filter grammar:** `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R`, `clear`/`!`, include/exclude and the `:`/`.` modifiers; `-f` is bound to `--filter`; a single `-F` transfers - `.rsync-filter` and `-FF` excludes it. + `.rsync-filter` and `-FF` excludes it. The xattr-name `x` modifier and the + merge-only `e`/`n`/`w` modifiers are **not implemented** and are rejected with + a clear error (audit cycle) instead of being silently ignored or folded into + the pattern. - **Absolute basis directories** are used verbatim (rsync semantics) and **`--link-dest`** relinks an already up-to-date destination. -- 2.54.0 From bc71a3c0a5f73434a6833c9fe201e6a96319088e Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 21:18:26 +0200 Subject: [PATCH 28/68] docs: correct README build deps, delete defaults, scanner, flags; AGENTS deps/CI --- AGENTS.md | 13 ++++---- README.md | 94 +++++++++++++++++++++++++++++++++++-------------------- 2 files changed, 67 insertions(+), 40 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 5ece0ec..82fc4dd 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -4,9 +4,9 @@ FastSync is a high-performance file synchronization system written in C11. It su ## Dependency installation -**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v11`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, Node.js, plus `rsync` 3.4.1 (with zstd/xxhash/lz4), `acl` and `attr` (setfacl/getfacl, setfattr/getfattr) for drop-in parity tests. +**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v11`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, zlib1g-dev, liblz4-dev, libxxhash-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, Node.js, plus `rsync` 3.4.1 (with zstd/xxhash/lz4), `acl` and `attr` (setfacl/getfacl, setfattr/getfattr) for drop-in parity tests. (CMake hard-requires zstd, zlib, and lz4; xxHash is fetched via `FetchContent`.) -**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity. +**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, zlib, lz4, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity. ```bash # Use the prebuilt CI image directly (faster, guaranteed CI parity) @@ -38,11 +38,12 @@ If a dependency is missing from the CI image, add it to the `Dockerfile` (and re When configuring for CI parity, use: ```bash cmake -B build -S . -DSTRICT_WARNINGS=ON # -Wextra -Wpedantic -Werror -cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan) -cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan) +cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan); in the CI matrix +cmake -B build -S . -DSANITIZER=undefined # UndefinedBehaviorSanitizer (UBSan); in the CI matrix +cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan); local-only, NOT in CI ``` -The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, sanitizer, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. The two `setpriv` privilege tests are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions. +The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, the `address`+`undefined` sanitizer matrix, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. TSan is not part of the CI matrix and is a local-only configuration. The four `setpriv` privilege tests are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions. ## Build @@ -105,7 +106,7 @@ Two main branches: `dev` (integration) and `main` (stable releases). ### Rules - **All PRs target `dev`** — never target `main` directly -- **`dev` is the default branch** in Gitea repo settings +- **`dev` is intended to be the default branch** in Gitea repo settings — verify in the repo settings, since this clone's `origin/HEAD` still points at `main` - **`main` is protected** — only merged from `dev` via PR with 2 approvals + full CI pass - **Feature/bug branches** branch from `dev`, PR back to `dev` - **`dev` → `main` merges** happen on-demand or weekly, requiring full CI + review diff --git a/README.md b/README.md index e7ae011..03a7ea2 100644 --- a/README.md +++ b/README.md @@ -75,8 +75,9 @@ matrix is classified as parity, caveat, or divergent in owner, group, devices, and special files — and does not imply compression or multithreading (see [Client](#client)). Ownership application is still privilege-gated: a receiver that cannot `chown` logs a warning and skips it. - Under `-p` the source mode is copied exactly, including setuid/setgid/sticky - and group/other-write bits (strict rsync parity; see + Under `-p` the source mode is copied exactly, including group/other-write + bits; setuid/setgid/sticky bits are copied only when super-user activities are + permitted, and are masked under `SUPER_MODE_OFF`/`--no-super` (see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). - Symlink transfer stores targets **verbatim** (`-l`/`--links`), including absolute and `..`-bearing targets, matching rsync. The receiver does not @@ -172,7 +173,7 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `--preserve` | Preserve mode and mtime (`-p` + `-t`; add `-o`/`-g` for owner/group or `-U`/`--atimes` for atime; `-N`/`--crtimes` captures birth time but cannot apply it) | | `-U, --atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. | | `-N, --crtimes` | Capture birth time; cannot be applied (documented divergence) | -| `-p, --perms` | Preserve permission bits. Strict rsync parity: the source mode is copied exactly, including setuid/setgid/sticky and group/other-write bits | +| `-p, --perms` | Preserve permission bits. The source mode is copied exactly, including group/other-write bits; setuid/setgid/sticky are copied only when super-user activities are permitted (`SUPER_MODE_OFF`/`--no-super` masks them) | | `-t, --times` | Preserve modification times | | `-o, --owner` | Preserve the source owner (privilege-gated; mapped by name on the receiver with a numeric fallback) | | `-g, --group` | Preserve the source group (privilege-gated; mapped by name on the receiver with a numeric fallback) | @@ -204,9 +205,10 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `-u, --update` | Skip files newer than the source on the receiver | | `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. | | `--existing` | Skip files not already present at the destination; update existing files normally. | -| `--compare-dest ` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`) | -| `--copy-dest ` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination | -| `--link-dest ` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win) | +| `--ignore-existing` | Skip files that already exist on the receiver; like rsync it does not apply to directories or symlinks. | +| `--compare-dest ` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`; a basis MISS above the 256 MiB whole-file payload bound is refused — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) | +| `--copy-dest ` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination (same basis-size caveat; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) | +| `--link-dest ` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win; same basis-size caveat; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) | | `--verify-basis` | FastSync-only: require a basis hit (`--compare-dest`/`--copy-dest`/`--link-dest`) to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync) | | `--delete` | Delete files on receiver not present in source (default timing: delete-during, matching rsync, so destination space is freed progressively). Scoped to the synchronized directories, so `--files-from` subsets are safe | | `--delete-before` | Delete extras before the transfer starts (implies `--delete`) | @@ -216,16 +218,16 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `--delete-commit` | FastSync-only: keep the pre-2.28 atomic timing — delete only after the whole transfer succeeded (identical timing to `--delete-after`) | | `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) | | `--max-delete ` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync | -| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication) | +| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication; the fixed `.fastsync-stage` staging name diverges from rsync — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) | | `-T, --temp-dir ` | Scratch directory for temp files before the atomic install; confined to the receive root (relative only), with an `EXDEV` non-atomic copy fallback | | `-n, --dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. | | `-v, --verbose` | Enable debug logging | | `-q, --quiet` | Suppress non-error output | -| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters (FastSync does not print rsync's leading `./` line) | +| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created) | | `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption | -| `--stats` | Print transfer statistics at end (bytes, files, timing), including the receiver-only counters reported over the wire; rsync's per-type `Number of files` breakdown is not reproduced | +| `--stats` | Print transfer statistics at end (bytes, files, timing), including the receiver-only counters reported over the wire; `Number of files` and `Number of created files` carry rsync's per-type breakdown (deleted files are reported as a single total) | | `-i, --itemize-changes` | Print an rsync-style per-file change line | -| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`) | +| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %c %C %i %M %%`) | | `--list-only` | List source files instead of transferring | | `--fsync` | Fsync every written file before publication | | `-h, --human-readable` | Format transfer byte/rate counts with rsync's decimal (base-1000) units | @@ -246,7 +248,7 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `-4, --ipv4` | Force IPv4 for destination resolution | | `-6, --ipv6` | Force IPv6 for destination resolution | | `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect (`TCP_NODELAY`, `SO_KEEPALIVE`, `SO_RCVBUF`, `SO_SNDBUF`, `SO_REUSEADDR`) | -| `--bwlimit ` | Bandwidth limit in kilobytes per second | +| `--bwlimit ` | Bandwidth limit in kilobytes per second; also paces `--sendfile` transfers | | `--chunk-size ` | Chunk size in bytes (default: 10485760) | | `--timeout ` | I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline. Default `0` = disabled (matching rsync); `0` disables it. `--no-timeout` is the negation. The value is not sent on the wire; the server side keeps its own safe floor. | | `--contimeout ` | Connection timeout in seconds (default: 60, matching rsync); `0` disables it (`--no-contimeout` is the negation) | @@ -314,17 +316,22 @@ transfer is never aborted. `timeout`, `contimeout`, `quiet`, `stats`, `max_depth`, and `log_file` are client-only. 5. **Queue** — thread-safe bounded queue with condition variables. -6. **DirectoryScanner** — recursive BFS traversal with exclude and include - pattern support, max-depth enforcement. +6. **DirectoryScanner** — recursive traversal that buffers and sorts each + directory (non-directories ascending, then directories ascending) and walks + depth-first in rsync flist order, with exclude and include pattern support and + max-depth enforcement. ### Key Algorithms -1. **File scanning** — BFS directory traversal; entries matched against exclude - and include patterns, with max-depth enforced. +1. **File scanning** — sorted depth-first traversal in rsync flist order (each + directory's non-directories ascending, then its directories ascending); + entries matched against exclude and include patterns, with max-depth + enforced. The `--threads` parallel scanner remains unordered. 2. **Chunking** — files accumulated until the `chunk_size` threshold (default 10 MiB) is reached, then flushed. 3. **Compression** — streaming zstd via `ZSTD_compressStream2()` / - `ZSTD_decompressStream()`. + `ZSTD_decompressStream()`, with lz4 and zlib/zlibx codecs also supported + (selectable with `--compress-choice`). 4. **Network protocol** — status-code-driven exchange with metadata packing, keep-alive, and abort support. 5. **Incremental check** — the client sends `STATUS_CHECK` + path + size + @@ -379,6 +386,8 @@ Received files are written to a temporary path (suffixed with `.tmp`) and then a - C11 compiler - CMake >= 3.22 - zstd library +- zlib library +- lz4 library - OpenSSL (development headers and libraries) - pthreads - SSH client (for SSH transport mode only) @@ -387,12 +396,12 @@ Received files are written to a temporary path (suffixed with `.tmp`) and then a **Ubuntu/Debian:** ```bash -sudo apt install cmake build-essential libzstd-dev libssl-dev openssh-client +sudo apt install cmake build-essential libzstd-dev zlib1g-dev liblz4-dev libssl-dev openssh-client ``` **Nix:** ```bash -nix-shell # provides zstd, openssl, cmake, gcc +nix-shell # provides zstd, zlib, lz4, openssl, cmake, gcc ``` ## Building @@ -526,9 +535,9 @@ features without changing the meaning of ordinary compatibility options. | `--server-host ` | Select the TCP server host. | | `--server-port ` | Select the TCP server port (`--port ` and `--port=` are rsync-friendly aliases). | | `--tls` | Enable TLS for TCP transport. | -| `--bwlimit ` | Apply token-bucket bandwidth limiting. | -| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters (FastSync omits rsync's leading `./` line). | -| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; rsync's per-type `Number of files` breakdown is not reproduced. | +| `--bwlimit ` | Apply token-bucket bandwidth limiting (also paces `--sendfile` transfers). | +| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created). | +| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; `Number of files`/`Number of created files` carry rsync's per-type breakdown (deleted files are a single total). | | `--timeout ` | Set the socket **and** per-message protocol I/O timeout. Default `0` = disabled (matching rsync); `0` disables it. | | `--contimeout ` | Connection timeout (default 60, matching rsync); `0` disables it. | @@ -562,23 +571,26 @@ remote SSH argv is already built injection-safe. | `--size-only` | Skip incremental files matching in size, ignoring mtime. | | `-I, --ignore-times` | Transfer files even when size and mtime match. | | `-u, --update` | Skip files newer than the source on the receiver. | +| `--ignore-existing` | Skip files that already exist on the receiver; like rsync it does not apply to directories or symlinks. | +| `-@, --modify-window ` | Modification-time tolerance (seconds) for the incremental/basis quick-check; `0` requires an exact mtime match. | | `-W, --whole-file` | Transfer changed files without delta processing (`--no-whole-file` clears it). | | `-B , --block-size ` | Delta block size in bytes (alias `--delta-block`). | | `-d, --dirs` | Transfer the named directory entries without recursing into their contents (aliases `--old-dirs`/`--old-d`). | | `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root. | | `--files-from ` | Read the source file list from FILE (paths relative to the source root). | -| `--delay-updates` | Put updated files into place only at the end of the transfer. | -| `--compare-dest ` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`). | -| `--copy-dest ` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination. | -| `--link-dest ` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win). | +| `-0, --from0` | Treat entries in `--files-from` files as NUL-delimited instead of newline-delimited. | +| `--delay-updates` | Put updated files into place only at the end of the transfer (the fixed `.fastsync-stage` staging name diverges from rsync; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). | +| `--compare-dest ` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`; a basis MISS above the 256 MiB whole-file payload bound is refused — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). | +| `--copy-dest ` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination (same basis-size caveat; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). | +| `--link-dest ` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win; same basis-size caveat; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). | | `--verify-basis` | FastSync-only: require a basis hit to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync). | | `--preallocate` | Allocate destination file space up front (fail-fast on a full disk). | | `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`). | | `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch). | -| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. Scoped to the synchronized directories, so `--files-from` subsets are safe. | +| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-during (matching rsync's `--del`): extras are removed per directory as the transfer proceeds, so destination space is freed progressively. Scoped to the synchronized directories, so `--files-from` subsets are safe. | | `--delete-before` | Delete extras before the transfer starts (implies `--delete`). | -| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). | -| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). | +| `--delete-during`, `--del` | Delete each directory's extras as that directory is processed (implies `--delete`). Since protocol 2.24.0 the sender streams a per-directory `STATUS_DELETE_PLAN` frame as it reaches each source directory; this is also the default timing of a plain `--delete`. | +| `--delete-delay` | Record extras per directory during the scan but remove them only after a successful transfer (implies `--delete`). Uses the same per-directory `STATUS_DELETE_PLAN` frames as `--delete-during`, applied late. | | `--delete-commit` | FastSync-only: atomic delete-after timing (only after the whole transfer succeeded). | | `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). | | `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected). | @@ -598,7 +610,7 @@ remote SSH argv is already built injection-safe. | `--backup-dir ` | Store backups under a separate directory (requires `--backup`). | | `--suffix ` | Set the backup filename suffix (default: `~`). | | `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir `, completed files are written under the partial directory and installed atomically. | -| `--partial-dir ` | Set a relative partial-transfer directory below the server destination root. Use with `--partial`. | +| `--partial-dir ` | Set a relative partial-transfer directory below the server destination root. Implies `--partial` (unless `--inplace`, which bypasses the partial/temp staging). | | `--inplace` | Write directly to the destination instead of using a temporary file. | | `--fsync` | Fsync every written file before publication. | | `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree. | @@ -614,8 +626,11 @@ remote SSH argv is already built injection-safe. | `--preserve` | Preserve mode and mtime (long form only; equivalent to `-p` + `-t`). Add `-o`/`-g` for owner/group, `-U`/`--atimes` for atime, or an identity flag (`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`) for mapped ownership. | | `-U`, `--atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. | | `-N`, `--crtimes` | Capture birth time and transmit it; it cannot be applied because no portable filesystem call can set a birth time (documented divergence). | -| `-p`, `--perms` | Preserve permission bits. One of the four per-attribute preserve flags (with `-t`/`-o`/`-g`); under `-p` the source mode is copied exactly (setuid/setgid/sticky and group/other-write included), matching rsync. | +| `-p`, `--perms` | Preserve permission bits. One of the four per-attribute preserve flags (with `-t`/`-o`/`-g`); under `-p` the source mode is copied exactly (group/other-write included; setuid/setgid/sticky included only when super-user activities are permitted, masked under `SUPER_MODE_OFF`/`--no-super`), matching rsync otherwise. | | `-t`, `--times` | Preserve modification times. Independent of the other attributes; `-O`/`--omit-dir-times` suppresses directories only. | +| `-O`, `--omit-dir-times` | Do not apply modification times to directories. | +| `-J`, `--omit-link-times` | Do not apply times to symlinks. | +| `--open-noatime` | Open source files with `O_NOATIME` so reading for a transfer does not update their access time (client-only). | | `-o`, `--owner` | Preserve the source owner (uid). Mapped by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); application is privilege-gated. | | `-g`, `--group` | Preserve the source group (gid). Same name-mapping/numeric-fallback and privilege gating as `-o`. | | `--no-perms`, `--no-times`, `--no-owner`, `--no-group` | Negate each per-attribute flag (also `--no-p`/`--no-t`/`--no-o`/`--no-g`); `--no-preserve` clears all four. | @@ -642,6 +657,7 @@ remote SSH argv is already built injection-safe. | `-D` | Preserve device and special files (implies `--devices --specials`). | | `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`). | | `--specials` | Recreate special files: FIFOs and unix sockets. | +| `--copy-devices` | Copy a source device's content as an ordinary regular file on the destination (rsync's non-privileged safe mode) instead of recreating the device node. | | `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). | ### Output and logging @@ -650,12 +666,17 @@ remote SSH argv is already built injection-safe. |---|---| | `-v`, `--verbose` | Enable debug logging. | | `-q`, `--quiet` | Suppress non-error output. | -| `--progress` | Show rsync-style per-file progress blocks (not rsync's leading `./` line). | -| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire. | -| `-i`, `--itemize-changes` | Print an rsync-style per-file change line. | -| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`). | +| `--progress` | Show rsync-style per-file progress blocks; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created). | +| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; `Number of files`/`Number of created files` carry rsync's per-type breakdown (deleted files are a single total). | +| `-i, --itemize-changes` | Print an rsync-style per-file change line. | +| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %c %C %i %M %%`). | | `--list-only` | List source files instead of transferring. | +| `--outbuf=MODE` | stdout/stderr buffering: `N` (none/unbuffered), `L` (line-buffered), or `B` (block-buffered, default). | | `--log-file ` | Write log output to a file. | +| `--log-file-format=FORMAT` | Per-file log-line format (requires `--log-file`). | +| `--stderr=MODE` | Route logging to stderr: `errors` or `all`. | +| `--msgs2stderr` | Route all messages to stderr (deprecated spelling of `--stderr=all`). | +| `--no-msgs2stderr` | Select errors-only stderr (deprecated spelling; the default). | | `-V`, `--version` | Print the FastSync protocol version. | | `--help` | Print command usage. | @@ -680,6 +701,11 @@ remote SSH argv is already built injection-safe. | `-4`, `--ipv4` | Force IPv4 for destination resolution. | | `-6`, `--ipv6` | Force IPv6 for destination resolution. | | `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect. | +| `--blocking-io` | SSH transport only: leave the socket without read/write timeouts so it blocks naturally (no effect on TCP). | +| `--protocol=NUM` | Force the wire protocol version; must equal the current `PROTOCOL_VERSION` (FastSync cannot speak older/virtual wire formats). | +| `--old-args` | Accepted for rsync CLI compatibility; no effect (the remote server path is always safely quoted). | +| `--iconv=LOCAL[,REMOTE]` | Convert file-name charsets at the wire boundary (`LOCAL` is our names' charset, `REMOTE` the peer's, defaulting to `LOCAL`). | +| `--no-iconv` | Disable `--iconv` charset conversion (same as `--iconv=-`). | | `--tls` | Enable TLS. Requires `--cert`, `--key`, and `--ca`. | | `--cert ` | TLS certificate file. | | `--key ` | TLS private key file. | -- 2.54.0 From 338c27db738e1694fd1e725aa525e70374ba71a3 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 21:28:36 +0200 Subject: [PATCH 29/68] fix: resolve cppcheck shadow/always-true findings --- src/shared/credentials.c | 6 +++--- tests/test_file.c | 6 ++---- 2 files changed, 5 insertions(+), 7 deletions(-) diff --git a/src/shared/credentials.c b/src/shared/credentials.c index e241271..45be2bb 100644 --- a/src/shared/credentials.c +++ b/src/shared/credentials.c @@ -141,9 +141,9 @@ static FILE* secret_file_open(const char* path, char* err, size_t err_size) { * most systems, but explicit. Failures here are ignored: O_NONBLOCK on a * regular file does not affect reads either way. */ if (S_ISREG(st.st_mode)) { - int flags = fcntl(fd, F_GETFL); - if (flags >= 0) - (void)fcntl(fd, F_SETFL, flags & ~O_NONBLOCK); + int status_flags = fcntl(fd, F_GETFL); + if (status_flags >= 0) + (void)fcntl(fd, F_SETFL, status_flags & ~O_NONBLOCK); } FILE* fp = fdopen(fd, "r"); if (!fp) { diff --git a/tests/test_file.c b/tests/test_file.c index 0546365..d0124de 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -411,8 +411,7 @@ static void test_file_open_temp_dir_symlink_confinement() { EXPECT_EQ_INT(mkdir(scratch, 0755), 0); int scratch_fd = file_open_temp_dir(scratch); EXPECT_TRUE(scratch_fd >= 0); - if (scratch_fd >= 0) - close(scratch_fd); + close(scratch_fd); /* A symlink whose target is outside the root is refused. */ char* escape = path_cat(root_abs, "escape"); @@ -426,8 +425,7 @@ static void test_file_open_temp_dir_symlink_confinement() { EXPECT_EQ_INT(symlink(scratch, inside_link), 0); int link_fd = file_open_temp_dir(inside_link); EXPECT_TRUE(link_fd >= 0); - if (link_fd >= 0) - close(link_fd); + close(link_fd); free(inside_link); free(escape); -- 2.54.0 From b1eddf0133cde5716d24d0c810bbf0bc4c37384b Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 21:59:58 +0200 Subject: [PATCH 30/68] fix: config leak, compression log, inplace+partial-dir rejection, umask/root test fixes --- src/client/client_validation.c | 7 +++ src/shared/compression.c | 3 +- src/shared/config.c | 2 +- tests/integration/test_daemon.py | 47 +++++++++++--- tests/test_client_cli.c | 7 ++- tests/test_file.c | 102 ++++++++++++++++++++----------- tests/test_protocol.c | 2 +- 7 files changed, 119 insertions(+), 51 deletions(-) diff --git a/src/client/client_validation.c b/src/client/client_validation.c index 2ff7868..eb71c3c 100644 --- a/src/client/client_validation.c +++ b/src/client/client_validation.c @@ -75,6 +75,13 @@ bool validate_config(const Config* config) { log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive"); return false; } + /* rsync 3.4.1 rejects --inplace together with --partial-dir (exit 1): the + inplace write path bypasses partial staging, so a partial-dir name would be + silently ignored. Match rsync's message and refuse before any I/O. */ + if (config->inplace && config->partial_dir) { + log_message(LOG_LEVEL_ERROR, "--inplace cannot be used with --partial-dir"); + return false; + } if (config->log_file_format && !config->log_file) { log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file"); return false; diff --git a/src/shared/compression.c b/src/shared/compression.c index c7bba22..a3d7d66 100644 --- a/src/shared/compression.c +++ b/src/shared/compression.c @@ -644,8 +644,7 @@ static Data* zstd_decompress(Data* compressed_data, size_t maximum_size) { } if (ret > 0 && output.pos == output.size) { if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) { - log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes", - (unsigned long long)MAX_DECOMPRESSED_SIZE); + log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes", hard_limit); data_destroy(uncompressed_data); uncompressed_data = NULL; goto cleanup; diff --git a/src/shared/config.c b/src/shared/config.c index ba3f34d..f4b255d 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -234,7 +234,7 @@ Config* config_create(void) { * server_host NULL and would crash later consumers, so fail the whole create * (every caller already handles a NULL return). */ if (!config->server_host) { - free(config); + config_delete(config); return NULL; } return config; diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 0212616..ae1d765 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -63,6 +63,10 @@ DETACH_MODULE = os.path.join(MODULE_ROOT, "detach") DETACH_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_detach.conf") DETACH_PORT = None +# A dedicated config for the umask test: the daemon must be launched in the real +# (double-fork) detach path, whose daemonize() applies umask(022). +UMASK_CONF = os.path.join(TEST_DATA_DIR, "fastsyncd_umask.conf") + # Passwords are never sent as plaintext and never logged; these literals are # only hashed into the server credential file / client password file. ALICE_PASS = "alice-s3cret" @@ -332,20 +336,43 @@ class TestDaemonModuleSelection: assert not missing, f"missing: {missing[:5]}" assert not mismatches, f"mismatch: {mismatches[:5]}" - def test_daemon_new_dirs_not_world_writable(self, daemon): + def test_daemon_new_dirs_not_world_writable(self): """The daemon must not force umask 0: implied parent directories created - without -p are the source default (0755 under a 022 umask), never - world-writable 0777.""" + without -p are the source default (0755 under the daemon's 022 umask), + never world-writable 0777. + + This drives the real double-fork detach path, where the umask(022) fix + lives (daemonize()); the --no-detach path never calls it. The launcher + is run with umask 0, so without the fix the daemon would inherit 0 and + create a 0777 directory; with the fix the assertion below fails only if + the fix regresses.""" + port = _find_free_port() + with open(UMASK_CONF, "w") as f: + f.write("port = %d\n\n[files]\npath = %s\n" % (port, FILES_MODULE)) sub = os.path.join(FILES_MODULE, "umask_check") shutil.rmtree(sub, ignore_errors=True) os.makedirs(sub, exist_ok=True) - result = _push("127.0.0.1::files/umask_check", daemon.port) - assert result.returncode == 0, result.stderr or result.stdout - received = get_dest_received_dir(sub, SOURCE_DIR) - nested = os.path.join(received, "nested") - assert os.path.isdir(nested), f"nested dir missing under {received}" - mode = stat.S_IMODE(os.stat(nested).st_mode) - assert (mode & 0o022) == 0, f"implied directory is group/other writable: {oct(mode)}" + log_path = os.path.join(TEST_DATA_DIR, "fastsyncd_umask.log") + log = open(log_path, "w") + cmd = SERVER_CMD + ["--daemon", "--config", UMASK_CONF, "--allow-unauthenticated"] + proc = subprocess.Popen(cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, + preexec_fn=lambda: os.umask(0)) + try: + _wait_for_port(port, timeout=15) + result = _push("127.0.0.1::files/umask_check", port) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(sub, SOURCE_DIR) + nested = os.path.join(received, "nested") + assert os.path.isdir(nested), f"nested dir missing under {received}" + mode = stat.S_IMODE(os.stat(nested).st_mode) + assert (mode & 0o022) == 0, f"implied directory is group/other writable: {oct(mode)}" + finally: + _kill_by_cmdline_marker(UMASK_CONF) + log.close() + try: + proc.wait(timeout=5) + except subprocess.TimeoutExpired: + proc.kill() class TestDaemonRejection: diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index b8b2ff9..d86f8f6 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -578,13 +578,18 @@ static void test_parse_args_partial_dir_implies_partial() { config_delete(cfg); } { - /* --inplace bypasses partial staging: the implication must not fire. */ + /* --inplace bypasses partial staging, so parse_args must not set the + implied --partial; the combination itself is invalid (rsync parity: + "--inplace cannot be used with --partial-dir"), so validation rejects. */ Config* cfg = config_create(); char* argv[] = {"fastsync", "--inplace", "--partial-dir=.partial", "/src", "/dst"}; int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); EXPECT_FALSE(cfg->partial); + cfg->send_directory = str_dup("/src"); + cfg->receive_root_directory = str_dup("/dst"); + EXPECT_FALSE(validate_config(cfg)); config_delete(cfg); } { diff --git a/tests/test_file.c b/tests/test_file.c index d0124de..69912f0 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -391,52 +391,82 @@ static void test_file_open_temp_dir_symlink_confinement() { rmdir("test_tempdir_link_root/scratch"); rmdir(root); rmdir(outside); - EXPECT_EQ_INT(mkdir(root, 0755), 0); - EXPECT_EQ_INT(mkdir(outside, 0755), 0); - EXPECT_NOT_NULL(realpath(root, root_abs)); - EXPECT_NOT_NULL(realpath(outside, outside_abs)); - int root_fd = open(root_abs, O_RDONLY | O_DIRECTORY | O_CLOEXEC); - EXPECT_TRUE(root_fd >= 0); - // cppcheck-suppress knownConditionTrueFalse - if (root_fd < 0) { - rmdir(root); - rmdir(outside); - return; + + int mkdir_root_ret = mkdir(root, 0755); + int mkdir_outside_ret = mkdir(outside, 0755); + bool root_resolved = realpath(root, root_abs) != NULL; + bool outside_resolved = realpath(outside, outside_abs) != NULL; + int root_fd = root_resolved ? open(root_abs, O_RDONLY | O_DIRECTORY | O_CLOEXEC) : -1; + + bool root_set = false; + bool scratch_ok = false; + int scratch_fd = -1; + bool escape_staged = false; + int escape_fd = 0; + bool inside_staged = false; + int inside_fd = -1; + char* scratch = NULL; + char* escape = NULL; + char* inside_link = NULL; + + /* Only touch the global authorized root and the scratch fixtures once the + setup succeeded; the teardown below always runs regardless. */ + if (root_fd >= 0 && outside_resolved) { + root_set = utils_set_authorized_root(root_fd, root_abs); + + /* An existing in-root scratch dir opens normally. */ + scratch = path_cat(root_abs, "scratch"); + if (scratch && mkdir(scratch, 0755) == 0) { + scratch_ok = true; + scratch_fd = file_open_temp_dir(scratch); + if (scratch_fd >= 0) + close(scratch_fd); + } + + /* A symlink whose target is outside the root is refused. */ + escape = path_cat(root_abs, "escape"); + if (escape && symlink(outside_abs, escape) == 0) { + escape_staged = true; + escape_fd = file_open_temp_dir(escape); + } + + /* A symlink that stays inside the root is accepted (EXDEV fallback). */ + inside_link = path_cat(root_abs, "inside_link"); + if (inside_link && scratch && symlink(scratch, inside_link) == 0) { + inside_staged = true; + inside_fd = file_open_temp_dir(inside_link); + if (inside_fd >= 0) + close(inside_fd); + } } - EXPECT_TRUE(utils_set_authorized_root(root_fd, root_abs)); - - /* An existing in-root scratch dir opens normally. */ - char* scratch = path_cat(root_abs, "scratch"); - EXPECT_NOT_NULL(scratch); - EXPECT_EQ_INT(mkdir(scratch, 0755), 0); - int scratch_fd = file_open_temp_dir(scratch); - EXPECT_TRUE(scratch_fd >= 0); - close(scratch_fd); - - /* A symlink whose target is outside the root is refused. */ - char* escape = path_cat(root_abs, "escape"); - EXPECT_NOT_NULL(escape); - EXPECT_EQ_INT(symlink(outside_abs, escape), 0); - EXPECT_EQ_INT(file_open_temp_dir(escape), -1); - - /* A symlink that stays inside the root is accepted (EXDEV fallback path). */ - char* inside_link = path_cat(root_abs, "inside_link"); - EXPECT_NOT_NULL(inside_link); - EXPECT_EQ_INT(symlink(scratch, inside_link), 0); - int link_fd = file_open_temp_dir(inside_link); - EXPECT_TRUE(link_fd >= 0); - close(link_fd); + /* Release the global authorized root and all fixtures BEFORE asserting: + EXPECT_* returns early on failure, so a failed assertion must not be able + to leave the process state poisoned or leak root_fd. */ + utils_set_authorized_root(-1, NULL); + if (root_fd >= 0) + close(root_fd); free(inside_link); free(escape); free(scratch); - utils_set_authorized_root(-1, NULL); - close(root_fd); unlink("test_tempdir_link_root/escape"); unlink("test_tempdir_link_root/inside_link"); rmdir("test_tempdir_link_root/scratch"); rmdir(root); rmdir(outside); + + EXPECT_EQ_INT(mkdir_root_ret, 0); + EXPECT_EQ_INT(mkdir_outside_ret, 0); + EXPECT_TRUE(root_resolved); + EXPECT_TRUE(outside_resolved); + EXPECT_TRUE(root_fd >= 0); + EXPECT_TRUE(root_set); + EXPECT_TRUE(scratch_ok); + EXPECT_TRUE(scratch_fd >= 0); + EXPECT_TRUE(escape_staged); + EXPECT_EQ_INT(escape_fd, -1); + EXPECT_TRUE(inside_staged); + EXPECT_TRUE(inside_fd >= 0); } /* Issue #251: file_save_to_disk_full must distinguish receiver-side skips diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 2665096..483ee82 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -741,7 +741,7 @@ static void test_protocol_throttle_bytes_unlimited() { clock_gettime(CLOCK_MONOTONIC, &now); long long elapsed_ms = (now.tv_sec - start.tv_sec) * 1000LL + (now.tv_nsec - start.tv_nsec) / 1000000LL; - EXPECT_TRUE(elapsed_ms < 50); + EXPECT_TRUE(elapsed_ms < 2000); protocol_session_unbind(); } -- 2.54.0 From 9f47b13712ecf7583c457686f472e8d140dabaa1 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 22:01:18 +0200 Subject: [PATCH 31/68] fix(credentials): bound-wait on FIFO reads so slow process substitution works secret_file_open() opened secret files with O_NONBLOCK and only cleared it for S_ISREG, so on a FIFO/process-substitution source (--password-file <(...), --early-input <(...)) fgets() failed immediately with EAGAIN when the writer had not yet produced data, breaking slow producers. Keep O_NONBLOCK at open() (a writer-less FIFO must not block the open) and route all three readers through a new secret_read_line() helper. It accumulates a line across reads and, on EAGAIN/EWOULDBLOCK (or a partial line) with no newline and no EOF, clearerr()s and polls for readability against one overall CLOCK_MONOTONIC deadline of CREDENTIAL_FIFO_READ_TIMEOUT_MS (3000 ms); on timeout or a real read error it fails with a clear message. EOF finishes normally. Regular files are left blocking and read exactly as before. Handles a line split across several write()s and keeps the owner/mode fstat gate, O_NOFOLLOW and the /dev/fd/N exception unchanged. --- src/shared/credentials.c | 168 ++++++++++++++++++++++++++++++++++++--- tests/test_credentials.c | 158 ++++++++++++++++++++++++++++++++++++ 2 files changed, 315 insertions(+), 11 deletions(-) diff --git a/src/shared/credentials.c b/src/shared/credentials.c index 45be2bb..d17b99f 100644 --- a/src/shared/credentials.c +++ b/src/shared/credentials.c @@ -8,11 +8,13 @@ #include #include #include +#include #include #include #include #include #include +#include #include /* One store entry: a username and its salted PBKDF2 verifier. The plaintext @@ -105,11 +107,16 @@ static bool is_fd_backed_path(const char* path) { * O_NOFOLLOW guards against, and requiring O_NOFOLLOW would break the * documented process-substitution/FIFO usage. For them only, O_NOFOLLOW is * omitted; the same fstat owner/mode gate still applies to the resolved inode. - * O_NONBLOCK keeps a FIFO - * from blocking the open/read forever: an empty or writer-less FIFO yields - * EOF/EAGAIN rather than hanging in fgets. Only regular files and FIFOs pass - * the ownership/mode checks; O_NONBLOCK is cleared for regular files, where it - * is a no-op anyway, so their stdio read path is byte-for-byte unchanged. + * O_NONBLOCK keeps the OPEN itself from + * blocking forever on a writer-less FIFO (a blocking O_RDONLY open would wait + * for a writer). The fd is left nonblocking for FIFOs so a read never blocks + * either; the read loop (secret_read_line) absorbs the resulting EAGAIN by + * waiting, under a bounded deadline, for the writer -- this is what makes a + * slow process substitution (`--password-file <(sleep 1; ...)`) work while a + * writer-less FIFO still fails after the deadline instead of hanging. Only + * regular files and FIFOs pass the ownership/mode checks; O_NONBLOCK is + * cleared for regular files, where it is a no-op anyway and no EAGAIN can + * occur, so their stdio read path is byte-for-byte unchanged. * * Returns a FILE* the caller must fclose, or NULL with `err` filled. */ static FILE* secret_file_open(const char* path, char* err, size_t err_size) { @@ -154,6 +161,123 @@ static FILE* secret_file_open(const char* path, char* err, size_t err_size) { return fp; } +/* Overall bound on how long the reader waits for a process-substitution/FIFO + * writer to produce data before giving up. It must comfortably exceed a + * producer's startup delay (e.g. `--password-file <(sleep 1; ...)`) while still + * bounding a writer-less FIFO, so a stray or hostile FIFO cannot stall the + * daemon or client indefinitely. */ +#define CREDENTIAL_FIFO_READ_TIMEOUT_MS 3000 + +/* Monotonic milliseconds, used only for the read deadline (wall-clock changes + * must not extend or shorten the wait). */ +static int64_t credential_monotonic_ms(void) { + struct timespec ts; + if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0) + return 0; + return (int64_t)ts.tv_sec * 1000 + (int64_t)(ts.tv_nsec / 1000000); +} + +/* Wait until `fd` is readable or the deadline passes. Returns true when it is + * readable, false on timeout or a poll error (err filled). EINTR is retried + * against the same deadline, so signals cannot extend the wait. */ +static bool credential_wait_readable(int fd, int64_t deadline, const char* label, const char* path, + char* err, size_t err_size) { + for (;;) { + int64_t remaining = deadline - credential_monotonic_ms(); + if (remaining <= 0) + break; + if (remaining > INT_MAX) + remaining = INT_MAX; + struct pollfd pfd = {.fd = fd, .events = POLLIN, .revents = 0}; + int rc = poll(&pfd, 1, (int)remaining); + if (rc > 0) + return true; + if (rc == 0) + break; + if (errno != EINTR) { + set_error(err, err_size, "error waiting for %s '%s': %s", label, path, strerror(errno)); + return false; + } + } + set_error(err, err_size, "timed out after %d ms waiting for %s '%s'", + CREDENTIAL_FIFO_READ_TIMEOUT_MS, label, path); + return false; +} + +typedef enum { + SECRET_READ_LINE, + SECRET_READ_EOF, + SECRET_READ_ERROR, +} SecretReadResult; + +/* Read one complete line from `fp` into `line` (capacity `cap`), including the + * trailing newline when present and always NUL-terminating. `*out_len` + * receives strlen(line). + * + * A regular file is read exactly as before: secret_file_open leaves it + * blocking, so fgets never sees EAGAIN. A FIFO stays nonblocking, so fgets + * returns NULL (or a partial line) with EAGAIN while the writer is still + * starting up; instead of treating that as a fatal error the loop clearerr()s + * and polls for readability against one overall deadline. The `used` + * accumulator reassembles a line that arrived in several write()s into a single + * line, so a split write is not misparsed as two entries. + * + * Returns SECRET_READ_LINE, SECRET_READ_EOF, or SECRET_READ_ERROR (err filled) + * on timeout or a genuine read error. */ +static SecretReadResult secret_read_line(char* line, size_t cap, FILE* fp, const char* label, + const char* path, size_t* out_len, char* err, + size_t err_size) { + int fd = fileno(fp); + int64_t deadline = credential_monotonic_ms() + CREDENTIAL_FIFO_READ_TIMEOUT_MS; + size_t used = 0; + line[0] = '\0'; + for (;;) { + errno = 0; + if (fgets(line + used, (int)(cap - used), fp)) { + used += strlen(line + used); + if (used > 0 && line[used - 1] == '\n') { + *out_len = used; + return SECRET_READ_LINE; + } + if (feof(fp)) { + *out_len = used; /* final unterminated line */ + return SECRET_READ_LINE; + } + /* No newline and not EOF. A full buffer is the caller's over-long-line + * case; otherwise the line is only partially available (a nonblocking + * FIFO under a slow writer), so any genuine read error fails and anything + * else waits for the rest. */ + if (used >= cap - 1) { + *out_len = used; + return SECRET_READ_LINE; + } + int e = ferror(fp) ? errno : 0; + if (e != 0 && e != EAGAIN && e != EWOULDBLOCK) { + set_error(err, err_size, "error reading %s '%s': %s", label, path, strerror(e)); + return SECRET_READ_ERROR; + } + clearerr(fp); + if (!credential_wait_readable(fd, deadline, label, path, err, err_size)) + return SECRET_READ_ERROR; + continue; + } + /* fgets returned NULL: EOF, a not-yet-readable FIFO, or a real error. */ + if (feof(fp)) { + *out_len = used; + return used > 0 ? SECRET_READ_LINE : SECRET_READ_EOF; + } + if (errno == EAGAIN || errno == EWOULDBLOCK) { + clearerr(fp); + if (!credential_wait_readable(fd, deadline, label, path, err, err_size)) + return SECRET_READ_ERROR; + continue; + } + set_error(err, err_size, "error reading %s '%s': %s", label, path, + errno != 0 ? strerror(errno) : "read failed"); + return SECRET_READ_ERROR; + } +} + /* Trim leading/trailing ASCII space and tab in place; returns the new start. */ static char* trim_space(char* s) { while (*s == ' ' || *s == '\t') @@ -549,9 +673,17 @@ static CredentialStore* load_store_file(const char* path, char* err, size_t err_ char line[CREDENTIAL_MAX_LINE + 2]; bool ok = true; - while (fgets(line, sizeof(line), fp)) { + for (;;) { + size_t len = 0; + SecretReadResult rr = + secret_read_line(line, sizeof(line), fp, "credential file", path, &len, err, err_size); + if (rr == SECRET_READ_EOF) + break; + if (rr == SECRET_READ_ERROR) { + ok = false; + break; + } line_no++; - size_t len = strlen(line); if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) { set_error(err, err_size, "credential file '%s' line %d exceeds the %d-byte limit", path, line_no, CREDENTIAL_MAX_LINE); @@ -1188,9 +1320,17 @@ int credentials_hash_file(const char* path, uint32_t iters, FILE* out, char* err int line_no = 0; int result = 0; char line[CREDENTIAL_MAX_LINE + 2]; - while (fgets(line, sizeof(line), fp)) { + for (;;) { + size_t len = 0; + SecretReadResult rr = + secret_read_line(line, sizeof(line), fp, "plaintext file", path, &len, err, err_size); + if (rr == SECRET_READ_EOF) + break; + if (rr == SECRET_READ_ERROR) { + result = -1; + break; + } line_no++; - size_t len = strlen(line); if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) { set_error(err, err_size, "plaintext file '%s' line %d exceeds the %d-byte limit", path, line_no, CREDENTIAL_MAX_LINE); @@ -1268,9 +1408,15 @@ int credentials_read_secret_file(const char* path, char** user_out, char** passw char line[CREDENTIAL_MAX_LINE + 2]; int result = -1; - while (fgets(line, sizeof(line), fp)) { + for (;;) { + size_t len = 0; + SecretReadResult rr = + secret_read_line(line, sizeof(line), fp, "password file", path, &len, err, err_size); + if (rr == SECRET_READ_EOF) + break; + if (rr == SECRET_READ_ERROR) + goto done; line_no++; - size_t len = strlen(line); if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) { set_error(err, err_size, "password file '%s' line %d exceeds the %d-byte limit", path, line_no, CREDENTIAL_MAX_LINE); diff --git a/tests/test_credentials.c b/tests/test_credentials.c index 7128b7f..f868802 100644 --- a/tests/test_credentials.c +++ b/tests/test_credentials.c @@ -10,6 +10,7 @@ #include #include #include +#include #include /* Known-answer vector, independently recomputed with Python @@ -764,6 +765,160 @@ static void test_credentials_read_secret_file_fifo_no_hang() { unlink(fifo); } +/* Write `s` fully to `fd`, retrying EINTR. */ +static void write_all_fd(int fd, const char* s) { + size_t total = strlen(s); + size_t off = 0; + while (off < total) { + ssize_t w = write(fd, s + off, total - off); + if (w < 0) { + if (errno == EINTR) + continue; + return; + } + off += (size_t)w; + } +} + +/* Deterministically model a slow process substitution (`--password-file + * <(sleep N; ...)`): attach a writer to the FIFO (so the reader sees EAGAIN -- + * the empty/no-writer FIFO instead yields an immediate EOF), have it sleep + * `delay_ms`, then write `first` and, after another `delay_ms`, `second` (NULL + * for a single write). Splitting across the delay exercises reassembly of a + * line delivered by several write()s. + * + * The parent keeps a spare read end open for the lifetime of the test so the + * writer always has a reader; the caller must close(*hold_out), waitpid() the + * returned pid and unlink the FIFO. Returns the child pid, or -1 on setup + * failure. */ +static pid_t fifo_writer_sleep_then_write(const char* fifo, const char* first, unsigned delay_ms, + const char* second, int* hold_out) { + int sync[2]; + if (pipe(sync) != 0) + return -1; + pid_t pid = fork(); + if (pid < 0) { + close(sync[0]); + close(sync[1]); + return -1; + } + if (pid == 0) { + close(sync[0]); + int wfd = open(fifo, O_WRONLY | O_CLOEXEC); + char ready = wfd >= 0 ? 1 : 0; + if (write(sync[1], &ready, 1) != 1) + _exit(1); + close(sync[1]); + if (wfd >= 0) { + usleep(delay_ms * 1000); + write_all_fd(wfd, first); + if (second) { + usleep(delay_ms * 1000); + write_all_fd(wfd, second); + } + close(wfd); + } + _exit(0); + } + close(sync[1]); + int hold = open(fifo, O_RDONLY | O_NONBLOCK | O_CLOEXEC); + char ready = 0; + ssize_t got = read(sync[0], &ready, 1); + close(sync[0]); + if (got != 1 || ready != 1) { + if (hold >= 0) + close(hold); + return -1; + } + *hold_out = hold; + return pid; +} + +/* A FIFO writer that produces its data after a short delay must be read + * successfully (the regression: O_NONBLOCK made fgets fail with EAGAIN before + * the writer ran). */ +static void test_credentials_read_secret_file_fifo_delayed_writer() { + char err[512]; + char fifo[256]; + snprintf(fifo, sizeof(fifo), "/tmp/fs_cred_pwfifo_slow_%d_%d", (int)getpid(), g_file_counter++); + unlink(fifo); + EXPECT_EQ_INT(mkfifo(fifo, 0600), 0); + + int hold = -1; + pid_t writer = + fifo_writer_sleep_then_write(fifo, "alice:correct horse battery staple\n", 250, NULL, &hold); + EXPECT_TRUE(writer > 0); + + char* user = NULL; + char* password = NULL; + EXPECT_EQ_INT(credentials_read_secret_file(fifo, &user, &password, err, sizeof(err)), 0); + EXPECT_EQ_STR(user, "alice"); + EXPECT_EQ_STR(password, "correct horse battery staple"); + free(user); + free(password); + + int status = 0; + waitpid(writer, &status, 0); + if (hold >= 0) + close(hold); + unlink(fifo); +} + +/* The same, but the line is written in two chunks separated by the delay: the + * reader must reassemble one line rather than parse the first chunk as an + * empty-password entry. */ +static void test_credentials_read_secret_file_fifo_split_write() { + char err[512]; + char fifo[256]; + snprintf(fifo, sizeof(fifo), "/tmp/fs_cred_pwfifo_split_%d_%d", (int)getpid(), g_file_counter++); + unlink(fifo); + EXPECT_EQ_INT(mkfifo(fifo, 0600), 0); + + int hold = -1; + pid_t writer = + fifo_writer_sleep_then_write(fifo, "alice:correct horse", 200, " battery staple\n", &hold); + EXPECT_TRUE(writer > 0); + + char* user = NULL; + char* password = NULL; + EXPECT_EQ_INT(credentials_read_secret_file(fifo, &user, &password, err, sizeof(err)), 0); + EXPECT_EQ_STR(user, "alice"); + EXPECT_EQ_STR(password, "correct horse battery staple"); + free(user); + free(password); + + int status = 0; + waitpid(writer, &status, 0); + if (hold >= 0) + close(hold); + unlink(fifo); +} + +/* The server-side store loader (--password-file / --early-input) must also + * accept a FIFO whose writer appears after a delay. */ +static void test_credentials_store_fifo_delayed_writer() { + char err[512]; + char fifo[256]; + snprintf(fifo, sizeof(fifo), "/tmp/fs_cred_storefifo_%d_%d", (int)getpid(), g_file_counter++); + unlink(fifo); + EXPECT_EQ_INT(mkfifo(fifo, 0600), 0); + + int hold = -1; + pid_t writer = fifo_writer_sleep_then_write(fifo, KAT_STORE_LINE "\n", 250, NULL, &hold); + EXPECT_TRUE(writer > 0); + + CredentialStore* store = credentials_load(fifo, NULL, err, sizeof(err)); + EXPECT_NOT_NULL(store); + EXPECT_EQ_INT(credentials_store_size(store), 1); + credentials_free(store); + + int status = 0; + waitpid(writer, &status, 0); + if (hold >= 0) + close(hold); + rm_temp(fifo); +} + /* fd-backed store paths (bash process substitution `<(...)`, i.e. /dev/fd/N and * /proc/self/fd/N) are symlinks, so the ordinary O_NOFOLLOW rule would reject * them with ELOOP. They name the calling process's own descriptors, so they @@ -1184,6 +1339,9 @@ void test_credentials(void) { test_credentials_read_secret_file_bad(); test_credentials_read_secret_file_symlink_rejected(); test_credentials_read_secret_file_fifo_no_hang(); + test_credentials_read_secret_file_fifo_delayed_writer(); + test_credentials_read_secret_file_fifo_split_write(); + test_credentials_store_fifo_delayed_writer(); test_credentials_read_secret_file_fd_backed_accepted(); test_credentials_hash_file(); test_credentials_rejects_group_or_other_accessible(); -- 2.54.0 From 06c4026b74992e42c482460e66b22ca150171c00 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 22:01:44 +0200 Subject: [PATCH 32/68] fix(filter): accept e/n/w/- merge modifiers on merge/dir-merge rules The earlier modifier-rejection change rejected e/n/w on all rules, but rsync 3.4.1 accepts them (plus the '-' merge-only modifier) on merge and dir-merge rules. Restrict the rejection to non-merge rules and consume the merge-file modifiers (e/n/w/-) so they no longer leak into the merge filename. - is_merge_rule()/is_merge_modifier_char() gate the merge-only modifiers. - scan vs consume sets: e/n/w still count as modifier-run chars on every rule (pure tokens like -new/-press stay rejected), but are only consumed on merge rules, preserving mixed-token parsing such as H,!secret -> ecret. - '-' is accepted/consumed only on merge/dir-merge (e.g. dir-merge,- .rules). - x remains rejected everywhere with its dedicated message. - e/n/w/- semantics remain unimplemented and are documented as accepted-but- ignored in filter.h. Tests: split the merge forms out of the rejection test into a new acceptance test asserting the merge file is read and dir_merge_names keeps the modifier- free basename; non-merge pure-modifier forms still rejected. --- src/shared/filter.c | 54 ++++++++++++++++++++++++++----------- src/shared/filter.h | 8 ++++-- tests/test_filter.c | 66 ++++++++++++++++++++++++++++++++++++++++++++- 3 files changed, 110 insertions(+), 18 deletions(-) diff --git a/src/shared/filter.c b/src/shared/filter.c index 4bfd462..56cc6ed 100644 --- a/src/shared/filter.c +++ b/src/shared/filter.c @@ -162,33 +162,57 @@ static bool is_modifier_char(char c) { return c == 's' || c == 'r' || c == 'p' || c == 'x' || c == '/' || c == '!' || c == 'C'; } -/* Modifiers rsync defines but FastSync does not implement. They must still be - * consumed as part of the modifier run so they are rejected explicitly instead - * of leaking into the pattern (which produced misleading failures such as - * "could not read merge file 'n file'"). */ +/* merge/dir-merge rules are the only rules rsync accepts the merge-file + * modifiers on. */ +static bool is_merge_rule(RuleKind kind) { + return kind == RULE_KIND_MERGE || kind == RULE_KIND_DIR_MERGE; +} + +/* Merge-file modifiers rsync defines but FastSync does not implement: + * 'e' exclude the merge file itself, 'n' do not inherit the merge file, 'w' + * word-split the merge file. They are recognized as part of a modifier run on + * every rule (so a pure e/n/w token is rejected rather than folded into the + * pattern), but are accepted (and ignored) only on merge/dir-merge rules. */ static bool is_unsupported_modifier_char(char c) { return c == 'e' || c == 'n' || c == 'w'; } -/* Characters that are part of a modifier run, whether supported or not. */ -static bool is_modifier_scan_char(char c) { - return is_modifier_char(c) || is_unsupported_modifier_char(c); +/* Merge-file modifiers rsync accepts on merge/dir-merge rules: 'e', 'n', 'w' + * and '-' (do not transfer the merge file). */ +static bool is_merge_modifier_char(char c) { + return c == 'e' || c == 'n' || c == 'w' || c == '-'; +} + +/* Characters that count as part of a modifier run for `kind` when deciding + * whether a token is a pure modifier run. e/n/w count on every rule so that a + * pure e/n/w token is rejected on non-merge rules; '-' only on merge rules. */ +static bool is_modifier_scan_char(char c, RuleKind kind) { + return is_modifier_char(c) || is_unsupported_modifier_char(c) || + (is_merge_rule(kind) && is_merge_modifier_char(c)); +} + +/* Characters actually consumed as modifiers for `kind`. The merge-file + * modifiers are consumed only on merge/dir-merge rules; elsewhere e/n/w fall + * through to the pattern (so mixed tokens such as "H,!secret" keep their + * historical "ecret" pattern). */ +static bool is_consumed_modifier_char(char c, RuleKind kind) { + return is_modifier_char(c) || (is_merge_rule(kind) && is_merge_modifier_char(c)); } /* Inspect the token that follows a rule name (up to the first space/underscore * or the end). If the token is composed *solely* of modifier characters and - * includes one FastSync does not implement, it is unambiguously a modifier run: + * includes one that is invalid for `kind`, it is unambiguously a modifier run: * return that character so the caller can reject it. A token that contains any * non-modifier character is a pattern (e.g. "-newfile") and returns '\0', which * keeps the historical parsing of mixed tokens such as "H,!secret" intact. */ -static char unsupported_modifier_in_token(const char* tok) { +static char unsupported_modifier_in_token(const char* tok, RuleKind kind) { if (*tok == '\0' || *tok == ' ' || *tok == '_') return '\0'; char bad = '\0'; for (const char* q = tok; *q != '\0' && *q != ' ' && *q != '_'; q++) { - if (!is_modifier_scan_char(*q)) + if (!is_modifier_scan_char(*q, kind)) return '\0'; - if (is_unsupported_modifier_char(*q)) + if (!is_merge_rule(kind) && is_unsupported_modifier_char(*q)) bad = *q; } return bad; @@ -238,9 +262,9 @@ static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides, Only commit a modifier run that terminates at a separator or the end, so a pattern such as "*.tmp" written as "-*.tmp" is not mistaken for modifiers. */ if (*p == ',') { - *bad_mod = unsupported_modifier_in_token(p + 1); + *bad_mod = unsupported_modifier_in_token(p + 1, *kind); } else if (is_short) { - *bad_mod = unsupported_modifier_in_token(p); + *bad_mod = unsupported_modifier_in_token(p, *kind); } if (*bad_mod != '\0') return false; @@ -250,12 +274,12 @@ static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides, if (*p == ',') { p++; mod_start = p; - while (is_modifier_char(*p)) + while (is_consumed_modifier_char(*p, *kind)) p++; mod_end = p; } else if (is_short) { const char* scan = p; - while (is_modifier_char(*scan)) + while (is_consumed_modifier_char(*scan, *kind)) scan++; if (*scan == '\0' || *scan == ' ' || *scan == '_') { mod_start = p; diff --git a/src/shared/filter.h b/src/shared/filter.h index bc9e9ca..c651afc 100644 --- a/src/shared/filter.h +++ b/src/shared/filter.h @@ -24,8 +24,12 @@ * clear/! clear the current rule list (takes no argument) * Modifiers: '/' absolute anchor, '!' negate match, 'C' inject CVS defaults, * 's' sender side, 'r' receiver side, 'p' perishable. The rsync 'x' - * (xattr-name) modifier and the merge-only 'e'/'n'/'w' modifiers are not - * implemented and are rejected explicitly. + * (xattr-name) modifier is not implemented and is rejected explicitly + * everywhere. The merge-file modifiers 'e' (exclude the merge file itself), + * 'n' (do not inherit the merge file), 'w' (word-split the merge file) and '-' + * (do not transfer the merge file) are accepted and consumed only on merge/ + * dir-merge rules (rejected on every other rule, matching rsync); their + * semantics are not implemented and they are otherwise ignored. * A trailing '/' makes a pattern match directories only. A leading '/' anchors * the pattern to its owner directory. */ diff --git a/tests/test_filter.c b/tests/test_filter.c index 4036cbd..ae426f9 100644 --- a/tests/test_filter.c +++ b/tests/test_filter.c @@ -40,11 +40,17 @@ static void test_filter_list_rejects_xattr_modifier() { } static void test_filter_list_rejects_unsupported_modifiers() { + /* The merge-file modifiers e/n/w/- are invalid on every non-merge rule; a + token made up solely of modifier characters is a modifier run, so it must + be rejected rather than folded into the pattern. */ static const char* const rules[] = { "-e foo", /* e: merge-only in rsync */ "-n foo", /* n: merge-only in rsync */ "-w foo", /* w: merge-only in rsync */ - "merge,n /tmp/x", ".e /tmp/x", "dir-merge,e .rules", "exclude,w foo", + "-new", /* pure modifier letters (n/e/w) */ + "-press", /* pure modifier letters (p/r/e/s) */ + "exclude,w foo", "exclude,e foo", "exclude,n foo", + "hide,w foo", "protect,n foo", "risk,e foo", }; for (size_t i = 0; i < sizeof(rules) / sizeof(rules[0]); i++) { FilterRuleList* list = filter_rule_list_create(); @@ -57,6 +63,63 @@ static void test_filter_list_rejects_unsupported_modifiers() { } } +/* rsync accepts the merge-file modifiers e/n/w/- on merge and dir-merge rules. + * They must be consumed so they never leak into the merge filename. */ +static void test_filter_list_accepts_merge_modifiers() { + char tmpl[] = "/tmp/fastsync_filter_mmod_XXXXXX"; + EXPECT_TRUE(mkdtemp(tmpl) != NULL); + char path[512]; + snprintf(path, sizeof(path), "%s/rules", tmpl); + FILE* fp = fopen(path, "w"); + EXPECT_NOT_NULL(fp); + fputs("- *.tmp\n", fp); + fclose(fp); + + /* merge with e/n/w/- consumes the modifiers and reads the right file. */ + static const char* const fmts[] = { + "merge,e %s", "merge,n %s", "merge,w %s", "merge,- %s", ".e %s", ".- %s", + }; + for (size_t i = 0; i < sizeof(fmts) / sizeof(fmts[0]); i++) { + FilterRuleList* list = filter_rule_list_create(); + EXPECT_NOT_NULL(list); + char rule[600]; + char err[256] = ""; + snprintf(rule, sizeof(rule), fmts[i], path); + bool ok = filter_rule_list_parse_append(list, rule, NULL, NULL, err, sizeof(err)); + if (!ok) + printf(" merge rule '%s' errored: %s\n", rule, err); + EXPECT_TRUE(ok); + EXPECT_EQ_INT(list->count, 1); + EXPECT_EQ_STR(list->items[0]->pattern, "*.tmp"); + filter_rule_list_free(list); + } + + /* dir-merge with e/n/w/- registers the basename without the modifiers. */ + static const struct { + const char* rule; + const char* want; + } drules[] = { + {"dir-merge,e .rules", ".rules"}, {"dir-merge,n .rules", ".rules"}, + {"dir-merge,w .rules", ".rules"}, {"dir-merge,- .rules", ".rules"}, + {":e .rules", ".rules"}, {":- .rules", ".rules"}, + }; + for (size_t i = 0; i < sizeof(drules) / sizeof(drules[0]); i++) { + FilterRuleList* list = filter_rule_list_create(); + EXPECT_NOT_NULL(list); + char err[256] = ""; + bool ok = filter_rule_list_parse_append(list, drules[i].rule, NULL, NULL, err, sizeof(err)); + if (!ok) + printf(" dir-merge rule '%s' errored: %s\n", drules[i].rule, err); + EXPECT_TRUE(ok); + EXPECT_EQ_INT(list->dir_merge_count, 1); + EXPECT_EQ_STR(list->dir_merge_names[0], drules[i].want); + filter_rule_list_free(list); + } + + unlink(path); + rmdir(tmpl); +} + static void test_filter_list_accepts_supported_rules_and_modifiers() { static const char* const rules[] = { "- *.tmp", "+ /a.txt", "-s foo", "-r foo", "-p foo", @@ -223,6 +286,7 @@ static void test_filter_rules_apply_supported_modifiers() { void test_filter() { test_filter_list_rejects_xattr_modifier(); test_filter_list_rejects_unsupported_modifiers(); + test_filter_list_accepts_merge_modifiers(); test_filter_list_accepts_supported_rules_and_modifiers(); test_filter_list_merge_file_still_supported(); test_filter_rule_parse_rejects_unsupported_and_keeps_supported(); -- 2.54.0 From 283f9f08236866ce1d24e6d70004bb77caca0996 Mon Sep 17 00:00:00 2001 From: TapTap Date: Mon, 21 Sep 2026 22:11:19 +0200 Subject: [PATCH 33/68] docs: reflect filter modifiers, inplace+partial-dir, credentials hardening Update the docs for the audit follow-up fixes: - --filter merge modifiers e/n/w/- are now accepted-and-consumed on merge/dir-merge rules (rejected on non-merge, x rejected everywhere); their semantics stay unimplemented, so the --filter row moves to Caveat and the tally becomes 119/11/27 = 157. - --inplace + --partial-dir is rejected with rsync's message. - secret_file_open() O_NOFOLLOW (symlinked credential paths fail closed; fd-backed paths exempt) and ~3 s bound-wait on FIFO reads. - AGENTS setpriv wording corrected to the collected instance count. - CHANGELOG [Unreleased] audit section extended with the follow-ups. --- AGENTS.md | 2 +- CHANGELOG.md | 52 +++++++++++++++++++++++++++++++++++-------------- README.md | 4 ++-- RSYNC_COMPAT.md | 43 +++++++++++++++++++++++----------------- 4 files changed, 65 insertions(+), 36 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 82fc4dd..e0cc853 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -43,7 +43,7 @@ cmake -B build -S . -DSANITIZER=undefined # UndefinedBehaviorSanitizer ( cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan); local-only, NOT in CI ``` -The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, the `address`+`undefined` sanitizer matrix, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. TSan is not part of the CI matrix and is a local-only configuration. The four `setpriv` privilege tests are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions. +The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, the `address`+`undefined` sanitizer matrix, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. TSan is not part of the CI matrix and is a local-only configuration. The `setpriv`-marked privilege tests (four decorated functions, collecting to eight instances because two are parametrized) are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions. ## Build diff --git a/CHANGELOG.md b/CHANGELOG.md index beed8f1..763c1c1 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -11,12 +11,17 @@ The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0). **120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows. An audit cycle follows on the same wire version (`PROTOCOL_VERSION` stays -2.28.0): a security-and-correctness pass over the parity-2.29 baseline. It fixes +2.28.0): a security-and-correctness pass over the parity-2.29 baseline, plus a +set of audit follow-ups (filter merge modifiers, the `--inplace`/`--partial-dir` +conflict, credential-file hardening, and small leak/log/test fixes). It fixes a `--temp-dir` symlink escape, gates client-controlled special permission bits, -corrects `--partial-dir`/`--bwlimit`/`-z` behavior, rejects unsupported filter -modifiers, and tightens client and wire validation. No parity row changes -classification, so the matrix stays **120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows; the -affected rows' notes and the summary tally in `RSYNC_COMPAT.md` were updated. +corrects `--partial-dir`/`--bwlimit`/`-z` behavior, handles unsupported filter +modifiers, and tightens client and wire validation. The only parity +reclassification is `--filter=RULE` moving ✅ → ⚠️, because its merge-only +`e`/`n`/`w`/`-` modifiers are now accepted and consumed but their semantics +remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ / +11 ⚠️ / 27 ❌** of 157 rows. The affected rows' notes and the summary tally in +`RSYNC_COMPAT.md` were updated. ### Changed @@ -73,8 +78,10 @@ affected rows' notes and the summary tally in `RSYNC_COMPAT.md` were updated. conventional `022`, so implied parent directories created without `-p` are no longer world-writable `0777`. - **Credentials and signal handling hardened.** Secret files are opened with - `O_NOFOLLOW|O_NONBLOCK` (while allowing fd-backed store paths), and signal - handlers use `sigaction` with async-signal-safe bodies. + `O_NOFOLLOW|O_NONBLOCK` (while allowing fd-backed store paths and bound-waiting + a FIFO read for ~3 s so a slow process substitution works but a connected-but- + silent FIFO cannot hang), and signal handlers use `sigaction` with + async-signal-safe bodies. ### Fixed @@ -90,20 +97,35 @@ affected rows' notes and the summary tally in `RSYNC_COMPAT.md` were updated. - **`--partial-dir` implies `--partial`.** Matching rsync 3.4.1 (which sets `keep_partial` after option parsing), `--partial-dir=DIR` alone retains an interrupted transfer's partial and wins over an explicit `--no-partial`; - `--inplace` still bypasses the partial machinery. -- **Unsupported filter modifiers rejected.** The `x` xattr-name modifier and the - merge-only `e`/`n`/`w` modifiers are rejected with a clear error instead of - being silently ignored (`x` on merge/dir-merge rules) or folded into the - pattern (producing misleading merge-file errors). Glued patterns (`-newfile`, - `-e2e`) and mixed tokens (`H,!secret`) keep their historical parsing. + `--inplace` still bypasses the partial machinery, and combining `--inplace` + with `--partial-dir` is now rejected up front with rsync's message + (`--inplace cannot be used with --partial-dir`). +- **Filter modifiers handled.** The `x` xattr-name modifier is rejected with a + clear error everywhere. The merge-only `e`/`n`/`w` and `-` modifiers are now + accepted and consumed on `merge`/`dir-merge` rules (so they no longer leak + into the merge filename) while still being rejected on non-merge rules, + matching rsync; their semantics remain unimplemented (accepted-but-ignored). + Glued patterns (`-newfile`, `-e2e`) and mixed tokens (`H,!secret`) keep their + historical parsing. +- **Credential-file reads hardened.** Secret files (`--password-file`/ + `--early-input`/`--hash-credentials` input) are opened with `O_NOFOLLOW`, so a + symlinked credential path now fails closed (`ELOOP`) instead of being followed + before the owner/mode gate; literal fd-backed paths (`/dev/fd/`, + `/proc/self/fd/`) are exempt so process substitution still works. A + FIFO/process-substitution read now waits under a bounded ~3 s deadline for its + writer, so a slow producer works while a connected-but-silent FIFO fails + instead of hanging. - **Miscellaneous correctness fixes:** `--filter` rule count is checked client-side against `MAX_FILTER_RULES` before any network I/O (the receiver still re-checks the expanded count); unknown wire `Status` values are rejected as protocol errors; a mutex leak on an init-failure path, an `errno` read after `free()` in deferred delete application, `log_perror` misuse for non-`errno` conditions, and a `NULL` `server_host`/`ssh_destination` - allocation path were fixed; `SSL_read` length is clamped and `sendfile` - `poll()` retries on `EINTR`. + allocation path were fixed (the `config_create` failure now releases through + `config_delete`); the decompression-limit log now prints the effective bound + rather than the compile-time ceiling; the daemon umask and root test fixtures + were hardened; `SSL_read` length is clamped and `sendfile` `poll()` retries on + `EINTR`. ### Refactored / Docs diff --git a/README.md b/README.md index 03a7ea2..22b44dd 100644 --- a/README.md +++ b/README.md @@ -610,8 +610,8 @@ remote SSH argv is already built injection-safe. | `--backup-dir ` | Store backups under a separate directory (requires `--backup`). | | `--suffix ` | Set the backup filename suffix (default: `~`). | | `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir `, completed files are written under the partial directory and installed atomically. | -| `--partial-dir ` | Set a relative partial-transfer directory below the server destination root. Implies `--partial` (unless `--inplace`, which bypasses the partial/temp staging). | -| `--inplace` | Write directly to the destination instead of using a temporary file. | +| `--partial-dir ` | Set a relative partial-transfer directory below the server destination root. Implies `--partial`. Rejected together with `--inplace` (`--inplace cannot be used with --partial-dir`, matching rsync), because the inplace path bypasses partial/temp staging. | +| `--inplace` | Write directly to the destination instead of using a temporary file. Cannot be combined with `--partial-dir`. | | `--fsync` | Fsync every written file before publication. | | `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree. | | `--only-write-batch=FILE` | Emit the batch file only (no destination, no server). | diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 437b113..7657532 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -6,8 +6,8 @@ This document maps rsync's full feature set to FastSync's current implementation | Status | Count | Description | |--------|-------|-------------| -| ✅ Parity | 120 | Reproduces rsync's semantics for this option's scope | -| ⚠️ Caveat | 10 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) | +| ✅ Parity | 119 | Reproduces rsync's semantics for this option's scope | +| ⚠️ Caveat | 11 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) | | ❌ Divergent | 27 | Rejected, an accepted no-op, deliberately non-rsync (native config/auth/batch, privileged namespaces, safe-subset privilege), or impossible on any portable filesystem call | | **Total** | **157** | One row per rsync option/feature group; a row may name several spellings | @@ -62,21 +62,23 @@ matrix is **111 ✅ / 13 ⚠️ / 33 ❌ = 157**. - **Fuzzy eligibility.** The `-y/--fuzzy` candidate search no longer inherits the ordinary delta engine's 16 KiB minimum or 10× ratio bound, so an oversized or sub-16-KiB sibling is reused as rsync reuses it (`test_parity_basis_fuzzy.py`). - **Output partials.** `--info=mount`/`--info=stats`, the `--stats` `dir:` breakdown under `-r`, and real `--debug` output for `flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send` were added (`test_parity_info_mount_stats.py`, `test_output_parity.py`, `test_parity_debug.py`); those rows stay ⚠️ for their remaining documented residuals. `--delete-before`'s phase-0 late-file divergence and the `--progress` root/ancestor/symlink feedback remain open (they need a receiver→sender event channel), and the >256 MiB single-file streaming limit (B4) was not addressed. The matrix is now **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. -**Audit cycle (no wire change; `PROTOCOL_VERSION` stays 2.28.0).** A security-and-correctness audit pass ran against the parity-2.29 baseline; none of the fixes changes a row's classification, so the matrix stays **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. The affected rows (`-z`/`--compress`, `--bwlimit`, `-T`/`--temp-dir`, `-p`/`--chmod`, `--partial-dir`, `--filter`) had their notes updated in place: +**Audit cycle (no wire change; `PROTOCOL_VERSION` stays 2.28.0).** A security-and-correctness audit pass ran against the parity-2.29 baseline, followed by a set of audit follow-ups (filter merge modifiers, the `--inplace`/`--partial-dir` conflict, credential-file hardening, and small leak/log/test fixes). The only classification change is `--filter=RULE` moving ✅ → ⚠️, because its merge-only `e`/`n`/`w`/`-` modifiers are now accepted and consumed but their semantics remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ / 11 ⚠️ / 27 ❌ = 157**. The affected rows (`-z`/`--compress`, `--bwlimit`, `-T`/`--temp-dir`, `-p`/`--chmod`, `--partial-dir`, `--filter`) had their notes updated in place: - **Decompression ceiling.** `MAX_DECOMPRESSED_SIZE` was 100 MiB while the receiver advertises and the sender compresses whole files up to `MAX_RECEIVE_WHOLE_FILE_SIZE` (256 MiB), so `-z` on a 100–256 MiB regular file failed with `Declared decompressed size exceeds 104857600 bytes`. The ceiling is now defined in terms of the protocol whole-file bound (still a real allocation-clamped bomb guard), so the two cannot drift; `-z` on 100–256 MiB files now works. - **`--bwlimit` with `--sendfile`.** The plaintext-TCP `--sendfile` fast path wrote through `sendfile(2)` without passing through the protocol's token bucket, so `--bwlimit` was ignored on that path. It is now paced through the same per-session leaky bucket, so TLS and plaintext transports share identical `--bwlimit` semantics. - **`--temp-dir` confinement.** The receiver's scratch dir was opened with a bare `open()`, so a client-planted symlink under the receive root could redirect receiver scratch files outside the authorized root. The opened directory is now judged by the real path of its fd (`/proc/self/fd` via `realpath`), and an escaping target is refused (`EACCES`, logged); an in-root link to another filesystem (the `EXDEV` fallback case) still works. - **Special-bit masking and daemon umask.** Setuid/setgid/sticky bits from the client (`--perms`, `--chmod`, symlink and special-node paths, deferred directory modes) were applied even when the connection forbade super-user activities. They are now stripped when the super policy is off (`FileAttrPolicy.super_permitted`), and exact rsync semantics are preserved when permitted. The daemon's forced `umask(0)` is now `umask(022)`, so implied parent directories are no longer world-writable `0777`. -- **`--partial-dir` implies `--partial`.** Matching rsync 3.4.1 (which sets `keep_partial` after option parsing), `--partial-dir=DIR` alone now retains an interrupted transfer's partial and wins over an explicit `--no-partial`; `--inplace` still bypasses the partial machinery. -- **Filter modifiers.** The `x` xattr-name modifier and the merge-only `e`/`n`/`w` modifiers are now rejected with a clear error instead of being silently ignored (`x` on merge/dir-merge rules) or folded into the pattern (producing misleading merge-file errors). Glued patterns (`-newfile`, `-e2e`) keep their historical parsing. +- **`--partial-dir` implies `--partial`.** Matching rsync 3.4.1 (which sets `keep_partial` after option parsing), `--partial-dir=DIR` alone now retains an interrupted transfer's partial and wins over an explicit `--no-partial`; `--inplace` still bypasses the partial machinery, and combining `--inplace` with `--partial-dir` is now rejected up front with rsync's message (`--inplace cannot be used with --partial-dir`) instead of silently ignoring the partial dir. +- **Filter modifiers.** The `x` xattr-name modifier is rejected everywhere with a clear error. The merge-only `e`/`n`/`w` and `-` modifiers are now accepted and consumed on `merge`/`dir-merge` rules (so they no longer leak into the merge filename) while still being rejected on non-merge rules, matching rsync; their semantics remain unimplemented (accepted-but-ignored). Glued patterns (`-newfile`, `-e2e`) and mixed tokens (`H,!secret`) keep their historical parsing. - **Bounds and wire validation.** `--filter` rule count is now checked client-side against `MAX_FILTER_RULES` (with an actionable message before any network I/O) rather than surfacing as an opaque receiver protocol error; `send_protect_entries()` still re-checks the expanded count. Unknown wire `Status` values are rejected as protocol errors (`status_is_valid()`), and the audit also fixed a mutex leak on an init-failure path, an `errno`-after-`free()` in deferred delete application, `log_perror` misuse for non-`errno` conditions, `SSL_read` length clamping, `sendfile` `poll` `EINTR` retry, and printf-format/attribute issues. +- **Credential-file hardening follow-up.** `secret_file_open()` now opens `--password-file`/`--early-input`/`--hash-credentials` inputs with `O_NOFOLLOW`, so a symlinked credential path fails closed (`ELOOP`) instead of being followed before the owner/mode gate; literal fd-backed paths (`/dev/fd/`, `/proc/self/fd/`, which is what a bash process substitution passes) are exempt, so process substitution still works. A FIFO/process-substitution read now waits under a bounded ~3 s deadline for its writer, so a slow producer works while a connected-but-silent FIFO fails instead of hanging. The follow-up also fixed a `config_create` allocation leak on its `server_host` failure path (`config_delete` now releases it), corrected the decompression-limit log message to print the effective bound rather than the compile-time ceiling, and hardened the daemon umask/root test fixtures. **Parity completion wave (protocol 2.23.0 → 2.26.0).** This wave closed the remaining gaps the rsync-parity wave left open (delete timing, wire counters and output, codec breadth, general `-R`/`-d`, the filter grammar (the unsupported -`x` xattr-name and merge-only `e`/`n`/`w` modifiers are explicitly rejected, not -silently accepted), receiver-side +`x` xattr-name modifier is explicitly rejected everywhere, while the merge-only +`e`/`n`/`w`/`-` modifiers are accepted and consumed on merge/dir-merge rules and +rejected elsewhere — see the audit-cycle follow-up note above), receiver-side name resolution, absolute basis dirs, and the remaining client quick wins) and reclassified the inherently non-rsync rows as **divergent** (native daemon config/auth, the non-interoperable batch container, `--fake-super`'s xattr @@ -134,7 +136,7 @@ Every one of those has an entry below with its remaining caveats. |------|-------------------|-----------------|-------| | `--exclude-from=FILE` | Read exclude patterns from file | ✅ Parity | Reads patterns from file | | `--include-from=FILE` | Read include patterns from file | ✅ Parity | Reads patterns from file | -| `--filter=RULE` | Add file-filtering rule | ✅ Parity | The short `-f` **is** bound to `--filter` (the old FastSync sendfile conflict is gone; sendfile is long-only `--sendfile`), and `-f RULE`, `-f=RULE`, `--filter=RULE` and the two-argument form all parse. Protocol 2.26.0 implements rsync's filter grammar: `+`/`-`, `include`/`exclude`, a leading `/` anchor (to the transfer root or a `.rsync-filter` file's directory), a trailing `/` dir-only rule, and the `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R` and `clear`/`!` words, including the `:`/`.` modifiers. The xattr-name `x` modifier and the merge-only `e`/`n`/`w` modifiers are **explicitly rejected with a clear error** (audit-cycle fix: the list parser used by `--filter`/`-f` previously silently ignored `x` on merge/dir-merge rules and folded `e`/`n`/`w` into the pattern, producing misleading failures); a token made up solely of modifier characters that names an unsupported modifier is rejected, while glued patterns (`-newfile`, `-e2e`) and mixed tokens (`H,!secret`) keep their historical parsing. First match wins; the filter layer is independent of `--exclude`/`--include`. **Track 4a (protocol 2.28.0) adds the receiver filter engine:** the sender compiles its root-level rules exactly as the scanner does (`filter_base_build`) and streams them as one bounded, self-describing config-frame block; the receiver reconstructs them and re-applies first-match-wins to every extraneous destination path during deletion, so a `P *.log` rule protects a destination-only `extra.log` (differential `filter_protect`/`filter_protect_during`/`filter_protect_delay` vs rsync 3.4.1, plus the `-n` would-delete enumeration) — matching rsync's dual-sided engine for the command-line rule set. **Remaining residual:** per-directory merge (`:`/`.`, and therefore `-F`) is not yet re-derived on the receiver; a destination-only entry that matches ONLY a per-directory merge rule is still protected only through the sender-derived source-mirror prefixes, not by the received base rule list | +| `--filter=RULE` | Add file-filtering rule | ⚠️ Caveat | The short `-f` **is** bound to `--filter` (the old FastSync sendfile conflict is gone; sendfile is long-only `--sendfile`), and `-f RULE`, `-f=RULE`, `--filter=RULE` and the two-argument form all parse. Protocol 2.26.0 implements rsync's filter grammar: `+`/`-`, `include`/`exclude`, a leading `/` anchor (to the transfer root or a `.rsync-filter` file's directory), a trailing `/` dir-only rule, and the `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R` and `clear`/`!` words, including the `:`/`.` modifiers. The xattr-name `x` modifier is **explicitly rejected everywhere with a clear error**. The merge-only `e`/`n`/`w` and `-` modifiers are **accepted and consumed on `merge`/`dir-merge` rules** (so they no longer leak into the merge filename) while still being **rejected on non-merge rules**, matching rsync; their semantics remain unimplemented, so they are accepted-but-ignored (the reason this row is a caveat rather than parity). A token made up solely of modifier characters that names an unsupported modifier is rejected on non-merge rules, while glued patterns (`-newfile`, `-e2e`) and mixed tokens (`H,!secret`) keep their historical parsing. First match wins; the filter layer is independent of `--exclude`/`--include`. **Track 4a (protocol 2.28.0) adds the receiver filter engine:** the sender compiles its root-level rules exactly as the scanner does (`filter_base_build`) and streams them as one bounded, self-describing config-frame block; the receiver reconstructs them and re-applies first-match-wins to every extraneous destination path during deletion, so a `P *.log` rule protects a destination-only `extra.log` (differential `filter_protect`/`filter_protect_during`/`filter_protect_delay` vs rsync 3.4.1, plus the `-n` would-delete enumeration) — matching rsync's dual-sided engine for the command-line rule set. **Remaining residual:** per-directory merge (`:`/`.`, and therefore `-F`) is not yet re-derived on the receiver; a destination-only entry that matches ONLY a per-directory merge rule is still protected only through the sender-derived source-mirror prefixes, not by the received base rule list | | `--files-from=FILE` | Read source file list from file | ✅ Parity | Entries are paths relative to the source root (leading `./` stripped, `..`/absolute rejected at parse time, blank lines ignored; NUL-delimited with `-0`). A listed regular file is transferred; a listed directory transfers its whole subtree (FastSync recursion is always on). Non-listed paths are pruned by the scanner; the delete manifest is scoped to the listed directory subtrees. A listed entry that does not exist is a hard error unless `--ignore-missing-args`/`--delete-missing-args` is given. **An empty list is a zero-transfer success (exit 0), matching rsync 3.4.1** — the earlier claim that rsync reports "no source files specified" was wrong. Scalability note: `file_list_affects` is O(list size) per scanned entry, so a very large list against a huge tree is quadratic (the documented bound) | | `-0`, `--from0` | Delimit *-from files with NULs | ✅ Parity | `--files-from` entries become NUL-delimited; the flag may appear before or after `--files-from` on the command line. NUL mode preserves entry bytes exactly (trailing CR/LF are part of the name; only newline mode trims them) | | `--max-size=SIZE` | Skip files larger than SIZE | ✅ Parity | `max_size` in scanner | @@ -182,7 +184,7 @@ Every one of those has an entry below with its remaining caveats. | `--delay-updates` | Put updated files in place at end | ❌ Divergent | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication, and **`--force` is honored at publication** (protocol 2.23.0): a staged regular file or symlink may replace a destination directory that blocks it. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). **Reclassified Divergent (differential evidence):** the staging name is fixed and a delayed run wipes a pre-existing destination tree of that name at start even without `--delete`, whereas rsync uses its own internal temp name and leaves a genuine destination entry named `.fastsync-stage` untouched (`test_delay_updates_staging_name_collision_residual`); deletion also runs before publication while rsync's `--delay-updates` implies `--delete-after`. Works in single-threaded and `-j`/`--threads` modes | | `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ❌ Divergent | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). A **relative** dir matches rsync exactly: it is resolved below the receive/destination root and must already exist (differentially verified: `rsync -a --temp-dir=scratch src/ dst/` and FastSync produce identical trees and an empty scratch dir). **Reclassified as a deliberate divergence because an absolute `--temp-dir` is rejected by the receiver** — it is resolved verbatim by rsync standalone (which will use `/tmp` or any other absolute directory, including one outside the destination), but FastSync's security-reviewed receiver confines the scratch dir to the authorized receive root and rejects any absolute path or one containing `..`. **Audit-cycle hardening:** the opened dir is additionally judged by the real path of its fd (`/proc/self/fd`), so a client-planted symlink under the receive root cannot redirect receiver scratch files outside the authorized root (an escaping target is refused with `EACCES`), while an in-root symlink to another filesystem — the `EXDEV` fallback case — still works. A differential test confirms rsync exits 0 using an absolute scratch dir while FastSync refuses before writing anything into it (the scratch dir stays empty). Its daemon mode also confines relative to the module, but standalone rsync's absolute-temp-dir behavior is not reproduced because it would let a client place receiver scratch files outside the sandbox. Temp copies use a unique name in the scratch dir and are atomically renamed into place; **on `EXDEV` (scratch dir and destination on different filesystems, reachable via a confined relative symlink) the receiver falls back to a non-atomic copy instead of aborting**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir | | `--partial` | Keep partially transferred files | ✅ Parity | On a failed/interrupted write the already-written temp file is retained at the destination path (best-effort rename instead of unlink) so a later `--append`/`--append-verify` run can resume it. Retention never runs when no data was actually written or under `--ignore-existing`/`--existing` (the destination is not ours to overwrite), and it only ever renames the already-written temp. A failed rename falls back to the normal unlink | -| `--partial-dir=DIR` | Keep partial files in DIR | ✅ Parity | The working file is written under the confined partial directory (a relative dir below the receive root) and atomically renamed into place once complete, so an interrupted transfer leaves a resumable copy there and completed transfers do not linger under it. `--inplace` bypasses the partial dir (rsync parity). **Implies `--partial`** (audit-cycle fix, matching rsync 3.4.1, which sets `keep_partial` after option parsing): `--partial-dir=DIR` alone retains an interrupted transfer's partial, and the implication wins over an explicit `--no-partial` regardless of order. `--inplace` is the exception — it writes the destination in place with no partial staging, so the implication is skipped | +| `--partial-dir=DIR` | Keep partial files in DIR | ✅ Parity | The working file is written under the confined partial directory (a relative dir below the receive root) and atomically renamed into place once complete, so an interrupted transfer leaves a resumable copy there and completed transfers do not linger under it. `--inplace` bypasses the partial dir (rsync parity), and combining `--inplace` with `--partial-dir` is now **rejected up front** with rsync's message (`--inplace cannot be used with --partial-dir`) instead of silently ignoring the partial dir. **Implies `--partial`** (audit-cycle fix, matching rsync 3.4.1, which sets `keep_partial` after option parsing): `--partial-dir=DIR` alone retains an interrupted transfer's partial, and the implication wins over an explicit `--no-partial` regardless of order | ## 7. Deletion @@ -726,8 +728,8 @@ targets verbatim, matching rsync. | `--config=FILE` | Alternate rsyncd.conf file | ❌ Divergent | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` | | `--dparam=OVERRIDE` | Override global daemon config | ❌ Divergent | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | | `--no-detach` | Don't detach from parent | ✅ Parity | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` | -| `--password-file=FILE` | Read daemon password from file | ❌ Divergent | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | -| `--early-input=FILE` | Use FILE for daemon early exec | ❌ Divergent | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | +| `--password-file=FILE` | Read daemon password from file | ❌ Divergent | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. **Hardening follow-up:** the file is opened with `O_NOFOLLOW`, so a symlinked credential path fails closed (`ELOOP`) instead of being followed before the owner/mode gate; literal fd-backed paths (`/dev/fd/`, `/proc/self/fd/`, which is what a bash process substitution passes) are exempt, so process substitution still works. A FIFO/process-substitution read now waits under a bounded ~3 s deadline for its writer, so a slow producer works while a connected-but-silent FIFO fails instead of hanging. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | +| `--early-input=FILE` | Use FILE for daemon early exec | ❌ Divergent | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. Opened with the same `O_NOFOLLOW` hardening as `--password-file` (a symlinked path fails closed with `ELOOP`; fd-backed `/dev/fd/N`/`/proc/self/fd/N` process-substitution paths are exempt) and a FIFO read is bound-waited (~3 s) so a slow producer works while a writer-less FIFO cannot hang. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | | `--hash-credentials=FILE`, `--iterations N` | Hash a plaintext credential file | ❌ Divergent | Server-only offline tool (A7): reads the `user:password` lines of FILE (same owner-only 0600 check) and prints one new-format store line per entry to stdout, then exits. `--iterations` sets the PBKDF2 work factor (default 600000, range 100000–10000000). Dependency-free and does not run a listener. Use its output as `--password-file` for `--daemon`. There is no auto-upgrade: a legacy store line is hard-rejected by the loader and must be regenerated | **Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. @@ -949,10 +951,12 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass.** ✅ Parity 120 / ⚠️ Caveat 10 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `--delete-before`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass and the audit-cycle follow-ups.** ✅ Parity 119 / ⚠️ Caveat 11 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `--delete-before`, `--filter`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that -receiver filter engine, flipping `--filter=RULE` back to ✅ (see above). The fs pass flips `-d/--dirs` and `--iconv` to ✅ — recursive transfers now recreate empty source directories (and replace a blocking destination non-directory with an incoming directory); `-R --no-implied-dirs --files-from` places a listed file under a missing implied parent with default attributes instead of refusing; and `--iconv` now reproduces rsync's push direction (destination charset = the spec's REMOTE half) — and reclassified six rows to ❌ after reproducing their exact residual with differential tests: `--temp-dir` (the receiver confines the scratch dir to the receive root, so an absolute temp dir is deliberately rejected although standalone rsync follows it), the three basis-dir options (FastSync xxHash-verifies a basis hit while rsync's `--size-only` quick check installs the wrong basis content), `--delay-updates` (fixed staging name wipes an unrelated destination entry of that name), and `--dry-run` (would-delete report over-reports). `--fuzzy` was also reclassified to ❌ (deterministic heuristic with a 10× size window, not rsync's matcher), but its residual is the candidate-selection heuristic itself: the final tree is byte-exact by design, so no destination differential can expose it and the row is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync differential. (Track 5b later showed the name heuristic is in fact rsync's own and moved the row ❌ → ⚠️, leaving only the narrower delta size window as the residual; see the track 5b paragraph above.) The remaining ⚠️ rows are the ones with a documented residual (see the row notes and the **Parity Completion Wave (protocol 2.26.0)** section below). +receiver filter engine, flipping `--filter=RULE` back to ✅ (see above; the +audit-cycle follow-ups later moved it to ⚠️ for the accepted-but-ignored merge +modifiers, see the audit-cycle note). The fs pass flips `-d/--dirs` and `--iconv` to ✅ — recursive transfers now recreate empty source directories (and replace a blocking destination non-directory with an incoming directory); `-R --no-implied-dirs --files-from` places a listed file under a missing implied parent with default attributes instead of refusing; and `--iconv` now reproduces rsync's push direction (destination charset = the spec's REMOTE half) — and reclassified six rows to ❌ after reproducing their exact residual with differential tests: `--temp-dir` (the receiver confines the scratch dir to the receive root, so an absolute temp dir is deliberately rejected although standalone rsync follows it), the three basis-dir options (FastSync xxHash-verifies a basis hit while rsync's `--size-only` quick check installs the wrong basis content), `--delay-updates` (fixed staging name wipes an unrelated destination entry of that name), and `--dry-run` (would-delete report over-reports). `--fuzzy` was also reclassified to ❌ (deterministic heuristic with a 10× size window, not rsync's matcher), but its residual is the candidate-selection heuristic itself: the final tree is byte-exact by design, so no destination differential can expose it and the row is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync differential. (Track 5b later showed the name heuristic is in fact rsync's own and moved the row ❌ → ⚠️, leaving only the narrower delta size window as the residual; see the track 5b paragraph above.) The remaining ⚠️ rows are the ones with a documented residual (see the row notes and the **Parity Completion Wave (protocol 2.26.0)** section below). **Preserve-attribute split (protocol 2.21.0 → 2.22.0) — ✅ implemented.** FastSync splits the former single metadata bundle into four independent, rsync-compatible per-attribute flags — `-p/--perms`, `-t/--times`, `-o/--owner`, `-g/--group` — each with a negation (`--no-perms`/`--no-times`/`--no-owner`/`--no-group`, short `--no-p`/`--no-t`/`--no-o`/`--no-g`), plus `--no-preserve` clearing all four. `-a/--archive` is now full rsync `-rlptgoD` (owner and group included, though their application stays privilege-gated), `-A/--acls` implies `-p`, `-X/--xattrs` does not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply `-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the user explicitly negated them. Wire: the binary config frame gains four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/`preserve_group`) after `omit_link_times`, so `PROTOCOL_VERSION` is bumped **2.21.0 → 2.22.0**; the fixed-width `FileMetadata` layout is unchanged and the receiver gates the metadata frame on a derived `use_metadata`. Receiver behavior: each attribute is applied independently, directory modes are applied under `-p` (at the end of the transfer, alongside dir times), symlink mode under `-p`, and `-O/--omit-dir-times` suppresses directory times only. Documented divergences as of 2.22.0, **all but (d)/(e) removed by the rsync-parity wave (protocol 2.23.0)**: (a) the mode-masking divergence is **gone** — under `-p` the source mode is now copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky; (b) a brand-new file without `-p` still gets `source_mode & ~umask` when metadata is present (else the historical fixed `0644`), and a new *directory* without `-p` still uses FastSync's `0755` default; (c) the `--chmod`-implies-`-p` divergence is **gone** — `--chmod` no longer implies `-p` (rsync parity); (d) `-o`/`-g` map by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); (e) a daemon module without `client owner = yes` does not refuse a plain `-a`/`-o`/`-g` — it forces super off, applies no ownership, and logs a warning, while explicit `--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused. @@ -1085,7 +1089,9 @@ These remain after the wave; they are the reasons a row above is ⚠️. count) are reported as 0; `--progress` is an aggregate line, not per-file. - **`--password-file`/`--early-input`/`--hash-credentials`/`--iterations` are FastSync-native** (SCRAM/PBKDF2), not rsync semantics; the batch format is not - rsync-interoperable. + rsync-interoperable. Credential files are opened with `O_NOFOLLOW` (a symlinked + path fails closed; fd-backed process-substitution paths are exempt) and a FIFO + read is bound-waited (~3 s). - **xattr/ACL namespace policy** permits only `user.*` and `system.posix_acl_*` when `-A` is negotiated (stricter than rsync). - **`--stop-at` remains a FastSync-flexible parser** (client-only, not @@ -1165,10 +1171,11 @@ wire protocol three times (full rationale in `src/shared/config.h`): - **Filter grammar:** `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R`, `clear`/`!`, include/exclude and the `:`/`.` modifiers; `-f` is bound to `--filter`; a single `-F` transfers - `.rsync-filter` and `-FF` excludes it. The xattr-name `x` modifier and the - merge-only `e`/`n`/`w` modifiers are **not implemented** and are rejected with - a clear error (audit cycle) instead of being silently ignored or folded into - the pattern. + `.rsync-filter` and `-FF` excludes it. The xattr-name `x` modifier is **not + implemented** and is rejected with a clear error everywhere. The merge-only + `e`/`n`/`w` and `-` modifiers are accepted and consumed on `merge`/`dir-merge` + rules (rejected elsewhere, matching rsync), but their semantics are **not + implemented** (accepted-but-ignored). - **Absolute basis directories** are used verbatim (rsync semantics) and **`--link-dest`** relinks an already up-to-date destination. -- 2.54.0 From 4638030288d85a45fde186ad3e3b75e5cfa807b9 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:34:55 +0200 Subject: [PATCH 34/68] refactor(client): split reporting/scan/manifest out of client_send Move the stats/progress reporting, scanner-preparation/scan helpers and manifest/list/dry-run senders out of the ~3.9k-line client_send.c into client_report.c, client_scan.c and client_manifest.c, sharing declarations through the new internal client_send_internal.h. client_send.c keeps the transfer orchestration and is now ~2.1k lines. Decompose the monolithic send_files into static phase helpers (send_files_prepare/_prepare_delete/_run/_finalize/_cleanup) driven by a single SendFilesState; ownership, ordering and exit codes are unchanged. No behavior change. --- CMakeLists.txt | 3 + src/client/client_manifest.c | 682 +++++++++ src/client/client_report.c | 704 +++++++++ src/client/client_scan.c | 454 ++++++ src/client/client_send.c | 2313 ++++------------------------- src/client/client_send_internal.h | 91 ++ 6 files changed, 2207 insertions(+), 2040 deletions(-) create mode 100644 src/client/client_manifest.c create mode 100644 src/client/client_report.c create mode 100644 src/client/client_scan.c create mode 100644 src/client/client_send_internal.h diff --git a/CMakeLists.txt b/CMakeLists.txt index a55d93e..f567378 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -136,6 +136,9 @@ set(SERVER_MAIN_SRCS src/server/server.c) # Client implementation (no main): everything except the CLI entry point. set(CLIENT_CORE_SRCS src/client/change_list.c + src/client/client_manifest.c + src/client/client_report.c + src/client/client_scan.c src/client/client_send.c src/client/client_validation.c src/client/scanner.c diff --git a/src/client/client_manifest.c b/src/client/client_manifest.c new file mode 100644 index 0000000..d476bb3 --- /dev/null +++ b/src/client/client_manifest.c @@ -0,0 +1,682 @@ +#include "client_send_internal.h" +#include "array_list.h" +#include "change_list.h" +#include "charset.h" +#include "config.h" +#include "data.h" +#include "delta.h" +#include "file.h" +#include "format.h" +#include "log.h" +#include "protocol.h" +#include "scanner.h" +#include "transport_tls.h" +#include "utils.h" +#include +#include +#include +#include +#include + +/* True when --dry-run should contact a receiver rather than running the + * client-side local manifest. Any target a real run would reach over the wire + * selects the server-contacting path: a remote (SSH host:path), a daemon + * (host::module/path), an explicit --server-host, --server-port/--port, TLS, or + * a source-bind --address. A plain local destination (none of these) keeps the + * original client-side behavior, which never dials the default 127.0.0.1:8080. */ +bool dry_run_targets_server(const Config* config) { + if (!config) + return false; + if (config->transport == TRANSPORT_SSH) + return true; + if (config->module && config->module[0] != '\0') + return true; + if (config->server_host_set || config->server_port_set) + return true; + if (config->use_tls) + return true; + if (config->address != NULL) + return true; + return false; +} + +bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk) { + if (!manifest) + return true; + for (int i = 0; i < chunk->element_count; i++) { + const char* path = file_wire_path(chunk->items[i]); + if (*path == '/') + path++; + char* entry = str_dup(path); + if (!entry) { + log_message(LOG_LEVEL_ERROR, "Failed to allocate manifest entry"); + return false; + } + if (!array_list_add(manifest, entry)) { + free(entry); + return false; + } + } + return true; +} + +/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */ +int send_dry_run_manifest(const Config* config) { + int skipped = 0; + ArrayList* missing_dest = NULL; + if (config->delete_missing_args) { + missing_dest = array_list_create(free); + if (!missing_dest) + return -1; + } + if (!files_from_list_check(config, missing_dest, &skipped)) { + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) { + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) { + prepared_scanner_destroy(&prepared); + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + Chunk* chunk; + int file_count = 0; + unsigned long long total_bytes = 0; + char size_buffer[32]; + if (!config->quiet) + printf("Dry run: files to be transferred\n"); + while ((chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + if (!config->quiet) { + char* escaped_path = + output_escape(file_wire_path(chunk->items[i]), config->eight_bit_output); + if (!escaped_path) { + chunk_destroy(chunk); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + if (missing_dest) + array_list_delete(missing_dest); + return -1; + } + if (config->human_readable) + printf( + " %s (%s)\n", escaped_path, + display_bytes(chunk->items[i]->data->size, true, size_buffer, sizeof(size_buffer))); + else + printf(" %s (%zu bytes)\n", escaped_path, chunk->items[i]->data->size); + free(escaped_path); + } + total_bytes += chunk->items[i]->data->size; + file_count++; + } + chunk_destroy(chunk); + } + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + /* --delete-missing-args: the missing entries' destination mirrors render as + would-be deletions (rsync's dry-run also lists its *deleting lines). */ + if (missing_dest && !config->quiet) { + for (int i = 0; i < missing_dest->size; i++) { + char* escaped = output_escape((char*)missing_dest->items[i], config->eight_bit_output); + printf(" %s (missing; would be deleted)\n", escaped ? escaped : ""); + free(escaped); + } + } + if (missing_dest) + array_list_delete(missing_dest); + if (!config->quiet) { + if (config->human_readable) + printf("Total: %d files, %s\n", file_count, + display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); + else + printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); + } + return 0; +} + +typedef struct { + char* name; /* transfer-relative name ("" == the source root) */ + mode_t mode; + unsigned long long size; + time_t mtime; + long mtime_nsec; + bool is_dir; + bool is_symlink; + char* link_target; +} ListEntry; + +static void list_entries_destroy(ListEntry* entries, size_t count) { + if (entries == NULL) + return; + for (size_t i = 0; i < count; i++) { + free(entries[i].name); + free(entries[i].link_target); + } + free(entries); +} + +static int compare_list_entries(const void* left, const void* right) { + const ListEntry* a = (const ListEntry*)left; + const ListEntry* b = (const ListEntry*)right; + return strcmp(a->name, b->name); +} + +/* Relative path of an entry below `root` ("" for the root itself). Mirrors + * change_list's relative_name for list-only rendering. */ +static char* list_relative_name(const char* root, const char* full) { + if (root == NULL || full == NULL) + return str_dup(full != NULL ? full : ""); + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(root, full, root_len) == 0) { + if (full[root_len] == '\0') + return str_dup(""); + if (full[root_len] == '/') + return str_dup(full + root_len + 1); + } + return str_dup(full); +} + +/* --list-only: print an ls-style listing of the entries that WOULD be + * transferred and exit without contacting the server or writing anything. + * Names are transfer-relative (rsync prints `a.txt`, `sub/b.txt`, `.`) and + * directory entries are included. Returns 0 on success, 1 on error. */ +int send_list_only(const Config* config) { + int skipped = 0; + if (!files_from_list_check(config, NULL, &skipped)) + return 1; + PreparedScanner prepared; + if (!prepare_scanner(config, 0, &prepared)) + return 1; + prepared.options.use_metadata = true; /* capture mode + mtime for the listing */ + prepared.options.list_dirs = true; + DirectoryScanner* scanner = + directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) { + prepared_scanner_destroy(&prepared); + return 1; + } + ListEntry* entries = NULL; + size_t count = 0; + size_t capacity = 0; + bool oom = false; + + /* rsync lists the source root itself (as "."). Only when the source is a + * directory and no --files-from subset is in effect. */ + if (config->files_from_set == NULL && config->send_directory != NULL) { + struct stat st; + if (stat(config->send_directory, &st) == 0 && S_ISDIR(st.st_mode)) { + capacity = 64; + entries = calloc(capacity, sizeof(ListEntry)); + if (entries == NULL) { + oom = true; + } else if ((entries[0].name = str_dup("")) == NULL) { + /* A NULL name would be dereferenced by qsort/render: fail the listing. */ + oom = true; + } else { + entries[0].mode = st.st_mode; + entries[0].mtime = st.st_mtime; + entries[0].mtime_nsec = st.st_mtim.tv_nsec; + entries[0].size = (unsigned long long)st.st_size; + entries[0].is_dir = true; + count = 1; + } + } + } + + Chunk* chunk; + while (!oom && (chunk = directory_scanner_next(scanner)) != NULL) { + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (f == NULL) + continue; + if (count == capacity) { + size_t new_capacity = capacity > 0 ? capacity * 2 : 64; + if (new_capacity <= capacity) { + oom = true; + break; + } + ListEntry* grown = realloc(entries, new_capacity * sizeof(ListEntry)); + if (!grown) { + oom = true; + break; + } + entries = grown; + memset(entries + capacity, 0, (new_capacity - capacity) * sizeof(ListEntry)); + capacity = new_capacity; + } + char* name = list_relative_name(config->send_directory, file_wire_path(f)); + if (!name) { + oom = true; + break; + } + mode_t mode = 0; + time_t mtime = 0; + long mtime_nsec = 0; + if (f->metadata != NULL) { + mode = f->metadata->mode; + mtime = f->metadata->mtime_sec; + mtime_nsec = f->metadata->mtime_nsec; + } else { + struct stat st; + if (lstat(f->path, &st) == 0) { + mode = st.st_mode; + mtime = st.st_mtime; + mtime_nsec = st.st_mtim.tv_nsec; + } + } + entries[count].name = name; + entries[count].mode = mode; + entries[count].mtime = mtime; + entries[count].mtime_nsec = mtime_nsec; + if (f->is_symlink) + entries[count].size = f->symlink_target != NULL ? strlen(f->symlink_target) : 0; + else if (f->is_dir) { + struct stat dir_st; + entries[count].size = stat(f->path, &dir_st) == 0 ? (unsigned long long)dir_st.st_size : 0; + } else + entries[count].size = f->data != NULL ? f->data->size : 0; + entries[count].is_dir = f->is_dir; + entries[count].is_symlink = f->is_symlink; + entries[count].link_target = + f->is_symlink && f->symlink_target ? str_dup(f->symlink_target) : NULL; + count++; + } + chunk_destroy(chunk); + } + bool failed = oom || directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner); + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + if (failed) { + list_entries_destroy(entries, count); + if (oom) + log_message(LOG_LEVEL_ERROR, "memory allocation failed while listing"); + return 1; + } + if (count > 1) + qsort(entries, count, sizeof(ListEntry), compare_list_entries); + for (size_t i = 0; i < count; i++) { + ChangeEvent event; + memset(&event, 0, sizeof(event)); + event.name = entries[i].name; + event.path = entries[i].name; + event.mode = entries[i].mode; + event.size = entries[i].size; + event.mtime_sec = entries[i].mtime; + event.mtime_nsec = entries[i].mtime_nsec; + event.is_directory = entries[i].is_dir; + event.is_symlink = entries[i].is_symlink; + event.symlink_target = entries[i].link_target; + char* line = change_render_list_line(config, &event); + if (line != NULL) { + char* escaped = output_escape(line, config->eight_bit_output); + printf("%s\n", escaped != NULL ? escaped : line); + free(escaped); + free(line); + } + } + list_entries_destroy(entries, count); + return 0; +} + +/* Send the delete manifest to the server. Returns 0 on success, -1 on + failure. It carries FOUR sections: the keep-set paths, the protected + excluded prefixes, the --delete-missing-args exact-delete paths, and the + destination-relative directories the sender synchronized this run. + When --delete-excluded is given `protected` is empty: excluded destination + mirrors are then ordinary extras and are removed. When + --delete-missing-args is active `missing_args` holds the destination mirrors + of missing --files-from entries: each is an explicit receiver-side deletion + request, independent of the extras walk. `synced_dirs` confines the extras + walk to entries directly inside a synchronized directory. A NULL + keep-set / protected / missing / dirs list transmits an empty section. All + four sections are unbounded on the sender; the receiver enforces + MAX_MANIFEST_ENTRIES per section and a single MAX_MANIFEST_BYTES budget + shared across the sections, rejecting (with STATUS_ERROR) an over-budget + frame. A heavily filtered source whose exclusion list is large therefore + fails the run cleanly on the receiver rather than being truncated. */ +int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs) { + if (!send_status(fd, STATUS_MANIFEST)) + return -1; + int keep_count = manifest ? manifest->size : 0; + if (!send_int(fd, keep_count)) + return -1; + for (int i = 0; i < keep_count; i++) { + if (!send_wire_str(fd, (char*)manifest->items[i])) + return -1; + } + /* The receiver has ONE protected-prefix section; filter-excluded prefixes + (dropped under --delete-excluded) and size-pruned prefixes (always + protected) are concatenated into it. */ + int protected_count = + (protected_prefixes ? protected_prefixes->size : 0) + (size_skipped ? size_skipped->size : 0); + if (!send_int(fd, protected_count)) + return -1; + if (protected_prefixes) { + for (int i = 0; i < protected_prefixes->size; i++) { + if (!send_wire_str(fd, (char*)protected_prefixes->items[i])) + return -1; + } + } + if (size_skipped) { + for (int i = 0; i < size_skipped->size; i++) { + if (!send_wire_str(fd, (char*)size_skipped->items[i])) + return -1; + } + } + int missing_count = missing_args ? missing_args->size : 0; + if (!send_int(fd, missing_count)) + return -1; + for (int i = 0; i < missing_count; i++) { + if (!send_wire_str(fd, (char*)missing_args->items[i])) + return -1; + } + int dirs_count = synced_dirs ? synced_dirs->size : 0; + if (!send_int(fd, dirs_count)) + return -1; + for (int i = 0; i < dirs_count; i++) { + if (!send_wire_str(fd, (char*)synced_dirs->items[i])) + return -1; + } + return 0; +} + +/* Transmit the keep-set manifest and wait for the receiver's verdict. Used by + --delete-before/--delete-during, where the extras are removed on the receiver + BEFORE the first byte of file data is sent: the receiver acknowledges with + STATUS_OK once the bounded delete committed, or STATUS_ERROR if it could not + (in which case the sender aborts without streaming any data). The ACK may + take much longer than an ordinary per-message round trip because the receiver + performs the whole bounded deletion walk (up to MAX_SERVER_DELETE_COUNT + unlinks) before replying, so the wait uses a generous explicit deadline + instead of the default 60 s receive window. */ +#define DELETE_ACK_TIMEOUT_SEC 3600 +/* While waiting for the (potentially slow) receiver-side deletion, send a + * STATUS_KEEPALIVE at most this often so the connection is demonstrably alive + * and neither side's per-message timeout trips. */ +#define DELETE_ACK_KEEPALIVE_SEC 10 + +bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, + ArrayList* synced_dirs) { + if (!client || !manifest) + return false; + if (send_delete_manifest(client->file_descriptor, manifest, protected_prefixes, size_skipped, + missing_args, synced_dirs) != 0) + return false; + Status ack; + /* The wait is long (up to an hour) and runs inline on this thread: a helper + * thread would race the non-thread-safe protocol send path, so keepalives are + * emitted from this wait loop itself. A Ctrl-C/SIGTERM abort flag also ends + * the wait; the caller then best-effort sends STATUS_ABORT. */ + if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, + DELETE_ACK_KEEPALIVE_SEC, client_abort_pending)) { + /* A Ctrl-C/SIGTERM abort ends the wait above; tell the receiver before the + caller tears the connection down (best-effort). */ + if (client_abort_pending()) { + log_info_message(LOG_INFO_MISC, + "Abort requested while awaiting delete ack; sending STATUS_ABORT"); + send_status(client->file_descriptor, STATUS_ABORT); + } + return false; + } + if (ack != STATUS_OK) { + log_server_rejection("Server failed to delete files before the transfer"); + return false; + } + return true; +} + +/* Server-contacting --dry-run. Connects to the configured remote/daemon and + * runs the normal per-file incremental decision WITHOUT transmitting any file + * data: the receiver (which also sees dry_run=true on the wire) answers + * STATUS_OK for an up-to-date file and STATUS_DRY_RUN_TRANSFER for a file it + * would otherwise write, mutating nothing on either side. The would-transfer + * set and the same trailer as the local dry-run are printed. A + * --compare-dest exact basis hit with no destination copy is reported as a + * skip by the receiver. + * + * Only regular files take the receiver-consulted check; directory / symlink / + * special / hard-link-sibling entries have no per-file content check, so they + * are reported conservatively as would-transfer and their frames are never + * sent (which is what keeps the receiver mutation-free). --delete* is + * deliberately NOT transmitted in dry-run, so no deletion can occur; the + * would-delete manifest report is a documented follow-up. + * + * Returns 0 on success, 1 on error. */ +int send_dry_run_remote(Config* config) { + int from_skipped = 0; + ArrayList* missing_args = NULL; + if (config->delete_missing_args) { + missing_args = array_list_create(free); + if (!missing_args) + return 1; + } + if (!files_from_list_check(config, missing_args, &from_skipped)) { + if (missing_args) + array_list_delete(missing_args); + return 1; + } + if (missing_args) + array_list_delete(missing_args); + /* A live session may follow, so arm graceful abort handling. */ + client_set_abort_armed(true); + Client* client = connect_transfer_client(config); + if (!client) { + if (config->transport == TRANSPORT_TCP) + log_message(LOG_LEVEL_ERROR, "could not connect to server%s", + config->use_tls ? " via TLS" : ""); + client_set_abort_armed(false); + return 1; + } + ProtocolSession session; + protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_io_timeout(&session, config->timeout); + protocol_session_set_ssl(&session, (SSL*)client->ssl); + protocol_session_bind(&session); + + int ret = 1; + time_t dry_start = time(NULL); + ReceiverStats dry_stats; + memset(&dry_stats, 0, sizeof(dry_stats)); + PreparedScanner prepared; + memset(&prepared, 0, sizeof(prepared)); + DirectoryScanner* scanner = NULL; + ArrayList* dry_manifest = NULL; + ArrayList* dry_dirs = NULL; + ArrayList* dry_excluded = NULL; + ArrayList* dry_size_skipped = NULL; + if (!config_send(client->file_descriptor, config)) + goto dry_fail; + receive_daemon_motd(client, config); + if (!prepare_scanner(config, 0, &prepared)) + goto dry_fail; + /* -n --delete: build the same keep-set manifest, protected prefixes, and + synchronized-directory scope a real run would send, so the receiver's + read-only extras walk enumerates exactly the deletions a real run makes. */ + if (config->use_delete) { + dry_manifest = array_list_create(free); + dry_dirs = array_list_create(free); + dry_size_skipped = array_list_create(free); + if (!dry_manifest || !dry_dirs || !dry_size_skipped) + goto dry_fail; + if (!config->delete_excluded) { + dry_excluded = array_list_create(free); + if (!dry_excluded) + goto dry_fail; + prepared.options.excluded_paths = dry_excluded; + } + prepared.options.size_skipped_paths = dry_size_skipped; + /* A --files-from subset confines the extras walk to the directories the + scan synchronized; a full recursive transfer marks the root itself. */ + if (config->files_from_set == NULL) { + char* root_marker = delete_scope_root_marker(config); + if (!root_marker || !array_list_add(dry_dirs, root_marker)) { + free(root_marker); + goto dry_fail; + } + } else { + prepared.options.synced_dirs = dry_dirs; + } + } + scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); + if (!scanner) + goto dry_fail; + + int file_count = 0; + unsigned long long total_bytes = 0; + char size_buffer[32]; + if (!config->quiet) + printf("Dry run: files to be transferred\n"); + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + if (dry_manifest && !add_chunk_to_manifest(dry_manifest, chunk)) { + chunk_destroy(chunk); + goto dry_fail; + } + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (!f) + continue; + unsigned long long fsize = f->data ? f->data->size : 0; + bool would; + if (f->is_dir || f->is_symlink || f->is_special || + (f->link_group != 0 && !f->link_first && f->hardlink_target != NULL)) { + /* No receiver-side content check exists for these frame types; a real + run would (re)create them, so report would-transfer and send no + frame (the receiver must stay mutation-free). */ + would = true; + } else if (fsize > MAX_RECEIVE_WHOLE_FILE_SIZE && !config->use_incremental && + !config_has_basis(config)) { + /* A non-incremental run streams a >whole-file-limit source without the + STATUS_CHECK handshake, so no read-only receiver decision is possible + (and none is needed: a real run would transfer it). */ + would = true; + } else { + DeltaSignature* sig = NULL; + unsigned long long resume_offset = 0; + int rc = incremental_check(client, f, config, &sig, &resume_offset); + delta_signature_destroy(sig); + if (rc < 0) { + chunk_destroy(chunk); + goto dry_fail; + } + if (rc == 1) + continue; /* up to date; nothing to report */ + if (rc != 4) { + log_message(LOG_LEVEL_ERROR, "Unexpected receiver reply during dry-run"); + chunk_destroy(chunk); + goto dry_fail; + } + would = true; + } + if (would) { + if (!config->quiet) { + char* escaped_path = output_escape(file_wire_path(f), config->eight_bit_output); + if (!escaped_path) { + chunk_destroy(chunk); + goto dry_fail; + } + if (config->human_readable) + printf(" %s (%s)\n", escaped_path, + display_bytes(fsize, true, size_buffer, sizeof(size_buffer))); + else + printf(" %s (%llu bytes)\n", escaped_path, fsize); + free(escaped_path); + } + total_bytes += fsize; + file_count++; + } + } + chunk_destroy(chunk); + } + bool io_error = directory_scanner_had_io_error(scanner); + if (directory_scanner_failed(scanner)) + goto dry_fail; + if (io_error) + log_message(LOG_LEVEL_WARNING, "source scan hit an unreadable directory"); + /* Send the keep-set manifest (no data frames) so the receiver can enumerate + the destination extras; an early-timing delete ACKs before it will accept + the terminal FINISHED. */ + bool early_delete = config->use_delete && config_delete_timing_early(config); + if (dry_manifest) { + if (send_delete_manifest(client->file_descriptor, dry_manifest, dry_excluded, dry_size_skipped, + NULL, dry_dirs) != 0) + goto dry_fail; + if (early_delete) { + Status ack; + if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, + DELETE_ACK_KEEPALIVE_SEC, client_abort_pending) || + ack != STATUS_OK) + goto dry_fail; + } + } + /* Terminate the stream so the receiver emits its success frame; no data frame + is ever sent in dry-run. */ + if (!send_status(client->file_descriptor, STATUS_FINISHED)) + goto dry_fail; + Status status; + if (!receive_status(client->file_descriptor, &status)) + goto dry_fail; + if (status == STATUS_STATS) { + ArrayList* would_delete = array_list_create(free); + if (!would_delete) + goto dry_fail; + if (!receive_stats_record(client->file_descriptor, &dry_stats, would_delete)) { + array_list_delete(would_delete); + goto dry_fail; + } + print_delete_reports(config, would_delete); + array_list_delete(would_delete); + if (!receive_status(client->file_descriptor, &status)) + goto dry_fail; + } + if (status != STATUS_OK) + goto dry_fail; + if (!config->quiet) { + if (config->human_readable) + printf("Total: %d files, %s\n", file_count, + display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); + else + printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); + } + { + TransferStats dry_transfer; + memset(&dry_transfer, 0, sizeof(dry_transfer)); + dry_transfer.flist_reg = (unsigned long long)file_count; + dry_transfer.total_file_size = total_bytes; + dry_transfer.transferred_regular = (unsigned long long)file_count; + dry_transfer.transferred_file_size = total_bytes; + dry_transfer.literal_data = total_bytes; + report_transfer_stats(config, &dry_transfer, dry_start, &dry_stats); + } + ret = io_error ? 1 : 0; + +dry_fail: + if (dry_manifest) + array_list_delete(dry_manifest); + if (dry_dirs) + array_list_delete(dry_dirs); + if (dry_excluded) + array_list_delete(dry_excluded); + if (dry_size_skipped) + array_list_delete(dry_size_skipped); + if (scanner) + directory_scanner_destroy(scanner); + prepared_scanner_destroy(&prepared); + disconnect_transfer_client(client); + protocol_session_unbind(); + client_set_abort_armed(false); + return ret; +} diff --git a/src/client/client_report.c b/src/client/client_report.c new file mode 100644 index 0000000..08050df --- /dev/null +++ b/src/client/client_report.c @@ -0,0 +1,704 @@ +#include "client_send_internal.h" +#include "array_list.h" +#include "change_list.h" +#include "charset.h" +#include "config.h" +#include "file.h" +#include "format.h" +#include "log.h" +#include "protocol.h" +#include "utils.h" +#include +#include +#include +#include +#include + +/* Surface a server rejection to the user. When the last status exchange + carried a STATUS_ERROR_DETAIL reason (protocol 2.21.0) it is appended to the + client-side context; a bare STATUS_ERROR still logs the context alone. */ +void log_server_rejection(const char* context) { + const char* detail = protocol_last_error(); + if (detail && detail[0] != '\0') { + /* The detail is peer-controlled: escape it so terminal/log-format + * metacharacters cannot be injected into the client's output. */ + char* escaped = output_escape(detail, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "%s: %s", context, escaped ? escaped : ""); + free(escaped); + } else { + log_message(LOG_LEVEL_ERROR, "%s", context); + } +} + +const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer, + size_t buffer_size) { + if (human_readable && format_human_size_decimal(bytes, buffer, buffer_size)) + return buffer; + snprintf(buffer, buffer_size, "%.1f MB", (double)bytes / (double)BYTES_PER_MIB); + return buffer; +} + +/* rsync byte count: human-readable decimal when -h was given, otherwise a + * comma-grouped integer (rsync's big_num in the C locale). */ +static const char* stats_bytes(const Config* config, unsigned long long bytes, char* buffer, + size_t buffer_size) { + if (!format_big_num(bytes, config->human_readable, buffer, buffer_size)) + snprintf(buffer, buffer_size, "%llu", bytes); + return buffer; +} + +/* Build rsync's per-type parenthetical: each non-zero category, in + reg/dir/link/special order. Empty when every count is zero. */ +static void type_breakdown(unsigned long long reg, unsigned long long dir, unsigned long long link, + unsigned long long special, char* out, size_t out_size) { + if (reg + dir + link + special == 0) { + out[0] = '\0'; + return; + } + out[0] = '\0'; + size_t used = 0; + const struct { + const char* name; + unsigned long long count; + } parts[4] = {{"reg", reg}, {"dir", dir}, {"link", link}, {"special", special}}; + bool first = true; + for (size_t i = 0; i < 4; i++) { + if (parts[i].count == 0) + continue; + int written = snprintf(out + used, out_size - used, "%s%s: %llu", first ? "(" : ", ", + parts[i].name, parts[i].count); + if (written < 0 || (size_t)written >= out_size - used) + break; + used += (size_t)written; + first = false; + } + if (!first && used + 1 < out_size) + out[used++] = ')'; + out[used] = '\0'; +} + +/* Build rsync's `Number of files` parenthetical from the scan's flist counts. */ +static void stats_type_breakdown(const TransferStats* stats, char* out, size_t out_size) { + type_breakdown(stats->flist_reg, stats->flist_dir, stats->flist_link, stats->flist_special, out, + out_size); +} + +/* rsync's `Number of files` counts every directory. A recursive scan that + preserves a directory attribute captures them in `dir_entries`; a `-r` scan + (no -t/-p) captures nothing, so fall back to the scanner's shared counter of + traversed directories that are not already represented by an inline + directory entry. The -d generator counts its explicit directory entries + inline and does not traverse, so it is excluded here. */ +unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries, + atomic_ullong* counter) { + if (config == NULL || config->dirs || config->list_only) + return 0; + if (dir_metadata_should_capture(config)) + return dir_entries != NULL ? (unsigned long long)dir_entries->size : 0; + return counter != NULL ? (unsigned long long)atomic_load(counter) : 0; +} + +/* Print the rsync `--stats` block on stdout. The source-side flist and + transferred counters come from `stats` (filled while scanning/sending), the + receiver-only counters from the STATUS_STATS frame, and the wire byte totals + from the process-wide protocol counters. The labels, layout and + rate/speedup formulas match rsync 3.4.1. Shared by the single-threaded and + multithreaded send paths. */ +void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start, + const ReceiverStats* recv) { + if (!config->stats || config->quiet) + return; + TransferStats empty = {0}; + if (stats == NULL) + stats = ∅ + ReceiverStats none = {0}; + if (recv == NULL) + recv = &none; + unsigned long long sent = protocol_bytes_written(); + unsigned long long received = protocol_bytes_read(); + /* rsync: bytes_per_sec = (written + read) / (0.5 + (end - start)). */ + double elapsed = difftime(time(NULL), start); + double rate = (double)(sent + received) / (0.5 + elapsed); + char total_buffer[32]; + char transferred_buffer[32]; + char literal_buffer[32]; + char matched_buffer[32]; + char sent_buffer[32]; + char recv_buffer[32]; + char rate_buffer[32] = {0}; + char human_rate[32] = {0}; + const char* total = + stats_bytes(config, stats->total_file_size, total_buffer, sizeof(total_buffer)); + const char* transferred = stats_bytes(config, stats->transferred_file_size, transferred_buffer, + sizeof(transferred_buffer)); + /* Protocol 2.28.0: the receiver reports the bytes it literally stored, which + is exact for a delta transfer (the sender's own literal_data counts each + stored file's whole source size and is only an upper bound). Fall back to + the sender total when the receiver reported no delta/literal accounting + (e.g. a local no-server path). */ + unsigned long long literal_bytes = (recv->literal_bytes != 0 || recv->matched_data != 0) + ? recv->literal_bytes + : stats->literal_data; + const char* literal = stats_bytes(config, literal_bytes, literal_buffer, sizeof(literal_buffer)); + const char* sent_s = stats_bytes(config, sent, sent_buffer, sizeof(sent_buffer)); + const char* recv_s = stats_bytes(config, received, recv_buffer, sizeof(recv_buffer)); + const char* rate_str = rate_buffer; + if (config->human_readable) { + if (!format_human_size_decimal((unsigned long long)rate, human_rate, sizeof(human_rate))) + snprintf(human_rate, sizeof(human_rate), "0"); + rate_str = human_rate; + } else { + snprintf(rate_buffer, sizeof(rate_buffer), "%.2f", rate); + } + double speedup = + (sent + received) > 0 ? (double)stats->total_file_size / (double)(sent + received) : 0.0; + char breakdown[128]; + stats_type_breakdown(stats, breakdown, sizeof(breakdown)); + unsigned long long flist_total = + stats->flist_reg + stats->flist_dir + stats->flist_link + stats->flist_special; + char created_breakdown[128]; + type_breakdown(recv->created_reg, recv->created_dir, recv->created_link, recv->created_special, + created_breakdown, sizeof(created_breakdown)); + unsigned long long created_total = + recv->created_reg + recv->created_dir + recv->created_link + recv->created_special; + printf("\n"); + if (breakdown[0] != '\0') + printf("Number of files: %llu %s\n", flist_total, breakdown); + else + printf("Number of files: %llu\n", flist_total); + /* Protocol 2.28.0: the receiver reports which destination entries it newly + created, split by type, so this line matches rsync exactly. */ + if (created_breakdown[0] != '\0') + printf("Number of created files: %llu %s\n", created_total, created_breakdown); + else + printf("Number of created files: %llu\n", created_total); + printf("Number of deleted files: %llu\n", recv->deleted_files); + printf("Number of regular files transferred: %llu\n", stats->transferred_regular); + printf("Total file size: %s bytes\n", total); + printf("Total transferred file size: %s bytes\n", transferred); + printf("Literal data: %s bytes\n", literal); + const char* matched = + stats_bytes(config, recv->matched_data, matched_buffer, sizeof(matched_buffer)); + printf("Matched data: %s bytes\n", matched); + printf("File list size: 0\n"); + printf("File list generation time: 0.000 seconds\n"); + printf("File list transfer time: 0.000 seconds\n"); + printf("Total bytes sent: %s\n", sent_s); + printf("Total bytes received: %s\n", recv_s); + printf("\n"); + printf("sent %s bytes received %s bytes %s bytes/sec\n", sent_s, recv_s, rate_str); + printf("total size is %s speedup is %.2f%s\n", total, speedup, + config->dry_run ? " (DRY RUN)" : ""); + fflush(stdout); +} + +/* Classify one scanned source entry into the rsync flist counters. Called for + every entry the sender walks, transferred or skipped. Directory entries are + counted here only for the explicit -d/--dirs generator; a recursive scan's + directories are accounted from the scanner's dir_entries list at report time. */ +void transfer_stats_note_entry(TransferStats* stats, const File* file) { + if (stats == NULL || file == NULL) + return; + if (file->is_dir) { + stats->flist_dir++; + return; + } + if (file->is_symlink) { + stats->flist_link++; + stats->total_file_size += file->symlink_target ? strlen(file->symlink_target) : 0; + return; + } + if (file->is_special) { + stats->flist_special++; + return; + } + stats->flist_reg++; + stats->total_file_size += file->data ? file->data->size : 0; +} + +/* Account for a regular file (or a whole-file append) the receiver actually + stored: rsync's transferred-file count and transferred/literal byte totals. + `literal_data` counts the whole source size, which is exact for a whole-file + send but an upper bound for a delta send (the receiver reuses basis blocks + the sender never ships); see TransferStats.literal_data in format.h. */ +void transfer_stats_note_transferred(TransferStats* stats, const File* file) { + if (stats == NULL || file == NULL) + return; + if (file->is_dir || file->is_symlink || file->is_special) + return; + if (file->link_group != 0 && !file->link_first) + return; + unsigned long long size = file->data ? file->data->size : 0; + stats->transferred_regular++; + stats->transferred_file_size += size; + stats->literal_data += size; +} + +/* ---- rsync-style per-file --progress ------------------------------------ + * rsync prints, for each transferred regular file, the file name followed by a + * two-frame progress line: the first at the initial 32 KiB read window (always + * 0.00 kB/s / 0:00:00 on a sub-second transfer) and a final 100% frame carrying + * `(xfr#N, to-chk=X/Y)`. Rates are wall-clock dependent, so only the final + * rate is measured here; the layout matches rsync 3.4.1's progress.c. */ +#define RSYNC_PROGRESS_IO_WINDOW (32ULL * 1024ULL) + +/* Paths-only pre-count of the source file list, built once at transfer start + * when progress output is requested. rsync's `to-chk` denominator is the whole + * file list -- every regular file, directory, symlink and special plus the + * transfer root -- while the streaming scan never emits directories. A + * metadata-only walk (no file reads, no hashing) supplies that total and the + * directory names, so the opt-in pass leaves non-progress runs untouched. */ +typedef struct { + unsigned long long total; + ArrayList* dir_paths; /* owned char* in transfer-relative display form */ +} ProgressPrecount; + +static bool g_progress_active; +static unsigned long long g_progress_xferred; +static unsigned long long g_progress_index; +static unsigned long long g_progress_total; +static struct timespec g_progress_file_start; +static ProgressPrecount g_progress_precount; +static PathIndex g_progress_dir_index; +static bool g_progress_dir_index_valid; +static StrHashSet g_progress_emitted; +static bool g_progress_emitted_valid; +static ArrayList* g_progress_emitted_keys; + +bool progress_requested(const Config* config) { + return config != NULL && !config->quiet && + (config->show_progress || (config->info_level & LOG_INFO_PROGRESS) != 0); +} + +static void progress_precount_dispose(ProgressPrecount* p) { + if (p->dir_paths != NULL) { + array_list_delete(p->dir_paths); + p->dir_paths = NULL; + } + p->total = 0; +} + +void client_progress_cleanup(void) { + if (g_progress_dir_index_valid) { + path_index_free(&g_progress_dir_index); + g_progress_dir_index_valid = false; + } + if (g_progress_emitted_valid) { + str_hash_set_free(&g_progress_emitted); + g_progress_emitted_valid = false; + } + if (g_progress_emitted_keys != NULL) { + array_list_delete(g_progress_emitted_keys); + g_progress_emitted_keys = NULL; + } + progress_precount_dispose(&g_progress_precount); + g_progress_active = false; + g_progress_total = 0; + g_progress_index = 0; + g_progress_xferred = 0; +} + +static void progress_first_frame(unsigned long long size, char* out, size_t out_size) { + char ofs_buf[32]; + unsigned long long ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; + if (!format_big_num(ofs, false, ofs_buf, sizeof(ofs_buf))) + snprintf(ofs_buf, sizeof(ofs_buf), "%llu", ofs); + int pct = size == 0 ? 100 : (ofs == size ? 100 : (int)(100.0 * (double)ofs / (double)size)); + snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s%s", ofs_buf, pct, 0.0, "kB/s", " 0:00:00", + " "); +} + +static void progress_final_frame(unsigned long long size, char* out, size_t out_size) { + char ofs_buf[32]; + char rembuf[32]; + unsigned long long last_ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; + if (!format_big_num(size, false, ofs_buf, sizeof(ofs_buf))) + snprintf(ofs_buf, sizeof(ofs_buf), "%llu", size); + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long diff_ms = (long long)(now.tv_sec - g_progress_file_start.tv_sec) * 1000 + + (now.tv_nsec - g_progress_file_start.tv_nsec) / 1000000; + if (diff_ms <= 0) + diff_ms = 1; + double rate = + size > last_ofs ? (double)(size - last_ofs) * 1000.0 / (double)diff_ms / 1024.0 : 0.0; + const char* units = "kB/s"; + if (rate > 1024.0 * 1024.0) { + rate /= 1024.0 * 1024.0; + units = "GB/s"; + } else if (rate > 1024.0) { + rate /= 1024.0; + units = "MB/s"; + } + unsigned long long remain = (unsigned long long)(diff_ms / 1000); + snprintf(rembuf, sizeof(rembuf), "%4u:%02u:%02u", (unsigned)(remain / 3600), + (unsigned)((remain / 60) % 60), (unsigned)(remain % 60)); + /* rsync's `to-chk` denominator is the whole file list (the pre-count); the + numerator falls as each entry is processed, root first. Without a + pre-count (the paths-only walk failed) fall back to the transferred-file + count so the single-file layout stays intact. */ + unsigned long long total = g_progress_total > 0 ? g_progress_total : g_progress_xferred + 1; + unsigned long long to_chk = total > g_progress_index ? total - g_progress_index - 1 : 0; + snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s (xfr#%llu, to-chk=%llu/%llu)\n", ofs_buf, 100, + rate, units, rembuf, g_progress_xferred, to_chk, total); +} + +bool info_flag_enabled(const Config* config, LogInfoFlag flag) { + return config != NULL && (config->info_level & flag) != 0; +} + +/* Print rsync's deletion lines for a received list of destination-relative + * paths: `*deleting PATH` when itemizing, the --out-format expansion when a + * format is set, else `deleting PATH` for --info=del. Used by both the dry-run + * would-delete report and the real --info=del report. */ +void print_delete_reports(const Config* config, const ArrayList* paths) { + if (!config || !paths || config->quiet) + return; + /* --debug=del is independent of the --info=del/itemize/out-format display: + emit the debug trace even when no deletion line would be printed. */ + if (log_debug_enabled(LOG_DEBUG_DEL)) { + for (int i = 0; i < paths->size; i++) { + const char* raw = (const char*)paths->items[i]; + const char* path = delete_display_path(config, raw); + log_debug_message(LOG_DEBUG_DEL, "del: %s", path ? path : raw); + } + } + if (!(config->itemize_changes || config->out_format != NULL || + info_flag_enabled(config, LOG_INFO_DEL))) + return; + for (int i = 0; i < paths->size; i++) { + const char* raw = (const char*)paths->items[i]; + const char* path = delete_display_path(config, raw); + if (config->out_format != NULL) { + ChangeEvent event; + memset(&event, 0, sizeof(event)); + event.decision = CHANGE_SENT; + event.deleted = true; + event.name = path; + event.path = path; + char* line = change_render_format(config->out_format, config, &event); + if (line) { + char* escaped = output_escape(line, config->eight_bit_output); + printf("%s\n", escaped ? escaped : line); + free(escaped); + free(line); + } + } else { + char* escaped = output_escape(path, config->eight_bit_output); + if (config->itemize_changes) + printf("*deleting %s\n", escaped ? escaped : path); + else + printf("deleting %s\n", escaped ? escaped : path); + free(escaped); + } + } + fflush(stdout); +} + +static void client_progress_emit_ancestors(const Config* config, const char* rel) { + if (!g_progress_dir_index_valid || !g_progress_emitted_valid || g_progress_emitted_keys == NULL || + rel == NULL) + return; + size_t rel_len = strlen(rel); + for (size_t i = 0; i < rel_len; i++) { + if (rel[i] != '/') + continue; + char* prefix = malloc(i + 1); + if (prefix == NULL) + return; + memcpy(prefix, rel, i); + prefix[i] = '\0'; + if (path_index_contains(&g_progress_dir_index, prefix) && + !str_hash_set_lookup(&g_progress_emitted, prefix)) { + char* key = str_dup(prefix); + if (key != NULL && array_list_add(g_progress_emitted_keys, key)) { + str_hash_set_insert_ref(&g_progress_emitted, key); + char* escaped = output_escape(prefix, config->eight_bit_output); + printf("%s/\n", escaped ? escaped : prefix); + free(escaped); + g_progress_index++; + } else { + free(key); + } + } + free(prefix); + } +} + +/* rsync's --info=name/progress line for one entry: transfer-relative name (a + * trailing slash for directories) plus the ` -> target` symlink suffix. */ +static char* progress_entry_line(const File* file, const char* rel) { + const char* arrow = NULL; + const char* target = NULL; + if (file->is_symlink && file->symlink_target != NULL) { + arrow = " -> "; + target = file->symlink_target; + } else if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { + arrow = " => "; + target = file->hardlink_target; + } + size_t rel_len = strlen(rel); + bool dir_slash = file->is_dir && (rel_len == 0 || rel[rel_len - 1] != '/'); + size_t extra = (dir_slash ? 1u : 0u) + (target != NULL ? 4u + strlen(target) : 0u); + char* line = malloc(rel_len + extra + 1); + if (line == NULL) + return NULL; + memcpy(line, rel, rel_len); + size_t off = rel_len; + if (dir_slash) + line[off++] = '/'; + if (target != NULL) { + memcpy(line + off, arrow, 4); + off += 4; + memcpy(line + off, target, strlen(target)); + off += strlen(target); + } + line[off] = '\0'; + return line; +} + +void client_progress_begin(const Config* config) { + change_reset_name_root(); + g_progress_active = progress_requested(config); + g_progress_xferred = 0; + g_progress_index = 1; /* the transfer root is file-list entry #0 */ + if (!g_progress_active) { + /* `--info=flist` prints rsync's file-list header even without progress. */ + if (!config->quiet && info_flag_enabled(config, LOG_INFO_FLIST)) { + printf("sending incremental file list\n"); + fflush(stdout); + } + return; + } + printf("sending incremental file list\n"); + /* rsync prints the transfer-root directory's name before the first file when + that directory is created; FastSync mirrors the source root below the + receive root and creates it on a fresh destination, so emit it here. */ + printf("./\n"); + fflush(stdout); +} + +/* Emit the name (unless itemize/out-format already did) and the two progress + * frames for one transferred regular file. */ +void client_progress_file(const Config* config, const File* file) { + if (!g_progress_active || file == NULL || !file->data) + return; + g_progress_xferred++; + unsigned long long size = file->data->size; + if (!config->itemize_changes && config->out_format == NULL) { + const char* rel = delete_display_path(config, file_wire_path(file)); + client_progress_emit_ancestors(config, rel); + char* escaped = output_escape(rel, config->eight_bit_output); + printf("%s\n", escaped ? escaped : (rel ? rel : "")); + free(escaped); + } + clock_gettime(CLOCK_MONOTONIC, &g_progress_file_start); + char frame[160]; + progress_first_frame(size, frame, sizeof(frame)); + fputs(frame, stdout); + progress_final_frame(size, frame, sizeof(frame)); + fputs(frame, stdout); + g_progress_index++; + fflush(stdout); +} + +/* Emit the name line for a transferred non-regular entry (directory, symlink, + * special or hard-link sibling): rsync prints these in the file list but has no + * progress frame for them. */ +void client_progress_name(const Config* config, const File* file) { + if (!g_progress_active || file == NULL) + return; + const char* rel = delete_display_path(config, file_wire_path(file)); + if (!config->itemize_changes && config->out_format == NULL) { + client_progress_emit_ancestors(config, rel); + char* line = progress_entry_line(file, rel ? rel : ""); + if (line != NULL) { + char* escaped = output_escape(line, config->eight_bit_output); + printf("%s\n", escaped ? escaped : line); + free(escaped); + free(line); + fflush(stdout); + } + } + g_progress_index++; +} + +/* An entry the receiver already had prints no name under --progress but still + * occupies a file-list slot in the `to-chk` numerator. */ +void client_progress_uptodate(const Config* config, const File* file) { + (void)config; + (void)file; + if (!g_progress_active) + return; + g_progress_index++; +} + +static bool progress_precount_add_dir(ProgressPrecount* p, const char* path) { + if (path == NULL || path[0] == '\0') + return true; + char* dup = str_dup(path); + if (dup == NULL) + return false; + if (array_list_add(p->dir_paths, dup)) + return true; + free(dup); + return false; +} + +/* Metadata-only walk collecting the full file-list total and every directory + * name. It uses its own scanner (fresh filter compilation and hard-link table) + * so the data pass's link-group state is never perturbed. */ +static bool progress_precount_scan(const Config* config, ProgressPrecount* out) { + out->dir_paths = array_list_create(free); + if (out->dir_paths == NULL) + return false; + out->total = 0; + PreparedScanner prepared; + memset(&prepared, 0, sizeof(prepared)); + if (!prepare_scanner(config, 0, &prepared)) { + progress_precount_dispose(out); + return false; + } + ScannerOptions local = prepared.options; + local.list_dirs = true; + local.note_nonreg = false; + local.note_mount = false; + local.dir_count = NULL; + local.use_metadata = false; + local.preserve_xattrs = false; + local.preserve_acls = false; + local.checksum = false; + local.capture_dir_times = false; + local.excluded_paths = NULL; + local.size_skipped_paths = NULL; + local.synced_dirs = NULL; + local.plan_dirs = NULL; + local.dir_entries = NULL; + local.dir_entries_mutex = NULL; + local.hardlinks = NULL; + DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); + bool ok = scanner != NULL; + if (scanner != NULL) { + Chunk* chunk; + while (ok && (chunk = directory_scanner_next(scanner)) != NULL) { + out->total += (unsigned long long)chunk->element_count; + for (int i = 0; i < chunk->element_count && ok; i++) { + const File* f = chunk->items[i]; + if (f != NULL && f->is_dir) + ok = progress_precount_add_dir(out, delete_display_path(config, file_wire_path(f))); + } + chunk_destroy(chunk); + } + if (ok && directory_scanner_failed(scanner)) + ok = false; + directory_scanner_destroy(scanner); + } + prepared_scanner_destroy(&prepared); + if (!ok) { + progress_precount_dispose(out); + return false; + } + out->total += 1; /* the transfer root "." */ + return true; +} + +/* Reuse the --delete-during/--delete-delay keep-set pre-scan: its traversed + * directory list already holds every directory and `non_dir_count` the entries + * counted during that same pass, so progress costs no second walk. */ +static bool progress_precount_from_plan_dirs(const Config* config, const ArrayList* plan_dirs, + unsigned long long non_dir_count, + ProgressPrecount* out) { + out->dir_paths = array_list_create(free); + if (out->dir_paths == NULL) + return false; + out->total = non_dir_count + 1; + for (int i = 0; i < plan_dirs->size; i++) { + const char* path = (const char*)plan_dirs->items[i]; + const char* rel = config->send_directory != NULL + ? utils_strip_transfer_root(path, config->send_directory) + : path; + if (!progress_precount_add_dir(out, rel)) { + progress_precount_dispose(out); + return false; + } + } + out->total += (unsigned long long)out->dir_paths->size; + return true; +} + +/* Build the optional progress pre-count. A failed pre-count is non-fatal: the + * transfer proceeds and the progress denominator falls back to the transferred + * file count. */ +void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, + unsigned long long plan_non_dir_count) { + client_progress_cleanup(); + g_progress_active = progress_requested(config); + if (!g_progress_active) + return; + bool ok = plan_dirs != NULL ? progress_precount_from_plan_dirs( + config, plan_dirs, plan_non_dir_count, &g_progress_precount) + : progress_precount_scan(config, &g_progress_precount); + if (!ok) { + g_progress_total = 0; + return; + } + g_progress_total = g_progress_precount.total; + if (g_progress_precount.dir_paths != NULL && g_progress_precount.dir_paths->size > 0 && + path_index_build(&g_progress_dir_index, + (const char* const*)g_progress_precount.dir_paths->items, + (size_t)g_progress_precount.dir_paths->size)) + g_progress_dir_index_valid = true; + if (str_hash_set_init(&g_progress_emitted, (size_t)(g_progress_precount.dir_paths != NULL + ? g_progress_precount.dir_paths->size + 1 + : 1))) + g_progress_emitted_valid = true; + g_progress_emitted_keys = array_list_create(free); +} + +/* Read the optional STATUS_STATS record (protocol 2.25.0) that the receiver + * sends just before its terminal status when report_stats was negotiated. + * Consumes the would-delete path list into `would_delete` (optional). */ +bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete) { + if (!format_stats_receive(fd, stats)) + return false; + int count = 0; + if (!receive_int(fd, &count) || count < 0 || count > MAX_MANIFEST_ENTRIES) + return false; + /* Mirror the delete-plan parser: every retained path must be a valid + destination-relative path, and the whole list shares one MAX_MANIFEST_BYTES + budget so a hostile peer cannot make the client retain unbounded memory. */ + size_t bytes = 0; + for (int i = 0; i < count; i++) { + char* path = receive_wire_str(fd); + if (!path) + return false; + if (path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) { + free(path); + return false; + } + if (would_delete) { + size_t entry_size = strlen(path) + sizeof(char*) + 16; + if (entry_size > MAX_MANIFEST_BYTES - bytes) { + free(path); + return false; + } + bytes += entry_size; + if (!array_list_add(would_delete, path)) { + free(path); + return false; + } + } else { + free(path); + } + } + return true; +} + +/* Strip the transfer-root prefix from a receiver-reported destination-relative + * delete path so a `*deleting` line matches rsync's transfer-relative name + * (FastSync's destination mirror includes the source's absolute path). */ +const char* delete_display_path(const Config* config, const char* path) { + if (!config || !path || !config->send_directory) + return path; + return utils_strip_transfer_root(path, config->send_directory); +} diff --git a/src/client/client_scan.c b/src/client/client_scan.c new file mode 100644 index 0000000..0a01530 --- /dev/null +++ b/src/client/client_scan.c @@ -0,0 +1,454 @@ +#include "client_send_internal.h" +#include "array_list.h" +#include "charset.h" +#include "config.h" +#include "delete_plan.h" +#include "file.h" +#include "file_list.h" +#include "filter.h" +#include "hardlink.h" +#include "log.h" +#include "scanner.h" +#include "utils.h" +#include +#include +#include +#include + +/* Build the scanner options for one scan. Returns false and logs on failure. */ +bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out) { + if (!out) + return false; + out->base_filters = NULL; + out->hardlinks = NULL; + out->relative_prefix = NULL; + memset(&out->options, 0, sizeof(out->options)); + + int rule_count = config->filters ? config->filters->size : 0; + const char** texts = NULL; + if (rule_count > 0) { + texts = malloc((size_t)rule_count * sizeof(char*)); + if (!texts) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed for filter rules"); + return false; + } + for (int i = 0; i < rule_count; i++) + texts[i] = (const char*)config->filters->items[i]; + } + if (rule_count > 0 || config->cvs_exclude) { + char err[160]; + out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, + config->delete_excluded, err, sizeof(err)); + free(texts); + if (!out->base_filters) { + log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); + return false; + } + } else { + free(texts); + } + + ScannerOptions* options = &out->options; + options->use_metadata = config->use_metadata; + options->preserve_atimes = config->preserve_atimes; + options->preserve_crtimes = config->preserve_crtimes; + options->preserve_xattrs = config->preserve_xattrs; + options->preserve_acls = config->preserve_acls; + options->chunk_size = config->chunk_size; + /* --exclude/--include are compiled, in command-line order, into the SAME + * ordered filter rule list as --filter/-f (see config_add_selection_rule), so + * the legacy per-kind arrays are deliberately NOT passed to the scanner: + * doing so would re-apply them with the old "excludes first, then includes as + * a mandatory whitelist" precedence and defeat rsync's first-match-wins + * ordering. The arrays remain populated purely for the Config API surface. */ + options->exclude_patterns = NULL; + options->exclude_count = 0; + options->include_patterns = NULL; + options->include_count = 0; + options->max_size = config->max_size; + options->min_size = config->min_size; + options->max_depth = config->max_depth; + options->num_threads = num_threads; + options->follow_symlinks = config->follow_symlinks; + options->copy_links = config->copy_links; + options->safe_links = config->safe_links; + options->copy_unsafe_links = config->copy_unsafe_links; + options->copy_dirlinks = config->copy_dirlinks; + options->munge_links = config->munge_links; + options->checksum = config->checksum; + options->one_file_system = config->one_file_system; + options->preserve_devices = config->preserve_devices; + options->preserve_specials = config->preserve_specials; + options->copy_devices = config->copy_devices; + options->file_list = (const FileListSet*)config->files_from_set; + options->base_filters = out->base_filters; + options->per_dir_filters = config->per_dir_filter; + options->delete_excluded = config->delete_excluded; + options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2; + options->dirs = config->dirs; + options->relative = config->relative; + /* A real recursive transfer recreates empty source directories (rsync + parity); low-level scanner users leave this off. */ + options->emit_empty_dirs = true; + /* --no-implied-dirs only has meaning with -R (rsync): without it the option + is a documented no-op, so the scanner must not suppress directory + metadata. */ + options->no_implied_dirs = config->no_implied_dirs && config->relative; + /* -R/--relative outside --files-from reconstructs every destination path from + * the source spec (rsync's '/./' cut point). With --files-from the listed + * entry already supplies the bare relative path, so no prefix is built. */ + if (config->relative && config->files_from_set == NULL && config->send_directory) { + out->relative_prefix = scanner_relative_prefix(config->send_directory); + if (!out->relative_prefix) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed building --relative path prefix"); + filter_rule_list_free(out->base_filters); + out->base_filters = NULL; + return false; + } + options->relative_prefix = out->relative_prefix; + } + options->prune_empty_dirs = config->prune_empty_dirs; + options->ignore_io_errors = config->ignore_errors; + options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args; + options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet; + options->note_mount = (config->info_level & LOG_INFO_MOUNT) != 0 && !config->quiet; + options->send_directory = config->send_directory; + options->eight_bit_output = config->eight_bit_output; + options->excluded_paths = NULL; + options->excluded_mutex = NULL; + options->size_skipped_paths = NULL; + options->synced_dirs = NULL; + options->hardlinks = NULL; + /* Set by the real send paths; NULL for the metadata-only scans (progress + pre-count, batch) that must not perturb the sender's --stats counter. */ + options->dir_count = NULL; + /* P7 Wave D: capture source directory metadata when a directory attribute is + requested (-p for modes, -t for times unless -O omits them). Whether they + are APPLIED is decided receiver-side. */ + options->capture_dir_times = dir_metadata_should_capture(config); + options->dir_entries = NULL; + options->dir_entries_mutex = NULL; + if (config->preserve_hard_links) { + out->hardlinks = hardlink_table_create(); + if (!out->hardlinks) { + filter_rule_list_free(out->base_filters); + out->base_filters = NULL; + return false; + } + options->hardlinks = out->hardlinks; + } + return true; +} + +void prepared_scanner_destroy(PreparedScanner* prepared) { + if (!prepared) + return; + filter_rule_list_free(prepared->base_filters); + prepared->base_filters = NULL; + hardlink_table_destroy(prepared->hardlinks); + prepared->hardlinks = NULL; + free(prepared->relative_prefix); + prepared->relative_prefix = NULL; +} + +/* -R/--relative implied directories: rsync transmits the metadata of the + * parent directories implied by the source path (every prefix component above + * the source root) so the receiver applies their attributes to the created + * parents. FastSync's scan only covers the source root and below, so append + * one metadata-only directory entry per implied ancestor. --no-implied-dirs + * suppresses this exactly like rsync. A missing ancestor is never fatal. */ +bool append_implied_dir_times(const Config* config, ArrayList* dir_entries) { + if (!dir_entries || !config->relative || config->files_from_set != NULL || + config->no_implied_dirs || !config->send_directory) + return true; + char* prefix = scanner_relative_prefix(config->send_directory); + if (!prefix) + return true; + int ncomp = 0; + for (const char* s = prefix; *s;) { + while (*s == '/') + s++; + if (!*s) + break; + while (*s && *s != '/') + s++; + ncomp++; + } + if (ncomp <= 1) { + free(prefix); + return true; + } + char* fs = str_dup(config->send_directory); + if (!fs) { + free(prefix); + return true; + } + size_t flen = strlen(fs); + while (flen > 1 && fs[flen - 1] == '/') + fs[--flen] = '\0'; + bool ok = true; + /* Walk the source path upwards one component at a time (fs is truncated in + place, so each step targets the next implied ancestor). */ + for (int depth = ncomp - 2; depth >= 0 && ok; depth--) { + char* slash = strrchr(fs, '/'); + if (!slash || slash == fs) + break; + *slash = '\0'; + char* p = prefix; + int c = 0; + while (c <= depth) { + while (*p == '/') + p++; + while (*p && *p != '/') + p++; + c++; + } + char saved = *p; + *p = '\0'; + struct stat st; + if (stat(fs, &st) == 0 && S_ISDIR(st.st_mode)) { + File* file = file_create(fs); + if (!file) { + ok = false; + } else { + file->is_dir = true; + file->metadata = + file_metadata_create(fs, &st, config->preserve_atimes, config->preserve_crtimes); + file->send_path = str_dup(prefix); + if (!file->metadata || !file->send_path || !array_list_add(dir_entries, file)) { + file_destroy(file); + ok = false; + } + } + } + *p = saved; + } + free(fs); + free(prefix); + return ok; +} + +/* The delete-walk root scope for a full (non---files-from) transfer: rsync + * confines --delete to the directories it actually transferred. A plain + * recursive run mirrors the source under the receive root, so "." (the whole + * tree) is correct; an -R run transfers only the reconstructed prefix subtree, + * so the walk is scoped to that prefix instead. Returns a malloc'd wire path + * (or "."), or NULL on allocation failure. */ +char* delete_scope_root_marker(const Config* config) { + if (config->relative && config->files_from_set == NULL && config->send_directory) { + char* prefix = scanner_relative_prefix(config->send_directory); + if (!prefix) + return NULL; + if (prefix[0] != '\0') + return prefix; + free(prefix); + } + return str_dup("."); +} + +/* The -R destination prefix that confines a per-directory delete walk, or NULL + * when the whole receive root is in scope. The marker was installed into + * `synced_dirs` by delete_scope_root_marker(); for a plain recursive transfer + * it is "." (whole root) and for --files-from the list is not a single prefix. */ +const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs) { + if (!config || config->files_from_set != NULL || !config->relative || !config->send_directory) + return NULL; + if (!synced_dirs || synced_dirs->size != 1) + return NULL; + const char* marker = (const char*)synced_dirs->items[0]; + if (marker[0] == '\0' || strcmp(marker, ".") == 0) + return NULL; + return marker; +} + +/* The destination-relative mirror path for a missing --files-from entry: where + a PRESENT entry with the same name would have been written. With -R that is + the entry's bare relative path (the bare wire path the receiver uses); + otherwise it is the full source mirror below the destination root + (`send_directory` joined to the entry, leading '/' stripped), exactly the + path the manifest records for a present sibling. Returns an owned string, or + NULL on allocation failure. */ +static char* files_from_missing_dest_path(const Config* config, const char* entry) { + if (config->relative) + return str_dup(entry); + char* joined = path_cat(config->send_directory, entry); + if (!joined) + return NULL; + const char* rel = *joined == '/' ? joined + 1 : joined; + char* dup = str_dup(rel); + free(joined); + return dup; +} + +/* --files-from semantics: every listed entry must resolve under the source + * root, otherwise rsync reports a hard error instead of silently transferring + * nothing. An entry of "." (the whole tree) and listed-but-empty directories + * are valid. An empty list is valid too: rsync transfers nothing and exits 0. + * With --ignore-missing-args + * (implied by --delete-missing-args) a listed-but-missing entry is instead + * skipped: nothing is transferred for it, it never enters the keep-set and the + * run succeeds for the rest (an all-missing non-empty list succeeds + * transferring nothing, matching rsync). With --delete-missing-args + * `missing_dest` (when non-NULL) collects the entry's destination-relative + * mirror for the receiver's exact-deletion request. Runs before any + * transfer so the failure/skip is surfaced uniformly in the single-threaded, + * -m, dry-run and --list-only paths. */ +bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out) { + *skipped_out = 0; + const FileListSet* set = (const FileListSet*)config->files_from_set; + if (!set) + return true; + if (!config->send_directory) { + log_message(LOG_LEVEL_ERROR, "--files-from requires a source directory"); + return false; + } + if (set->count == 0) { + /* rsync treats an empty --files-from list as "nothing to transfer" and + exits 0 (the source directory is still a valid source arg), so this is + not an error. Nothing passes the (empty) allow-set, so no file is sent + and no keep-set entry is produced. */ + return true; + } + bool ignore = config->ignore_missing_args || config->delete_missing_args; + for (int i = 0; i < set->count; i++) { + const char* entry = set->entries[i]; + if (entry[0] == '\0') + continue; /* "." == list the whole tree */ + char* full = path_cat(config->send_directory, entry); + if (!full) { + log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); + return false; + } + struct stat st; + if (lstat(full, &st) != 0) { + free(full); + if (ignore) { + (*skipped_out)++; + char* escaped_entry = output_escape(entry, log_get_8_bit_output()); + log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'", + escaped_entry ? escaped_entry : ""); + free(escaped_entry); + if (config->delete_missing_args && missing_dest) { + char* mirror = files_from_missing_dest_path(config, entry); + if (!mirror || !array_list_add(missing_dest, mirror)) { + free(mirror); + log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); + return false; + } + } + continue; + } + char* escaped_entry = output_escape(entry, log_get_8_bit_output()); + char* escaped_src = output_escape(config->send_directory, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "--files-from entry '%s' not found in source '%s'", + escaped_entry ? escaped_entry : "", + escaped_src ? escaped_src : ""); + free(escaped_entry); + free(escaped_src); + return false; + } + free(full); + } + if (*skipped_out > 0) { + if (config->delete_missing_args) { + /* --list-only never deletes and a --dry-run only shows intent, so the + summary must not claim a real deletion happened in those modes. */ + if (config->list_only) + log_message(LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s skipped (--list-only " + "never deletes)", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + else if (config->dry_run) + log_message(LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s would be deleted from " + "the destination (dry run)", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + else + log_message( + LOG_LEVEL_WARNING, + "--delete-missing-args: %d missing --files-from entr%s will be deleted from the " + "destination", + *skipped_out, *skipped_out == 1 ? "y" : "ies"); + } else if (config->ignore_missing_args) + log_message(LOG_LEVEL_WARNING, + "--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out, + *skipped_out == 1 ? "y" : "ies"); + } + return true; +} + +/* Walk the whole source tree once collecting only destination-relative wire + paths, loading and sending nothing. --delete-before/--delete-during need the + complete keep-set manifest before the first data byte, so it is built by a + dedicated pre-scan pass and transmitted early; the data pass then re-scans + with a fresh scanner. A source I/O error is fatal unless the options carry + --ignore-errors, in which case the scan continues past the unreadable + directory and *io_error_out reports it (the caller still performs the + deletion but reports the run as errored). */ +bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest, + DeletePlanSender* plans, bool* io_error_out, + unsigned long long* non_dir_count_out) { + if (io_error_out) + *io_error_out = false; + if (non_dir_count_out) + *non_dir_count_out = 0; + ScannerOptions local = *options; + /* The pre-scan is a paths-only pass with no client output; it must not emit + --info=nonreg lines (the data pass does that once). */ + local.note_nonreg = false; + DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); + if (!scanner) + return false; + bool ok = true; + Chunk* chunk; + while ((chunk = directory_scanner_next(scanner)) != NULL) { + if (non_dir_count_out) { + for (int i = 0; i < chunk->element_count; i++) { + const File* f = chunk->items[i]; + if (f && !f->is_dir) + (*non_dir_count_out)++; + } + } + if (manifest && !add_chunk_to_manifest(manifest, chunk)) { + ok = false; + chunk_destroy(chunk); + break; + } + if (plans) { + for (int i = 0; i < chunk->element_count; i++) { + File* f = chunk->items[i]; + if (!f) + continue; + const char* path = file_wire_path(f); + if (!delete_plan_sender_add(plans, path, f->is_dir)) { + ok = false; + break; + } + } + if (!ok) { + chunk_destroy(chunk); + break; + } + } + chunk_destroy(chunk); + } + if (ok) { + /* Keep every traversed source directory, including empty ones, so a plan + no longer removes the destination directory itself. Their own plans are + emitted after the data stream (no file frame triggers them). */ + if (plans && options->plan_dirs) { + for (int i = 0; i < options->plan_dirs->size; i++) { + if (!delete_plan_sender_add(plans, (const char*)options->plan_dirs->items[i], true)) { + ok = false; + break; + } + } + } + } + if (ok && directory_scanner_failed(scanner)) + ok = false; + if (io_error_out) + *io_error_out = directory_scanner_had_io_error(scanner); + directory_scanner_destroy(scanner); + return ok; +} diff --git a/src/client/client_send.c b/src/client/client_send.c index e5c71f4..3b77503 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1,4 +1,5 @@ #include "client_send.h" +#include "client_send_internal.h" #include "array_list.h" #include "batch.h" #include "change_list.h" @@ -44,26 +45,6 @@ receiver's RECEIVER_QUEUE_MAX_BYTES). */ #define SENDER_QUEUE_MAX_BYTES (MAX_CONNECTION_MEMORY - 2 * MAX_CHUNK_SIZE) -/* One mebibyte in bytes; the unit used by the --stats/--progress lines. - Always cast to double when dividing so the output stays fractional. */ -#define BYTES_PER_MIB (1024ULL * 1024ULL) - -/* Surface a server rejection to the user. When the last status exchange - carried a STATUS_ERROR_DETAIL reason (protocol 2.21.0) it is appended to the - client-side context; a bare STATUS_ERROR still logs the context alone. */ -static void log_server_rejection(const char* context) { - const char* detail = protocol_last_error(); - if (detail && detail[0] != '\0') { - /* The detail is peer-controlled: escape it so terminal/log-format - * metacharacters cannot be injected into the client's output. */ - char* escaped = output_escape(detail, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "%s: %s", context, escaped ? escaped : ""); - free(escaped); - } else { - log_message(LOG_LEVEL_ERROR, "%s", context); - } -} - /* rsync's --ignore-errors semantics: an I/O error during the transfer normally * suppresses deletion entirely ("IO error encountered -- skipping file * deletion"); --ignore-errors lets the deletion run anyway. FastSync always @@ -74,1007 +55,6 @@ bool ignore_errors_allows_delete(const Config* config, bool had_io_error) { return !had_io_error || (config && config->ignore_errors); } -static const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer, - size_t buffer_size) { - if (human_readable && format_human_size_decimal(bytes, buffer, buffer_size)) - return buffer; - snprintf(buffer, buffer_size, "%.1f MB", (double)bytes / (double)BYTES_PER_MIB); - return buffer; -} - -/* rsync byte count: human-readable decimal when -h was given, otherwise a - * comma-grouped integer (rsync's big_num in the C locale). */ -static const char* stats_bytes(const Config* config, unsigned long long bytes, char* buffer, - size_t buffer_size) { - if (!format_big_num(bytes, config->human_readable, buffer, buffer_size)) - snprintf(buffer, buffer_size, "%llu", bytes); - return buffer; -} - -/* Build rsync's per-type parenthetical: each non-zero category, in - reg/dir/link/special order. Empty when every count is zero. */ -static void type_breakdown(unsigned long long reg, unsigned long long dir, unsigned long long link, - unsigned long long special, char* out, size_t out_size) { - if (reg + dir + link + special == 0) { - out[0] = '\0'; - return; - } - out[0] = '\0'; - size_t used = 0; - const struct { - const char* name; - unsigned long long count; - } parts[4] = {{"reg", reg}, {"dir", dir}, {"link", link}, {"special", special}}; - bool first = true; - for (size_t i = 0; i < 4; i++) { - if (parts[i].count == 0) - continue; - int written = snprintf(out + used, out_size - used, "%s%s: %llu", first ? "(" : ", ", - parts[i].name, parts[i].count); - if (written < 0 || (size_t)written >= out_size - used) - break; - used += (size_t)written; - first = false; - } - if (!first && used + 1 < out_size) - out[used++] = ')'; - out[used] = '\0'; -} - -/* Build rsync's `Number of files` parenthetical from the scan's flist counts. */ -static void stats_type_breakdown(const TransferStats* stats, char* out, size_t out_size) { - type_breakdown(stats->flist_reg, stats->flist_dir, stats->flist_link, stats->flist_special, out, - out_size); -} - -/* rsync's `Number of files` counts every directory. A recursive scan that - preserves a directory attribute captures them in `dir_entries`; a `-r` scan - (no -t/-p) captures nothing, so fall back to the scanner's shared counter of - traversed directories that are not already represented by an inline - directory entry. The -d generator counts its explicit directory entries - inline and does not traverse, so it is excluded here. */ -static unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries, - atomic_ullong* counter) { - if (config == NULL || config->dirs || config->list_only) - return 0; - if (dir_metadata_should_capture(config)) - return dir_entries != NULL ? (unsigned long long)dir_entries->size : 0; - return counter != NULL ? (unsigned long long)atomic_load(counter) : 0; -} - -/* Print the rsync `--stats` block on stdout. The source-side flist and - transferred counters come from `stats` (filled while scanning/sending), the - receiver-only counters from the STATUS_STATS frame, and the wire byte totals - from the process-wide protocol counters. The labels, layout and - rate/speedup formulas match rsync 3.4.1. Shared by the single-threaded and - multithreaded send paths. */ -static void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start, - const ReceiverStats* recv) { - if (!config->stats || config->quiet) - return; - TransferStats empty = {0}; - if (stats == NULL) - stats = ∅ - ReceiverStats none = {0}; - if (recv == NULL) - recv = &none; - unsigned long long sent = protocol_bytes_written(); - unsigned long long received = protocol_bytes_read(); - /* rsync: bytes_per_sec = (written + read) / (0.5 + (end - start)). */ - double elapsed = difftime(time(NULL), start); - double rate = (double)(sent + received) / (0.5 + elapsed); - char total_buffer[32]; - char transferred_buffer[32]; - char literal_buffer[32]; - char matched_buffer[32]; - char sent_buffer[32]; - char recv_buffer[32]; - char rate_buffer[32] = {0}; - char human_rate[32] = {0}; - const char* total = - stats_bytes(config, stats->total_file_size, total_buffer, sizeof(total_buffer)); - const char* transferred = stats_bytes(config, stats->transferred_file_size, transferred_buffer, - sizeof(transferred_buffer)); - /* Protocol 2.28.0: the receiver reports the bytes it literally stored, which - is exact for a delta transfer (the sender's own literal_data counts each - stored file's whole source size and is only an upper bound). Fall back to - the sender total when the receiver reported no delta/literal accounting - (e.g. a local no-server path). */ - unsigned long long literal_bytes = (recv->literal_bytes != 0 || recv->matched_data != 0) - ? recv->literal_bytes - : stats->literal_data; - const char* literal = stats_bytes(config, literal_bytes, literal_buffer, sizeof(literal_buffer)); - const char* sent_s = stats_bytes(config, sent, sent_buffer, sizeof(sent_buffer)); - const char* recv_s = stats_bytes(config, received, recv_buffer, sizeof(recv_buffer)); - const char* rate_str = rate_buffer; - if (config->human_readable) { - if (!format_human_size_decimal((unsigned long long)rate, human_rate, sizeof(human_rate))) - snprintf(human_rate, sizeof(human_rate), "0"); - rate_str = human_rate; - } else { - snprintf(rate_buffer, sizeof(rate_buffer), "%.2f", rate); - } - double speedup = - (sent + received) > 0 ? (double)stats->total_file_size / (double)(sent + received) : 0.0; - char breakdown[128]; - stats_type_breakdown(stats, breakdown, sizeof(breakdown)); - unsigned long long flist_total = - stats->flist_reg + stats->flist_dir + stats->flist_link + stats->flist_special; - char created_breakdown[128]; - type_breakdown(recv->created_reg, recv->created_dir, recv->created_link, recv->created_special, - created_breakdown, sizeof(created_breakdown)); - unsigned long long created_total = - recv->created_reg + recv->created_dir + recv->created_link + recv->created_special; - printf("\n"); - if (breakdown[0] != '\0') - printf("Number of files: %llu %s\n", flist_total, breakdown); - else - printf("Number of files: %llu\n", flist_total); - /* Protocol 2.28.0: the receiver reports which destination entries it newly - created, split by type, so this line matches rsync exactly. */ - if (created_breakdown[0] != '\0') - printf("Number of created files: %llu %s\n", created_total, created_breakdown); - else - printf("Number of created files: %llu\n", created_total); - printf("Number of deleted files: %llu\n", recv->deleted_files); - printf("Number of regular files transferred: %llu\n", stats->transferred_regular); - printf("Total file size: %s bytes\n", total); - printf("Total transferred file size: %s bytes\n", transferred); - printf("Literal data: %s bytes\n", literal); - const char* matched = - stats_bytes(config, recv->matched_data, matched_buffer, sizeof(matched_buffer)); - printf("Matched data: %s bytes\n", matched); - printf("File list size: 0\n"); - printf("File list generation time: 0.000 seconds\n"); - printf("File list transfer time: 0.000 seconds\n"); - printf("Total bytes sent: %s\n", sent_s); - printf("Total bytes received: %s\n", recv_s); - printf("\n"); - printf("sent %s bytes received %s bytes %s bytes/sec\n", sent_s, recv_s, rate_str); - printf("total size is %s speedup is %.2f%s\n", total, speedup, - config->dry_run ? " (DRY RUN)" : ""); - fflush(stdout); -} - -/* Classify one scanned source entry into the rsync flist counters. Called for - every entry the sender walks, transferred or skipped. Directory entries are - counted here only for the explicit -d/--dirs generator; a recursive scan's - directories are accounted from the scanner's dir_entries list at report time. */ -static void transfer_stats_note_entry(TransferStats* stats, const File* file) { - if (stats == NULL || file == NULL) - return; - if (file->is_dir) { - stats->flist_dir++; - return; - } - if (file->is_symlink) { - stats->flist_link++; - stats->total_file_size += file->symlink_target ? strlen(file->symlink_target) : 0; - return; - } - if (file->is_special) { - stats->flist_special++; - return; - } - stats->flist_reg++; - stats->total_file_size += file->data ? file->data->size : 0; -} - -/* Account for a regular file (or a whole-file append) the receiver actually - stored: rsync's transferred-file count and transferred/literal byte totals. - `literal_data` counts the whole source size, which is exact for a whole-file - send but an upper bound for a delta send (the receiver reuses basis blocks - the sender never ships); see TransferStats.literal_data in format.h. */ -static void transfer_stats_note_transferred(TransferStats* stats, const File* file) { - if (stats == NULL || file == NULL) - return; - if (file->is_dir || file->is_symlink || file->is_special) - return; - if (file->link_group != 0 && !file->link_first) - return; - unsigned long long size = file->data ? file->data->size : 0; - stats->transferred_regular++; - stats->transferred_file_size += size; - stats->literal_data += size; -} - -/* ---- rsync-style per-file --progress ------------------------------------ - * rsync prints, for each transferred regular file, the file name followed by a - * two-frame progress line: the first at the initial 32 KiB read window (always - * 0.00 kB/s / 0:00:00 on a sub-second transfer) and a final 100% frame carrying - * `(xfr#N, to-chk=X/Y)`. Rates are wall-clock dependent, so only the final - * rate is measured here; the layout matches rsync 3.4.1's progress.c. */ -#define RSYNC_PROGRESS_IO_WINDOW (32ULL * 1024ULL) - -static const char* delete_display_path(const Config* config, const char* path); - -/* Paths-only pre-count of the source file list, built once at transfer start - * when progress output is requested. rsync's `to-chk` denominator is the whole - * file list -- every regular file, directory, symlink and special plus the - * transfer root -- while the streaming scan never emits directories. A - * metadata-only walk (no file reads, no hashing) supplies that total and the - * directory names, so the opt-in pass leaves non-progress runs untouched. */ -typedef struct { - unsigned long long total; - ArrayList* dir_paths; /* owned char* in transfer-relative display form */ -} ProgressPrecount; - -static bool g_progress_active; -static unsigned long long g_progress_xferred; -static unsigned long long g_progress_index; -static unsigned long long g_progress_total; -static struct timespec g_progress_file_start; -static ProgressPrecount g_progress_precount; -static PathIndex g_progress_dir_index; -static bool g_progress_dir_index_valid; -static StrHashSet g_progress_emitted; -static bool g_progress_emitted_valid; -static ArrayList* g_progress_emitted_keys; - -static bool progress_requested(const Config* config) { - return config != NULL && !config->quiet && - (config->show_progress || (config->info_level & LOG_INFO_PROGRESS) != 0); -} - -static void progress_precount_dispose(ProgressPrecount* p) { - if (p->dir_paths != NULL) { - array_list_delete(p->dir_paths); - p->dir_paths = NULL; - } - p->total = 0; -} - -static void client_progress_cleanup(void) { - if (g_progress_dir_index_valid) { - path_index_free(&g_progress_dir_index); - g_progress_dir_index_valid = false; - } - if (g_progress_emitted_valid) { - str_hash_set_free(&g_progress_emitted); - g_progress_emitted_valid = false; - } - if (g_progress_emitted_keys != NULL) { - array_list_delete(g_progress_emitted_keys); - g_progress_emitted_keys = NULL; - } - progress_precount_dispose(&g_progress_precount); - g_progress_active = false; - g_progress_total = 0; - g_progress_index = 0; - g_progress_xferred = 0; -} - -static void progress_first_frame(unsigned long long size, char* out, size_t out_size) { - char ofs_buf[32]; - unsigned long long ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; - if (!format_big_num(ofs, false, ofs_buf, sizeof(ofs_buf))) - snprintf(ofs_buf, sizeof(ofs_buf), "%llu", ofs); - int pct = size == 0 ? 100 : (ofs == size ? 100 : (int)(100.0 * (double)ofs / (double)size)); - snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s%s", ofs_buf, pct, 0.0, "kB/s", " 0:00:00", - " "); -} - -static void progress_final_frame(unsigned long long size, char* out, size_t out_size) { - char ofs_buf[32]; - char rembuf[32]; - unsigned long long last_ofs = size < RSYNC_PROGRESS_IO_WINDOW ? size : RSYNC_PROGRESS_IO_WINDOW; - if (!format_big_num(size, false, ofs_buf, sizeof(ofs_buf))) - snprintf(ofs_buf, sizeof(ofs_buf), "%llu", size); - struct timespec now; - clock_gettime(CLOCK_MONOTONIC, &now); - long long diff_ms = (long long)(now.tv_sec - g_progress_file_start.tv_sec) * 1000 + - (now.tv_nsec - g_progress_file_start.tv_nsec) / 1000000; - if (diff_ms <= 0) - diff_ms = 1; - double rate = - size > last_ofs ? (double)(size - last_ofs) * 1000.0 / (double)diff_ms / 1024.0 : 0.0; - const char* units = "kB/s"; - if (rate > 1024.0 * 1024.0) { - rate /= 1024.0 * 1024.0; - units = "GB/s"; - } else if (rate > 1024.0) { - rate /= 1024.0; - units = "MB/s"; - } - unsigned long long remain = (unsigned long long)(diff_ms / 1000); - snprintf(rembuf, sizeof(rembuf), "%4u:%02u:%02u", (unsigned)(remain / 3600), - (unsigned)((remain / 60) % 60), (unsigned)(remain % 60)); - /* rsync's `to-chk` denominator is the whole file list (the pre-count); the - numerator falls as each entry is processed, root first. Without a - pre-count (the paths-only walk failed) fall back to the transferred-file - count so the single-file layout stays intact. */ - unsigned long long total = g_progress_total > 0 ? g_progress_total : g_progress_xferred + 1; - unsigned long long to_chk = total > g_progress_index ? total - g_progress_index - 1 : 0; - snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s (xfr#%llu, to-chk=%llu/%llu)\n", ofs_buf, 100, - rate, units, rembuf, g_progress_xferred, to_chk, total); -} - -static bool info_flag_enabled(const Config* config, LogInfoFlag flag) { - return config != NULL && (config->info_level & flag) != 0; -} - -/* Print rsync's deletion lines for a received list of destination-relative - * paths: `*deleting PATH` when itemizing, the --out-format expansion when a - * format is set, else `deleting PATH` for --info=del. Used by both the dry-run - * would-delete report and the real --info=del report. */ -static void print_delete_reports(const Config* config, const ArrayList* paths) { - if (!config || !paths || config->quiet) - return; - /* --debug=del is independent of the --info=del/itemize/out-format display: - emit the debug trace even when no deletion line would be printed. */ - if (log_debug_enabled(LOG_DEBUG_DEL)) { - for (int i = 0; i < paths->size; i++) { - const char* raw = (const char*)paths->items[i]; - const char* path = delete_display_path(config, raw); - log_debug_message(LOG_DEBUG_DEL, "del: %s", path ? path : raw); - } - } - if (!(config->itemize_changes || config->out_format != NULL || - info_flag_enabled(config, LOG_INFO_DEL))) - return; - for (int i = 0; i < paths->size; i++) { - const char* raw = (const char*)paths->items[i]; - const char* path = delete_display_path(config, raw); - if (config->out_format != NULL) { - ChangeEvent event; - memset(&event, 0, sizeof(event)); - event.decision = CHANGE_SENT; - event.deleted = true; - event.name = path; - event.path = path; - char* line = change_render_format(config->out_format, config, &event); - if (line) { - char* escaped = output_escape(line, config->eight_bit_output); - printf("%s\n", escaped ? escaped : line); - free(escaped); - free(line); - } - } else { - char* escaped = output_escape(path, config->eight_bit_output); - if (config->itemize_changes) - printf("*deleting %s\n", escaped ? escaped : path); - else - printf("deleting %s\n", escaped ? escaped : path); - free(escaped); - } - } - fflush(stdout); -} - -static void client_progress_emit_ancestors(const Config* config, const char* rel) { - if (!g_progress_dir_index_valid || !g_progress_emitted_valid || g_progress_emitted_keys == NULL || - rel == NULL) - return; - size_t rel_len = strlen(rel); - for (size_t i = 0; i < rel_len; i++) { - if (rel[i] != '/') - continue; - char* prefix = malloc(i + 1); - if (prefix == NULL) - return; - memcpy(prefix, rel, i); - prefix[i] = '\0'; - if (path_index_contains(&g_progress_dir_index, prefix) && - !str_hash_set_lookup(&g_progress_emitted, prefix)) { - char* key = str_dup(prefix); - if (key != NULL && array_list_add(g_progress_emitted_keys, key)) { - str_hash_set_insert_ref(&g_progress_emitted, key); - char* escaped = output_escape(prefix, config->eight_bit_output); - printf("%s/\n", escaped ? escaped : prefix); - free(escaped); - g_progress_index++; - } else { - free(key); - } - } - free(prefix); - } -} - -/* rsync's --info=name/progress line for one entry: transfer-relative name (a - * trailing slash for directories) plus the ` -> target` symlink suffix. */ -static char* progress_entry_line(const File* file, const char* rel) { - const char* arrow = NULL; - const char* target = NULL; - if (file->is_symlink && file->symlink_target != NULL) { - arrow = " -> "; - target = file->symlink_target; - } else if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { - arrow = " => "; - target = file->hardlink_target; - } - size_t rel_len = strlen(rel); - bool dir_slash = file->is_dir && (rel_len == 0 || rel[rel_len - 1] != '/'); - size_t extra = (dir_slash ? 1u : 0u) + (target != NULL ? 4u + strlen(target) : 0u); - char* line = malloc(rel_len + extra + 1); - if (line == NULL) - return NULL; - memcpy(line, rel, rel_len); - size_t off = rel_len; - if (dir_slash) - line[off++] = '/'; - if (target != NULL) { - memcpy(line + off, arrow, 4); - off += 4; - memcpy(line + off, target, strlen(target)); - off += strlen(target); - } - line[off] = '\0'; - return line; -} - -static void client_progress_begin(const Config* config) { - change_reset_name_root(); - g_progress_active = progress_requested(config); - g_progress_xferred = 0; - g_progress_index = 1; /* the transfer root is file-list entry #0 */ - if (!g_progress_active) { - /* `--info=flist` prints rsync's file-list header even without progress. */ - if (!config->quiet && info_flag_enabled(config, LOG_INFO_FLIST)) { - printf("sending incremental file list\n"); - fflush(stdout); - } - return; - } - printf("sending incremental file list\n"); - /* rsync prints the transfer-root directory's name before the first file when - that directory is created; FastSync mirrors the source root below the - receive root and creates it on a fresh destination, so emit it here. */ - printf("./\n"); - fflush(stdout); -} - -/* Emit the name (unless itemize/out-format already did) and the two progress - * frames for one transferred regular file. */ -static void client_progress_file(const Config* config, const File* file) { - if (!g_progress_active || file == NULL || !file->data) - return; - g_progress_xferred++; - unsigned long long size = file->data->size; - if (!config->itemize_changes && config->out_format == NULL) { - const char* rel = delete_display_path(config, file_wire_path(file)); - client_progress_emit_ancestors(config, rel); - char* escaped = output_escape(rel, config->eight_bit_output); - printf("%s\n", escaped ? escaped : (rel ? rel : "")); - free(escaped); - } - clock_gettime(CLOCK_MONOTONIC, &g_progress_file_start); - char frame[160]; - progress_first_frame(size, frame, sizeof(frame)); - fputs(frame, stdout); - progress_final_frame(size, frame, sizeof(frame)); - fputs(frame, stdout); - g_progress_index++; - fflush(stdout); -} - -/* Emit the name line for a transferred non-regular entry (directory, symlink, - * special or hard-link sibling): rsync prints these in the file list but has no - * progress frame for them. */ -static void client_progress_name(const Config* config, const File* file) { - if (!g_progress_active || file == NULL) - return; - const char* rel = delete_display_path(config, file_wire_path(file)); - if (!config->itemize_changes && config->out_format == NULL) { - client_progress_emit_ancestors(config, rel); - char* line = progress_entry_line(file, rel ? rel : ""); - if (line != NULL) { - char* escaped = output_escape(line, config->eight_bit_output); - printf("%s\n", escaped ? escaped : line); - free(escaped); - free(line); - fflush(stdout); - } - } - g_progress_index++; -} - -/* An entry the receiver already had prints no name under --progress but still - * occupies a file-list slot in the `to-chk` numerator. */ -static void client_progress_uptodate(const Config* config, const File* file) { - (void)config; - (void)file; - if (!g_progress_active) - return; - g_progress_index++; -} - -/* Compiled scanner inputs that are shared read-only across scanner instances - * and, in -m mode, across worker threads. `base_filters` owns the compiled - * command-line + -C rules; the FileListSet allow-set lives in the Config. - * `hardlinks` owns the --hard-links/-H link-group detection table (NULL when - * off) and is shared (mutex-guarded) across every scanner/worker of one scan. */ -typedef struct { - ScannerOptions options; - FilterRuleList* base_filters; /* owned; may be NULL */ - HardLinkTable* hardlinks; /* owned; may be NULL */ - char* relative_prefix; /* owned -R prefix; may be NULL */ -} PreparedScanner; - -/* Build the scanner options for one scan. Returns false and logs on failure. */ -static bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out) { - if (!out) - return false; - out->base_filters = NULL; - out->hardlinks = NULL; - out->relative_prefix = NULL; - memset(&out->options, 0, sizeof(out->options)); - - int rule_count = config->filters ? config->filters->size : 0; - const char** texts = NULL; - if (rule_count > 0) { - texts = malloc((size_t)rule_count * sizeof(char*)); - if (!texts) { - log_message(LOG_LEVEL_ERROR, "memory allocation failed for filter rules"); - return false; - } - for (int i = 0; i < rule_count; i++) - texts[i] = (const char*)config->filters->items[i]; - } - if (rule_count > 0 || config->cvs_exclude) { - char err[160]; - out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude, - config->delete_excluded, err, sizeof(err)); - free(texts); - if (!out->base_filters) { - log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err); - return false; - } - } else { - free(texts); - } - - ScannerOptions* options = &out->options; - options->use_metadata = config->use_metadata; - options->preserve_atimes = config->preserve_atimes; - options->preserve_crtimes = config->preserve_crtimes; - options->preserve_xattrs = config->preserve_xattrs; - options->preserve_acls = config->preserve_acls; - options->chunk_size = config->chunk_size; - /* --exclude/--include are compiled, in command-line order, into the SAME - * ordered filter rule list as --filter/-f (see config_add_selection_rule), so - * the legacy per-kind arrays are deliberately NOT passed to the scanner: - * doing so would re-apply them with the old "excludes first, then includes as - * a mandatory whitelist" precedence and defeat rsync's first-match-wins - * ordering. The arrays remain populated purely for the Config API surface. */ - options->exclude_patterns = NULL; - options->exclude_count = 0; - options->include_patterns = NULL; - options->include_count = 0; - options->max_size = config->max_size; - options->min_size = config->min_size; - options->max_depth = config->max_depth; - options->num_threads = num_threads; - options->follow_symlinks = config->follow_symlinks; - options->copy_links = config->copy_links; - options->safe_links = config->safe_links; - options->copy_unsafe_links = config->copy_unsafe_links; - options->copy_dirlinks = config->copy_dirlinks; - options->munge_links = config->munge_links; - options->checksum = config->checksum; - options->one_file_system = config->one_file_system; - options->preserve_devices = config->preserve_devices; - options->preserve_specials = config->preserve_specials; - options->copy_devices = config->copy_devices; - options->file_list = (const FileListSet*)config->files_from_set; - options->base_filters = out->base_filters; - options->per_dir_filters = config->per_dir_filter; - options->delete_excluded = config->delete_excluded; - options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2; - options->dirs = config->dirs; - options->relative = config->relative; - /* A real recursive transfer recreates empty source directories (rsync - parity); low-level scanner users leave this off. */ - options->emit_empty_dirs = true; - /* --no-implied-dirs only has meaning with -R (rsync): without it the option - is a documented no-op, so the scanner must not suppress directory - metadata. */ - options->no_implied_dirs = config->no_implied_dirs && config->relative; - /* -R/--relative outside --files-from reconstructs every destination path from - * the source spec (rsync's '/./' cut point). With --files-from the listed - * entry already supplies the bare relative path, so no prefix is built. */ - if (config->relative && config->files_from_set == NULL && config->send_directory) { - out->relative_prefix = scanner_relative_prefix(config->send_directory); - if (!out->relative_prefix) { - log_message(LOG_LEVEL_ERROR, "memory allocation failed building --relative path prefix"); - filter_rule_list_free(out->base_filters); - out->base_filters = NULL; - return false; - } - options->relative_prefix = out->relative_prefix; - } - options->prune_empty_dirs = config->prune_empty_dirs; - options->ignore_io_errors = config->ignore_errors; - options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args; - options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet; - options->note_mount = (config->info_level & LOG_INFO_MOUNT) != 0 && !config->quiet; - options->send_directory = config->send_directory; - options->eight_bit_output = config->eight_bit_output; - options->excluded_paths = NULL; - options->excluded_mutex = NULL; - options->size_skipped_paths = NULL; - options->synced_dirs = NULL; - options->hardlinks = NULL; - /* Set by the real send paths; NULL for the metadata-only scans (progress - pre-count, batch) that must not perturb the sender's --stats counter. */ - options->dir_count = NULL; - /* P7 Wave D: capture source directory metadata when a directory attribute is - requested (-p for modes, -t for times unless -O omits them). Whether they - are APPLIED is decided receiver-side. */ - options->capture_dir_times = dir_metadata_should_capture(config); - options->dir_entries = NULL; - options->dir_entries_mutex = NULL; - if (config->preserve_hard_links) { - out->hardlinks = hardlink_table_create(); - if (!out->hardlinks) { - filter_rule_list_free(out->base_filters); - out->base_filters = NULL; - return false; - } - options->hardlinks = out->hardlinks; - } - return true; -} - -static void prepared_scanner_destroy(PreparedScanner* prepared) { - if (!prepared) - return; - filter_rule_list_free(prepared->base_filters); - prepared->base_filters = NULL; - hardlink_table_destroy(prepared->hardlinks); - prepared->hardlinks = NULL; - free(prepared->relative_prefix); - prepared->relative_prefix = NULL; -} - -static bool progress_precount_add_dir(ProgressPrecount* p, const char* path) { - if (path == NULL || path[0] == '\0') - return true; - char* dup = str_dup(path); - if (dup == NULL) - return false; - if (array_list_add(p->dir_paths, dup)) - return true; - free(dup); - return false; -} - -/* Metadata-only walk collecting the full file-list total and every directory - * name. It uses its own scanner (fresh filter compilation and hard-link table) - * so the data pass's link-group state is never perturbed. */ -static bool progress_precount_scan(const Config* config, ProgressPrecount* out) { - out->dir_paths = array_list_create(free); - if (out->dir_paths == NULL) - return false; - out->total = 0; - PreparedScanner prepared; - memset(&prepared, 0, sizeof(prepared)); - if (!prepare_scanner(config, 0, &prepared)) { - progress_precount_dispose(out); - return false; - } - ScannerOptions local = prepared.options; - local.list_dirs = true; - local.note_nonreg = false; - local.note_mount = false; - local.dir_count = NULL; - local.use_metadata = false; - local.preserve_xattrs = false; - local.preserve_acls = false; - local.checksum = false; - local.capture_dir_times = false; - local.excluded_paths = NULL; - local.size_skipped_paths = NULL; - local.synced_dirs = NULL; - local.plan_dirs = NULL; - local.dir_entries = NULL; - local.dir_entries_mutex = NULL; - local.hardlinks = NULL; - DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); - bool ok = scanner != NULL; - if (scanner != NULL) { - Chunk* chunk; - while (ok && (chunk = directory_scanner_next(scanner)) != NULL) { - out->total += (unsigned long long)chunk->element_count; - for (int i = 0; i < chunk->element_count && ok; i++) { - const File* f = chunk->items[i]; - if (f != NULL && f->is_dir) - ok = progress_precount_add_dir(out, delete_display_path(config, file_wire_path(f))); - } - chunk_destroy(chunk); - } - if (ok && directory_scanner_failed(scanner)) - ok = false; - directory_scanner_destroy(scanner); - } - prepared_scanner_destroy(&prepared); - if (!ok) { - progress_precount_dispose(out); - return false; - } - out->total += 1; /* the transfer root "." */ - return true; -} - -/* Reuse the --delete-during/--delete-delay keep-set pre-scan: its traversed - * directory list already holds every directory and `non_dir_count` the entries - * counted during that same pass, so progress costs no second walk. */ -static bool progress_precount_from_plan_dirs(const Config* config, const ArrayList* plan_dirs, - unsigned long long non_dir_count, - ProgressPrecount* out) { - out->dir_paths = array_list_create(free); - if (out->dir_paths == NULL) - return false; - out->total = non_dir_count + 1; - for (int i = 0; i < plan_dirs->size; i++) { - const char* path = (const char*)plan_dirs->items[i]; - const char* rel = config->send_directory != NULL - ? utils_strip_transfer_root(path, config->send_directory) - : path; - if (!progress_precount_add_dir(out, rel)) { - progress_precount_dispose(out); - return false; - } - } - out->total += (unsigned long long)out->dir_paths->size; - return true; -} - -/* Build the optional progress pre-count. A failed pre-count is non-fatal: the - * transfer proceeds and the progress denominator falls back to the transferred - * file count. */ -static void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, - unsigned long long plan_non_dir_count) { - client_progress_cleanup(); - g_progress_active = progress_requested(config); - if (!g_progress_active) - return; - bool ok = plan_dirs != NULL ? progress_precount_from_plan_dirs( - config, plan_dirs, plan_non_dir_count, &g_progress_precount) - : progress_precount_scan(config, &g_progress_precount); - if (!ok) { - g_progress_total = 0; - return; - } - g_progress_total = g_progress_precount.total; - if (g_progress_precount.dir_paths != NULL && g_progress_precount.dir_paths->size > 0 && - path_index_build(&g_progress_dir_index, - (const char* const*)g_progress_precount.dir_paths->items, - (size_t)g_progress_precount.dir_paths->size)) - g_progress_dir_index_valid = true; - if (str_hash_set_init(&g_progress_emitted, (size_t)(g_progress_precount.dir_paths != NULL - ? g_progress_precount.dir_paths->size + 1 - : 1))) - g_progress_emitted_valid = true; - g_progress_emitted_keys = array_list_create(free); -} - -/* -R/--relative implied directories: rsync transmits the metadata of the - * parent directories implied by the source path (every prefix component above - * the source root) so the receiver applies their attributes to the created - * parents. FastSync's scan only covers the source root and below, so append - * one metadata-only directory entry per implied ancestor. --no-implied-dirs - * suppresses this exactly like rsync. A missing ancestor is never fatal. */ -static bool append_implied_dir_times(const Config* config, ArrayList* dir_entries) { - if (!dir_entries || !config->relative || config->files_from_set != NULL || - config->no_implied_dirs || !config->send_directory) - return true; - char* prefix = scanner_relative_prefix(config->send_directory); - if (!prefix) - return true; - int ncomp = 0; - for (const char* s = prefix; *s;) { - while (*s == '/') - s++; - if (!*s) - break; - while (*s && *s != '/') - s++; - ncomp++; - } - if (ncomp <= 1) { - free(prefix); - return true; - } - char* fs = str_dup(config->send_directory); - if (!fs) { - free(prefix); - return true; - } - size_t flen = strlen(fs); - while (flen > 1 && fs[flen - 1] == '/') - fs[--flen] = '\0'; - bool ok = true; - /* Walk the source path upwards one component at a time (fs is truncated in - place, so each step targets the next implied ancestor). */ - for (int depth = ncomp - 2; depth >= 0 && ok; depth--) { - char* slash = strrchr(fs, '/'); - if (!slash || slash == fs) - break; - *slash = '\0'; - char* p = prefix; - int c = 0; - while (c <= depth) { - while (*p == '/') - p++; - while (*p && *p != '/') - p++; - c++; - } - char saved = *p; - *p = '\0'; - struct stat st; - if (stat(fs, &st) == 0 && S_ISDIR(st.st_mode)) { - File* file = file_create(fs); - if (!file) { - ok = false; - } else { - file->is_dir = true; - file->metadata = - file_metadata_create(fs, &st, config->preserve_atimes, config->preserve_crtimes); - file->send_path = str_dup(prefix); - if (!file->metadata || !file->send_path || !array_list_add(dir_entries, file)) { - file_destroy(file); - ok = false; - } - } - } - *p = saved; - } - free(fs); - free(prefix); - return ok; -} - -/* The delete-walk root scope for a full (non---files-from) transfer: rsync - * confines --delete to the directories it actually transferred. A plain - * recursive run mirrors the source under the receive root, so "." (the whole - * tree) is correct; an -R run transfers only the reconstructed prefix subtree, - * so the walk is scoped to that prefix instead. Returns a malloc'd wire path - * (or "."), or NULL on allocation failure. */ -static char* delete_scope_root_marker(const Config* config) { - if (config->relative && config->files_from_set == NULL && config->send_directory) { - char* prefix = scanner_relative_prefix(config->send_directory); - if (!prefix) - return NULL; - if (prefix[0] != '\0') - return prefix; - free(prefix); - } - return str_dup("."); -} - -/* The -R destination prefix that confines a per-directory delete walk, or NULL - * when the whole receive root is in scope. The marker was installed into - * `synced_dirs` by delete_scope_root_marker(); for a plain recursive transfer - * it is "." (whole root) and for --files-from the list is not a single prefix. */ -static const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs) { - if (!config || config->files_from_set != NULL || !config->relative || !config->send_directory) - return NULL; - if (!synced_dirs || synced_dirs->size != 1) - return NULL; - const char* marker = (const char*)synced_dirs->items[0]; - if (marker[0] == '\0' || strcmp(marker, ".") == 0) - return NULL; - return marker; -} - -/* The destination-relative mirror path for a missing --files-from entry: where - a PRESENT entry with the same name would have been written. With -R that is - the entry's bare relative path (the bare wire path the receiver uses); - otherwise it is the full source mirror below the destination root - (`send_directory` joined to the entry, leading '/' stripped), exactly the - path the manifest records for a present sibling. Returns an owned string, or - NULL on allocation failure. */ -static char* files_from_missing_dest_path(const Config* config, const char* entry) { - if (config->relative) - return str_dup(entry); - char* joined = path_cat(config->send_directory, entry); - if (!joined) - return NULL; - const char* rel = *joined == '/' ? joined + 1 : joined; - char* dup = str_dup(rel); - free(joined); - return dup; -} - -/* --files-from semantics: every listed entry must resolve under the source - * root, otherwise rsync reports a hard error instead of silently transferring - * nothing. An entry of "." (the whole tree) and listed-but-empty directories - * are valid. An empty list is valid too: rsync transfers nothing and exits 0. - * With --ignore-missing-args - * (implied by --delete-missing-args) a listed-but-missing entry is instead - * skipped: nothing is transferred for it, it never enters the keep-set and the - * run succeeds for the rest (an all-missing non-empty list succeeds - * transferring nothing, matching rsync). With --delete-missing-args - * `missing_dest` (when non-NULL) collects the entry's destination-relative - * mirror for the receiver's exact-deletion request. Runs before any - * transfer so the failure/skip is surfaced uniformly in the single-threaded, - * -m, dry-run and --list-only paths. */ -static bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out) { - *skipped_out = 0; - const FileListSet* set = (const FileListSet*)config->files_from_set; - if (!set) - return true; - if (!config->send_directory) { - log_message(LOG_LEVEL_ERROR, "--files-from requires a source directory"); - return false; - } - if (set->count == 0) { - /* rsync treats an empty --files-from list as "nothing to transfer" and - exits 0 (the source directory is still a valid source arg), so this is - not an error. Nothing passes the (empty) allow-set, so no file is sent - and no keep-set entry is produced. */ - return true; - } - bool ignore = config->ignore_missing_args || config->delete_missing_args; - for (int i = 0; i < set->count; i++) { - const char* entry = set->entries[i]; - if (entry[0] == '\0') - continue; /* "." == list the whole tree */ - char* full = path_cat(config->send_directory, entry); - if (!full) { - log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); - return false; - } - struct stat st; - if (lstat(full, &st) != 0) { - free(full); - if (ignore) { - (*skipped_out)++; - char* escaped_entry = output_escape(entry, log_get_8_bit_output()); - log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'", - escaped_entry ? escaped_entry : ""); - free(escaped_entry); - if (config->delete_missing_args && missing_dest) { - char* mirror = files_from_missing_dest_path(config, entry); - if (!mirror || !array_list_add(missing_dest, mirror)) { - free(mirror); - log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from"); - return false; - } - } - continue; - } - char* escaped_entry = output_escape(entry, log_get_8_bit_output()); - char* escaped_src = output_escape(config->send_directory, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "--files-from entry '%s' not found in source '%s'", - escaped_entry ? escaped_entry : "", - escaped_src ? escaped_src : ""); - free(escaped_entry); - free(escaped_src); - return false; - } - free(full); - } - if (*skipped_out > 0) { - if (config->delete_missing_args) { - /* --list-only never deletes and a --dry-run only shows intent, so the - summary must not claim a real deletion happened in those modes. */ - if (config->list_only) - log_message(LOG_LEVEL_WARNING, - "--delete-missing-args: %d missing --files-from entr%s skipped (--list-only " - "never deletes)", - *skipped_out, *skipped_out == 1 ? "y" : "ies"); - else if (config->dry_run) - log_message(LOG_LEVEL_WARNING, - "--delete-missing-args: %d missing --files-from entr%s would be deleted from " - "the destination (dry run)", - *skipped_out, *skipped_out == 1 ? "y" : "ies"); - else - log_message( - LOG_LEVEL_WARNING, - "--delete-missing-args: %d missing --files-from entr%s will be deleted from the " - "destination", - *skipped_out, *skipped_out == 1 ? "y" : "ies"); - } else if (config->ignore_missing_args) - log_message(LOG_LEVEL_WARNING, - "--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out, - *skipped_out == 1 ? "y" : "ies"); - } - return true; -} - /* Read the daemon's MOTD frame and, unless --no-motd, display it on stdout. * * The daemon sends the MOTD as the first thing after the config-frame STATUS_OK @@ -1085,7 +65,7 @@ static bool files_from_list_check(const Config* config, ArrayList* missing_dest, * has no MOTD frame. The text is rendered through motd_render so a hostile * server cannot inject terminal escape sequences. A read failure is not fatal * here: the transfer that follows surfaces the real connection error. */ -static void receive_daemon_motd(Client* client, const Config* config) { +void receive_daemon_motd(Client* client, const Config* config) { if (!config->module || config->module[0] == '\0') return; char* motd = motd_receive(client->file_descriptor); @@ -1106,7 +86,7 @@ static void receive_daemon_motd(Client* client, const Config* config) { } /* Select the configured transport for both transfer execution paths. */ -static Client* connect_transfer_client(const Config* config) { +Client* connect_transfer_client(const Config* config) { if (config->transport == TRANSPORT_SSH) { if (config->use_sendfile) { log_message(LOG_LEVEL_ERROR, "--sendfile is not supported with SSH transport"); @@ -1144,55 +124,13 @@ static Client* connect_transfer_client(const Config* config) { return client; } -static void disconnect_transfer_client(Client* client) { +void disconnect_transfer_client(Client* client) { if (!client) return; client_disconnect(client); client_delete(client); } -/* True when --dry-run should contact a receiver rather than running the - * client-side local manifest. Any target a real run would reach over the wire - * selects the server-contacting path: a remote (SSH host:path), a daemon - * (host::module/path), an explicit --server-host, --server-port/--port, TLS, or - * a source-bind --address. A plain local destination (none of these) keeps the - * original client-side behavior, which never dials the default 127.0.0.1:8080. */ -static bool dry_run_targets_server(const Config* config) { - if (!config) - return false; - if (config->transport == TRANSPORT_SSH) - return true; - if (config->module && config->module[0] != '\0') - return true; - if (config->server_host_set || config->server_port_set) - return true; - if (config->use_tls) - return true; - if (config->address != NULL) - return true; - return false; -} - -static bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk) { - if (!manifest) - return true; - for (int i = 0; i < chunk->element_count; i++) { - const char* path = file_wire_path(chunk->items[i]); - if (*path == '/') - path++; - char* entry = str_dup(path); - if (!entry) { - log_message(LOG_LEVEL_ERROR, "Failed to allocate manifest entry"); - return false; - } - if (!array_list_add(manifest, entry)) { - free(entry); - return false; - } - } - return true; -} - /* (finalize_transfer is defined after the SourceFile helpers below.) */ typedef struct SourceFile { @@ -1298,54 +236,6 @@ static void mark_sender_done(PipelineContextSender* context) { mtx_unlock(&context->mutex_progress); } -/* Read the optional STATUS_STATS record (protocol 2.25.0) that the receiver - * sends just before its terminal status when report_stats was negotiated. - * Consumes the would-delete path list into `would_delete` (optional). */ -static bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete) { - if (!format_stats_receive(fd, stats)) - return false; - int count = 0; - if (!receive_int(fd, &count) || count < 0 || count > MAX_MANIFEST_ENTRIES) - return false; - /* Mirror the delete-plan parser: every retained path must be a valid - destination-relative path, and the whole list shares one MAX_MANIFEST_BYTES - budget so a hostile peer cannot make the client retain unbounded memory. */ - size_t bytes = 0; - for (int i = 0; i < count; i++) { - char* path = receive_wire_str(fd); - if (!path) - return false; - if (path[0] == '\0' || path[0] == '/' || has_path_traversal(path)) { - free(path); - return false; - } - if (would_delete) { - size_t entry_size = strlen(path) + sizeof(char*) + 16; - if (entry_size > MAX_MANIFEST_BYTES - bytes) { - free(path); - return false; - } - bytes += entry_size; - if (!array_list_add(would_delete, path)) { - free(path); - return false; - } - } else { - free(path); - } - } - return true; -} - -/* Strip the transfer-root prefix from a receiver-reported destination-relative - * delete path so a `*deleting` line matches rsync's transfer-relative name - * (FastSync's destination mirror includes the source's absolute path). */ -static const char* delete_display_path(const Config* config, const char* path) { - if (!config || !path || !config->send_directory) - return path; - return utils_strip_transfer_root(path, config->send_directory); -} - /* Send the final STATUS_FINISHED frame and await the receiver's verdict. When --remove-source-files is active the receiver acknowledges each data file it processed, in send order: STATUS_NEXT means the file was written, @@ -1429,463 +319,8 @@ static void pipeline_cancel(PipelineContextSender* context) { mtx_unlock(&context->mutex_scanner); } -/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */ -static int send_dry_run_manifest(const Config* config) { - int skipped = 0; - ArrayList* missing_dest = NULL; - if (config->delete_missing_args) { - missing_dest = array_list_create(free); - if (!missing_dest) - return -1; - } - if (!files_from_list_check(config, missing_dest, &skipped)) { - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - PreparedScanner prepared; - if (!prepare_scanner(config, 0, &prepared)) { - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - DirectoryScanner* scanner = - directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) { - prepared_scanner_destroy(&prepared); - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - Chunk* chunk; - int file_count = 0; - unsigned long long total_bytes = 0; - char size_buffer[32]; - if (!config->quiet) - printf("Dry run: files to be transferred\n"); - while ((chunk = directory_scanner_next(scanner)) != NULL) { - for (int i = 0; i < chunk->element_count; i++) { - if (!config->quiet) { - char* escaped_path = - output_escape(file_wire_path(chunk->items[i]), config->eight_bit_output); - if (!escaped_path) { - chunk_destroy(chunk); - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - if (missing_dest) - array_list_delete(missing_dest); - return -1; - } - if (config->human_readable) - printf( - " %s (%s)\n", escaped_path, - display_bytes(chunk->items[i]->data->size, true, size_buffer, sizeof(size_buffer))); - else - printf(" %s (%zu bytes)\n", escaped_path, chunk->items[i]->data->size); - free(escaped_path); - } - total_bytes += chunk->items[i]->data->size; - file_count++; - } - chunk_destroy(chunk); - } - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - /* --delete-missing-args: the missing entries' destination mirrors render as - would-be deletions (rsync's dry-run also lists its *deleting lines). */ - if (missing_dest && !config->quiet) { - for (int i = 0; i < missing_dest->size; i++) { - char* escaped = output_escape((char*)missing_dest->items[i], config->eight_bit_output); - printf(" %s (missing; would be deleted)\n", escaped ? escaped : ""); - free(escaped); - } - } - if (missing_dest) - array_list_delete(missing_dest); - if (!config->quiet) { - if (config->human_readable) - printf("Total: %d files, %s\n", file_count, - display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); - else - printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); - } - return 0; -} - -typedef struct { - char* name; /* transfer-relative name ("" == the source root) */ - mode_t mode; - unsigned long long size; - time_t mtime; - long mtime_nsec; - bool is_dir; - bool is_symlink; - char* link_target; -} ListEntry; - -static void list_entries_destroy(ListEntry* entries, size_t count) { - if (entries == NULL) - return; - for (size_t i = 0; i < count; i++) { - free(entries[i].name); - free(entries[i].link_target); - } - free(entries); -} - -static int compare_list_entries(const void* left, const void* right) { - const ListEntry* a = (const ListEntry*)left; - const ListEntry* b = (const ListEntry*)right; - return strcmp(a->name, b->name); -} - -/* Relative path of an entry below `root` ("" for the root itself). Mirrors - * change_list's relative_name for list-only rendering. */ -static char* list_relative_name(const char* root, const char* full) { - if (root == NULL || full == NULL) - return str_dup(full != NULL ? full : ""); - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(root, full, root_len) == 0) { - if (full[root_len] == '\0') - return str_dup(""); - if (full[root_len] == '/') - return str_dup(full + root_len + 1); - } - return str_dup(full); -} - -/* --list-only: print an ls-style listing of the entries that WOULD be - * transferred and exit without contacting the server or writing anything. - * Names are transfer-relative (rsync prints `a.txt`, `sub/b.txt`, `.`) and - * directory entries are included. Returns 0 on success, 1 on error. */ -static int send_list_only(const Config* config) { - int skipped = 0; - if (!files_from_list_check(config, NULL, &skipped)) - return 1; - PreparedScanner prepared; - if (!prepare_scanner(config, 0, &prepared)) - return 1; - prepared.options.use_metadata = true; /* capture mode + mtime for the listing */ - prepared.options.list_dirs = true; - DirectoryScanner* scanner = - directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) { - prepared_scanner_destroy(&prepared); - return 1; - } - ListEntry* entries = NULL; - size_t count = 0; - size_t capacity = 0; - bool oom = false; - - /* rsync lists the source root itself (as "."). Only when the source is a - * directory and no --files-from subset is in effect. */ - if (config->files_from_set == NULL && config->send_directory != NULL) { - struct stat st; - if (stat(config->send_directory, &st) == 0 && S_ISDIR(st.st_mode)) { - capacity = 64; - entries = calloc(capacity, sizeof(ListEntry)); - if (entries == NULL) { - oom = true; - } else if ((entries[0].name = str_dup("")) == NULL) { - /* A NULL name would be dereferenced by qsort/render: fail the listing. */ - oom = true; - } else { - entries[0].mode = st.st_mode; - entries[0].mtime = st.st_mtime; - entries[0].mtime_nsec = st.st_mtim.tv_nsec; - entries[0].size = (unsigned long long)st.st_size; - entries[0].is_dir = true; - count = 1; - } - } - } - - Chunk* chunk; - while (!oom && (chunk = directory_scanner_next(scanner)) != NULL) { - for (int i = 0; i < chunk->element_count; i++) { - File* f = chunk->items[i]; - if (f == NULL) - continue; - if (count == capacity) { - size_t new_capacity = capacity > 0 ? capacity * 2 : 64; - if (new_capacity <= capacity) { - oom = true; - break; - } - ListEntry* grown = realloc(entries, new_capacity * sizeof(ListEntry)); - if (!grown) { - oom = true; - break; - } - entries = grown; - memset(entries + capacity, 0, (new_capacity - capacity) * sizeof(ListEntry)); - capacity = new_capacity; - } - char* name = list_relative_name(config->send_directory, file_wire_path(f)); - if (!name) { - oom = true; - break; - } - mode_t mode = 0; - time_t mtime = 0; - long mtime_nsec = 0; - if (f->metadata != NULL) { - mode = f->metadata->mode; - mtime = f->metadata->mtime_sec; - mtime_nsec = f->metadata->mtime_nsec; - } else { - struct stat st; - if (lstat(f->path, &st) == 0) { - mode = st.st_mode; - mtime = st.st_mtime; - mtime_nsec = st.st_mtim.tv_nsec; - } - } - entries[count].name = name; - entries[count].mode = mode; - entries[count].mtime = mtime; - entries[count].mtime_nsec = mtime_nsec; - if (f->is_symlink) - entries[count].size = f->symlink_target != NULL ? strlen(f->symlink_target) : 0; - else if (f->is_dir) { - struct stat dir_st; - entries[count].size = stat(f->path, &dir_st) == 0 ? (unsigned long long)dir_st.st_size : 0; - } else - entries[count].size = f->data != NULL ? f->data->size : 0; - entries[count].is_dir = f->is_dir; - entries[count].is_symlink = f->is_symlink; - entries[count].link_target = - f->is_symlink && f->symlink_target ? str_dup(f->symlink_target) : NULL; - count++; - } - chunk_destroy(chunk); - } - bool failed = oom || directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner); - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - if (failed) { - list_entries_destroy(entries, count); - if (oom) - log_message(LOG_LEVEL_ERROR, "memory allocation failed while listing"); - return 1; - } - if (count > 1) - qsort(entries, count, sizeof(ListEntry), compare_list_entries); - for (size_t i = 0; i < count; i++) { - ChangeEvent event; - memset(&event, 0, sizeof(event)); - event.name = entries[i].name; - event.path = entries[i].name; - event.mode = entries[i].mode; - event.size = entries[i].size; - event.mtime_sec = entries[i].mtime; - event.mtime_nsec = entries[i].mtime_nsec; - event.is_directory = entries[i].is_dir; - event.is_symlink = entries[i].is_symlink; - event.symlink_target = entries[i].link_target; - char* line = change_render_list_line(config, &event); - if (line != NULL) { - char* escaped = output_escape(line, config->eight_bit_output); - printf("%s\n", escaped != NULL ? escaped : line); - free(escaped); - free(line); - } - } - list_entries_destroy(entries, count); - return 0; -} - -/* Send the delete manifest to the server. Returns 0 on success, -1 on - failure. It carries FOUR sections: the keep-set paths, the protected - excluded prefixes, the --delete-missing-args exact-delete paths, and the - destination-relative directories the sender synchronized this run. - When --delete-excluded is given `protected` is empty: excluded destination - mirrors are then ordinary extras and are removed. When - --delete-missing-args is active `missing_args` holds the destination mirrors - of missing --files-from entries: each is an explicit receiver-side deletion - request, independent of the extras walk. `synced_dirs` confines the extras - walk to entries directly inside a synchronized directory. A NULL - keep-set / protected / missing / dirs list transmits an empty section. All - four sections are unbounded on the sender; the receiver enforces - MAX_MANIFEST_ENTRIES per section and a single MAX_MANIFEST_BYTES budget - shared across the sections, rejecting (with STATUS_ERROR) an over-budget - frame. A heavily filtered source whose exclusion list is large therefore - fails the run cleanly on the receiver rather than being truncated. */ -static int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes, - ArrayList* size_skipped, ArrayList* missing_args, - ArrayList* synced_dirs) { - if (!send_status(fd, STATUS_MANIFEST)) - return -1; - int keep_count = manifest ? manifest->size : 0; - if (!send_int(fd, keep_count)) - return -1; - for (int i = 0; i < keep_count; i++) { - if (!send_wire_str(fd, (char*)manifest->items[i])) - return -1; - } - /* The receiver has ONE protected-prefix section; filter-excluded prefixes - (dropped under --delete-excluded) and size-pruned prefixes (always - protected) are concatenated into it. */ - int protected_count = - (protected_prefixes ? protected_prefixes->size : 0) + (size_skipped ? size_skipped->size : 0); - if (!send_int(fd, protected_count)) - return -1; - if (protected_prefixes) { - for (int i = 0; i < protected_prefixes->size; i++) { - if (!send_wire_str(fd, (char*)protected_prefixes->items[i])) - return -1; - } - } - if (size_skipped) { - for (int i = 0; i < size_skipped->size; i++) { - if (!send_wire_str(fd, (char*)size_skipped->items[i])) - return -1; - } - } - int missing_count = missing_args ? missing_args->size : 0; - if (!send_int(fd, missing_count)) - return -1; - for (int i = 0; i < missing_count; i++) { - if (!send_wire_str(fd, (char*)missing_args->items[i])) - return -1; - } - int dirs_count = synced_dirs ? synced_dirs->size : 0; - if (!send_int(fd, dirs_count)) - return -1; - for (int i = 0; i < dirs_count; i++) { - if (!send_wire_str(fd, (char*)synced_dirs->items[i])) - return -1; - } - return 0; -} - -/* Transmit the keep-set manifest and wait for the receiver's verdict. Used by - --delete-before/--delete-during, where the extras are removed on the receiver - BEFORE the first byte of file data is sent: the receiver acknowledges with - STATUS_OK once the bounded delete committed, or STATUS_ERROR if it could not - (in which case the sender aborts without streaming any data). The ACK may - take much longer than an ordinary per-message round trip because the receiver - performs the whole bounded deletion walk (up to MAX_SERVER_DELETE_COUNT - unlinks) before replying, so the wait uses a generous explicit deadline - instead of the default 60 s receive window. */ -#define DELETE_ACK_TIMEOUT_SEC 3600 -/* While waiting for the (potentially slow) receiver-side deletion, send a - * STATUS_KEEPALIVE at most this often so the connection is demonstrably alive - * and neither side's per-message timeout trips. */ -#define DELETE_ACK_KEEPALIVE_SEC 10 - -static bool send_delete_manifest_early(Client* client, ArrayList* manifest, - ArrayList* protected_prefixes, ArrayList* size_skipped, - ArrayList* missing_args, ArrayList* synced_dirs) { - if (!client || !manifest) - return false; - if (send_delete_manifest(client->file_descriptor, manifest, protected_prefixes, size_skipped, - missing_args, synced_dirs) != 0) - return false; - Status ack; - /* The wait is long (up to an hour) and runs inline on this thread: a helper - * thread would race the non-thread-safe protocol send path, so keepalives are - * emitted from this wait loop itself. A Ctrl-C/SIGTERM abort flag also ends - * the wait; the caller then best-effort sends STATUS_ABORT. */ - if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, - DELETE_ACK_KEEPALIVE_SEC, client_abort_pending)) { - /* A Ctrl-C/SIGTERM abort ends the wait above; tell the receiver before the - caller tears the connection down (best-effort). */ - if (client_abort_pending()) { - log_info_message(LOG_INFO_MISC, - "Abort requested while awaiting delete ack; sending STATUS_ABORT"); - send_status(client->file_descriptor, STATUS_ABORT); - } - return false; - } - if (ack != STATUS_OK) { - log_server_rejection("Server failed to delete files before the transfer"); - return false; - } - return true; -} - -/* Walk the whole source tree once collecting only destination-relative wire - paths, loading and sending nothing. --delete-before/--delete-during need the - complete keep-set manifest before the first data byte, so it is built by a - dedicated pre-scan pass and transmitted early; the data pass then re-scans - with a fresh scanner. A source I/O error is fatal unless the options carry - --ignore-errors, in which case the scan continues past the unreadable - directory and *io_error_out reports it (the caller still performs the - deletion but reports the run as errored). */ -static bool scan_paths_only(const Config* config, const ScannerOptions* options, - ArrayList* manifest, DeletePlanSender* plans, bool* io_error_out, - unsigned long long* non_dir_count_out) { - if (io_error_out) - *io_error_out = false; - if (non_dir_count_out) - *non_dir_count_out = 0; - ScannerOptions local = *options; - /* The pre-scan is a paths-only pass with no client output; it must not emit - --info=nonreg lines (the data pass does that once). */ - local.note_nonreg = false; - DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); - if (!scanner) - return false; - bool ok = true; - Chunk* chunk; - while ((chunk = directory_scanner_next(scanner)) != NULL) { - if (non_dir_count_out) { - for (int i = 0; i < chunk->element_count; i++) { - const File* f = chunk->items[i]; - if (f && !f->is_dir) - (*non_dir_count_out)++; - } - } - if (manifest && !add_chunk_to_manifest(manifest, chunk)) { - ok = false; - chunk_destroy(chunk); - break; - } - if (plans) { - for (int i = 0; i < chunk->element_count; i++) { - File* f = chunk->items[i]; - if (!f) - continue; - const char* path = file_wire_path(f); - if (!delete_plan_sender_add(plans, path, f->is_dir)) { - ok = false; - break; - } - } - if (!ok) { - chunk_destroy(chunk); - break; - } - } - chunk_destroy(chunk); - } - if (ok) { - /* Keep every traversed source directory, including empty ones, so a plan - no longer removes the destination directory itself. Their own plans are - emitted after the data stream (no file frame triggers them). */ - if (plans && options->plan_dirs) { - for (int i = 0; i < options->plan_dirs->size; i++) { - if (!delete_plan_sender_add(plans, (const char*)options->plan_dirs->items[i], true)) { - ok = false; - break; - } - } - } - } - if (ok && directory_scanner_failed(scanner)) - ok = false; - if (io_error_out) - *io_error_out = directory_scanner_had_io_error(scanner); - directory_scanner_destroy(scanner); - return ok; -} - -static int incremental_check(Client* client, File* file, const Config* config, - DeltaSignature** out_sig, unsigned long long* resume_offset) { +int incremental_check(Client* client, File* file, const Config* config, DeltaSignature** out_sig, + unsigned long long* resume_offset) { *out_sig = NULL; if (resume_offset) *resume_offset = 0; @@ -2141,249 +576,6 @@ static int send_append(const Client* client, File* file, Config* config, return ok ? 0 : -1; } -/* Server-contacting --dry-run. Connects to the configured remote/daemon and - * runs the normal per-file incremental decision WITHOUT transmitting any file - * data: the receiver (which also sees dry_run=true on the wire) answers - * STATUS_OK for an up-to-date file and STATUS_DRY_RUN_TRANSFER for a file it - * would otherwise write, mutating nothing on either side. The would-transfer - * set and the same trailer as the local dry-run are printed. A - * --compare-dest exact basis hit with no destination copy is reported as a - * skip by the receiver. - * - * Only regular files take the receiver-consulted check; directory / symlink / - * special / hard-link-sibling entries have no per-file content check, so they - * are reported conservatively as would-transfer and their frames are never - * sent (which is what keeps the receiver mutation-free). --delete* is - * deliberately NOT transmitted in dry-run, so no deletion can occur; the - * would-delete manifest report is a documented follow-up. - * - * Returns 0 on success, 1 on error. */ -static int send_dry_run_remote(Config* config) { - int from_skipped = 0; - ArrayList* missing_args = NULL; - if (config->delete_missing_args) { - missing_args = array_list_create(free); - if (!missing_args) - return 1; - } - if (!files_from_list_check(config, missing_args, &from_skipped)) { - if (missing_args) - array_list_delete(missing_args); - return 1; - } - if (missing_args) - array_list_delete(missing_args); - /* A live session may follow, so arm graceful abort handling. */ - client_set_abort_armed(true); - Client* client = connect_transfer_client(config); - if (!client) { - if (config->transport == TRANSPORT_TCP) - log_message(LOG_LEVEL_ERROR, "could not connect to server%s", - config->use_tls ? " via TLS" : ""); - client_set_abort_armed(false); - return 1; - } - ProtocolSession session; - protocol_session_init(&session, client->file_descriptor, client->file_descriptor); - protocol_session_set_io_timeout(&session, config->timeout); - protocol_session_set_ssl(&session, (SSL*)client->ssl); - protocol_session_bind(&session); - - int ret = 1; - time_t dry_start = time(NULL); - ReceiverStats dry_stats; - memset(&dry_stats, 0, sizeof(dry_stats)); - PreparedScanner prepared; - memset(&prepared, 0, sizeof(prepared)); - DirectoryScanner* scanner = NULL; - ArrayList* dry_manifest = NULL; - ArrayList* dry_dirs = NULL; - ArrayList* dry_excluded = NULL; - ArrayList* dry_size_skipped = NULL; - if (!config_send(client->file_descriptor, config)) - goto dry_fail; - receive_daemon_motd(client, config); - if (!prepare_scanner(config, 0, &prepared)) - goto dry_fail; - /* -n --delete: build the same keep-set manifest, protected prefixes, and - synchronized-directory scope a real run would send, so the receiver's - read-only extras walk enumerates exactly the deletions a real run makes. */ - if (config->use_delete) { - dry_manifest = array_list_create(free); - dry_dirs = array_list_create(free); - dry_size_skipped = array_list_create(free); - if (!dry_manifest || !dry_dirs || !dry_size_skipped) - goto dry_fail; - if (!config->delete_excluded) { - dry_excluded = array_list_create(free); - if (!dry_excluded) - goto dry_fail; - prepared.options.excluded_paths = dry_excluded; - } - prepared.options.size_skipped_paths = dry_size_skipped; - /* A --files-from subset confines the extras walk to the directories the - scan synchronized; a full recursive transfer marks the root itself. */ - if (config->files_from_set == NULL) { - char* root_marker = delete_scope_root_marker(config); - if (!root_marker || !array_list_add(dry_dirs, root_marker)) { - free(root_marker); - goto dry_fail; - } - } else { - prepared.options.synced_dirs = dry_dirs; - } - } - scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) - goto dry_fail; - - int file_count = 0; - unsigned long long total_bytes = 0; - char size_buffer[32]; - if (!config->quiet) - printf("Dry run: files to be transferred\n"); - Chunk* chunk; - while ((chunk = directory_scanner_next(scanner)) != NULL) { - if (dry_manifest && !add_chunk_to_manifest(dry_manifest, chunk)) { - chunk_destroy(chunk); - goto dry_fail; - } - for (int i = 0; i < chunk->element_count; i++) { - File* f = chunk->items[i]; - if (!f) - continue; - unsigned long long fsize = f->data ? f->data->size : 0; - bool would; - if (f->is_dir || f->is_symlink || f->is_special || - (f->link_group != 0 && !f->link_first && f->hardlink_target != NULL)) { - /* No receiver-side content check exists for these frame types; a real - run would (re)create them, so report would-transfer and send no - frame (the receiver must stay mutation-free). */ - would = true; - } else if (fsize > MAX_RECEIVE_WHOLE_FILE_SIZE && !config->use_incremental && - !config_has_basis(config)) { - /* A non-incremental run streams a >whole-file-limit source without the - STATUS_CHECK handshake, so no read-only receiver decision is possible - (and none is needed: a real run would transfer it). */ - would = true; - } else { - DeltaSignature* sig = NULL; - unsigned long long resume_offset = 0; - int rc = incremental_check(client, f, config, &sig, &resume_offset); - delta_signature_destroy(sig); - if (rc < 0) { - chunk_destroy(chunk); - goto dry_fail; - } - if (rc == 1) - continue; /* up to date; nothing to report */ - if (rc != 4) { - log_message(LOG_LEVEL_ERROR, "Unexpected receiver reply during dry-run"); - chunk_destroy(chunk); - goto dry_fail; - } - would = true; - } - if (would) { - if (!config->quiet) { - char* escaped_path = output_escape(file_wire_path(f), config->eight_bit_output); - if (!escaped_path) { - chunk_destroy(chunk); - goto dry_fail; - } - if (config->human_readable) - printf(" %s (%s)\n", escaped_path, - display_bytes(fsize, true, size_buffer, sizeof(size_buffer))); - else - printf(" %s (%llu bytes)\n", escaped_path, fsize); - free(escaped_path); - } - total_bytes += fsize; - file_count++; - } - } - chunk_destroy(chunk); - } - bool io_error = directory_scanner_had_io_error(scanner); - if (directory_scanner_failed(scanner)) - goto dry_fail; - if (io_error) - log_message(LOG_LEVEL_WARNING, "source scan hit an unreadable directory"); - /* Send the keep-set manifest (no data frames) so the receiver can enumerate - the destination extras; an early-timing delete ACKs before it will accept - the terminal FINISHED. */ - bool early_delete = config->use_delete && config_delete_timing_early(config); - if (dry_manifest) { - if (send_delete_manifest(client->file_descriptor, dry_manifest, dry_excluded, dry_size_skipped, - NULL, dry_dirs) != 0) - goto dry_fail; - if (early_delete) { - Status ack; - if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC, - DELETE_ACK_KEEPALIVE_SEC, client_abort_pending) || - ack != STATUS_OK) - goto dry_fail; - } - } - /* Terminate the stream so the receiver emits its success frame; no data frame - is ever sent in dry-run. */ - if (!send_status(client->file_descriptor, STATUS_FINISHED)) - goto dry_fail; - Status status; - if (!receive_status(client->file_descriptor, &status)) - goto dry_fail; - if (status == STATUS_STATS) { - ArrayList* would_delete = array_list_create(free); - if (!would_delete) - goto dry_fail; - if (!receive_stats_record(client->file_descriptor, &dry_stats, would_delete)) { - array_list_delete(would_delete); - goto dry_fail; - } - print_delete_reports(config, would_delete); - array_list_delete(would_delete); - if (!receive_status(client->file_descriptor, &status)) - goto dry_fail; - } - if (status != STATUS_OK) - goto dry_fail; - if (!config->quiet) { - if (config->human_readable) - printf("Total: %d files, %s\n", file_count, - display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer))); - else - printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB); - } - { - TransferStats dry_transfer; - memset(&dry_transfer, 0, sizeof(dry_transfer)); - dry_transfer.flist_reg = (unsigned long long)file_count; - dry_transfer.total_file_size = total_bytes; - dry_transfer.transferred_regular = (unsigned long long)file_count; - dry_transfer.transferred_file_size = total_bytes; - dry_transfer.literal_data = total_bytes; - report_transfer_stats(config, &dry_transfer, dry_start, &dry_stats); - } - ret = io_error ? 1 : 0; - -dry_fail: - if (dry_manifest) - array_list_delete(dry_manifest); - if (dry_dirs) - array_list_delete(dry_dirs); - if (dry_excluded) - array_list_delete(dry_excluded); - if (dry_size_skipped) - array_list_delete(dry_size_skipped); - if (scanner) - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); - disconnect_transfer_client(client); - protocol_session_unbind(); - client_set_abort_armed(false); - return ret; -} - // Send a single file directly (non-incremental path). static bool send_file_direct(File* file, int fd, bool use_metadata, int compression_level, const Config* config) { @@ -3185,133 +1377,104 @@ int apply_batch_to_dest(const Config* config, const char* batch_path, const char return rc; } -int send_files(Config* config) { - if (config->list_only) - return send_list_only(config); - if (config->dry_run) - return dry_run_targets_server(config) ? send_dry_run_remote(config) - : send_dry_run_manifest(config); - ArrayList* missing_args = NULL; - int skipped = 0; - if (config->delete_missing_args) { - missing_args = array_list_create(free); - if (!missing_args) - return 1; - } - if (!files_from_list_check(config, missing_args, &skipped)) { - if (missing_args) - array_list_delete(missing_args); - return 1; - } - - /* From here on a server session may be live, so Ctrl-C/SIGTERM should set the - abort flag (and be forwarded as STATUS_ABORT) instead of terminating. */ - client_set_abort_armed(true); - Client* client = connect_transfer_client(config); - if (!client) { - if (config->transport == TRANSPORT_TCP) - log_message(LOG_LEVEL_ERROR, "could not connect to server%s", - config->use_tls ? " via TLS" : ""); - if (missing_args) - array_list_delete(missing_args); - return 1; - } - ProtocolSession session; - protocol_session_init(&session, client->file_descriptor, client->file_descriptor); - protocol_session_set_io_timeout(&session, config->timeout); - protocol_session_set_ssl(&session, (SSL*)client->ssl); - protocol_session_bind(&session); - int ret = 1; - DirectoryScanner* scanner = NULL; - ArrayList* manifest = NULL; - DeletePlanSender* plan_sender = NULL; - ArrayList* remove_sources = NULL; - /* P7 Wave D: captured source directory times, transmitted in trailing - STATUS_DIR_TIMES frame(s) (only when metadata rides the wire). */ - ArrayList* dir_entries = NULL; - /* --stats directory accounting for the no-metadata (-r) case. */ +/* Resources and phase state threaded through the single-threaded send path. + * The send_files_* helpers populate it incrementally and send_files_cleanup + * releases every owned field; the members mirror the locals of the original + * monolithic send_files, so ownership and destruction order are unchanged. */ +typedef struct { + Client* client; + DirectoryScanner* scanner; + ArrayList* manifest; + DeletePlanSender* plan_sender; + ArrayList* remove_sources; + ArrayList* dir_entries; atomic_ullong dir_count; - atomic_init(&dir_count, 0); - /* Protected excluded prefixes (delete-excluded default protection). */ - ArrayList* excluded = NULL; - /* Size-pruned prefixes (always protected) and synchronized directories. */ - ArrayList* size_skipped = NULL; - ArrayList* synced_dirs = NULL; - /* Traversed source directories for the per-directory delete keep set. */ - ArrayList* plan_dirs = NULL; - bool delete_early = config->use_delete && config_delete_timing_early(config); - /* --delete-during/--delete-delay use per-directory plans for every transfer - shape. For -d/--dirs the generator records only the directories whose - direct children it actually enumerated, so the plan removes extras directly - inside a listed directory while an untraversed (kept) subdirectory is - shielded -- rsync's `-d DIR/ --delete`. */ - bool delete_per_dir = config->use_delete && config_delete_timing_per_dir(config); - bool send_failed = false; - bool had_scan_io = false; - unsigned long long per_dir_non_dir_count = 0; + ArrayList* excluded; + ArrayList* size_skipped; + ArrayList* synced_dirs; + ArrayList* plan_dirs; + ArrayList* missing_args; PreparedScanner prepared; - memset(&prepared, 0, sizeof(prepared)); + StopCondition stop; + TransferStats transfer_stats; + time_t start; + bool delete_early; + bool delete_per_dir; + bool had_scan_io; + bool scan_stopped_early; + unsigned long long per_dir_non_dir_count; +} SendFilesState; + +/* Post-connect setup: negotiate the protocol, consume the daemon MOTD, build + * the scanner and allocate the delete/keep-set/remove-source list containers. */ +static bool send_files_prepare(Config* config, SendFilesState* state) { + Client* client = state->client; if (!config_send(client->file_descriptor, config)) - goto send_fail; + return false; receive_daemon_motd(client, config); - if (!prepare_scanner(config, 0, &prepared)) - goto send_fail; + if (!prepare_scanner(config, 0, &state->prepared)) + return false; if (dir_metadata_should_capture(config)) { - dir_entries = array_list_create(file_destroy); - if (!dir_entries) - goto send_fail; - if (!append_implied_dir_times(config, dir_entries)) - goto send_fail; + state->dir_entries = array_list_create(file_destroy); + if (!state->dir_entries) + return false; + if (!append_implied_dir_times(config, state->dir_entries)) + return false; } if (config->remove_source_files) - remove_sources = array_list_create(source_file_destroy); - if (config->remove_source_files && !remove_sources) - goto send_fail; + state->remove_sources = array_list_create(source_file_destroy); + if (config->remove_source_files && !state->remove_sources) + return false; /* Unless --delete-excluded opts out, collect the paths the source scan prunes by user-selection rules so the receiver protects their destination mirrors from --delete (rsync's default). Only scans that build the keep-set get the sink attached (prescan for early timing, the streaming data pass otherwise). */ if (config->use_delete) { if (!config->delete_excluded) { - excluded = array_list_create(free); - if (!excluded) - goto send_fail; - prepared.options.excluded_paths = excluded; + state->excluded = array_list_create(free); + if (!state->excluded) + return false; + state->prepared.options.excluded_paths = state->excluded; } - size_skipped = array_list_create(free); - synced_dirs = array_list_create(free); - if (!size_skipped || !synced_dirs) - goto send_fail; - prepared.options.size_skipped_paths = size_skipped; + state->size_skipped = array_list_create(free); + state->synced_dirs = array_list_create(free); + if (!state->size_skipped || !state->synced_dirs) + return false; + state->prepared.options.size_skipped_paths = state->size_skipped; /* Only a --files-from subset confines the extras walk to the directories the scan synchronized; a full recursive transfer deletes throughout the receive root, so mark the root itself (the "." sentinel) and let the scanner record nothing extra. */ if (config->files_from_set == NULL) { char* root_marker = delete_scope_root_marker(config); - if (!root_marker || !array_list_add(synced_dirs, root_marker)) { + if (!root_marker || !array_list_add(state->synced_dirs, root_marker)) { free(root_marker); - goto send_fail; + return false; } } else { - prepared.options.synced_dirs = synced_dirs; + state->prepared.options.synced_dirs = state->synced_dirs; } } - /* The late-timing modes (--delete-after/--delete-commit) build the manifest - while streaming and send it after the last data frame. --delete-before - sends a whole-tree keep-set up front; --delete-during/--delete-delay build - the complete per-directory plan set up front (paths only) and transmit it - all before the first data frame, so a mid-transfer abort has already - applied every planned removal. */ - if (delete_early) { + return true; +} + +/* Delete-timing pre-pass. The late-timing modes (--delete-after/--delete-commit) + * build the manifest while streaming (handled by the run/finalize phases); + * --delete-before sends a whole-tree keep-set up front, and --delete-during/ + * --delete-delay build the complete per-directory plan set up front (paths only) + * and transmit it all before the first data frame, so a mid-transfer abort has + * already applied every planned removal. */ +static bool send_files_prepare_delete(Config* config, SendFilesState* state) { + Client* client = state->client; + if (state->delete_early) { /* Pass 1: collect the complete keep-set (paths only, no data loaded) and transmit it now, before any file data. The receiver removes extras and acks; the transfer aborts here if the deletion could not commit. */ ArrayList* early_manifest = array_list_create(free); if (!early_manifest) - goto send_fail; - bool prescan_ok = - scan_paths_only(config, &prepared.options, early_manifest, NULL, &had_scan_io, NULL); + return false; + bool prescan_ok = scan_paths_only(config, &state->prepared.options, early_manifest, NULL, + &state->had_scan_io, NULL); bool early_ok = false; bool skip_delete = false; if (prescan_ok) { @@ -3320,84 +1483,92 @@ int send_files(Config* config) { and an empty keep-set would delete the whole destination. Refuse to delete; the genuine-empty-source case has no io_error and still sends its (empty) keep-set. */ - if (had_scan_io && early_manifest->size == 0) { + if (state->had_scan_io && early_manifest->size == 0) { log_message(LOG_LEVEL_ERROR, "source scan hit an I/O error before finding any file; refusing to delete " "with an empty keep-set (--delete)"); prescan_ok = false; - } else if (!ignore_errors_allows_delete(config, had_scan_io)) { + } else if (!ignore_errors_allows_delete(config, state->had_scan_io)) { /* rsync default: an I/O error suppresses deletion unless --ignore-errors. Skip the manifest; the transfer still proceeds. */ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion"); skip_delete = true; } else { - early_ok = send_delete_manifest_early(client, early_manifest, excluded, size_skipped, - missing_args, synced_dirs); + early_ok = + send_delete_manifest_early(client, early_manifest, state->excluded, state->size_skipped, + state->missing_args, state->synced_dirs); } } array_list_delete(early_manifest); /* The keep-set (and its protected prefixes and synchronized directories) are already on the wire; the data pass must not append to those lists again. */ - prepared.options.excluded_paths = NULL; - prepared.options.size_skipped_paths = NULL; - prepared.options.synced_dirs = NULL; + state->prepared.options.excluded_paths = NULL; + state->prepared.options.size_skipped_paths = NULL; + state->prepared.options.synced_dirs = NULL; if (!prescan_ok || (!early_ok && !skip_delete)) - goto send_fail; - } else if (delete_per_dir) { + return false; + } else if (state->delete_per_dir) { /* --delete-during/--delete-delay: build one plan per source directory from a path-only pre-scan and transmit the COMPLETE plan set now, before any data, so every planned removal has already been applied when a later transfer phase fails -- exactly like rsync's generator, whose deletion list runs ahead of its throttled sender. A completed run is unaffected. */ - plan_sender = delete_plan_sender_create(); - plan_dirs = array_list_create(free); - if (!plan_sender || !plan_dirs) - goto send_fail; - prepared.options.plan_dirs = plan_dirs; - bool prescan_ok = scan_paths_only(config, &prepared.options, NULL, plan_sender, &had_scan_io, - &per_dir_non_dir_count); + state->plan_sender = delete_plan_sender_create(); + state->plan_dirs = array_list_create(free); + if (!state->plan_sender || !state->plan_dirs) + return false; + state->prepared.options.plan_dirs = state->plan_dirs; + bool prescan_ok = scan_paths_only(config, &state->prepared.options, NULL, state->plan_sender, + &state->had_scan_io, &state->per_dir_non_dir_count); bool plans_ok = false; bool skip_delete = false; if (prescan_ok) { - const char* walk_root = delete_plan_walk_root(config, synced_dirs); + const char* walk_root = delete_plan_walk_root(config, state->synced_dirs); const ArrayList* scope = - config->files_from_set ? synced_dirs : (walk_root ? synced_dirs : NULL); - delete_plan_sender_finalize(plan_sender, scope, walk_root); - delete_plan_sender_set_config(plan_sender, excluded, size_skipped, missing_args); - if (had_scan_io && delete_plan_sender_empty(plan_sender)) { + config->files_from_set ? state->synced_dirs : (walk_root ? state->synced_dirs : NULL); + delete_plan_sender_finalize(state->plan_sender, scope, walk_root); + delete_plan_sender_set_config(state->plan_sender, state->excluded, state->size_skipped, + state->missing_args); + if (state->had_scan_io && delete_plan_sender_empty(state->plan_sender)) { log_message(LOG_LEVEL_ERROR, "source scan hit an I/O error before finding any file; refusing to delete " "with an empty keep-set (--delete)"); prescan_ok = false; - } else if (!ignore_errors_allows_delete(config, had_scan_io)) { + } else if (!ignore_errors_allows_delete(config, state->had_scan_io)) { /* rsync default: an I/O error suppresses deletion unless --ignore-errors. Drop the plans; the transfer still proceeds. */ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion"); - delete_plan_sender_destroy(plan_sender); - plan_sender = NULL; - array_list_delete(plan_dirs); - plan_dirs = NULL; + delete_plan_sender_destroy(state->plan_sender); + state->plan_sender = NULL; + array_list_delete(state->plan_dirs); + state->plan_dirs = NULL; skip_delete = true; } else { - plans_ok = delete_plan_send_all(client->file_descriptor, plan_sender, plan_dirs) == 0; + plans_ok = delete_plan_send_all(client->file_descriptor, state->plan_sender, + state->plan_dirs) == 0; } } - prepared.options.excluded_paths = NULL; - prepared.options.size_skipped_paths = NULL; - prepared.options.synced_dirs = NULL; - prepared.options.plan_dirs = NULL; + state->prepared.options.excluded_paths = NULL; + state->prepared.options.size_skipped_paths = NULL; + state->prepared.options.synced_dirs = NULL; + state->prepared.options.plan_dirs = NULL; if (!prescan_ok || (!plans_ok && !skip_delete)) - goto send_fail; + return false; } else if (config->use_delete) { - manifest = array_list_create(free); - if (!manifest) - goto send_fail; + state->manifest = array_list_create(free); + if (!state->manifest) + return false; } - /* --progress/--info=progress: pre-count the file list for rsync's to-chk - denominator. When a --delete-during/--delete-delay pre-scan already ran, - reuse its traversed directory list instead of walking the tree again. */ + return true; +} + +/* Streaming run phase: pre-count progress, arm the stop deadline, scan the + * source and transmit every chunk. Returns false on a fatal error (the caller + * runs the shared cleanup). */ +static bool send_files_run(Config* config, SendFilesState* state) { + Client* client = state->client; if (progress_requested(config)) - client_progress_prepare(config, plan_dirs, per_dir_non_dir_count); + client_progress_prepare(config, state->plan_dirs, state->per_dir_non_dir_count); /* Phase 6: compute the client-only stop deadline once at transfer start. The early-delete pre-scan above deliberately ignores it so the keep-set (and its committed deletion) is always complete and correct. */ @@ -3406,27 +1577,27 @@ int send_files(Config* config) { now_mono.tv_sec = 0; now_mono.tv_nsec = 0; } - StopCondition stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, - config->stop_at_set, config->stop_at, now_mono); - prepared.options.stop_condition = &stop; + state->stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, + config->stop_at_set, config->stop_at, now_mono); + state->prepared.options.stop_condition = &state->stop; /* The early-delete pre-scan above already ran; only the data pass should feed the directory-time list (otherwise every directory would be captured twice). */ - prepared.options.dir_entries = dir_entries; - prepared.options.dir_count = config->stats ? &dir_count : NULL; - scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options); - if (!scanner) - goto send_fail; + state->prepared.options.dir_entries = state->dir_entries; + state->prepared.options.dir_count = config->stats ? &state->dir_count : NULL; + state->scanner = + directory_scanner_create_with_options(config->send_directory, &state->prepared.options); + if (!state->scanner) + return false; Chunk* current_chunk; - TransferStats transfer_stats; - memset(&transfer_stats, 0, sizeof(transfer_stats)); - time_t start = time(NULL); + memset(&state->transfer_stats, 0, sizeof(state->transfer_stats)); + state->start = time(NULL); client_progress_begin(config); /* True when the stop deadline cut the scan short so the keep-set manifest is only a prefix of the source. */ - bool scan_stopped_early = false; - while ((current_chunk = directory_scanner_next(scanner)) != NULL) { + bool send_failed = false; + while ((current_chunk = directory_scanner_next(state->scanner)) != NULL) { /* Graceful abort (Ctrl-C/SIGTERM): notify the receiver and clean up. The session is active (config_send already succeeded); a send failure here is fine because the client is exiting anyway. */ @@ -3434,21 +1605,21 @@ int send_files(Config* config) { log_info_message(LOG_INFO_MISC, "Abort requested; sending STATUS_ABORT to server"); chunk_destroy(current_chunk); send_status(client->file_descriptor, STATUS_ABORT); - goto send_fail; + return false; } /* Phase 6: stop-elegantly at the next chunk boundary once the deadline has passed. The scanner may also have stopped early itself; either way the completion tail below keeps everything already sent. */ - if (stop_condition_reached(&stop)) { + if (stop_condition_reached(&state->stop)) { chunk_destroy(current_chunk); log_info_message(LOG_INFO_MISC, "Stop deadline reached; stopping transfer at the next chunk boundary"); - scan_stopped_early = true; + state->scan_stopped_early = true; break; } - if (manifest && !add_chunk_to_manifest(manifest, current_chunk)) { + if (state->manifest && !add_chunk_to_manifest(state->manifest, current_chunk)) { chunk_destroy(current_chunk); - goto send_fail; + return false; } if (!config->use_sendfile) { bool load_ok = true; @@ -3464,11 +1635,11 @@ int send_files(Config* config) { } if (!load_ok) { chunk_destroy(current_chunk); - goto send_fail; + return false; } } - if (send_chunk_with_removal(client, current_chunk, config, remove_sources, &transfer_stats) != - 0) { + if (send_chunk_with_removal(client, current_chunk, config, state->remove_sources, + &state->transfer_stats) != 0) { log_message(LOG_LEVEL_ERROR, "Failed to send chunk"); chunk_destroy(current_chunk); send_failed = true; @@ -3476,31 +1647,32 @@ int send_files(Config* config) { } chunk_destroy(current_chunk); } - if (send_failed) { - if (manifest) { - array_list_delete(manifest); - manifest = NULL; - } - goto send_fail; - } - if (directory_scanner_failed(scanner)) - goto send_fail; - if (directory_scanner_had_io_error(scanner)) - had_scan_io = true; + return !send_failed; +} + +/* Completion tail: send the late delete manifest and captured directory times, + * finalize the receiver handshake, remove transferred sources and report stats. + * Returns the rsync-compatible exit code. */ +static int send_files_finalize(Config* config, SendFilesState* state) { + Client* client = state->client; + if (directory_scanner_failed(state->scanner)) + return 1; + if (directory_scanner_had_io_error(state->scanner)) + state->had_scan_io = true; /* An abort that arrived after the last chunk must still stop the completion tail (manifest/finalize) rather than let it run to success. */ if (client_abort_pending()) { log_info_message(LOG_INFO_MISC, "Abort requested; sending STATUS_ABORT to server"); send_status(client->file_descriptor, STATUS_ABORT); - goto send_fail; + return 1; } /* Phase 6: the scanner may have stopped early (returning NULL without a failure) as soon as the deadline passed, so reflect that here too. A deadline that cut the scan short leaves an incomplete keep-set; transmitting it would make the receiver --delete the unscanned source mirrors (data loss), so the late delete manifest is suppressed below. */ - scan_stopped_early = scan_stopped_early || stop_condition_reached(&stop); - if (scan_stopped_early) { + state->scan_stopped_early = state->scan_stopped_early || stop_condition_reached(&state->stop); + if (state->scan_stopped_early) { if (config->use_delete || config->delete_missing_args) log_message(LOG_LEVEL_WARNING, "transfer stopped early (stop deadline); skipping --delete keep-set so " @@ -3508,25 +1680,25 @@ int send_files(Config* config) { else log_message(LOG_LEVEL_WARNING, "transfer stopped early (stop deadline)"); } else { - if (had_scan_io && manifest && manifest->size == 0) { + if (state->had_scan_io && state->manifest && state->manifest->size == 0) { /* A scan that hit an I/O error and produced no keep entries is ambiguous; an empty keep-set would delete the whole destination. Refuse to delete (see the early-timing comment above). */ log_message(LOG_LEVEL_ERROR, "source scan hit an I/O error before finding any file; refusing to delete with " "an empty keep-set (--delete)"); - goto send_fail; + return 1; } /* rsync default: a scan I/O error suppresses deletion unless --ignore-errors, even in the late (commit) modes. Drop the keep-set so the receiver removes nothing; the readable tree still transferred. */ - bool late_delete = - (manifest || config->delete_missing_args) && !delete_early && !delete_per_dir; - if (late_delete && !ignore_errors_allows_delete(config, had_scan_io)) { + bool late_delete = (state->manifest || config->delete_missing_args) && !state->delete_early && + !state->delete_per_dir; + if (late_delete && !ignore_errors_allows_delete(config, state->had_scan_io)) { log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion"); - if (manifest) { - array_list_delete(manifest); - manifest = NULL; + if (state->manifest) { + array_list_delete(state->manifest); + state->manifest = NULL; } } else if (late_delete) { /* Late (commit) ordering: all file data is out; transmit the manifest so @@ -3535,83 +1707,144 @@ int send_files(Config* config) { succeeds. In the early modes (--delete-before) and the per-directory modes the deletion already went out with the data, so nothing is re-sent here. */ - if (send_delete_manifest(client->file_descriptor, manifest, excluded, size_skipped, - missing_args, synced_dirs) != 0) { - if (manifest) { - array_list_delete(manifest); - manifest = NULL; + if (send_delete_manifest(client->file_descriptor, state->manifest, state->excluded, + state->size_skipped, state->missing_args, state->synced_dirs) != 0) { + if (state->manifest) { + array_list_delete(state->manifest); + state->manifest = NULL; } - goto send_fail; + return 1; } - if (manifest) { - array_list_delete(manifest); - manifest = NULL; + if (state->manifest) { + array_list_delete(state->manifest); + state->manifest = NULL; } } } /* P7 Wave D: every directory has now been traversed (or the scan stopped early), so transmit the captured directory times last. The receiver defers applying them until after its own deletion/publication phase. */ - if (!send_dir_times(client, config, dir_entries)) - goto send_fail; + if (!send_dir_times(client, config, state->dir_entries)) + return 1; bool delete_limit = false; ReceiverStats recv_stats; memset(&recv_stats, 0, sizeof(recv_stats)); - bool ok = finalize_transfer(client, config, remove_sources, &delete_limit, &recv_stats); + bool ok = finalize_transfer(client, config, state->remove_sources, &delete_limit, &recv_stats); if (!ok && config->use_delete) log_message(LOG_LEVEL_ERROR, "server reported a deletion failure (--delete); see the server log for the reason"); if (ok) - remove_transferred_sources(config, remove_sources); + remove_transferred_sources(config, state->remove_sources); /* A recursive -a scan has no directory entries in its chunks; account them from the scanner's captured directory list (present whenever a directory attribute is preserved, e.g. -a/-t/-p). The -d generator counts its explicit directory entries inline instead. */ - transfer_stats.flist_dir += dir_count_for_stats(config, dir_entries, &dir_count); - report_transfer_stats(config, &transfer_stats, start, &recv_stats); + state->transfer_stats.flist_dir += + dir_count_for_stats(config, state->dir_entries, &state->dir_count); + report_transfer_stats(config, &state->transfer_stats, state->start, &recv_stats); log_info_message(LOG_INFO_STATS, "Transfer summary: %llu files, %.1f MB", - transfer_stats.transferred_regular, - (double)transfer_stats.transferred_file_size / (double)BYTES_PER_MIB); + state->transfer_stats.transferred_regular, + (double)state->transfer_stats.transferred_file_size / (double)BYTES_PER_MIB); /* A skipped source entry (--ignore-errors past an unreadable directory, or a dereferenced symlink with no referent) makes rsync report a partial transfer (exit 23) even though the rest of the run succeeded. A --max-delete-capped commit is a successful transfer that rsync reports with exit code 25. */ if (!ok) - ret = 1; - else if (had_scan_io) - ret = 23; - else - ret = delete_limit ? 25 : 0; + return 1; + if (state->had_scan_io) + return 23; + return delete_limit ? 25 : 0; +} -send_fail: - /* Single cleanup path for all exits. The manifest is intentionally deleted - here even on success without --delete, fixing a pre-existing leak. */ - if (manifest) - array_list_delete(manifest); - if (plan_sender) - delete_plan_sender_destroy(plan_sender); - if (excluded) - array_list_delete(excluded); - if (size_skipped) - array_list_delete(size_skipped); - if (synced_dirs) - array_list_delete(synced_dirs); - if (plan_dirs) - array_list_delete(plan_dirs); - if (missing_args) - array_list_delete(missing_args); - if (remove_sources) - array_list_delete(remove_sources); - if (dir_entries) - array_list_delete(dir_entries); - if (scanner) - directory_scanner_destroy(scanner); - prepared_scanner_destroy(&prepared); +/* Single cleanup path for all exits. The manifest is intentionally deleted + * here even on success without --delete, fixing a pre-existing leak. */ +static void send_files_cleanup(SendFilesState* state) { + if (state->manifest) + array_list_delete(state->manifest); + if (state->plan_sender) + delete_plan_sender_destroy(state->plan_sender); + if (state->excluded) + array_list_delete(state->excluded); + if (state->size_skipped) + array_list_delete(state->size_skipped); + if (state->synced_dirs) + array_list_delete(state->synced_dirs); + if (state->plan_dirs) + array_list_delete(state->plan_dirs); + if (state->missing_args) + array_list_delete(state->missing_args); + if (state->remove_sources) + array_list_delete(state->remove_sources); + if (state->dir_entries) + array_list_delete(state->dir_entries); + if (state->scanner) + directory_scanner_destroy(state->scanner); + prepared_scanner_destroy(&state->prepared); client_progress_cleanup(); - disconnect_transfer_client(client); + disconnect_transfer_client(state->client); protocol_session_unbind(); client_set_abort_armed(false); +} + +int send_files(Config* config) { + if (config->list_only) + return send_list_only(config); + if (config->dry_run) + return dry_run_targets_server(config) ? send_dry_run_remote(config) + : send_dry_run_manifest(config); + SendFilesState state; + memset(&state, 0, sizeof(state)); + atomic_init(&state.dir_count, 0); + state.delete_early = config->use_delete && config_delete_timing_early(config); + /* --delete-during/--delete-delay use per-directory plans for every transfer + shape. For -d/--dirs the generator records only the directories whose + direct children it actually enumerated, so the plan removes extras directly + inside a listed directory while an untraversed (kept) subdirectory is + shielded -- rsync's `-d DIR/ --delete`. */ + state.delete_per_dir = config->use_delete && config_delete_timing_per_dir(config); + int skipped = 0; + if (config->delete_missing_args) { + state.missing_args = array_list_create(free); + if (!state.missing_args) + return 1; + } + if (!files_from_list_check(config, state.missing_args, &skipped)) { + if (state.missing_args) + array_list_delete(state.missing_args); + return 1; + } + + /* From here on a server session may be live, so Ctrl-C/SIGTERM should set the + abort flag (and be forwarded as STATUS_ABORT) instead of terminating. */ + client_set_abort_armed(true); + Client* client = connect_transfer_client(config); + if (!client) { + if (config->transport == TRANSPORT_TCP) + log_message(LOG_LEVEL_ERROR, "could not connect to server%s", + config->use_tls ? " via TLS" : ""); + if (state.missing_args) + array_list_delete(state.missing_args); + return 1; + } + state.client = client; + ProtocolSession session; + protocol_session_init(&session, client->file_descriptor, client->file_descriptor); + protocol_session_set_io_timeout(&session, config->timeout); + protocol_session_set_ssl(&session, (SSL*)client->ssl); + protocol_session_bind(&session); + + int ret = 1; + if (!send_files_prepare(config, &state)) + goto send_fail; + if (!send_files_prepare_delete(config, &state)) + goto send_fail; + if (!send_files_run(config, &state)) + goto send_fail; + ret = send_files_finalize(config, &state); + +send_fail: + send_files_cleanup(&state); return ret; } diff --git a/src/client/client_send_internal.h b/src/client/client_send_internal.h new file mode 100644 index 0000000..6d042b5 --- /dev/null +++ b/src/client/client_send_internal.h @@ -0,0 +1,91 @@ +#ifndef CLIENT_SEND_INTERNAL_H +#define CLIENT_SEND_INTERNAL_H + +/* Declarations shared between the client_send.c transfer orchestration and the + * reporting (client_report.c), scanner-preparation (client_scan.c) and + * manifest/list/dry-run (client_manifest.c) translation units that were split + * out of it. Nothing here is part of the public client_send.h facade. */ + +#include "array_list.h" +#include "client_send.h" +#include "config.h" +#include "delete_plan.h" +#include "delta.h" +#include "format.h" +#include "log.h" +#include "scanner.h" +#include +#include +#include +#include + +/* One mebibyte in bytes; the unit used by the --stats/--progress lines. + Always cast to double when dividing so the output stays fractional. */ +#define BYTES_PER_MIB (1024ULL * 1024ULL) + +/* Compiled scanner inputs that are shared read-only across scanner instances + * and, in -m mode, across worker threads. `base_filters` owns the compiled + * command-line + -C rules; the FileListSet allow-set lives in the Config. + * `hardlinks` owns the --hard-links/-H link-group detection table (NULL when + * off) and is shared (mutex-guarded) across every scanner/worker of one scan. */ +typedef struct { + ScannerOptions options; + FilterRuleList* base_filters; /* owned; may be NULL */ + HardLinkTable* hardlinks; /* owned; may be NULL */ + char* relative_prefix; /* owned -R prefix; may be NULL */ +} PreparedScanner; + +/* client_scan.c */ +bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out); +void prepared_scanner_destroy(PreparedScanner* prepared); +bool append_implied_dir_times(const Config* config, ArrayList* dir_entries); +char* delete_scope_root_marker(const Config* config); +const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs); +bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out); +bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest, + DeletePlanSender* plans, bool* io_error_out, + unsigned long long* non_dir_count_out); + +/* client_report.c */ +void log_server_rejection(const char* context); +const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer, + size_t buffer_size); +unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries, + atomic_ullong* counter); +void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start, + const ReceiverStats* recv); +void transfer_stats_note_entry(TransferStats* stats, const File* file); +void transfer_stats_note_transferred(TransferStats* stats, const File* file); +bool info_flag_enabled(const Config* config, LogInfoFlag flag); +void print_delete_reports(const Config* config, const ArrayList* paths); +const char* delete_display_path(const Config* config, const char* path); +bool progress_requested(const Config* config); +void client_progress_cleanup(void); +void client_progress_begin(const Config* config); +void client_progress_file(const Config* config, const File* file); +void client_progress_name(const Config* config, const File* file); +void client_progress_uptodate(const Config* config, const File* file); +void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, + unsigned long long plan_non_dir_count); +bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete); + +/* client_send.c */ +void receive_daemon_motd(Client* client, const Config* config); +Client* connect_transfer_client(const Config* config); +void disconnect_transfer_client(Client* client); +int incremental_check(Client* client, File* file, const Config* config, DeltaSignature** out_sig, + unsigned long long* resume_offset); + +/* client_manifest.c */ +bool dry_run_targets_server(const Config* config); +bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk); +int send_dry_run_manifest(const Config* config); +int send_list_only(const Config* config); +int send_dry_run_remote(Config* config); +int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs); +bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes, + ArrayList* size_skipped, ArrayList* missing_args, + ArrayList* synced_dirs); + +#endif -- 2.54.0 From a6659472feb3cdacd513829dec071e3a2d176385 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:38:05 +0200 Subject: [PATCH 35/68] refactor(scanner): split into filter/sequential/parallel TUs MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Move the filter rule-tree/context helpers and entry inspection into scanner_filter.c, the parallel scanner into scanner_parallel.c, and keep the sequential scanner in scanner.c. Shared internal declarations live in the new scanner_internal.h; scanner.h stays the public façade. Decompose directory_scanner_next into static helpers (skipped-entry, selection-protection, mount/-x, directory-finish and per-entry handlers) with no semantic change. --- CMakeLists.txt | 2 + src/client/scanner.c | 1929 ++++++--------------------------- src/client/scanner_filter.c | 671 ++++++++++++ src/client/scanner_internal.h | 107 ++ src/client/scanner_parallel.c | 703 ++++++++++++ 5 files changed, 1789 insertions(+), 1623 deletions(-) create mode 100644 src/client/scanner_filter.c create mode 100644 src/client/scanner_internal.h create mode 100644 src/client/scanner_parallel.c diff --git a/CMakeLists.txt b/CMakeLists.txt index a55d93e..dc99f62 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -139,6 +139,8 @@ set(CLIENT_CORE_SRCS src/client/client_send.c src/client/client_validation.c src/client/scanner.c + src/client/scanner_filter.c + src/client/scanner_parallel.c src/client/usage.c ) set(CLIENT_MAIN_SRCS src/client/client_cli.c) diff --git a/src/client/scanner.c b/src/client/scanner.c index 1adbab8..55f38f1 100644 --- a/src/client/scanner.c +++ b/src/client/scanner.c @@ -1,5 +1,6 @@ #include "log.h" #include "scanner.h" +#include "scanner_internal.h" #include "array_list.h" #include "chunk.h" #include "file.h" @@ -17,194 +18,6 @@ #include "xattr.h" -typedef struct { - char* path; - int depth; - FilterNode* context; /* inherited per-directory filter context */ -} DirEntry; - -/* A chain node: `own` holds the .rsync-filter rules of one directory, `parent` - * the context that directory inherited (nearest ancestor with a filter file). - * The chain for a directory's contents runs from that directory's own node up - * to the root; the command-line base rules are evaluated after the whole - * chain. */ -struct FilterNode { - FilterNode* parent; - FilterRuleList* own; -}; - -static void filter_node_destroy(void* item) { - if (item) { - FilterNode* node = (FilterNode*)item; - if (node->own) - filter_rule_list_free(node->own); - free(node); - } -} - -static FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) { - FilterNode* node = malloc(sizeof(FilterNode)); - if (!node) - return NULL; - node->parent = parent; - node->own = own; - return node; -} - -/* Evaluate a rule chain for one entry. rsync precedence, highest first: the - * innermost (current) directory's .rsync-filter rules, then each ancestor's, - * then the root's, and finally the command-line base rules (--filter/-C). The - * sender-side verdict decides whether the entry is hidden from the transfer; - * the receiver-side verdict decides whether its destination mirror is protected - * from --delete. Each side takes the FIRST matching rule independently. */ -typedef struct { - bool hide; /* sender-side exclude matched */ - bool protect; /* receiver-side exclude matched */ -} FilterOutcome; - -static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel, - const char* leaf, bool is_dir, FilterOutcome* out) { - memset(out, 0, sizeof(*out)); - bool sender_decided = false; - bool receiver_decided = false; - const FilterNode* n = node; - while (!sender_decided || !receiver_decided) { - const FilterRuleList* list = n ? n->own : base; - if (list) { - if (!sender_decided) { - FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER); - if (action != FILTER_ACTION_NONE) { - out->hide = action == FILTER_ACTION_EXCLUDE; - sender_decided = true; - } - } - if (!receiver_decided) { - FilterAction action = - filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER); - if (action != FILTER_ACTION_NONE) { - out->protect = action == FILTER_ACTION_PROTECT; - receiver_decided = true; - } - } - } - if (!n) - break; - n = n->parent; - } -} - -static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel, - const char* leaf, bool is_dir, bool exclude_filter_files, - bool* protect_out) { - /* -FF: per-directory .rsync-filter files are never transferred (single -F - transfers them, matching rsync). */ - if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) { - if (protect_out) - *protect_out = false; - return false; - } - FilterOutcome outcome; - chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome); - if (protect_out) - *protect_out = outcome.protect; - return !outcome.hide; -} - -static void dir_entry_destroy(void* item) { - if (item) { - DirEntry* de = (DirEntry*)item; - free(de->path); - free(de); - } -} - -static DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context) { - DirEntry* de = malloc(sizeof(DirEntry)); - if (!de) - return NULL; - de->path = str_dup(path); - if (!de->path) { - free(de); - return NULL; - } - de->depth = depth; - de->context = context; - return de; -} - -/* How rsync's readlink_stat()/generator resolves one source symlink. */ -typedef enum { - LINK_ACTION_SKIP, /* not transferred (no link option) */ - LINK_ACTION_SKIP_PROTECTED, /* ignored as unsafe by --safe-links; rsync keeps - it in the transfer, so its destination mirror - must be protected from --delete */ - LINK_ACTION_DEREF, /* follow the referent (--copy-links, an unsafe - target under --copy-unsafe-links, or -k dir) */ - LINK_ACTION_CARRY, /* transmit the link itself (-l) */ -} LinkAction; - -/* Apply rsync's symlink-resolution precedence to one S_ISLNK entry: - * --copy-links dereferences every symlink; - * --copy-unsafe-links dereferences only targets unsafe_symlink() flags; - * -k/--copy-dirlinks dereferences only a symlink whose referent is a dir; - * --safe-links (receiver-side in rsync; modelled here) ignores an unsafe - * target that would otherwise be carried; with --munge-links - * every stored target becomes absolute, so --safe-links then - * ignores every symlink, exactly as rsync documents; - * -l/--links carries the link. - * `link_rel` is the symlink's transfer-relative path (incl. name) and is used - * only for the lexical unsafe test. `target` receives the raw link value. */ -static LinkAction scanner_link_action(const ScannerOptions* options, const char* path, - const char* link_rel, char* target, size_t target_size) { - if (!options->follow_symlinks && !options->copy_links && !options->safe_links && - !options->copy_unsafe_links && !options->copy_dirlinks) - return LINK_ACTION_SKIP; - ssize_t length = readlink(path, target, target_size - 1); - if (length < 0) - return LINK_ACTION_SKIP; - target[length] = '\0'; - - bool unsafe = file_symlink_unsafe(target, link_rel); - if (options->copy_links || (options->copy_unsafe_links && unsafe)) - return LINK_ACTION_DEREF; - if (options->copy_dirlinks) { - struct stat ref; - if (stat(path, &ref) == 0 && S_ISDIR(ref.st_mode)) - return LINK_ACTION_DEREF; - } - if (options->safe_links && (unsafe || options->munge_links)) - return LINK_ACTION_SKIP_PROTECTED; - if (!options->follow_symlinks || target[0] == '\0') - return LINK_ACTION_SKIP; - return LINK_ACTION_CARRY; -} - -typedef struct { - char* path; - struct stat stats; - bool is_directory; - /* True when the entry should be carried through as a SYMLINK (is_symlink) - rather than a dereferenced file/directory. When true, `link_target` holds - the owned target string to transmit (sender-munged under --munge-links); - ownership transfers to the File built from this entry. */ - bool is_symlink; - char* link_target; - /* True when the entry was pruned by a user selection rule (--filter/-C/per-dir - rules or the --exclude/--include layer) rather than skipped for another - reason (unreadable, symlink policy, not applicable). */ - bool excluded; - /* True when the entry was skipped specifically by --max-size/--min-size. - Size pruning protects the destination mirror even under --delete-excluded, - so it is recorded into a separate sink from `excluded`. */ - bool size_excluded; - /* True when a symlink selected for dereferencing (-L/--copy-links or an - unsafe target under --copy-unsafe-links) had no usable referent (a broken - link or a stat() failure). rsync still reports this as a partial transfer - (exit 23) even though the entry is skipped, so the scanner records it as a - non-fatal I/O error. */ - bool referent_error; -} ScannerEntry; - /* One inspected directory entry buffered so the sequential scanner can emit the stream in rsync's flist order. `name` is the raw dirent name (owned here); `entry` is the scanner_inspect_entry() result whose path/link_target are owned @@ -216,522 +29,6 @@ typedef struct { int inspection; } SortedEntry; -/* --one-file-system (-x) decision. Only directories can carry a different - * device than their parent (mount points), so this is checked when a child - * directory is about to be descended into. */ -bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device) { - return one_file_system <= 0 || entry_device == root_device; -} - -/* Build a payload-less directory File carrying the captured metadata (when - * requested). Used by -x mount-point emission and --list-only directory - * entries. Returns NULL on allocation failure. */ -static File* scanner_build_dir_file(const char* path, const struct stat* stats, - const ScannerOptions* options) { - File* dir = file_create(path); - if (dir == NULL) - return NULL; - dir->is_dir = true; - if (options->use_metadata) { - dir->metadata = - file_metadata_create(dir->path, stats, options->preserve_atimes, options->preserve_crtimes); - if (!dir->metadata) { - file_destroy(dir); - return NULL; - } - } - return dir; -} - -/* Relative path of an on-disk path below `root`. The transfer root may be - * given with a trailing slash; the returned rel path never has one and is "" - * for the root itself. A root of "/" is handled (its children start at "/"). - * Exposed so tests can exercise the mapping directly. */ -char* scanner_path_relative(const char* root, const char* fs_path) { - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(root, fs_path, root_len) != 0) - return NULL; - if (root_len == 1 && root[0] == '/') { - if (fs_path[1] == '\0') - return str_dup(""); - return str_dup(fs_path + 1); - } - if (fs_path[root_len] == '\0') - return str_dup(""); - if (fs_path[root_len] != '/') - return NULL; - return str_dup(fs_path + root_len + 1); -} - -/* -R/--relative destination-relative prefix reconstructed from a source spec: - * everything after the first '.' path component (rsync's '/./' cut point), - * with leading/trailing slashes removed; or the whole spec (normalized) when - * there is no cut. Returns "" for the receive root. Exposed for tests. */ -char* scanner_relative_prefix(const char* spec) { - if (!spec || spec[0] == '\0') - return NULL; - const char* after = spec; - if (spec[0] == '.' && spec[1] == '/') { - after = spec + 2; - } else { - const char* cut = strstr(spec, "/./"); - if (cut) - after = cut + 3; - } - size_t cap = strlen(spec) + 1; - char* out = malloc(cap); - if (!out) - return NULL; - size_t len = 0; - for (const char* s = after; *s;) { - while (*s == '/') - s++; - const char* comp = s; - while (*s && *s != '/') - s++; - size_t clen = (size_t)(s - comp); - if (clen == 0 || (clen == 1 && comp[0] == '.')) - continue; - if (len) - out[len++] = '/'; - memcpy(out + len, comp, clen); - len += clen; - } - out[len] = '\0'; - return out; -} - -/* Relative path of a child entry below the current directory. */ -static char* child_rel_path(const char* parent_rel, const char* name) { - if (!parent_rel || parent_rel[0] == '\0') - return str_dup(name); - return path_cat(parent_rel, name); -} - -/* Destination-relative wire path for an entry under an -R prefix. */ -static char* scanner_prefix_send_path(const char* prefix, const char* rel) { - if (prefix[0] == '\0') - return str_dup(rel); - if (rel[0] == '\0') - return str_dup(prefix); - return path_cat(prefix, rel); -} - -/* Apply the --files-from allow-set and the filter layer to one entry. On - * return `*protect_out` is true when a receiver-side rule protects the entry's - * destination mirror from deletion. */ -static bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, - const FilterNode* node, const char* rel, const char* leaf, - bool is_dir, bool per_dir_filters, bool exclude_filter_files, - bool* protect_out) { - if (protect_out) - *protect_out = false; - if (file_list && !file_list_affects(file_list, rel)) - return false; - if (base || per_dir_filters) - return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out); - return true; -} - -/* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to - * read xattrs is non-fatal: the file is transferred without them. */ -static void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) { - if (!scanner || !file || !(scanner->options.preserve_xattrs || scanner->options.preserve_acls)) - return; - file->xattrs = xattr_capture_path(file->path, scanner->options.preserve_acls); -} - -/* Apply --hard-links (-H) detection to one regular File. On a sibling (a - * later member of an already-seen source inode) the File keeps the group id - * and the first member's wire path but carries NO data payload (size 0); the - * first member is left untouched (data present, link_first). Allocation - * failure is fatal: the scanner is marked failed. */ -static void scanner_assign_hardlink(DirectoryScanner* scanner, HardLinkTable* table, File* file, - const struct stat* stats) { - if (!table || !file || !stats) - return; - int gid; - bool is_first; - char* first_path = NULL; - if (!hardlink_table_assign(table, file_wire_path(file), stats->st_dev, stats->st_ino, &gid, - &is_first, &first_path)) { - if (scanner) - scanner->failed = true; - return; - } - file->link_group = gid; - file->link_first = is_first; - if (!is_first) { - file->hardlink_target = first_path; - file->data->size = 0; - } else { - free(first_path); - } -} - -/* Phase 4 special/devices decision for one non-regular entry, matching rsync: - - a char/block device is RECREATED as a node under -D/--devices, unless - --copy-devices asks for its content to be copied into a regular file; - - a FIFO/socket is RECREATED under --specials; - - when the matching flag is absent the entry is SKIPPED ("skipping - non-regular file"), exactly like rsync's default, instead of being - silently copied as a zero-length regular file; - - anything else (regular/directory) is left to the normal data path. */ -typedef enum { - SCANNER_SPECIAL_REGULAR, /* ordinary file: transfer content */ - SCANNER_SPECIAL_RECREATE, /* is_special node to recreate on the receiver */ - SCANNER_SPECIAL_SKIP, /* non-regular entry not requested: skip */ -} ScannerSpecial; - -static ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials, - bool copy_devices, File* file, - const struct stat* stats) { - if (!file || !stats) - return SCANNER_SPECIAL_REGULAR; - bool is_device = S_ISCHR(stats->st_mode) || S_ISBLK(stats->st_mode); - bool is_fifo = S_ISFIFO(stats->st_mode); - bool is_socket = S_ISSOCK(stats->st_mode); - if (!is_device && !is_fifo && !is_socket) - return SCANNER_SPECIAL_REGULAR; - if (is_device && copy_devices) - return SCANNER_SPECIAL_REGULAR; /* copy device content as a regular file */ - bool preserve = is_device ? preserve_devices : preserve_specials; - if (!preserve) - return SCANNER_SPECIAL_SKIP; - file->is_special = true; - file->data->size = 0; - file->data->data = NULL; - if (is_device) { - file->rdev_major = (int32_t)major(stats->st_rdev); - file->rdev_minor = (int32_t)minor(stats->st_rdev); - } - return SCANNER_SPECIAL_RECREATE; -} - -/* Append `rel` to the caller's exclusion sink, taking `mtx` when shared across - parallel worker threads. Returns false on allocation failure (list left - unchanged). */ -static bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) { - if (!list) - return true; - char* dup = str_dup(rel); - if (!dup) - return false; - if (mtx) - mtx_lock(mtx); - bool ok = array_list_add(list, dup); - if (mtx) - mtx_unlock(mtx); - if (!ok) - free(dup); - return ok; -} - -/* Record one pruned filesystem path in a delete-protection sink. The stored - form is the entry's wire/destination-relative path (a single leading '/' - removed, exactly how manifest keep entries are stored), so the receiver's - walker prefixes match the destination layout. An allocation failure is a - fatal scan error. */ -static void scanner_record_protected(DirectoryScanner* scanner, const char* fs_path, - ArrayList* sink) { - if (!sink || !fs_path) - return; - const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path; - if (!excluded_sink_append(sink, scanner->options.excluded_mutex, rel)) - scanner->failed = true; -} - -/* rsync's `--info=nonreg` line for a non-regular entry that is not being - * preserved: `skipping non-regular file "NAME"`. The name is the path relative - * to the transfer root, so it matches rsync's displayed name. */ -static void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path) { - if (!options || !options->note_nonreg || !fs_path) - return; - const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); - char* escaped = output_escape(rel, options->eight_bit_output); - printf("skipping non-regular file \"%s\"\n", escaped ? escaped : rel); - free(escaped); - fflush(stdout); -} - -/* rsync 3.4.1's `--info=mount` line, emitted when `-xx` drops a mount-point - * directory: `[sender] skipping mount-point dir NAME` (the client is the - * sender). Plain `-x` keeps the empty directory and prints nothing, matching - * rsync. */ -static void scanner_note_mount(const ScannerOptions* options, const char* fs_path) { - if (!options || !options->note_mount || !fs_path) - return; - const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); - char* escaped = output_escape(rel, options->eight_bit_output); - printf("[sender] skipping mount-point dir %s\n", escaped ? escaped : rel); - free(escaped); - fflush(stdout); -} - -/* --debug=filter: a selection/filter decision dropped an entry. */ -static void scanner_note_filter(const ScannerOptions* options, const char* name) { - if (!options || !log_debug_enabled(LOG_DEBUG_FILTER) || !name) - return; - log_debug_message(LOG_DEBUG_FILTER, "filter: excluded %s", name); -} - -/* Account for a directory that will not be represented by an inline directory - * entry. Paired with scanner_dir_count_uncount for empty directories that are - * emitted inline, so every traversed directory is counted exactly once. */ -static void scanner_dir_count_count(const ScannerOptions* options) { - if (options && options->dir_count) - atomic_fetch_add(options->dir_count, 1); -} - -static void scanner_dir_count_uncount(const ScannerOptions* options) { - if (options && options->dir_count) - atomic_fetch_sub(options->dir_count, 1); -} - -/* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */ -static void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) { - scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths); -} - -/* A --max-size/--min-size prune (always protected, even under --delete-excluded). */ -static void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path) { - scanner_record_protected(scanner, fs_path, scanner->options.size_skipped_paths); -} - -/* Record a directory the scan synchronized. `fs_path` is its absolute path and - `rel` its path relative to the transfer root ("" for the root); the stored - form matches the wire layout (the bare relative path in -R+--files-from, else - the source path with a leading '/' removed, with "." for the receive root). - Returns false on allocation failure. */ -static bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, - const char* rel, bool relative_mode) { - if (!options->synced_dirs && !options->plan_dirs) - return true; - if (!file_list_dir_in_scope(options->file_list, rel)) - return true; - char* prefixed = NULL; - const char* dest; - if (relative_mode) { - dest = rel; - } else if (options->relative_prefix) { - prefixed = scanner_prefix_send_path(options->relative_prefix, rel); - if (!prefixed) - return false; - dest = prefixed; - } else { - dest = fs_path; - } - if (dest[0] == '/') - dest++; - if (dest[0] == '\0') - dest = "."; - bool ok = true; - if (options->synced_dirs) - ok = excluded_sink_append(options->synced_dirs, options->excluded_mutex, dest); - /* The delete-plan keep set needs an entry for every traversed source - directory, including empty ones, so its destination mirror is kept rather - than deleted as an extra; the receive root (".") is implicit. */ - if (ok && options->plan_dirs && strcmp(dest, ".") != 0) - ok = excluded_sink_append(options->plan_dirs, options->excluded_mutex, dest); - free(prefixed); - return ok; -} - -/* Read every per-directory filter file that applies to `dir_path` (its - * .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a - * fresh list. Returns NULL on allocation/parse failure (message in `err`); - * returns an empty list (and *any_exists=false) when no file exists. */ -static FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path, - const char* rel, bool* any_exists, char* err, - size_t err_size) { - if (err && err_size > 0) - err[0] = '\0'; - const FilterRuleList* base = options->base_filters; - bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0); - if (any_exists) - *any_exists = false; - if (!have_names) - return NULL; - FilterRuleList* own = filter_rule_list_create(); - if (!own) { - snprintf(err, err_size, "memory allocation failed"); - return NULL; - } - FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false}; - bool exists = false; - if (options->per_dir_filters) { - if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size)) - goto fail; - if (exists && any_exists) - *any_exists = true; - } - if (base) { - for (int i = 0; i < base->dir_merge_count; i++) { - if (!filter_file_append(own, dir_path, base->dir_merge_names[i], rel, &opts, &exists, err, - err_size)) - goto fail; - if (exists && any_exists) - *any_exists = true; - } - } - return own; -fail: - filter_rule_list_free(own); - return NULL; -} - -/* Merge the open directory's own per-directory filter files (the default - * .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the - * base rule list) into the inherited context, returning the context used for - * this directory's entries. On a parse error the scanner is marked failed. - * Returns 0 on success, -1 on failure. */ -static int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { - char err[256]; - bool any_exists = false; - FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path, - scanner->current_rel ? scanner->current_rel : "", - &any_exists, err, sizeof(err)); - if (!own) { - /* read_dir_filters() leaves `err` set on a parse/allocation failure even - when an earlier merge file in the same directory existed (any_exists true); - key off the error text rather than any_exists so an invalid per-directory - filter file can never be silently ignored. */ - if (err[0] == '\0') { - scanner->current_node = (FilterNode*)inherited; - return 0; - } - char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", - escaped_path ? escaped_path : "", err); - free(escaped_path); - scanner->failed = true; - return -1; - } - if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { - FilterNode* node = filter_node_alloc((FilterNode*)inherited, own); - if (!node || !array_list_add(scanner->filter_nodes, node)) { - filter_node_destroy(node); - scanner->failed = true; - return -1; - } - scanner->current_node = node; - } else { - filter_rule_list_free(own); - scanner->current_node = (FilterNode*)inherited; - } - return 0; -} - -/* Inspect symlinks, resolve the entry type, and apply file filters once for both scanners. - * `link_rel` is the entry's path relative to the transfer root (including its - * name), used for the lexical rsync unsafe-symlink test. */ -static int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir, - const char* link_rel, const char* name, ScannerEntry* entry) { - entry->excluded = false; - entry->size_excluded = false; - entry->referent_error = false; - entry->is_symlink = false; - entry->link_target = NULL; - entry->path = path_cat(containing_dir, name); - if (!entry->path) - return -1; - - struct stat link_stats; - if (lstat(entry->path, &link_stats) != 0) { - free(entry->path); - return 0; - } - if (!S_ISLNK(link_stats.st_mode)) - goto regular; - - char link_target[4096]; - switch (scanner_link_action(options, entry->path, link_rel, link_target, sizeof(link_target))) { - case LINK_ACTION_SKIP: - goto skip; - case LINK_ACTION_SKIP_PROTECTED: - /* --safe-links ignored the link, but rsync still counts it as present in - the transfer, so its destination mirror survives --delete. Record it as - an excluded path (the same delete-protection channel as a filter prune). */ - entry->excluded = true; - goto skip; - case LINK_ACTION_DEREF: - if (stat(entry->path, &entry->stats) != 0) { - /* rsync reports "symlink has no referent" and continues with a partial - transfer (exit 23); record the error so the run exits 23 too. */ - char* escaped = output_escape(entry->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "symlink has no referent: %s", - escaped ? escaped : ""); - free(escaped); - entry->referent_error = true; - goto skip; - } - entry->is_directory = S_ISDIR(entry->stats.st_mode); - if (entry->is_directory) - return 1; - goto apply_filters; - case LINK_ACTION_CARRY: - break; - } - - /* Carry the link as a symlink. --munge-links is applied by the RECEIVER (it - prefixes every stored target with /rsyncd-munged/); when the SOURCE already - holds a munged value the sender strips it so the receiver re-munges a clean - target, round-tripping a munged tree exactly like rsync. */ - entry->is_symlink = true; - entry->stats = link_stats; - entry->is_directory = false; - entry->link_target = str_dup(link_target); - if (!entry->link_target) - goto skip; - if (options->munge_links) - file_symlink_unmunge(entry->link_target); - goto apply_filters; - -regular: - /* Not a symlink: the lstat() above already described this entry, and lstat - and stat are identical for every non-symlink, so reuse that result instead - of issuing a redundant stat() on the scanner hot path. stat() is still - used on the dereference paths above/below for actual symlinks (copy-links, - safe/copy-unsafe links, and -k symlinks-to-directories). */ - entry->stats = link_stats; - entry->is_directory = S_ISDIR(link_stats.st_mode); - if (entry->is_directory) - return 1; - -apply_filters: - for (int i = 0; i < options->exclude_count; i++) - if (glob_match(options->exclude_patterns[i], name)) { - entry->excluded = true; - goto skip; - } - if (options->include_count > 0) { - bool included = false; - for (int i = 0; i < options->include_count; i++) - if (glob_match(options->include_patterns[i], name)) - included = true; - if (!included) { - entry->excluded = true; - goto skip; - } - } - if ((options->max_size > 0 && (unsigned long long)entry->stats.st_size > options->max_size) || - (options->min_size > 0 && (unsigned long long)entry->stats.st_size < options->min_size)) { - entry->excluded = true; - entry->size_excluded = true; - goto skip; - } - return 1; - -skip: - free(entry->path); - entry->path = NULL; - free(entry->link_target); - entry->link_target = NULL; - return 0; -} - static void sorted_entry_destroy(void* item) { SortedEntry* se = (SortedEntry*)item; if (!se) @@ -1012,12 +309,11 @@ static Chunk* chunk_data_to_chunk(ArrayList* chunk_data) { * append for the parallel scanner's shared workers. An unstattable or * non-directory path is silently skipped (the transfer is unaffected); an * allocation failure is fatal and reported to the caller. */ -static bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path, - const char* fs_path, bool relative_mode, - const char* relative_prefix, bool preserve_atimes, - bool preserve_crtimes, bool preserve_xattrs, - bool preserve_acls, bool no_implied_dirs, - const FileListSet* file_list) { +bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path, + const char* fs_path, bool relative_mode, const char* relative_prefix, + bool preserve_atimes, bool preserve_crtimes, bool preserve_xattrs, + bool preserve_acls, bool no_implied_dirs, + const FileListSet* file_list) { if (!dir_entries || !root_path || !fs_path) return true; struct stat st; @@ -1588,6 +884,296 @@ static Chunk* directory_scanner_next_dirs(DirectoryScanner* scanner) { return dirs_flush_batch(scanner); } +/* Result of processing one inspected entry inside directory_scanner_next(). */ +typedef enum { + SCANNER_ACTION_CONTINUE, /* advance to the next buffered entry */ + SCANNER_ACTION_BREAK, /* stop the scan loop (failure recorded) */ + SCANNER_ACTION_CHUNK, /* return the Chunk produced in *out_chunk */ +} ScannerAction; + +/* Reconstruct the delete-protection path for a skipped (inspection == 0) entry + * whose destination mirror must be protected. */ +static char* scanner_entry_protected_path(DirectoryScanner* scanner, const char* name) { + if (scanner->relative_mode) + return child_rel_path(scanner->current_rel, name); + if (scanner->options.relative_prefix) { + char* relc = child_rel_path(scanner->current_rel, name); + char* prefixed = relc ? scanner_prefix_send_path(scanner->options.relative_prefix, relc) : NULL; + free(relc); + return prefixed; + } + return path_cat(scanner->current_path, name); +} + +/* Handle a buffered entry that scanner_inspect_entry() skipped (inspection == + * 0): record a partial-transfer I/O error and protect the destination mirror + * of a user-selection or size prune. Returns 0 to continue, -1 on failure. */ +static int scanner_handle_skipped_entry(DirectoryScanner* scanner, const ScannerEntry* inspected, + const char* name) { + /* A dereferenced symlink with no referent is a partial-transfer error + (rsync exit 23): record it as a non-fatal scan I/O error. */ + if (inspected->referent_error) + scanner->io_error = true; + /* A user-selection exclude protects its destination mirror from --delete + unless --delete-excluded; a size prune is always protected. Other + skips (unreadable, symlink policy) protect nothing. Under -R + + --files-from the protected prefix must be the entry's bare relative + wire path, not its source path (which would not match the destination + layout and would leave the mirror deletable). */ + if (inspected->excluded) { + char* protected_path = scanner_entry_protected_path(scanner, name); + if (!protected_path) { + scanner->failed = true; + return -1; + } + if (inspected->size_excluded) + scanner_record_size_skipped(scanner, protected_path); + else + scanner_record_excluded(scanner, protected_path); + free(protected_path); + } + return 0; +} + +/* Record the delete-protection prefix for an entry dropped by the --files-from + * allow-set or a filter rule (sender-hide or receiver-protect). Returns 0 on + * success, -1 on allocation failure (caller reports it). */ +static int scanner_record_selection_protection(DirectoryScanner* scanner, bool protect, + bool passes_selection, const char* rel, + const char* cur_path) { + if (passes_selection && !protect) + return 0; + /* --files-from subset pruning is not a filter exclusion: its delete + semantics stay keep-set-only (an unlisted source path is treated as + absent, so its destination mirror is a deletable extra). A rule-based + exclusion is recorded as a protected prefix. -R + --files-from bare + wire paths are never recorded (see ScannerOptions.excluded_paths). */ + bool files_from_prune = + scanner->options.file_list && !file_list_affects(scanner->options.file_list, rel); + if (protect && scanner->relative_mode) { + /* -R + --files-from: the destination/wire path is the bare relative + name, so the protected mirror prefix must be `rel` (not the source + path) for the delete walker to match it. */ + scanner_record_excluded(scanner, rel); + } else if (!files_from_prune && !scanner->relative_mode) { + if (scanner->options.relative_prefix) { + char* wrel = scanner_prefix_send_path(scanner->options.relative_prefix, rel); + if (!wrel) + return -1; + scanner_record_excluded(scanner, wrel); + free(wrel); + } else { + scanner_record_excluded(scanner, cur_path); + } + } + return 0; +} + +/* -x/--one-file-system handling for a directory entry: 1 when the entry was + * fully handled (caller continues), 0 when it is on the same filesystem as the + * root (caller descends), -1 on a fatal allocation failure. */ +static int scanner_handle_mount_dir(DirectoryScanner* scanner, ArrayList* chunk_data, + const char* cur_path, const struct stat* stats) { + if (scanner_same_filesystem(scanner->options.one_file_system, scanner->root_dev, stats->st_dev)) + return 0; + if (scanner->options.one_file_system > 1) { + /* rsync's -xx drops the mount-point directory entirely (the plain -x + path below keeps it as an empty directory) and prints the + --info=mount line when that category is enabled. */ + scanner_note_mount(&scanner->options, cur_path); + return 1; + } + /* rsync's -x/--one-file-system emits the mount-point directory entry + itself (so the destination gets an empty directory) but does NOT + descend into it. Build a payload-less directory File and hand it to + the caller; never enqueue it for traversal. */ + File* mount = scanner_build_dir_file(cur_path, stats, &scanner->options); + if (mount == NULL || !array_list_add(chunk_data, mount)) { + file_destroy(mount); + scanner->failed = true; + return -1; + } + scanner->current_dir_produced = true; + return 1; +} + +/* Finish the open directory (exhausted entries): emit an empty-directory entry + * when appropriate, push its pending children and reset the per-directory + * state. Returns false when the scanner failed. */ +static bool scanner_finish_current_directory(DirectoryScanner* scanner, ArrayList* chunk_data) { + /* The directory is exhausted: if nothing was transferred or descended + from it, recreate it at the destination as an explicit entry. */ + if (scanner->options.emit_empty_dirs && !scanner->current_dir_produced && + !scanner->options.prune_empty_dirs && !scanner->options.list_dirs && + scanner->options.file_list == NULL) { + if (!scanner_emit_empty_dir(scanner, chunk_data)) + scanner->failed = true; + } + scanner_push_pending_dirs(scanner); + closedir(scanner->current_dir); + scanner->current_dir = NULL; + free(scanner->current_path); + scanner->current_path = NULL; + scanner_free_sorted(scanner); + return !scanner->failed; +} + +/* Handle one kept buffered entry (inspection == 1): selection recording, + * directory descent and regular-file emission. `*chunk_data_size` tracks the + * accumulated payload so a chunk is cut at the same point as before. */ +static ScannerAction scanner_process_entry(DirectoryScanner* scanner, ArrayList* chunk_data, + unsigned long long* chunk_data_size, SortedEntry* sorted, + Chunk** out_chunk) { + const char* name = sorted->name; + ScannerEntry* inspected = &sorted->entry; + char* cur_path = inspected->path; + struct stat stats = inspected->stats; + + /* --files-from allow-set and the filter layer apply to files and to + * directories (an excluded directory is not descended into). */ + bool is_dir = inspected->is_directory; + char* rel = child_rel_path(scanner->current_rel, name); + if (!rel) { + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + bool protect = false; + bool passes_selection = entry_passes_selection( + scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, name, + is_dir, scanner->options.per_dir_filters, scanner->options.exclude_per_dir_filter_files, + &protect); + /* A sender-side hide leaves the entry out of the transfer; an independent + receiver-side protect rule keeps a transferred entry's destination mirror + from being deleted. Both are recorded in the same protection set. */ + if (!passes_selection || protect) { + if (scanner_record_selection_protection(scanner, protect, passes_selection, rel, cur_path) != + 0) { + free(rel); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + } + /* With -R the wire/destination path is a reconstructed relative path, not + the source path; keep `rel` alive to build it for a transferred file. */ + bool needs_rel = scanner->relative_mode || scanner->options.relative_prefix != NULL; + char* rel_copy = needs_rel ? str_dup(rel) : NULL; + free(rel); + if (rel_copy == NULL && needs_rel) { + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + if (!passes_selection) { + scanner_note_filter(&scanner->options, name); + free(rel_copy); + return SCANNER_ACTION_CONTINUE; + } + + if (is_dir) { + free(rel_copy); + int mount = scanner_handle_mount_dir(scanner, chunk_data, cur_path, &stats); + if (mount < 0) + return SCANNER_ACTION_BREAK; + if (mount > 0) + return SCANNER_ACTION_CONTINUE; + /* --list-only: list directory entries too (rsync prints them), even + though a real transfer never sends them explicitly. */ + if (scanner->options.list_dirs) { + File* dir = scanner_build_dir_file(cur_path, &stats, &scanner->options); + if (dir == NULL || !array_list_add(chunk_data, dir)) { + file_destroy(dir); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + } + scanner->current_dir_produced = true; + int next_depth = scanner->current_depth + 1; + if (scanner->options.max_depth <= 0 || next_depth < scanner->options.max_depth) { + DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node); + if (!de || !array_list_add((ArrayList*)scanner->pending_dirs, de)) { + dir_entry_destroy(de); + scanner->failed = true; + } + } + return SCANNER_ACTION_CONTINUE; + } + + if (scanner->options.max_depth > 0 && scanner->current_depth + 1 > scanner->options.max_depth) { + free(rel_copy); + return SCANNER_ACTION_CONTINUE; + } + File* file = file_create(cur_path); + if (file == NULL) { + free(rel_copy); + free(inspected->link_target); + inspected->link_target = NULL; + scanner->failed = true; + return SCANNER_ACTION_CONTINUE; + } + if (inspected->is_symlink) { + file->is_symlink = true; + file->symlink_target = inspected->link_target; + inspected->link_target = NULL; + } else { + file->data->size = stats.st_size; + } + if (scanner->relative_mode) { + file->send_path = rel_copy; + rel_copy = NULL; + } else if (scanner->options.relative_prefix) { + file->send_path = scanner_prefix_send_path(scanner->options.relative_prefix, rel_copy); + free(rel_copy); + rel_copy = NULL; + if (!file->send_path) { + file_destroy(file); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + } + /* --devices/--specials: a device/FIFO/socket entry marked for preservation + becomes a node to recreate (is_special, no data, rdev captured); an + unrequested non-regular entry is skipped (rsync default). */ + ScannerSpecial special = + scanner_prepare_special(scanner->options.preserve_devices, scanner->options.preserve_specials, + scanner->options.copy_devices, file, &stats); + if (special == SCANNER_SPECIAL_SKIP) { + scanner_note_nonreg(&scanner->options, file->path); + free(rel_copy); + file_destroy(file); + return SCANNER_ACTION_CONTINUE; + } + if (scanner->options.hardlinks && S_ISREG(stats.st_mode)) + scanner_assign_hardlink(scanner, scanner->options.hardlinks, file, &stats); + if (scanner->options.use_metadata) + file->metadata = file_metadata_create(file->path, &stats, scanner->options.preserve_atimes, + scanner->options.preserve_crtimes); + if (scanner->options.use_metadata && !file->metadata) { + free(rel_copy); + file_destroy(file); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + if (!(file->link_group != 0 && !file->link_first)) + scanner_capture_xattrs(scanner, file); + if (!array_list_add(chunk_data, file)) { + free(rel_copy); + file_destroy(file); + scanner->failed = true; + return SCANNER_ACTION_BREAK; + } + scanner->current_dir_produced = true; + *chunk_data_size += file->data->size; + if (*chunk_data_size > scanner->options.chunk_size) { + free(rel_copy); + Chunk* result = chunk_data_to_chunk(chunk_data); + if (!result) + scanner->failed = true; + *out_chunk = result; + return SCANNER_ACTION_CHUNK; + } + free(rel_copy); + return SCANNER_ACTION_CONTINUE; +} + Chunk* directory_scanner_next(DirectoryScanner* scanner) { if (scanner && scanner->options.dirs) return directory_scanner_next_dirs(scanner); @@ -1613,21 +1199,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { } if (scanner->sorted_index >= scanner->sorted_count) { - /* The directory is exhausted: if nothing was transferred or descended - from it, recreate it at the destination as an explicit entry. */ - if (scanner->options.emit_empty_dirs && !scanner->current_dir_produced && - !scanner->options.prune_empty_dirs && !scanner->options.list_dirs && - scanner->options.file_list == NULL) { - if (!scanner_emit_empty_dir(scanner, chunk_data)) - scanner->failed = true; - } - scanner_push_pending_dirs(scanner); - closedir(scanner->current_dir); - scanner->current_dir = NULL; - free(scanner->current_path); - scanner->current_path = NULL; - scanner_free_sorted(scanner); - if (scanner->failed) { + if (!scanner_finish_current_directory(scanner, chunk_data)) { array_list_delete(chunk_data); return NULL; } @@ -1635,226 +1207,21 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) { } SortedEntry* sorted = &((SortedEntry*)scanner->sorted_entries)[scanner->sorted_index++]; - const char* name = sorted->name; - ScannerEntry* inspected = &sorted->entry; int inspection = sorted->inspection; if (inspection == 0) { - /* A dereferenced symlink with no referent is a partial-transfer error - (rsync exit 23): record it as a non-fatal scan I/O error. */ - if (inspected->referent_error) - scanner->io_error = true; - /* A user-selection exclude protects its destination mirror from --delete - unless --delete-excluded; a size prune is always protected. Other - skips (unreadable, symlink policy) protect nothing. Under -R + - --files-from the protected prefix must be the entry's bare relative - wire path, not its source path (which would not match the destination - layout and would leave the mirror deletable). */ - if (inspected->excluded) { - char* protected_path; - if (scanner->relative_mode) { - protected_path = child_rel_path(scanner->current_rel, name); - } else if (scanner->options.relative_prefix) { - char* relc = child_rel_path(scanner->current_rel, name); - protected_path = - relc ? scanner_prefix_send_path(scanner->options.relative_prefix, relc) : NULL; - free(relc); - } else { - protected_path = path_cat(scanner->current_path, name); - } - if (!protected_path) { - scanner->failed = true; - break; - } - if (inspected->size_excluded) - scanner_record_size_skipped(scanner, protected_path); - else - scanner_record_excluded(scanner, protected_path); - free(protected_path); - } - continue; - } - char* cur_path = inspected->path; - struct stat stats = inspected->stats; - - /* --files-from allow-set and the filter layer apply to files and to - * directories (an excluded directory is not descended into). */ - bool is_dir = inspected->is_directory; - char* rel = child_rel_path(scanner->current_rel, name); - if (!rel) { - scanner->failed = true; - break; - } - bool protect = false; - bool passes_selection = entry_passes_selection( - scanner->options.file_list, scanner->options.base_filters, scanner->current_node, rel, name, - is_dir, scanner->options.per_dir_filters, scanner->options.exclude_per_dir_filter_files, - &protect); - /* A sender-side hide leaves the entry out of the transfer; an independent - receiver-side protect rule keeps a transferred entry's destination mirror - from being deleted. Both are recorded in the same protection set. */ - if (!passes_selection || protect) { - /* --files-from subset pruning is not a filter exclusion: its delete - semantics stay keep-set-only (an unlisted source path is treated as - absent, so its destination mirror is a deletable extra). A rule-based - exclusion is recorded as a protected prefix. -R + --files-from bare - wire paths are never recorded (see ScannerOptions.excluded_paths). */ - bool files_from_prune = - scanner->options.file_list && !file_list_affects(scanner->options.file_list, rel); - if (protect && scanner->relative_mode) { - /* -R + --files-from: the destination/wire path is the bare relative - name, so the protected mirror prefix must be `rel` (not the source - path) for the delete walker to match it. */ - scanner_record_excluded(scanner, rel); - } else if (!files_from_prune && !scanner->relative_mode) { - if (scanner->options.relative_prefix) { - char* wrel = scanner_prefix_send_path(scanner->options.relative_prefix, rel); - if (!wrel) { - free(rel); - scanner->failed = true; - break; - } - scanner_record_excluded(scanner, wrel); - free(wrel); - } else { - scanner_record_excluded(scanner, cur_path); - } - } - } - /* With -R the wire/destination path is a reconstructed relative path, not - the source path; keep `rel` alive to build it for a transferred file. */ - bool needs_rel = scanner->relative_mode || scanner->options.relative_prefix != NULL; - char* rel_copy = needs_rel ? str_dup(rel) : NULL; - free(rel); - if (rel_copy == NULL && needs_rel) { - scanner->failed = true; - break; - } - if (!passes_selection) { - scanner_note_filter(&scanner->options, name); - free(rel_copy); + if (scanner_handle_skipped_entry(scanner, &sorted->entry, sorted->name) != 0) + break; continue; } - if (is_dir) { - free(rel_copy); - if (!scanner_same_filesystem(scanner->options.one_file_system, scanner->root_dev, - stats.st_dev)) { - if (scanner->options.one_file_system > 1) { - /* rsync's -xx drops the mount-point directory entirely (the plain -x - path below keeps it as an empty directory) and prints the - --info=mount line when that category is enabled. */ - scanner_note_mount(&scanner->options, cur_path); - continue; - } - /* rsync's -x/--one-file-system emits the mount-point directory entry - itself (so the destination gets an empty directory) but does NOT - descend into it. Build a payload-less directory File and hand it to - the caller; never enqueue it for traversal. */ - File* mount = scanner_build_dir_file(cur_path, &stats, &scanner->options); - if (mount == NULL || !array_list_add(chunk_data, mount)) { - file_destroy(mount); - scanner->failed = true; - break; - } - scanner->current_dir_produced = true; - continue; - } - /* --list-only: list directory entries too (rsync prints them), even - though a real transfer never sends them explicitly. */ - if (scanner->options.list_dirs) { - File* dir = scanner_build_dir_file(cur_path, &stats, &scanner->options); - if (dir == NULL || !array_list_add(chunk_data, dir)) { - file_destroy(dir); - scanner->failed = true; - break; - } - } - scanner->current_dir_produced = true; - int next_depth = scanner->current_depth + 1; - if (scanner->options.max_depth <= 0 || next_depth < scanner->options.max_depth) { - DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node); - if (!de || !array_list_add((ArrayList*)scanner->pending_dirs, de)) { - dir_entry_destroy(de); - scanner->failed = true; - } - } - } else { - if (scanner->options.max_depth > 0 && - scanner->current_depth + 1 > scanner->options.max_depth) { - free(rel_copy); - continue; - } - File* file = file_create(cur_path); - if (file == NULL) { - free(rel_copy); - free(inspected->link_target); - inspected->link_target = NULL; - scanner->failed = true; - continue; - } - if (inspected->is_symlink) { - file->is_symlink = true; - file->symlink_target = inspected->link_target; - inspected->link_target = NULL; - } else { - file->data->size = stats.st_size; - } - if (scanner->relative_mode) { - file->send_path = rel_copy; - rel_copy = NULL; - } else if (scanner->options.relative_prefix) { - file->send_path = scanner_prefix_send_path(scanner->options.relative_prefix, rel_copy); - free(rel_copy); - rel_copy = NULL; - if (!file->send_path) { - file_destroy(file); - scanner->failed = true; - break; - } - } - /* --devices/--specials: a device/FIFO/socket entry marked for preservation - becomes a node to recreate (is_special, no data, rdev captured); an - unrequested non-regular entry is skipped (rsync default). */ - ScannerSpecial special = scanner_prepare_special(scanner->options.preserve_devices, - scanner->options.preserve_specials, - scanner->options.copy_devices, file, &stats); - if (special == SCANNER_SPECIAL_SKIP) { - scanner_note_nonreg(&scanner->options, file->path); - free(rel_copy); - file_destroy(file); - continue; - } - if (scanner->options.hardlinks && S_ISREG(stats.st_mode)) - scanner_assign_hardlink(scanner, scanner->options.hardlinks, file, &stats); - if (scanner->options.use_metadata) - file->metadata = file_metadata_create(file->path, &stats, scanner->options.preserve_atimes, - scanner->options.preserve_crtimes); - if (scanner->options.use_metadata && !file->metadata) { - free(rel_copy); - file_destroy(file); - scanner->failed = true; - break; - } - if (!(file->link_group != 0 && !file->link_first)) - scanner_capture_xattrs(scanner, file); - if (!array_list_add(chunk_data, file)) { - free(rel_copy); - file_destroy(file); - scanner->failed = true; - break; - } - scanner->current_dir_produced = true; - chunk_data_size += file->data->size; - if (chunk_data_size > scanner->options.chunk_size) { - free(rel_copy); - Chunk* result = chunk_data_to_chunk(chunk_data); - if (!result) - scanner->failed = true; - return result; - } - free(rel_copy); - } + Chunk* result = NULL; + ScannerAction action = + scanner_process_entry(scanner, chunk_data, &chunk_data_size, sorted, &result); + if (action == SCANNER_ACTION_CHUNK) + return result; + if (action == SCANNER_ACTION_BREAK) + break; } if (chunk_data->size > 0) { @@ -1874,687 +1241,3 @@ bool directory_scanner_failed(const DirectoryScanner* scanner) { bool directory_scanner_had_io_error(const DirectoryScanner* scanner) { return scanner != NULL && (scanner->io_error || scanner->root_io_error); } - -typedef struct { - ParallelScanner* ps; - char** dirs; - int dir_count; - char* root_dir; /* the transfer root, for relative-path computation */ - ScannerOptions options; - ProtocolSession* allocation_session; -} ParallelWorkerArg; - -static int parallel_worker_thread(void* arg) { - ParallelWorkerArg* wa = (ParallelWorkerArg*)arg; - ProtocolSession* allocation_session = wa->allocation_session; - if (allocation_session) - protocol_session_bind(allocation_session); - for (int i = 0; i < wa->dir_count; i++) { - DirectoryScanner* ds = directory_scanner_create_with_options(wa->dirs[i], &wa->options); - if (!ds) { - mtx_lock(&wa->ps->result_mutex); - wa->ps->failed = true; - atomic_store(&wa->ps->cancelled, true); - cnd_broadcast(&wa->ps->result_not_empty); - cnd_broadcast(&wa->ps->result_not_full); - mtx_unlock(&wa->ps->result_mutex); - for (int j = i; j < wa->dir_count; j++) - free(wa->dirs[j]); - break; - } - /* Root .rsync-filter rules (parsed by the parallel scanner) apply to the - * contents of every assigned subdirectory. Relative paths (used by the - * allow-set and per-directory rules) are computed against the transfer - * root, not the subdirectory the worker is seeded with. Exclusion - * recording shares one caller-owned list across the workers. */ - free(ds->root_path); - ds->root_path = str_dup(wa->root_dir); - ds->seed_node = wa->ps->root_filter_node; - ds->options.excluded_mutex = &wa->ps->result_mutex; - Chunk* chunk; - while ((chunk = directory_scanner_next(ds)) != NULL) { - if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex, - &wa->ps->result_not_empty, &wa->ps->result_not_full, - &wa->ps->cancelled)) { - chunk_destroy(chunk); - break; - } - } - if (directory_scanner_failed(ds)) { - mtx_lock(&wa->ps->result_mutex); - wa->ps->failed = true; - atomic_store(&wa->ps->cancelled, true); - cnd_broadcast(&wa->ps->result_not_empty); - cnd_broadcast(&wa->ps->result_not_full); - mtx_unlock(&wa->ps->result_mutex); - } else if (directory_scanner_had_io_error(ds)) { - /* --ignore-errors path: an unreadable directory was skipped, not fatal. */ - mtx_lock(&wa->ps->result_mutex); - wa->ps->io_error = true; - mtx_unlock(&wa->ps->result_mutex); - } - directory_scanner_destroy(ds); - free(wa->dirs[i]); - } - ParallelScanner* ps = wa->ps; - free(wa->root_dir); - free(wa->dirs); - free(wa); - mtx_lock(&ps->result_mutex); - ps->completed++; - if (ps->completed >= ps->expected_threads) { - ps->done = true; - cnd_signal(&ps->result_not_empty); - } - mtx_unlock(&ps->result_mutex); - if (allocation_session) - protocol_session_unbind(); - return thrd_success; -} - -static void parallel_scanner_creation_failed(ParallelScanner* ps) { - mtx_lock(&ps->result_mutex); - ps->failed = true; - atomic_store(&ps->cancelled, true); - ps->expected_threads = ps->created_threads; - if (ps->completed >= ps->expected_threads) - ps->done = true; - cnd_broadcast(&ps->result_not_empty); - cnd_broadcast(&ps->result_not_full); - mtx_unlock(&ps->result_mutex); -} - -/* Initialize result queue and synchronization primitives. Returns true on success. */ -static bool parallel_scanner_init(ParallelScanner* ps) { - ps->result_queue = queue_create(100, chunk_destroy); - if (!ps->result_queue) - return false; - atomic_init(&ps->cancelled, false); - int init = 0; - bool ok = true; - if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success) - ok = false; - if (ok) { - init++; - if (cnd_init(&ps->result_not_empty) != thrd_success) - ok = false; - } - if (ok) { - // cppcheck-suppress unreadVariable - init++; - if (cnd_init(&ps->result_not_full) != thrd_success) - ok = false; - } - if (!ok) { - if (init >= 3) - cnd_destroy(&ps->result_not_full); - if (init >= 2) - cnd_destroy(&ps->result_not_empty); - if (init >= 1) - mtx_destroy(&ps->result_mutex); - queue_destroy(ps->result_queue); - ps->result_queue = NULL; - return false; - } - return true; -} - -/* Split files into chunks of roughly chunk_size bytes. Returns the first chunk (also stored - * chunks beyond the first are enqueued on `queue`). Nulls out consumed entries in `files`. - * Sets *failed on allocation/enqueue errors. */ -static Chunk* batch_files(ArrayList* files, unsigned long long chunk_size, Queue* queue, - bool* failed) { - Chunk* first = NULL; - if (files->size <= 0) - return NULL; - ArrayList* batch = array_list_create(NULL); - if (!batch) { - *failed = true; - return NULL; - } - unsigned long long batch_size = 0; - for (int i = 0; i < files->size; i++) { - File* f = (File*)files->items[i]; - if (!array_list_add(batch, f)) { - *failed = true; - break; - } - batch_size += f->data->size; - if (batch_size >= chunk_size || i == files->size - 1) { - void** items = array_list_to_array(batch); - if (!items) { - *failed = true; - array_list_delete(batch); - batch = NULL; - break; - } - Chunk* c = chunk_create((File**)items, batch->size); - free(items); - if (!c) { - *failed = true; - array_list_delete(batch); - batch = NULL; - break; - } - int batch_start = i - batch->size + 1; - for (int j = batch_start; j <= i; j++) - files->items[j] = NULL; - batch->item_destroyer = NULL; - array_list_delete(batch); - batch = NULL; - if (!first) { - first = c; - } else { - if (!queue_enqueue(queue, c)) { - chunk_destroy(c); - *failed = true; - } - } - if (i < files->size - 1) { - batch = array_list_create(NULL); - if (!batch) { - *failed = true; - break; - } - batch_size = 0; - } - } - } - if (batch) { - batch->item_destroyer = NULL; - array_list_delete(batch); - } - return first; -} - -/* Scan one root-directory entry into either the subdirs or files list. */ -static void scan_root_entry(const ScannerOptions* options, const FilterNode* root_node, - const char* root_directory, const struct dirent* entry, - ArrayList* root_files, ArrayList* subdirs, dev_t root_dev, - ParallelScanner* ps) { - ScannerEntry inspected; - int inspection = - scanner_inspect_entry(options, root_directory, entry->d_name, entry->d_name, &inspected); - if (inspection < 0) { - ps->failed = true; - return; - } - if (inspection == 0) { - if (inspected.referent_error) - ps->io_error = true; - ArrayList* sink = NULL; - if (inspected.excluded) - sink = inspected.size_excluded ? options->size_skipped_paths : options->excluded_paths; - if (sink) { - /* A root-level prune protects the destination mirror of the entry's wire - path: under -R + --files-from that is the bare relative name, otherwise - it is the full source path with a leading '/' removed (matching the - send_path/file_wire_path the scanner hands the sender). */ - if (options->relative && options->file_list != NULL) { - if (!excluded_sink_append(sink, options->excluded_mutex, entry->d_name)) - ps->failed = true; - } else if (options->relative_prefix) { - char* wrel = scanner_prefix_send_path(options->relative_prefix, entry->d_name); - if (!wrel) { - ps->failed = true; - } else { - if (!excluded_sink_append(sink, options->excluded_mutex, wrel)) - ps->failed = true; - free(wrel); - } - } else { - char* abs_path = path_cat(root_directory, entry->d_name); - if (!abs_path) { - ps->failed = true; - } else { - const char* rel = *abs_path == '/' ? abs_path + 1 : abs_path; - if (!excluded_sink_append(sink, options->excluded_mutex, rel)) - ps->failed = true; - free(abs_path); - } - } - } - return; - } - char* cur_path = inspected.path; - struct stat st = inspected.stats; - bool is_dir = inspected.is_directory; - char* rel = str_dup(entry->d_name); - if (!rel) { - free(cur_path); - ps->failed = true; - return; - } - bool protect = false; - bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel, - entry->d_name, is_dir, options->per_dir_filters, - options->exclude_per_dir_filter_files, &protect); - /* -R + --files-from: root-level files keep their bare relative send path. */ - bool use_rel = options->relative && options->file_list != NULL; - if (!passes || protect) { - /* --files-from subset pruning is not a filter exclusion; -R bare-wire-path - exclusions are never recorded (see ScannerOptions.excluded_paths). */ - bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel); - if ((!files_from_prune && !use_rel) || protect) { - const char* rel_path; - char* prefixed = NULL; - if (use_rel) { - /* -R + --files-from: the destination/wire path is the bare relative - name, not the source path. */ - rel_path = rel; - } else if (options->relative_prefix) { - prefixed = scanner_prefix_send_path(options->relative_prefix, entry->d_name); - if (!prefixed) { - free(rel); - free(cur_path); - ps->failed = true; - return; - } - rel_path = prefixed; - } else { - rel_path = *cur_path == '/' ? cur_path + 1 : cur_path; - } - if (options->excluded_paths && - !excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) - ps->failed = true; - free(prefixed); - } - if (!passes) { - scanner_note_filter(options, entry->d_name); - free(rel); - free(cur_path); - return; - } - } - if (is_dir) { - if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { - if (options->one_file_system > 1) { - /* -xx: drop the mount-point directory entirely (rsync) and print the - --info=mount line when enabled. */ - scanner_note_mount(options, cur_path); - free(rel); - free(cur_path); - return; - } - /* -x/--one-file-system: emit the mount-point directory entry (empty) but - do not descend into it (see the sequential scanner for the same rule). */ - File* mount = file_create(cur_path); - free(cur_path); - if (mount == NULL) { - free(rel); - ps->failed = true; - return; - } - mount->is_dir = true; - if (options->use_metadata) { - mount->metadata = file_metadata_create(mount->path, &st, options->preserve_atimes, - options->preserve_crtimes); - if (!mount->metadata) { - free(rel); - file_destroy(mount); - ps->failed = true; - return; - } - } - if (options->relative_prefix) { - mount->send_path = scanner_prefix_send_path(options->relative_prefix, rel); - if (!mount->send_path) { - free(rel); - file_destroy(mount); - ps->failed = true; - return; - } - } - free(rel); - if (!array_list_add(root_files, mount)) { - file_destroy(mount); - ps->failed = true; - } - return; - } - free(rel); - if (!array_list_add(subdirs, cur_path)) { - free(cur_path); - ps->failed = true; - } - return; - } - File* file = file_create(cur_path); - free(cur_path); - if (!file) { - free(rel); - free(inspected.link_target); - inspected.link_target = NULL; - ps->failed = true; - return; - } - if (inspected.is_symlink) { - file->is_symlink = true; - file->symlink_target = inspected.link_target; - inspected.link_target = NULL; - } else { - file->data->size = st.st_size; - } - if (use_rel) { - file->send_path = rel; - rel = NULL; - } else if (options->relative_prefix) { - file->send_path = scanner_prefix_send_path(options->relative_prefix, rel); - free(rel); - rel = NULL; - if (!file->send_path) { - file_destroy(file); - ps->failed = true; - return; - } - } - ScannerSpecial special = scanner_prepare_special( - options->preserve_devices, options->preserve_specials, options->copy_devices, file, &st); - if (special == SCANNER_SPECIAL_SKIP) { - scanner_note_nonreg(ps->options, file->path); - free(rel); - file_destroy(file); - return; - } - if (options->hardlinks && S_ISREG(st.st_mode)) { - int gid; - bool is_first; - char* first_path = NULL; - if (!hardlink_table_assign((HardLinkTable*)options->hardlinks, file_wire_path(file), st.st_dev, - st.st_ino, &gid, &is_first, &first_path)) { - ps->failed = true; - } else { - file->link_group = gid; - file->link_first = is_first; - if (!is_first) { - file->hardlink_target = first_path; - file->data->size = 0; - } else { - free(first_path); - } - } - } - if (options->use_metadata) - file->metadata = - file_metadata_create(file->path, &st, options->preserve_atimes, options->preserve_crtimes); - if (options->use_metadata && !file->metadata) { - free(rel); - file_destroy(file); - ps->failed = true; - return; - } - if ((options->preserve_xattrs || options->preserve_acls) && - !(file->link_group != 0 && !file->link_first)) - file->xattrs = xattr_capture_path(file->path, options->preserve_acls); - if (!array_list_add(root_files, file)) { - free(rel); - file_destroy(file); - ps->failed = true; - return; - } - free(rel); -} - -/* Scan the root directory itself, collecting root files and subdirectories. - * Returns false if the root directory could not be opened. */ -static bool scan_root_directory(ParallelScanner* ps, const char* root_directory, - const ScannerOptions* options, const FilterNode* root_node, - dev_t root_dev, ArrayList* root_files, ArrayList* subdirs) { - DIR* dir = opendir(root_directory); - if (!dir) { - log_perror("Could not open root directory for parallel scan"); - return false; - } - /* The parallel scanner opens the transfer root directly (not through - open_next_directory), so record it as synchronized here. */ - if (!scanner_record_synced_dir(options, root_directory, "", - options->relative && options->file_list != NULL)) { - closedir(dir); - ps->failed = true; - return false; - } - log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", root_directory); - const struct dirent* entry; - while ((entry = readdir(dir)) != NULL) { - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; - scan_root_entry(options, root_node, root_directory, entry, root_files, subdirs, root_dev, ps); - } - closedir(dir); - return true; -} - -/* Spawn worker threads, one per group of subdirectories. */ -static void spawn_parallel_workers(ParallelScanner* ps, ArrayList* subdirs, - const ScannerOptions* options, const char* root_directory, - unsigned long long cs) { - if (subdirs->size <= 0) - return; - int n = options->num_threads > 0 ? options->num_threads : 4; - if (n > subdirs->size) - n = subdirs->size; - - ps->num_threads = n; - ps->expected_threads = n; - ps->threads = calloc(n, sizeof(thrd_t)); - if (!ps->threads) { - ps->num_threads = 0; - ps->expected_threads = 0; - ps->failed = true; - return; - } - int dirs_per_thread = subdirs->size / n; - int remainder = subdirs->size % n; - int start = 0; - ps->num_threads = 0; - for (int t = 0; t < n; t++) { - int count = dirs_per_thread + (t < remainder ? 1 : 0); - if (count == 0) - break; - ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg)); - if (!wa) { - parallel_scanner_creation_failed(ps); - break; - } - wa->ps = ps; - wa->dirs = calloc(count, sizeof(char*)); - wa->root_dir = str_dup(root_directory); - if (!wa->dirs || !wa->root_dir) { - free(wa->root_dir); - free(wa->dirs); - free(wa); - parallel_scanner_creation_failed(ps); - break; - } - bool dup_ok = true; - for (int j = 0; j < count; j++) { - wa->dirs[j] = str_dup((char*)subdirs->items[start + j]); - if (!wa->dirs[j]) - dup_ok = false; - } - if (!dup_ok) { - for (int j = 0; j < count; j++) - free(wa->dirs[j]); - free(wa->root_dir); - free(wa->dirs); - free(wa); - parallel_scanner_creation_failed(ps); - break; - } - wa->dir_count = count; - wa->options = *options; - wa->options.chunk_size = cs; - wa->allocation_session = ps->allocation_session; - start += count; - if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) { - for (int j = 0; j < count; j++) - free(wa->dirs[j]); - free(wa->root_dir); - free(wa->dirs); - free(wa); - parallel_scanner_creation_failed(ps); - break; - } - ps->num_threads++; - ps->created_threads++; - } -} - -ParallelScanner* parallel_scanner_create_with_options(const char* root_directory, - const ScannerOptions* options, - ProtocolSession* allocation_session) { - if (!root_directory || !options) - return NULL; - ParallelScanner* ps = calloc(1, sizeof(ParallelScanner)); - if (!ps) - return NULL; - if (!parallel_scanner_init(ps)) { - free(ps); - return NULL; - } - ps->allocation_session = allocation_session; - ps->options = options; - - ArrayList* root_files = array_list_create(file_destroy); - ArrayList* subdirs = array_list_create(free); - if (!root_files || !subdirs) { - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - - dev_t root_dev = 0; - if (options->one_file_system) { - struct stat root_stats; - if (stat(root_directory, &root_stats) != 0) { - log_perror("Could not stat source directory"); - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - root_dev = root_stats.st_dev; - } - - /* Build the root directory's per-directory filter context once; workers seed - * their scanners with it so per-dir rules behave identically to the sequential - * scanner. */ - FilterNode* root_node = NULL; - { - char err[256]; - bool any_exists = false; - FilterRuleList* own = - read_dir_filters(options, root_directory, "", &any_exists, err, sizeof(err)); - if (!own) { - /* A parse/allocation failure must fail the scan even when an earlier - merge file in the same directory existed (see the sequential scanner). */ - if (err[0] != '\0') { - log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err); - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - /* no files exist: leave root_node NULL */ - } else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { - root_node = filter_node_alloc(NULL, own); - if (!root_node) { - filter_rule_list_free(own); - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - } else { - filter_rule_list_free(own); - } - } - ps->root_filter_node = root_node; - - if (!scan_root_directory(ps, root_directory, options, root_node, root_dev, root_files, subdirs)) { - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - /* The root itself is a traversed directory (rsync counts it in - `Number of files`); the worker DirectoryScanners account for every - subdirectory below it. */ - scanner_dir_count_count(options); - /* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the - transfer root itself (it hands the root's immediate subdirectories to - workers), so capture the root's directory time here. */ - if (options->capture_dir_times && - !scanner_capture_dir_time( - options->dir_entries, options->dir_entries_mutex, root_directory, root_directory, - options->relative && options->file_list != NULL, options->relative_prefix, - options->preserve_atimes, options->preserve_crtimes, options->preserve_xattrs, - options->preserve_acls, options->no_implied_dirs, options->file_list)) { - array_list_delete(root_files); - array_list_delete(subdirs); - parallel_scanner_destroy(ps); - return NULL; - } - - unsigned long long cs = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE; - ps->initial_chunk = batch_files(root_files, cs, ps->result_queue, &ps->failed); - array_list_delete(root_files); - - spawn_parallel_workers(ps, subdirs, options, root_directory, cs); - array_list_delete(subdirs); - return ps; -} - -Chunk* parallel_scanner_next(ParallelScanner* ps) { - if (ps->initial_chunk) { - Chunk* c = ps->initial_chunk; - ps->initial_chunk = NULL; - return c; - } - if (ps->num_threads == 0) { - mtx_lock(&ps->result_mutex); - if (!queue_is_empty(ps->result_queue)) { - Chunk* chunk = queue_dequeue(ps->result_queue); - mtx_unlock(&ps->result_mutex); - return chunk; - } - ps->done = true; - mtx_unlock(&ps->result_mutex); - return NULL; - } - Chunk* chunk = queue_dequeue_multithreaded( - ps->result_queue, &ps->result_mutex, &ps->result_not_empty, &ps->result_not_full, &ps->done); - return chunk; -} - -bool parallel_scanner_failed(const ParallelScanner* ps) { - return ps == NULL || ps->failed; -} - -bool parallel_scanner_had_io_error(const ParallelScanner* ps) { - return ps != NULL && ps->io_error; -} - -void parallel_scanner_destroy(ParallelScanner* ps) { - if (!ps) - return; - mtx_lock(&ps->result_mutex); - ps->done = true; - atomic_store(&ps->cancelled, true); - cnd_broadcast(&ps->result_not_empty); - cnd_broadcast(&ps->result_not_full); - mtx_unlock(&ps->result_mutex); - for (int i = 0; i < ps->num_threads; i++) - thrd_join(ps->threads[i], NULL); - free(ps->threads); - if (ps->root_filter_node) - filter_node_destroy(ps->root_filter_node); - if (ps->initial_chunk) - chunk_destroy(ps->initial_chunk); - queue_destroy(ps->result_queue); - mtx_destroy(&ps->result_mutex); - cnd_destroy(&ps->result_not_empty); - cnd_destroy(&ps->result_not_full); - free(ps); -} diff --git a/src/client/scanner_filter.c b/src/client/scanner_filter.c new file mode 100644 index 0000000..225d9f2 --- /dev/null +++ b/src/client/scanner_filter.c @@ -0,0 +1,671 @@ +#include "log.h" +#include "scanner.h" +#include "scanner_internal.h" +#include "array_list.h" +#include "chunk.h" +#include "file.h" +#include "queue.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "xattr.h" + +/* A chain node: `own` holds the .rsync-filter rules of one directory, `parent` + * the context that directory inherited (nearest ancestor with a filter file). + * The chain for a directory's contents runs from that directory's own node up + * to the root; the command-line base rules are evaluated after the whole + * chain. */ +struct FilterNode { + FilterNode* parent; + FilterRuleList* own; +}; + +void filter_node_destroy(void* item) { + if (item) { + FilterNode* node = (FilterNode*)item; + if (node->own) + filter_rule_list_free(node->own); + free(node); + } +} + +FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) { + FilterNode* node = malloc(sizeof(FilterNode)); + if (!node) + return NULL; + node->parent = parent; + node->own = own; + return node; +} + +/* Evaluate a rule chain for one entry. rsync precedence, highest first: the + * innermost (current) directory's .rsync-filter rules, then each ancestor's, + * then the root's, and finally the command-line base rules (--filter/-C). The + * sender-side verdict decides whether the entry is hidden from the transfer; + * the receiver-side verdict decides whether its destination mirror is protected + * from --delete. Each side takes the FIRST matching rule independently. */ +typedef struct { + bool hide; /* sender-side exclude matched */ + bool protect; /* receiver-side exclude matched */ +} FilterOutcome; + +static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel, + const char* leaf, bool is_dir, FilterOutcome* out) { + memset(out, 0, sizeof(*out)); + bool sender_decided = false; + bool receiver_decided = false; + const FilterNode* n = node; + while (!sender_decided || !receiver_decided) { + const FilterRuleList* list = n ? n->own : base; + if (list) { + if (!sender_decided) { + FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER); + if (action != FILTER_ACTION_NONE) { + out->hide = action == FILTER_ACTION_EXCLUDE; + sender_decided = true; + } + } + if (!receiver_decided) { + FilterAction action = + filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER); + if (action != FILTER_ACTION_NONE) { + out->protect = action == FILTER_ACTION_PROTECT; + receiver_decided = true; + } + } + } + if (!n) + break; + n = n->parent; + } +} + +static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel, + const char* leaf, bool is_dir, bool exclude_filter_files, + bool* protect_out) { + /* -FF: per-directory .rsync-filter files are never transferred (single -F + transfers them, matching rsync). */ + if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) { + if (protect_out) + *protect_out = false; + return false; + } + FilterOutcome outcome; + chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome); + if (protect_out) + *protect_out = outcome.protect; + return !outcome.hide; +} + +void dir_entry_destroy(void* item) { + if (item) { + DirEntry* de = (DirEntry*)item; + free(de->path); + free(de); + } +} + +DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context) { + DirEntry* de = malloc(sizeof(DirEntry)); + if (!de) + return NULL; + de->path = str_dup(path); + if (!de->path) { + free(de); + return NULL; + } + de->depth = depth; + de->context = context; + return de; +} + +/* Apply rsync's symlink-resolution precedence to one S_ISLNK entry: + * --copy-links dereferences every symlink; + * --copy-unsafe-links dereferences only targets unsafe_symlink() flags; + * -k/--copy-dirlinks dereferences only a symlink whose referent is a dir; + * --safe-links (receiver-side in rsync; modelled here) ignores an unsafe + * target that would otherwise be carried; with --munge-links + * every stored target becomes absolute, so --safe-links then + * ignores every symlink, exactly as rsync documents; + * -l/--links carries the link. + * `link_rel` is the symlink's transfer-relative path (incl. name) and is used + * only for the lexical unsafe test. `target` receives the raw link value. */ +LinkAction scanner_link_action(const ScannerOptions* options, const char* path, + const char* link_rel, char* target, size_t target_size) { + if (!options->follow_symlinks && !options->copy_links && !options->safe_links && + !options->copy_unsafe_links && !options->copy_dirlinks) + return LINK_ACTION_SKIP; + ssize_t length = readlink(path, target, target_size - 1); + if (length < 0) + return LINK_ACTION_SKIP; + target[length] = '\0'; + + bool unsafe = file_symlink_unsafe(target, link_rel); + if (options->copy_links || (options->copy_unsafe_links && unsafe)) + return LINK_ACTION_DEREF; + if (options->copy_dirlinks) { + struct stat ref; + if (stat(path, &ref) == 0 && S_ISDIR(ref.st_mode)) + return LINK_ACTION_DEREF; + } + if (options->safe_links && (unsafe || options->munge_links)) + return LINK_ACTION_SKIP_PROTECTED; + if (!options->follow_symlinks || target[0] == '\0') + return LINK_ACTION_SKIP; + return LINK_ACTION_CARRY; +} + +/* --one-file-system (-x) decision. Only directories can carry a different + * device than their parent (mount points), so this is checked when a child + * directory is about to be descended into. */ +bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device) { + return one_file_system <= 0 || entry_device == root_device; +} + +/* Build a payload-less directory File carrying the captured metadata (when + * requested). Used by -x mount-point emission and --list-only directory + * entries. Returns NULL on allocation failure. */ +File* scanner_build_dir_file(const char* path, const struct stat* stats, + const ScannerOptions* options) { + File* dir = file_create(path); + if (dir == NULL) + return NULL; + dir->is_dir = true; + if (options->use_metadata) { + dir->metadata = + file_metadata_create(dir->path, stats, options->preserve_atimes, options->preserve_crtimes); + if (!dir->metadata) { + file_destroy(dir); + return NULL; + } + } + return dir; +} + +/* Relative path of an on-disk path below `root`. The transfer root may be + * given with a trailing slash; the returned rel path never has one and is "" + * for the root itself. A root of "/" is handled (its children start at "/"). + * Exposed so tests can exercise the mapping directly. */ +char* scanner_path_relative(const char* root, const char* fs_path) { + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(root, fs_path, root_len) != 0) + return NULL; + if (root_len == 1 && root[0] == '/') { + if (fs_path[1] == '\0') + return str_dup(""); + return str_dup(fs_path + 1); + } + if (fs_path[root_len] == '\0') + return str_dup(""); + if (fs_path[root_len] != '/') + return NULL; + return str_dup(fs_path + root_len + 1); +} + +/* -R/--relative destination-relative prefix reconstructed from a source spec: + * everything after the first '.' path component (rsync's '/./' cut point), + * with leading/trailing slashes removed; or the whole spec (normalized) when + * there is no cut. Returns "" for the receive root. Exposed for tests. */ +char* scanner_relative_prefix(const char* spec) { + if (!spec || spec[0] == '\0') + return NULL; + const char* after = spec; + if (spec[0] == '.' && spec[1] == '/') { + after = spec + 2; + } else { + const char* cut = strstr(spec, "/./"); + if (cut) + after = cut + 3; + } + size_t cap = strlen(spec) + 1; + char* out = malloc(cap); + if (!out) + return NULL; + size_t len = 0; + for (const char* s = after; *s;) { + while (*s == '/') + s++; + const char* comp = s; + while (*s && *s != '/') + s++; + size_t clen = (size_t)(s - comp); + if (clen == 0 || (clen == 1 && comp[0] == '.')) + continue; + if (len) + out[len++] = '/'; + memcpy(out + len, comp, clen); + len += clen; + } + out[len] = '\0'; + return out; +} + +/* Relative path of a child entry below the current directory. */ +char* child_rel_path(const char* parent_rel, const char* name) { + if (!parent_rel || parent_rel[0] == '\0') + return str_dup(name); + return path_cat(parent_rel, name); +} + +/* Destination-relative wire path for an entry under an -R prefix. */ +char* scanner_prefix_send_path(const char* prefix, const char* rel) { + if (prefix[0] == '\0') + return str_dup(rel); + if (rel[0] == '\0') + return str_dup(prefix); + return path_cat(prefix, rel); +} + +/* Apply the --files-from allow-set and the filter layer to one entry. On + * return `*protect_out` is true when a receiver-side rule protects the entry's + * destination mirror from deletion. */ +bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, + const FilterNode* node, const char* rel, const char* leaf, bool is_dir, + bool per_dir_filters, bool exclude_filter_files, bool* protect_out) { + if (protect_out) + *protect_out = false; + if (file_list && !file_list_affects(file_list, rel)) + return false; + if (base || per_dir_filters) + return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out); + return true; +} + +/* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to + * read xattrs is non-fatal: the file is transferred without them. */ +void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) { + if (!scanner || !file || !(scanner->options.preserve_xattrs || scanner->options.preserve_acls)) + return; + file->xattrs = xattr_capture_path(file->path, scanner->options.preserve_acls); +} + +/* Apply --hard-links (-H) detection to one regular File. On a sibling (a + * later member of an already-seen source inode) the File keeps the group id + * and the first member's wire path but carries NO data payload (size 0); the + * first member is left untouched (data present, link_first). Allocation + * failure is fatal: the scanner is marked failed. */ +void scanner_assign_hardlink(DirectoryScanner* scanner, HardLinkTable* table, File* file, + const struct stat* stats) { + if (!table || !file || !stats) + return; + int gid; + bool is_first; + char* first_path = NULL; + if (!hardlink_table_assign(table, file_wire_path(file), stats->st_dev, stats->st_ino, &gid, + &is_first, &first_path)) { + if (scanner) + scanner->failed = true; + return; + } + file->link_group = gid; + file->link_first = is_first; + if (!is_first) { + file->hardlink_target = first_path; + file->data->size = 0; + } else { + free(first_path); + } +} + +/* Phase 4 special/devices decision for one non-regular entry, matching rsync: + - a char/block device is RECREATED as a node under -D/--devices, unless + --copy-devices asks for its content to be copied into a regular file; + - a FIFO/socket is RECREATED under --specials; + - when the matching flag is absent the entry is SKIPPED ("skipping + non-regular file"), exactly like rsync's default, instead of being + silently copied as a zero-length regular file; + - anything else (regular/directory) is left to the normal data path. */ +ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials, + bool copy_devices, File* file, const struct stat* stats) { + if (!file || !stats) + return SCANNER_SPECIAL_REGULAR; + bool is_device = S_ISCHR(stats->st_mode) || S_ISBLK(stats->st_mode); + bool is_fifo = S_ISFIFO(stats->st_mode); + bool is_socket = S_ISSOCK(stats->st_mode); + if (!is_device && !is_fifo && !is_socket) + return SCANNER_SPECIAL_REGULAR; + if (is_device && copy_devices) + return SCANNER_SPECIAL_REGULAR; /* copy device content as a regular file */ + bool preserve = is_device ? preserve_devices : preserve_specials; + if (!preserve) + return SCANNER_SPECIAL_SKIP; + file->is_special = true; + file->data->size = 0; + file->data->data = NULL; + if (is_device) { + file->rdev_major = (int32_t)major(stats->st_rdev); + file->rdev_minor = (int32_t)minor(stats->st_rdev); + } + return SCANNER_SPECIAL_RECREATE; +} + +/* Append `rel` to the caller's exclusion sink, taking `mtx` when shared across + parallel worker threads. Returns false on allocation failure (list left + unchanged). */ +bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) { + if (!list) + return true; + char* dup = str_dup(rel); + if (!dup) + return false; + if (mtx) + mtx_lock(mtx); + bool ok = array_list_add(list, dup); + if (mtx) + mtx_unlock(mtx); + if (!ok) + free(dup); + return ok; +} + +/* Record one pruned filesystem path in a delete-protection sink. The stored + form is the entry's wire/destination-relative path (a single leading '/' + removed, exactly how manifest keep entries are stored), so the receiver's + walker prefixes match the destination layout. An allocation failure is a + fatal scan error. */ +static void scanner_record_protected(DirectoryScanner* scanner, const char* fs_path, + ArrayList* sink) { + if (!sink || !fs_path) + return; + const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path; + if (!excluded_sink_append(sink, scanner->options.excluded_mutex, rel)) + scanner->failed = true; +} + +/* rsync's `--info=nonreg` line for a non-regular entry that is not being + * preserved: `skipping non-regular file "NAME"`. The name is the path relative + * to the transfer root, so it matches rsync's displayed name. */ +void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path) { + if (!options || !options->note_nonreg || !fs_path) + return; + const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); + char* escaped = output_escape(rel, options->eight_bit_output); + printf("skipping non-regular file \"%s\"\n", escaped ? escaped : rel); + free(escaped); + fflush(stdout); +} + +/* rsync 3.4.1's `--info=mount` line, emitted when `-xx` drops a mount-point + * directory: `[sender] skipping mount-point dir NAME` (the client is the + * sender). Plain `-x` keeps the empty directory and prints nothing, matching + * rsync. */ +void scanner_note_mount(const ScannerOptions* options, const char* fs_path) { + if (!options || !options->note_mount || !fs_path) + return; + const char* rel = utils_strip_transfer_root(fs_path, options->send_directory); + char* escaped = output_escape(rel, options->eight_bit_output); + printf("[sender] skipping mount-point dir %s\n", escaped ? escaped : rel); + free(escaped); + fflush(stdout); +} + +/* --debug=filter: a selection/filter decision dropped an entry. */ +void scanner_note_filter(const ScannerOptions* options, const char* name) { + if (!options || !log_debug_enabled(LOG_DEBUG_FILTER) || !name) + return; + log_debug_message(LOG_DEBUG_FILTER, "filter: excluded %s", name); +} + +/* Account for a directory that will not be represented by an inline directory + * entry. Paired with scanner_dir_count_uncount for empty directories that are + * emitted inline, so every traversed directory is counted exactly once. */ +void scanner_dir_count_count(const ScannerOptions* options) { + if (options && options->dir_count) + atomic_fetch_add(options->dir_count, 1); +} + +void scanner_dir_count_uncount(const ScannerOptions* options) { + if (options && options->dir_count) + atomic_fetch_sub(options->dir_count, 1); +} + +/* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */ +void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) { + scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths); +} + +/* A --max-size/--min-size prune (always protected, even under --delete-excluded). */ +void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path) { + scanner_record_protected(scanner, fs_path, scanner->options.size_skipped_paths); +} + +/* Record a directory the scan synchronized. `fs_path` is its absolute path and + `rel` its path relative to the transfer root ("" for the root); the stored + form matches the wire layout (the bare relative path in -R+--files-from, else + the source path with a leading '/' removed, with "." for the receive root). + Returns false on allocation failure. */ +bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel, + bool relative_mode) { + if (!options->synced_dirs && !options->plan_dirs) + return true; + if (!file_list_dir_in_scope(options->file_list, rel)) + return true; + char* prefixed = NULL; + const char* dest; + if (relative_mode) { + dest = rel; + } else if (options->relative_prefix) { + prefixed = scanner_prefix_send_path(options->relative_prefix, rel); + if (!prefixed) + return false; + dest = prefixed; + } else { + dest = fs_path; + } + if (dest[0] == '/') + dest++; + if (dest[0] == '\0') + dest = "."; + bool ok = true; + if (options->synced_dirs) + ok = excluded_sink_append(options->synced_dirs, options->excluded_mutex, dest); + /* The delete-plan keep set needs an entry for every traversed source + directory, including empty ones, so its destination mirror is kept rather + than deleted as an extra; the receive root (".") is implicit. */ + if (ok && options->plan_dirs && strcmp(dest, ".") != 0) + ok = excluded_sink_append(options->plan_dirs, options->excluded_mutex, dest); + free(prefixed); + return ok; +} + +/* Read every per-directory filter file that applies to `dir_path` (its + * .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a + * fresh list. Returns NULL on allocation/parse failure (message in `err`); + * returns an empty list (and *any_exists=false) when no file exists. */ +FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path, + const char* rel, bool* any_exists, char* err, size_t err_size) { + if (err && err_size > 0) + err[0] = '\0'; + const FilterRuleList* base = options->base_filters; + bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0); + if (any_exists) + *any_exists = false; + if (!have_names) + return NULL; + FilterRuleList* own = filter_rule_list_create(); + if (!own) { + snprintf(err, err_size, "memory allocation failed"); + return NULL; + } + FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false}; + bool exists = false; + if (options->per_dir_filters) { + if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size)) + goto fail; + if (exists && any_exists) + *any_exists = true; + } + if (base) { + for (int i = 0; i < base->dir_merge_count; i++) { + if (!filter_file_append(own, dir_path, base->dir_merge_names[i], rel, &opts, &exists, err, + err_size)) + goto fail; + if (exists && any_exists) + *any_exists = true; + } + } + return own; +fail: + filter_rule_list_free(own); + return NULL; +} + +/* Merge the open directory's own per-directory filter files (the default + * .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the + * base rule list) into the inherited context, returning the context used for + * this directory's entries. On a parse error the scanner is marked failed. + * Returns 0 on success, -1 on failure. */ +int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) { + char err[256]; + bool any_exists = false; + FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path, + scanner->current_rel ? scanner->current_rel : "", + &any_exists, err, sizeof(err)); + if (!own) { + /* read_dir_filters() leaves `err` set on a parse/allocation failure even + when an earlier merge file in the same directory existed (any_exists true); + key off the error text rather than any_exists so an invalid per-directory + filter file can never be silently ignored. */ + if (err[0] == '\0') { + scanner->current_node = (FilterNode*)inherited; + return 0; + } + char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", + escaped_path ? escaped_path : "", err); + free(escaped_path); + scanner->failed = true; + return -1; + } + if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { + FilterNode* node = filter_node_alloc((FilterNode*)inherited, own); + if (!node || !array_list_add(scanner->filter_nodes, node)) { + filter_node_destroy(node); + scanner->failed = true; + return -1; + } + scanner->current_node = node; + } else { + filter_rule_list_free(own); + scanner->current_node = (FilterNode*)inherited; + } + return 0; +} + +/* Inspect symlinks, resolve the entry type, and apply file filters once for both scanners. + * `link_rel` is the entry's path relative to the transfer root (including its + * name), used for the lexical rsync unsafe-symlink test. */ +int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir, + const char* link_rel, const char* name, ScannerEntry* entry) { + entry->excluded = false; + entry->size_excluded = false; + entry->referent_error = false; + entry->is_symlink = false; + entry->link_target = NULL; + entry->path = path_cat(containing_dir, name); + if (!entry->path) + return -1; + + struct stat link_stats; + if (lstat(entry->path, &link_stats) != 0) { + free(entry->path); + return 0; + } + if (!S_ISLNK(link_stats.st_mode)) + goto regular; + + char link_target[4096]; + switch (scanner_link_action(options, entry->path, link_rel, link_target, sizeof(link_target))) { + case LINK_ACTION_SKIP: + goto skip; + case LINK_ACTION_SKIP_PROTECTED: + /* --safe-links ignored the link, but rsync still counts it as present in + the transfer, so its destination mirror survives --delete. Record it as + an excluded path (the same delete-protection channel as a filter prune). */ + entry->excluded = true; + goto skip; + case LINK_ACTION_DEREF: + if (stat(entry->path, &entry->stats) != 0) { + /* rsync reports "symlink has no referent" and continues with a partial + transfer (exit 23); record the error so the run exits 23 too. */ + char* escaped = output_escape(entry->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "symlink has no referent: %s", + escaped ? escaped : ""); + free(escaped); + entry->referent_error = true; + goto skip; + } + entry->is_directory = S_ISDIR(entry->stats.st_mode); + if (entry->is_directory) + return 1; + goto apply_filters; + case LINK_ACTION_CARRY: + break; + } + + /* Carry the link as a symlink. --munge-links is applied by the RECEIVER (it + prefixes every stored target with /rsyncd-munged/); when the SOURCE already + holds a munged value the sender strips it so the receiver re-munges a clean + target, round-tripping a munged tree exactly like rsync. */ + entry->is_symlink = true; + entry->stats = link_stats; + entry->is_directory = false; + entry->link_target = str_dup(link_target); + if (!entry->link_target) + goto skip; + if (options->munge_links) + file_symlink_unmunge(entry->link_target); + goto apply_filters; + +regular: + /* Not a symlink: the lstat() above already described this entry, and lstat + and stat are identical for every non-symlink, so reuse that result instead + of issuing a redundant stat() on the scanner hot path. stat() is still + used on the dereference paths above/below for actual symlinks (copy-links, + safe/copy-unsafe links, and -k symlinks-to-directories). */ + entry->stats = link_stats; + entry->is_directory = S_ISDIR(link_stats.st_mode); + if (entry->is_directory) + return 1; + +apply_filters: + for (int i = 0; i < options->exclude_count; i++) + if (glob_match(options->exclude_patterns[i], name)) { + entry->excluded = true; + goto skip; + } + if (options->include_count > 0) { + bool included = false; + for (int i = 0; i < options->include_count; i++) + if (glob_match(options->include_patterns[i], name)) + included = true; + if (!included) { + entry->excluded = true; + goto skip; + } + } + if ((options->max_size > 0 && (unsigned long long)entry->stats.st_size > options->max_size) || + (options->min_size > 0 && (unsigned long long)entry->stats.st_size < options->min_size)) { + entry->excluded = true; + entry->size_excluded = true; + goto skip; + } + return 1; + +skip: + free(entry->path); + entry->path = NULL; + free(entry->link_target); + entry->link_target = NULL; + return 0; +} diff --git a/src/client/scanner_internal.h b/src/client/scanner_internal.h new file mode 100644 index 0000000..c772d9d --- /dev/null +++ b/src/client/scanner_internal.h @@ -0,0 +1,107 @@ +#ifndef SCANNER_INTERNAL_H +#define SCANNER_INTERNAL_H + +/* Internal declarations shared between the scanner translation units + * (scanner_filter.c, scanner.c, scanner_parallel.c). Nothing here is part of + * the public scanner façade (scanner.h); every symbol stays internal to the + * client module. */ + +#include "array_list.h" +#include "file.h" +#include "scanner.h" +#include +#include +#include + +typedef struct { + char* path; + int depth; + FilterNode* context; /* inherited per-directory filter context */ +} DirEntry; + +/* How rsync's readlink_stat()/generator resolves one source symlink. */ +typedef enum { + LINK_ACTION_SKIP, /* not transferred (no link option) */ + LINK_ACTION_SKIP_PROTECTED, /* ignored as unsafe by --safe-links; rsync keeps + it in the transfer, so its destination mirror + must be protected from --delete */ + LINK_ACTION_DEREF, /* follow the referent (--copy-links, an unsafe + target under --copy-unsafe-links, or -k dir) */ + LINK_ACTION_CARRY, /* transmit the link itself (-l) */ +} LinkAction; + +typedef struct { + char* path; + struct stat stats; + bool is_directory; + /* True when the entry should be carried through as a SYMLINK (is_symlink) + rather than a dereferenced file/directory. When true, `link_target` holds + the owned target string to transmit (sender-munged under --munge-links); + ownership transfers to the File built from this entry. */ + bool is_symlink; + char* link_target; + /* True when the entry was pruned by a user selection rule (--filter/-C/per-dir + rules or the --exclude/--include layer) rather than skipped for another + reason (unreadable, symlink policy, not applicable). */ + bool excluded; + /* True when the entry was skipped specifically by --max-size/--min-size. + Size pruning protects the destination mirror even under --delete-excluded, + so it is recorded into a separate sink from `excluded`. */ + bool size_excluded; + /* True when a symlink selected for dereferencing (-L/--copy-links or an + unsafe target under --copy-unsafe-links) had no usable referent (a broken + link or a stat() failure). rsync still reports this as a partial transfer + (exit 23) even though the entry is skipped, so the scanner records it as a + non-fatal I/O error. */ + bool referent_error; +} ScannerEntry; + +typedef enum { + SCANNER_SPECIAL_REGULAR, /* ordinary file: transfer content */ + SCANNER_SPECIAL_RECREATE, /* is_special node to recreate on the receiver */ + SCANNER_SPECIAL_SKIP, /* non-regular entry not requested: skip */ +} ScannerSpecial; + +/* scanner_filter.c */ +void filter_node_destroy(void* item); +FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own); +void dir_entry_destroy(void* item); +DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context); +LinkAction scanner_link_action(const ScannerOptions* options, const char* path, + const char* link_rel, char* target, size_t target_size); +File* scanner_build_dir_file(const char* path, const struct stat* stats, + const ScannerOptions* options); +char* child_rel_path(const char* parent_rel, const char* name); +char* scanner_prefix_send_path(const char* prefix, const char* rel); +bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base, + const FilterNode* node, const char* rel, const char* leaf, bool is_dir, + bool per_dir_filters, bool exclude_filter_files, bool* protect_out); +void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file); +void scanner_assign_hardlink(DirectoryScanner* scanner, HardLinkTable* table, File* file, + const struct stat* stats); +ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials, + bool copy_devices, File* file, const struct stat* stats); +bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel); +void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path); +void scanner_note_mount(const ScannerOptions* options, const char* fs_path); +void scanner_note_filter(const ScannerOptions* options, const char* name); +void scanner_dir_count_count(const ScannerOptions* options); +void scanner_dir_count_uncount(const ScannerOptions* options); +void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path); +void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path); +bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel, + bool relative_mode); +FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path, + const char* rel, bool* any_exists, char* err, size_t err_size); +int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited); +int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir, + const char* link_rel, const char* name, ScannerEntry* entry); + +/* scanner.c */ +bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path, + const char* fs_path, bool relative_mode, const char* relative_prefix, + bool preserve_atimes, bool preserve_crtimes, bool preserve_xattrs, + bool preserve_acls, bool no_implied_dirs, + const FileListSet* file_list); + +#endif diff --git a/src/client/scanner_parallel.c b/src/client/scanner_parallel.c new file mode 100644 index 0000000..9fa5151 --- /dev/null +++ b/src/client/scanner_parallel.c @@ -0,0 +1,703 @@ +#include "log.h" +#include "scanner.h" +#include "scanner_internal.h" +#include "array_list.h" +#include "chunk.h" +#include "file.h" +#include "queue.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "xattr.h" + +typedef struct { + ParallelScanner* ps; + char** dirs; + int dir_count; + char* root_dir; /* the transfer root, for relative-path computation */ + ScannerOptions options; + ProtocolSession* allocation_session; +} ParallelWorkerArg; + +static int parallel_worker_thread(void* arg) { + ParallelWorkerArg* wa = (ParallelWorkerArg*)arg; + ProtocolSession* allocation_session = wa->allocation_session; + if (allocation_session) + protocol_session_bind(allocation_session); + for (int i = 0; i < wa->dir_count; i++) { + DirectoryScanner* ds = directory_scanner_create_with_options(wa->dirs[i], &wa->options); + if (!ds) { + mtx_lock(&wa->ps->result_mutex); + wa->ps->failed = true; + atomic_store(&wa->ps->cancelled, true); + cnd_broadcast(&wa->ps->result_not_empty); + cnd_broadcast(&wa->ps->result_not_full); + mtx_unlock(&wa->ps->result_mutex); + for (int j = i; j < wa->dir_count; j++) + free(wa->dirs[j]); + break; + } + /* Root .rsync-filter rules (parsed by the parallel scanner) apply to the + * contents of every assigned subdirectory. Relative paths (used by the + * allow-set and per-directory rules) are computed against the transfer + * root, not the subdirectory the worker is seeded with. Exclusion + * recording shares one caller-owned list across the workers. */ + free(ds->root_path); + ds->root_path = str_dup(wa->root_dir); + ds->seed_node = wa->ps->root_filter_node; + ds->options.excluded_mutex = &wa->ps->result_mutex; + Chunk* chunk; + while ((chunk = directory_scanner_next(ds)) != NULL) { + if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex, + &wa->ps->result_not_empty, &wa->ps->result_not_full, + &wa->ps->cancelled)) { + chunk_destroy(chunk); + break; + } + } + if (directory_scanner_failed(ds)) { + mtx_lock(&wa->ps->result_mutex); + wa->ps->failed = true; + atomic_store(&wa->ps->cancelled, true); + cnd_broadcast(&wa->ps->result_not_empty); + cnd_broadcast(&wa->ps->result_not_full); + mtx_unlock(&wa->ps->result_mutex); + } else if (directory_scanner_had_io_error(ds)) { + /* --ignore-errors path: an unreadable directory was skipped, not fatal. */ + mtx_lock(&wa->ps->result_mutex); + wa->ps->io_error = true; + mtx_unlock(&wa->ps->result_mutex); + } + directory_scanner_destroy(ds); + free(wa->dirs[i]); + } + ParallelScanner* ps = wa->ps; + free(wa->root_dir); + free(wa->dirs); + free(wa); + mtx_lock(&ps->result_mutex); + ps->completed++; + if (ps->completed >= ps->expected_threads) { + ps->done = true; + cnd_signal(&ps->result_not_empty); + } + mtx_unlock(&ps->result_mutex); + if (allocation_session) + protocol_session_unbind(); + return thrd_success; +} + +static void parallel_scanner_creation_failed(ParallelScanner* ps) { + mtx_lock(&ps->result_mutex); + ps->failed = true; + atomic_store(&ps->cancelled, true); + ps->expected_threads = ps->created_threads; + if (ps->completed >= ps->expected_threads) + ps->done = true; + cnd_broadcast(&ps->result_not_empty); + cnd_broadcast(&ps->result_not_full); + mtx_unlock(&ps->result_mutex); +} + +/* Initialize result queue and synchronization primitives. Returns true on success. */ +static bool parallel_scanner_init(ParallelScanner* ps) { + ps->result_queue = queue_create(100, chunk_destroy); + if (!ps->result_queue) + return false; + atomic_init(&ps->cancelled, false); + int init = 0; + bool ok = true; + if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success) + ok = false; + if (ok) { + init++; + if (cnd_init(&ps->result_not_empty) != thrd_success) + ok = false; + } + if (ok) { + // cppcheck-suppress unreadVariable + init++; + if (cnd_init(&ps->result_not_full) != thrd_success) + ok = false; + } + if (!ok) { + if (init >= 3) + cnd_destroy(&ps->result_not_full); + if (init >= 2) + cnd_destroy(&ps->result_not_empty); + if (init >= 1) + mtx_destroy(&ps->result_mutex); + queue_destroy(ps->result_queue); + ps->result_queue = NULL; + return false; + } + return true; +} + +/* Split files into chunks of roughly chunk_size bytes. Returns the first chunk (also stored + * chunks beyond the first are enqueued on `queue`). Nulls out consumed entries in `files`. + * Sets *failed on allocation/enqueue errors. */ +static Chunk* batch_files(ArrayList* files, unsigned long long chunk_size, Queue* queue, + bool* failed) { + Chunk* first = NULL; + if (files->size <= 0) + return NULL; + ArrayList* batch = array_list_create(NULL); + if (!batch) { + *failed = true; + return NULL; + } + unsigned long long batch_size = 0; + for (int i = 0; i < files->size; i++) { + File* f = (File*)files->items[i]; + if (!array_list_add(batch, f)) { + *failed = true; + break; + } + batch_size += f->data->size; + if (batch_size >= chunk_size || i == files->size - 1) { + void** items = array_list_to_array(batch); + if (!items) { + *failed = true; + array_list_delete(batch); + batch = NULL; + break; + } + Chunk* c = chunk_create((File**)items, batch->size); + free(items); + if (!c) { + *failed = true; + array_list_delete(batch); + batch = NULL; + break; + } + int batch_start = i - batch->size + 1; + for (int j = batch_start; j <= i; j++) + files->items[j] = NULL; + batch->item_destroyer = NULL; + array_list_delete(batch); + batch = NULL; + if (!first) { + first = c; + } else { + if (!queue_enqueue(queue, c)) { + chunk_destroy(c); + *failed = true; + } + } + if (i < files->size - 1) { + batch = array_list_create(NULL); + if (!batch) { + *failed = true; + break; + } + batch_size = 0; + } + } + } + if (batch) { + batch->item_destroyer = NULL; + array_list_delete(batch); + } + return first; +} + +/* Scan one root-directory entry into either the subdirs or files list. */ +static void scan_root_entry(const ScannerOptions* options, const FilterNode* root_node, + const char* root_directory, const struct dirent* entry, + ArrayList* root_files, ArrayList* subdirs, dev_t root_dev, + ParallelScanner* ps) { + ScannerEntry inspected; + int inspection = + scanner_inspect_entry(options, root_directory, entry->d_name, entry->d_name, &inspected); + if (inspection < 0) { + ps->failed = true; + return; + } + if (inspection == 0) { + if (inspected.referent_error) + ps->io_error = true; + ArrayList* sink = NULL; + if (inspected.excluded) + sink = inspected.size_excluded ? options->size_skipped_paths : options->excluded_paths; + if (sink) { + /* A root-level prune protects the destination mirror of the entry's wire + path: under -R + --files-from that is the bare relative name, otherwise + it is the full source path with a leading '/' removed (matching the + send_path/file_wire_path the scanner hands the sender). */ + if (options->relative && options->file_list != NULL) { + if (!excluded_sink_append(sink, options->excluded_mutex, entry->d_name)) + ps->failed = true; + } else if (options->relative_prefix) { + char* wrel = scanner_prefix_send_path(options->relative_prefix, entry->d_name); + if (!wrel) { + ps->failed = true; + } else { + if (!excluded_sink_append(sink, options->excluded_mutex, wrel)) + ps->failed = true; + free(wrel); + } + } else { + char* abs_path = path_cat(root_directory, entry->d_name); + if (!abs_path) { + ps->failed = true; + } else { + const char* rel = *abs_path == '/' ? abs_path + 1 : abs_path; + if (!excluded_sink_append(sink, options->excluded_mutex, rel)) + ps->failed = true; + free(abs_path); + } + } + } + return; + } + char* cur_path = inspected.path; + struct stat st = inspected.stats; + bool is_dir = inspected.is_directory; + char* rel = str_dup(entry->d_name); + if (!rel) { + free(cur_path); + ps->failed = true; + return; + } + bool protect = false; + bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel, + entry->d_name, is_dir, options->per_dir_filters, + options->exclude_per_dir_filter_files, &protect); + /* -R + --files-from: root-level files keep their bare relative send path. */ + bool use_rel = options->relative && options->file_list != NULL; + if (!passes || protect) { + /* --files-from subset pruning is not a filter exclusion; -R bare-wire-path + exclusions are never recorded (see ScannerOptions.excluded_paths). */ + bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel); + if ((!files_from_prune && !use_rel) || protect) { + const char* rel_path; + char* prefixed = NULL; + if (use_rel) { + /* -R + --files-from: the destination/wire path is the bare relative + name, not the source path. */ + rel_path = rel; + } else if (options->relative_prefix) { + prefixed = scanner_prefix_send_path(options->relative_prefix, entry->d_name); + if (!prefixed) { + free(rel); + free(cur_path); + ps->failed = true; + return; + } + rel_path = prefixed; + } else { + rel_path = *cur_path == '/' ? cur_path + 1 : cur_path; + } + if (options->excluded_paths && + !excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path)) + ps->failed = true; + free(prefixed); + } + if (!passes) { + scanner_note_filter(options, entry->d_name); + free(rel); + free(cur_path); + return; + } + } + if (is_dir) { + if (!scanner_same_filesystem(options->one_file_system, root_dev, st.st_dev)) { + if (options->one_file_system > 1) { + /* -xx: drop the mount-point directory entirely (rsync) and print the + --info=mount line when enabled. */ + scanner_note_mount(options, cur_path); + free(rel); + free(cur_path); + return; + } + /* -x/--one-file-system: emit the mount-point directory entry (empty) but + do not descend into it (see the sequential scanner for the same rule). */ + File* mount = file_create(cur_path); + free(cur_path); + if (mount == NULL) { + free(rel); + ps->failed = true; + return; + } + mount->is_dir = true; + if (options->use_metadata) { + mount->metadata = file_metadata_create(mount->path, &st, options->preserve_atimes, + options->preserve_crtimes); + if (!mount->metadata) { + free(rel); + file_destroy(mount); + ps->failed = true; + return; + } + } + if (options->relative_prefix) { + mount->send_path = scanner_prefix_send_path(options->relative_prefix, rel); + if (!mount->send_path) { + free(rel); + file_destroy(mount); + ps->failed = true; + return; + } + } + free(rel); + if (!array_list_add(root_files, mount)) { + file_destroy(mount); + ps->failed = true; + } + return; + } + free(rel); + if (!array_list_add(subdirs, cur_path)) { + free(cur_path); + ps->failed = true; + } + return; + } + File* file = file_create(cur_path); + free(cur_path); + if (!file) { + free(rel); + free(inspected.link_target); + inspected.link_target = NULL; + ps->failed = true; + return; + } + if (inspected.is_symlink) { + file->is_symlink = true; + file->symlink_target = inspected.link_target; + inspected.link_target = NULL; + } else { + file->data->size = st.st_size; + } + if (use_rel) { + file->send_path = rel; + rel = NULL; + } else if (options->relative_prefix) { + file->send_path = scanner_prefix_send_path(options->relative_prefix, rel); + free(rel); + rel = NULL; + if (!file->send_path) { + file_destroy(file); + ps->failed = true; + return; + } + } + ScannerSpecial special = scanner_prepare_special( + options->preserve_devices, options->preserve_specials, options->copy_devices, file, &st); + if (special == SCANNER_SPECIAL_SKIP) { + scanner_note_nonreg(ps->options, file->path); + free(rel); + file_destroy(file); + return; + } + if (options->hardlinks && S_ISREG(st.st_mode)) { + int gid; + bool is_first; + char* first_path = NULL; + if (!hardlink_table_assign((HardLinkTable*)options->hardlinks, file_wire_path(file), st.st_dev, + st.st_ino, &gid, &is_first, &first_path)) { + ps->failed = true; + } else { + file->link_group = gid; + file->link_first = is_first; + if (!is_first) { + file->hardlink_target = first_path; + file->data->size = 0; + } else { + free(first_path); + } + } + } + if (options->use_metadata) + file->metadata = + file_metadata_create(file->path, &st, options->preserve_atimes, options->preserve_crtimes); + if (options->use_metadata && !file->metadata) { + free(rel); + file_destroy(file); + ps->failed = true; + return; + } + if ((options->preserve_xattrs || options->preserve_acls) && + !(file->link_group != 0 && !file->link_first)) + file->xattrs = xattr_capture_path(file->path, options->preserve_acls); + if (!array_list_add(root_files, file)) { + free(rel); + file_destroy(file); + ps->failed = true; + return; + } + free(rel); +} + +/* Scan the root directory itself, collecting root files and subdirectories. + * Returns false if the root directory could not be opened. */ +static bool scan_root_directory(ParallelScanner* ps, const char* root_directory, + const ScannerOptions* options, const FilterNode* root_node, + dev_t root_dev, ArrayList* root_files, ArrayList* subdirs) { + DIR* dir = opendir(root_directory); + if (!dir) { + log_perror("Could not open root directory for parallel scan"); + return false; + } + /* The parallel scanner opens the transfer root directly (not through + open_next_directory), so record it as synchronized here. */ + if (!scanner_record_synced_dir(options, root_directory, "", + options->relative && options->file_list != NULL)) { + closedir(dir); + ps->failed = true; + return false; + } + log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", root_directory); + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + scan_root_entry(options, root_node, root_directory, entry, root_files, subdirs, root_dev, ps); + } + closedir(dir); + return true; +} + +/* Spawn worker threads, one per group of subdirectories. */ +static void spawn_parallel_workers(ParallelScanner* ps, ArrayList* subdirs, + const ScannerOptions* options, const char* root_directory, + unsigned long long cs) { + if (subdirs->size <= 0) + return; + int n = options->num_threads > 0 ? options->num_threads : 4; + if (n > subdirs->size) + n = subdirs->size; + + ps->num_threads = n; + ps->expected_threads = n; + ps->threads = calloc(n, sizeof(thrd_t)); + if (!ps->threads) { + ps->num_threads = 0; + ps->expected_threads = 0; + ps->failed = true; + return; + } + int dirs_per_thread = subdirs->size / n; + int remainder = subdirs->size % n; + int start = 0; + ps->num_threads = 0; + for (int t = 0; t < n; t++) { + int count = dirs_per_thread + (t < remainder ? 1 : 0); + if (count == 0) + break; + ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg)); + if (!wa) { + parallel_scanner_creation_failed(ps); + break; + } + wa->ps = ps; + wa->dirs = calloc(count, sizeof(char*)); + wa->root_dir = str_dup(root_directory); + if (!wa->dirs || !wa->root_dir) { + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + bool dup_ok = true; + for (int j = 0; j < count; j++) { + wa->dirs[j] = str_dup((char*)subdirs->items[start + j]); + if (!wa->dirs[j]) + dup_ok = false; + } + if (!dup_ok) { + for (int j = 0; j < count; j++) + free(wa->dirs[j]); + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + wa->dir_count = count; + wa->options = *options; + wa->options.chunk_size = cs; + wa->allocation_session = ps->allocation_session; + start += count; + if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) { + for (int j = 0; j < count; j++) + free(wa->dirs[j]); + free(wa->root_dir); + free(wa->dirs); + free(wa); + parallel_scanner_creation_failed(ps); + break; + } + ps->num_threads++; + ps->created_threads++; + } +} + +ParallelScanner* parallel_scanner_create_with_options(const char* root_directory, + const ScannerOptions* options, + ProtocolSession* allocation_session) { + if (!root_directory || !options) + return NULL; + ParallelScanner* ps = calloc(1, sizeof(ParallelScanner)); + if (!ps) + return NULL; + if (!parallel_scanner_init(ps)) { + free(ps); + return NULL; + } + ps->allocation_session = allocation_session; + ps->options = options; + + ArrayList* root_files = array_list_create(file_destroy); + ArrayList* subdirs = array_list_create(free); + if (!root_files || !subdirs) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + + dev_t root_dev = 0; + if (options->one_file_system) { + struct stat root_stats; + if (stat(root_directory, &root_stats) != 0) { + log_perror("Could not stat source directory"); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + root_dev = root_stats.st_dev; + } + + /* Build the root directory's per-directory filter context once; workers seed + * their scanners with it so per-dir rules behave identically to the sequential + * scanner. */ + FilterNode* root_node = NULL; + { + char err[256]; + bool any_exists = false; + FilterRuleList* own = + read_dir_filters(options, root_directory, "", &any_exists, err, sizeof(err)); + if (!own) { + /* A parse/allocation failure must fail the scan even when an earlier + merge file in the same directory existed (see the sequential scanner). */ + if (err[0] != '\0') { + log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + /* no files exist: leave root_node NULL */ + } else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) { + root_node = filter_node_alloc(NULL, own); + if (!root_node) { + filter_rule_list_free(own); + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + } else { + filter_rule_list_free(own); + } + } + ps->root_filter_node = root_node; + + if (!scan_root_directory(ps, root_directory, options, root_node, root_dev, root_files, subdirs)) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + /* The root itself is a traversed directory (rsync counts it in + `Number of files`); the worker DirectoryScanners account for every + subdirectory below it. */ + scanner_dir_count_count(options); + /* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the + transfer root itself (it hands the root's immediate subdirectories to + workers), so capture the root's directory time here. */ + if (options->capture_dir_times && + !scanner_capture_dir_time( + options->dir_entries, options->dir_entries_mutex, root_directory, root_directory, + options->relative && options->file_list != NULL, options->relative_prefix, + options->preserve_atimes, options->preserve_crtimes, options->preserve_xattrs, + options->preserve_acls, options->no_implied_dirs, options->file_list)) { + array_list_delete(root_files); + array_list_delete(subdirs); + parallel_scanner_destroy(ps); + return NULL; + } + + unsigned long long cs = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE; + ps->initial_chunk = batch_files(root_files, cs, ps->result_queue, &ps->failed); + array_list_delete(root_files); + + spawn_parallel_workers(ps, subdirs, options, root_directory, cs); + array_list_delete(subdirs); + return ps; +} + +Chunk* parallel_scanner_next(ParallelScanner* ps) { + if (ps->initial_chunk) { + Chunk* c = ps->initial_chunk; + ps->initial_chunk = NULL; + return c; + } + if (ps->num_threads == 0) { + mtx_lock(&ps->result_mutex); + if (!queue_is_empty(ps->result_queue)) { + Chunk* chunk = queue_dequeue(ps->result_queue); + mtx_unlock(&ps->result_mutex); + return chunk; + } + ps->done = true; + mtx_unlock(&ps->result_mutex); + return NULL; + } + Chunk* chunk = queue_dequeue_multithreaded( + ps->result_queue, &ps->result_mutex, &ps->result_not_empty, &ps->result_not_full, &ps->done); + return chunk; +} + +bool parallel_scanner_failed(const ParallelScanner* ps) { + return ps == NULL || ps->failed; +} + +bool parallel_scanner_had_io_error(const ParallelScanner* ps) { + return ps != NULL && ps->io_error; +} + +void parallel_scanner_destroy(ParallelScanner* ps) { + if (!ps) + return; + mtx_lock(&ps->result_mutex); + ps->done = true; + atomic_store(&ps->cancelled, true); + cnd_broadcast(&ps->result_not_empty); + cnd_broadcast(&ps->result_not_full); + mtx_unlock(&ps->result_mutex); + for (int i = 0; i < ps->num_threads; i++) + thrd_join(ps->threads[i], NULL); + free(ps->threads); + if (ps->root_filter_node) + filter_node_destroy(ps->root_filter_node); + if (ps->initial_chunk) + chunk_destroy(ps->initial_chunk); + queue_destroy(ps->result_queue); + mtx_destroy(&ps->result_mutex); + cnd_destroy(&ps->result_not_empty); + cnd_destroy(&ps->result_not_full); + free(ps); +} -- 2.54.0 From d2d1b63f4478266126a410df34a1370b6bb2eb5e Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:38:54 +0200 Subject: [PATCH 36/68] refactor(receive): split file_save/incremental_check/delete_commit out Pure structural split of src/shared/file_receive.c into focused translation units behind the unchanged file_receive.h facade: - file_save.c : save-to-disk, special nodes, --delay-updates staging - incremental_check.c : xattr/delta/basis/fuzzy receive + check state machine - delete_commit.c : manifest receive + delete budget walkers - file_receive.c : wire receive dispatch + deferred dir metadata The shared receive_file_xattrs helper and MAX_FILE_DATA_SIZE are declared in incremental_check.h. file_save_to_disk_full_ex is decomposed into static helpers (validation, special dispatch, dir/symlink creation, path resolution, pre-write policies, data install) routed through one cleanup epilogue. No behavior change. --- CMakeLists.txt | 3 + src/shared/delete_commit.c | 635 ++++++ src/shared/delete_commit.h | 110 ++ src/shared/file_receive.c | 3287 +------------------------------- src/shared/file_receive.h | 151 +- src/shared/file_save.c | 1164 +++++++++++ src/shared/file_save.h | 40 + src/shared/incremental_check.c | 1676 ++++++++++++++++ src/shared/incremental_check.h | 41 + 9 files changed, 3680 insertions(+), 3427 deletions(-) create mode 100644 src/shared/delete_commit.c create mode 100644 src/shared/delete_commit.h create mode 100644 src/shared/file_save.c create mode 100644 src/shared/file_save.h create mode 100644 src/shared/incremental_check.c create mode 100644 src/shared/incremental_check.h diff --git a/CMakeLists.txt b/CMakeLists.txt index a55d93e..dbddf3b 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -99,17 +99,20 @@ set(SHARED_SRCS src/shared/daemon_limits.c src/shared/data.c src/shared/delay_updates.c + src/shared/delete_commit.c src/shared/delete_plan.c src/shared/delta.c src/shared/file.c src/shared/file_list.c src/shared/file_receive.c + src/shared/file_save.c src/shared/file_send.c src/shared/file_store.c src/shared/filter.c src/shared/format.c src/shared/hardlink.c src/shared/identity.c + src/shared/incremental_check.c src/shared/log.c src/shared/metadata.c src/shared/motd.c diff --git a/src/shared/delete_commit.c b/src/shared/delete_commit.c new file mode 100644 index 0000000..a522dd8 --- /dev/null +++ b/src/shared/delete_commit.c @@ -0,0 +1,635 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "array_list.h" +#include "charset.h" +#include "chmod.h" +#include "chunk.h" +#include "compression.h" +#include "config.h" +#include "data.h" +#include "delay_updates.h" +#include "delete_commit.h" +#include "delta.h" +#include "file.h" +#include "format.h" +#include "identity.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include "xattr.h" + +#define MAX_SERVER_DELETE_COUNT 100000U +/* Retained cost of one delete-manifest entry beyond its path bytes: the + ArrayList pointer slot plus an approximate malloc header/rounding for the + heap copy. Charged against MAX_MANIFEST_BYTES so a frame full of tiny paths + cannot retain far more than the byte budget (B5). */ +#define MANIFEST_ENTRY_OVERHEAD (sizeof(char*) + 16) + +/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already + been consumed): a keep-set entry count followed by that many + destination-relative paths, then a protected-prefix count followed by that + many destination-relative prefixes, then a missing-args count followed by that + many destination-relative delete paths, then (protocol 2.23.0) a + synchronized-directory count followed by that many destination-relative + directory paths (the receive root is the "." sentinel). The frame is + self-delimiting (the counts are authoritative), so the caller decides what to + do next and continues reading the following STATUS_* frame. Every section is + validated identically: an entry must be non-empty, relative and traversal-free + and the aggregate length across ALL sections is capped by MAX_MANIFEST_BYTES + (so the missing-args deletion requests are confined like the rest of the + manifest). Returns an owned DeleteManifest, or NULL after sending STATUS_ERROR + when the frame is malformed (bad count, empty/absolute path, path traversal, + or an aggregate size beyond MAX_MANIFEST_BYTES). */ +static bool receive_manifest_section(int fd, ArrayList* list, size_t* manifest_bytes, + size_t* manifest_entries) { + int count; + if (!receive_int(fd, &count)) { + send_status(fd, STATUS_ERROR); + return false; + } + if (count < 0 || count > MAX_MANIFEST_ENTRIES || + (size_t)count > MAX_MANIFEST_ENTRIES - *manifest_entries) { + send_status(fd, STATUS_ERROR); + return false; + } + for (int i = 0; i < count; i++) { + char* s = receive_wire_str(fd); + size_t entry_size = s ? strlen(s) + MANIFEST_ENTRY_OVERHEAD : 0; + if (!s || s[0] == '\0' || s[0] == '/' || has_path_traversal(s) || + entry_size > MAX_MANIFEST_BYTES - *manifest_bytes || + (*manifest_bytes += entry_size) > MAX_MANIFEST_BYTES || !array_list_add(list, s)) { + free(s); + send_status(fd, STATUS_ERROR); + return false; + } + } + *manifest_entries += (size_t)count; + return true; +} + +DeleteManifest* receive_manifest_entries(int fd) { + DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest)); + if (!manifest) { + send_status(fd, STATUS_ERROR); + return NULL; + } + manifest->keeps = array_list_create(free); + manifest->protected = array_list_create(free); + manifest->missing = array_list_create(free); + manifest->dirs = array_list_create(free); + if (!manifest->keeps || !manifest->protected || !manifest->missing || !manifest->dirs) { + delete_manifest_free(manifest); + send_status(fd, STATUS_ERROR); + return NULL; + } + size_t manifest_bytes = 0; + size_t manifest_entries = 0; + if (!receive_manifest_section(fd, manifest->keeps, &manifest_bytes, &manifest_entries) || + !receive_manifest_section(fd, manifest->protected, &manifest_bytes, &manifest_entries) || + !receive_manifest_section(fd, manifest->missing, &manifest_bytes, &manifest_entries) || + !receive_manifest_section(fd, manifest->dirs, &manifest_bytes, &manifest_entries)) { + delete_manifest_free(manifest); + return NULL; + } + return manifest; +} + +void delete_manifest_free(DeleteManifest* manifest) { + if (!manifest) + return; + array_list_delete(manifest->keeps); + array_list_delete(manifest->protected); + array_list_delete(manifest->missing); + array_list_delete(manifest->dirs); + free(manifest); +} + +/* Shared --max-delete budget for one receiver-side deletion commit. Both the + --delete-missing-args exact-path removals and the ordinary extras walk draw + from the same tally, matching rsync (whose --max-delete counts every deleted + file or directory). `max_delete` is SIZE_MAX for an unlimited budget. */ +typedef struct { + size_t max_delete; + size_t deleted; + size_t skipped; + bool limit_hit; +} DeleteBudgetState; + +/* Build the delete-walk protection prefix for one basis directory. The walker + compares paths relative to the receive root, so a relative entry is already + in the right form; an absolute entry that lies below the root is converted to + its root-relative form, and one outside the root returns NULL (the walk + cannot reach it, and it is not protected data beneath the root). Exposed so + tests can exercise the root-of-"/" child mapping directly. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path) { + if (!path) + return NULL; + if (path[0] != '/') + return str_dup(path); + const char* root = config->receive_root_directory; + if (!root || root[0] != '/') + return NULL; + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(path, root, root_len) != 0) + return NULL; + if (root_len == 1) { + /* `root` is "/" (the only single-character absolute root): every absolute + path is below it, and the child relative form is everything after the + leading '/'. */ + if (path[1] == '\0') + return NULL; /* identical to the root, not a child */ + return str_dup(path + 1); + } + if (path[root_len] != '/') + return NULL; /* identical or a sibling sharing a name prefix */ + return str_dup(path + root_len + 1); +} + +/* Remove every destination entry under the receive root that is not in the + keep-set, bounded by the shared budget (a smaller client --max-delete=NUM + replaces the server hard bound; rsync deletes up to the bound and skips the + rest). With --delay-updates the not-yet-published staging directory is a + direct child of the receive root and must not be treated as a set of extras; + the manifest's protected prefixes (paths excluded on the source), the + size-pruned prefixes (--max-size/--min-size, always protected) and the + alternate basis directories are never destination content and are skipped at + any depth. Returns true unless a traversal/unlink error aborted the walk; + the budget's limit_hit/skipped fields report a cap-stopped run. */ +static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest* manifest, + DeleteBudgetState* budget, DeletePathObserver observer, + void* observer_context) { + if (!config || !manifest || !manifest->keeps) + return false; + fprintf(stderr, "Deleting files not in manifest...\n"); + /* Protected entries: + - the --delay-updates staging name, protected only as a DIRECT child of the + receive root (a nested destination directory that happens to be named + .fastsync-stage is ordinary content); + - alternate basis directories (--compare-dest / --copy-dest / --link-dest) + at any depth: they are extra comparison snapshots the user pointed at, + not destination content, and deleting them would destroy the very files a + --link-dest run just linked into place; + - the sender-side protected prefixes (source paths excluded by filters and + paths pruned by --max-size/--min-size), at any depth, so their destination + mirror survives --delete unless --delete-excluded opts back into removing + the filter-excluded ones (size-pruned entries are always protected). */ + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + + (manifest->protected ? manifest->protected->size : 0); + DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); + return false; + } + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + /* An absolute basis outside the receive root is unreachable by this walk, + so it contributes no protection prefix (and no slot). */ + char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; + skips[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < manifest->protected->size; i++) { + skips[idx].prefix = (const char*)manifest->protected->items[i]; + skips[idx].top_level_only = false; + idx++; + } + used = idx; + } + /* Clamp rather than subtract: an accounting bug where deleted already exceeds + max_delete must never underflow into an effectively unlimited budget. */ + size_t remaining; + if (budget->max_delete == SIZE_MAX) + remaining = SIZE_MAX; + else if (budget->deleted >= budget->max_delete) + remaining = 0; + else + remaining = budget->max_delete - budget->deleted; + size_t deleted = 0; + size_t skipped = 0; + DeleteWalkResult result = delete_extras_limited_observed( + config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips, used, + config->protect_rules, &deleted, &skipped, observer, observer_context); + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); + free(skips); + budget->deleted += deleted; + budget->skipped += skipped; + if (result == DELETE_WALK_LIMIT_REACHED) { + budget->limit_hit = true; + return true; + } + if (result != DELETE_WALK_OK) { + log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files"); + return false; + } + return true; +} + +static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest, + DeleteBudgetState* budget) { + return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL); +} + +/* Prefixes every observed path with a fixed subtree root, so a nested walk + (a recursively removed missing-arg directory) reports receive-root-relative + names like the rest of the delete output. */ +typedef struct { + DeletePathObserver inner; + void* inner_context; + const char* prefix; +} PrefixedDeleteObserver; + +static void prefixed_delete_observer(void* context, const char* rel) { + PrefixedDeleteObserver* prefixed = context; + if (!prefixed->inner || !rel) + return; + char* joined = path_cat((char*)prefixed->prefix, rel); + if (joined) { + prefixed->inner(prefixed->inner_context, joined); + free(joined); + } +} + +/* --delete-missing-args exact-path deletions: each destination mirror in + manifest->missing is an explicit user request, so it is removed even when the + ordinary extras walk (with its protected prefixes) would leave it alone. The + --delay-updates staging directory and basis snapshots are receiver artifacts + and stay protected exactly as in the extras walker. A regular file or + symlink is unlinked, an empty directory removed, and a NON-empty directory is + removed recursively only when --delete or --force is in effect (rsync parity: + the man page says a non-empty directory mirror is only deleted with --force + or --delete); otherwise it is left with a warning and the run continues. A + mirror that does not exist is a no-op. Each removal draws from the shared + --max-delete budget: once it is exhausted the remaining requests are skipped + and counted. Returns false only on a genuine error (a confinement failure on + a validated path or an I/O error), which fails the run. */ +static bool delete_missing_args_budgeted_observed(const Config* config, DeleteManifest* manifest, + DeleteBudgetState* budget, + DeletePathObserver observer, + void* observer_context) { + if (!config || !manifest) + return false; + if (!manifest->missing || manifest->missing->size == 0) + return true; + fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; + DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); + return false; + } + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; + skips[idx].top_level_only = false; + idx++; + } + used = idx; + } + bool ok = true; + for (int i = 0; i < manifest->missing->size; i++) { + const char* rel = (const char*)manifest->missing->items[i]; + if (!rel || *rel == '\0' || *rel == '/' || has_path_traversal(rel)) { + /* Defensive only: receive_manifest_entries already validated every + section identically, so a controlled peer never reaches this branch. */ + log_message(LOG_LEVEL_ERROR, "invalid missing-args delete path"); + ok = false; + continue; + } + bool at_root = strchr(rel, '/') == NULL; + if (path_under_skip_prefix(rel, at_root, skips, used)) { + char* escaped = output_escape(rel, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "missing-args path '%s' is protected (staging directory or basis snapshot); " + "not deleting", + escaped ? escaped : ""); + free(escaped); + continue; + } + char* full = path_cat(config->receive_root_directory, rel); + if (!full) { + ok = false; + continue; + } + char* leaf = NULL; + int parent_fd = file_open_secure_parent(full, &leaf, false); + if (parent_fd < 0) { + /* The mirror's parent directory may itself not exist on the destination + (a deeper missing entry whose leading directories were never created). + That is a no-op -- there is nothing to delete -- matching + file_remove_tree_secure's absent-path handling; only a genuine I/O + error (EACCES, a symlink loop, ...) fails the run. */ + bool absent = errno == ENOENT || errno == ENOTDIR; + free(full); + free(leaf); + if (!absent) + ok = false; + continue; + } + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) { + /* Already absent: nothing to delete (a no-op, not a deletion). */ + if (errno != ENOENT) + ok = false; + close(parent_fd); + free(leaf); + free(full); + continue; + } + /* An entry that exists is one deletion: skip it (and count it) when the + shared --max-delete budget is already exhausted. */ + if (budget->deleted >= budget->max_delete) { + budget->limit_hit = true; + budget->skipped++; + close(parent_fd); + free(leaf); + free(full); + continue; + } + bool removed = false; + if (S_ISDIR(st.st_mode)) { + if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) { + removed = true; + } else if (errno == ENOTEMPTY || errno == EEXIST) { + close(parent_fd); + parent_fd = -1; + free(leaf); + leaf = NULL; + if (config->use_delete || config->force_delete) { + /* Remove the contents entry-by-entry through the budgeted extras + walker so every deleted file/dir counts toward --max-delete (rsync + parity); the now-empty directory itself costs one more. A run that + hits the cap leaves the remaining entries in place. */ + ArrayList* no_keeps = array_list_create(free); + /* Never let an accounting slip (deleted > max_delete) underflow the + remaining budget into SIZE_MAX, which would grant unlimited + deletions. */ + size_t remaining = + budget->deleted >= budget->max_delete ? 0 : budget->max_delete - budget->deleted; + size_t contents_deleted = 0; + size_t contents_skipped = 0; + PrefixedDeleteObserver nested = {observer, observer_context, rel}; + DeleteWalkResult walk = + no_keeps ? delete_extras_limited_observed(full, no_keeps, NULL, remaining, NULL, 0, + NULL, &contents_deleted, &contents_skipped, + observer ? prefixed_delete_observer : NULL, + observer ? &nested : NULL) + : DELETE_WALK_ERROR; + if (no_keeps) + array_list_delete(no_keeps); + budget->deleted += contents_deleted; + budget->skipped += contents_skipped; + if (walk == DELETE_WALK_LIMIT_REACHED) { + budget->limit_hit = true; + } else if (walk != DELETE_WALK_OK) { + ok = false; + } else if (budget->deleted >= budget->max_delete) { + budget->limit_hit = true; + budget->skipped++; + } else if (file_remove_tree_secure(full)) { + /* The shared `if (removed)` tail charges this directory exactly + once; counting it here too would consume two budget units. */ + removed = true; + } else { + ok = false; + } + } else { + char* escaped = output_escape(rel, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "missing-args destination '%s' is a non-empty directory; use --force or " + "--delete to remove it", + escaped ? escaped : ""); + free(escaped); + } + } else if (errno != ENOENT) { + ok = false; + } + } else { + if (unlinkat(parent_fd, leaf, 0) == 0) { + removed = true; + } else if (errno != ENOENT) { + ok = false; + } + } + if (removed) { + budget->deleted++; + if (observer) + observer(observer_context, rel); + char* escaped = output_escape(rel, log_get_8_bit_output()); + fprintf(stderr, " Deleted: %s\n", escaped ? escaped : ""); + free(escaped); + } + if (parent_fd >= 0) + close(parent_fd); + free(leaf); + free(full); + if (!ok) + break; + } + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); + free(skips); + return ok; +} + +/* Public wrappers used outside the commit path (and by unit tests): no + --max-delete budget. */ +bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, + size_t* count_out) { + if (count_out) + *count_out = 0; + if (!config || !manifest || !manifest->keeps || !out) + return false; + int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + + (manifest->protected ? manifest->protected->size : 0); + DeleteSkipEntry* skips = NULL; + char** owned_prefixes = NULL; + int used = 0; + if (skip_count > 0) { + skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); + owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!skips || (config->basis_count > 0 && !owned_prefixes)) { + free(skips); + free(owned_prefixes); + return false; + } + int idx = 0; + if (config->delay_updates) { + skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; + skips[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + /* Normalize exactly like the real commit path: a relative entry is + already root-relative, an absolute one inside the receive root is + converted, and one outside contributes no protection prefix. */ + char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!prefix) + continue; + owned_prefixes[i] = prefix; + skips[idx].prefix = prefix; + skips[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < manifest->protected->size; i++) { + skips[idx].prefix = (const char*)manifest->protected->items[i]; + skips[idx].top_level_only = false; + idx++; + } + used = idx; + } + bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs, + skips, used, config->protect_rules, out, count_out); + if (owned_prefixes) { + for (int i = 0; i < config->basis_count; i++) + free(owned_prefixes[i]); + } + free(owned_prefixes); + free(skips); + return ok; +} + +bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) { + DeleteBudgetState budget = { + .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; + return delete_extras_budgeted(config, manifest, &budget); +} + +bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) { + DeleteBudgetState budget = { + .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; + return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL); +} + +bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, size_t* skipped, + bool* limit_hit) { + return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted, + skipped, limit_hit, NULL, NULL); +} + +bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, + DeletePathObserver observer, + void* observer_context) { + DeleteBudgetState budget = { + .max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; + bool ok = + delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context); + if (deleted) + *deleted = budget.deleted; + if (skipped) + *skipped = budget.skipped; + if (limit_hit) + *limit_hit = budget.limit_hit; + return ok; +} + +/* Commit every deletion family the manifest carries. The --delete-missing-args + exact-path deletions run FIRST: they are explicit user requests and must not + be blocked by the extras walker's filter-exclusion protection (a protected + leftover inside a missing-argument directory must not make that user-requested + removal fail). The ordinary extras walk then runs when --delete is active. + Both draw from one --max-delete budget; the result reports a cap-stopped + (partial) commit distinctly so the client can exit 25 like rsync. */ +DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest) { + return manifest_delete_all_counted(config, manifest, NULL); +} + +DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, + size_t* deleted) { + return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL); +} + +DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, + size_t* deleted, DeletePathObserver observer, + void* observer_context) { + if (deleted) + *deleted = 0; + if (!config || !manifest) + return DELETE_COMMIT_ERROR; + /* Central no-mutation guard: a dry-run never deletes. No manifest is sent on + the dry-run path, but a hostile/buggy peer could; treat it as a no-op so + the receiver can never remove anything. */ + if (config->dry_run) + return DELETE_COMMIT_OK; + /* A client --max-delete=NUM smaller than the server's hard bound replaces it + for this run; both still bound the commit. */ + bool user_limited = + config->max_delete >= 0 && (size_t)config->max_delete < MAX_SERVER_DELETE_COUNT; + DeleteBudgetState budget = {.max_delete = user_limited ? (size_t)config->max_delete + : MAX_SERVER_DELETE_COUNT, + .deleted = 0, + .skipped = 0, + .limit_hit = false}; + if (config->delete_missing_args && + !delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context)) + return DELETE_COMMIT_ERROR; + if (config->use_delete && + !delete_extras_budgeted_observed(config, manifest, &budget, observer, observer_context)) + return DELETE_COMMIT_ERROR; + if (deleted) + *deleted = budget.deleted; + if (budget.limit_hit) { + if (user_limited) { + log_message(LOG_LEVEL_ERROR, "Deletions stopped due to --max-delete limit (%zu skipped)", + budget.skipped); + } else { + log_message(LOG_LEVEL_ERROR, + "Deletions stopped due to the server deletion limit of %u (%zu skipped)", + (unsigned)MAX_SERVER_DELETE_COUNT, budget.skipped); + } + return DELETE_COMMIT_LIMIT_REACHED; + } + return DELETE_COMMIT_OK; +} diff --git a/src/shared/delete_commit.h b/src/shared/delete_commit.h new file mode 100644 index 0000000..c2d374b --- /dev/null +++ b/src/shared/delete_commit.h @@ -0,0 +1,110 @@ +#ifndef DELETE_COMMIT_H +#define DELETE_COMMIT_H + +#include "array_list.h" +#include "config.h" +#include "utils.h" +#include + +/* Delete-commit module: delete-manifest receive plus the budgeted extras and + * --delete-missing-args walkers. These declarations are re-exported by the + * file_receive.h facade. */ + +/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative + paths the sender transferred/keeps) plus `protected`, destination-relative + prefixes the sender asks the receiver never to delete (paths excluded on the + source, protected at any depth). When --delete-excluded is given the sender + transmits an empty protected list so excluded destination mirrors are treated + as ordinary extras. With --delete-missing-args a third section (`missing`) + carries the destination mirrors of explicitly-listed source entries that do + not exist: each is an exact deletion request, independent of the ordinary + extras walk (never blocked by the protected prefixes) and processed when the + manifest is committed. */ +typedef struct DeleteManifest { + ArrayList* keeps; + ArrayList* protected; + ArrayList* missing; + /* Destination-relative paths of the directories the sender synchronized for + this run. The extras walker only removes entries directly inside one of + these (the receive root is the "." sentinel); `--files-from` runs therefore + leave untransmitted directories and the unlisted parts of listed ones + alone, matching rsync's "delete only in synchronized directories". */ + ArrayList* dirs; +} DeleteManifest; + +void delete_manifest_free(DeleteManifest* manifest); +/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then + protected count + protected prefixes, then missing count + missing paths, + then synchronized-directory count + directory paths (self-delimiting; the + leading STATUS_MANIFEST code has been consumed). Returns an owned + DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */ +DeleteManifest* receive_manifest_entries(int fd); +/* Remove destination entries under config->receive_root_directory that are not + in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and + protected-prefix skips). `--max-delete` and `--force` are honored here. The + caller decides WHEN to run it based on the negotiated delete timing. Returns + false (and the transfer fails) when the deletion cannot be committed. */ +bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); +/* --delete-missing-args exact-path deletions: remove each destination mirror + in `manifest->missing` (never blocked by the protected prefixes, staging dir + and basis dirs excluded). A regular file/symlink is unlinked; an empty + directory is removed; a NON-empty directory is removed recursively only when + --delete or --force is in effect, otherwise it is left with a warning (rsync + parity). A missing path is a no-op. Returns false only on a genuine + confinement or I/O error (the run then fails); tolerated per-path cases are + reported and skipped. */ +bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest); +/* Budgeted form of manifest_delete_missing_args for the per-directory delete + session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited) + and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set + when the budget stopped the pass with entries left over. Returns false only + on a genuine deletion error. */ +bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, size_t* skipped, + bool* limit_hit); +/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may + be NULL) is invoked for every destination-relative path truly removed. */ +bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, + size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, + DeletePathObserver observer, + void* observer_context); +/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's + partial --max-delete result: the budget allowed some deletions and the rest + were skipped (the run still stores all file data but the client exits 25). */ +typedef enum { + DELETE_COMMIT_OK = 0, + DELETE_COMMIT_LIMIT_REACHED, + DELETE_COMMIT_ERROR +} DeleteCommitResult; + +/* Run every deletion family the manifest carries: the --delete-missing-args + exact-path deletions first (user requests are not blocked by exclusion + protection), then the ordinary extras walk when --delete is active. Both + share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to + do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget + stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */ +DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest); +/* Like manifest_delete_all, but reports how many destination entries the commit + removed (for the end-of-transfer wire stats). `deleted` may be NULL. */ +DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, + size_t* deleted); +/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL) + is invoked for every destination-relative path truly removed. */ +DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, + size_t* deleted, DeletePathObserver observer, + void* observer_context); + +/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as + the delete pass would and append (strdup'd) destination-relative paths that + WOULD be removed to `out`, without touching disk. Uses the same staging-dir, + basis-dir and protected-prefix skips as the real commit. Returns true on a + clean walk; `*count_out` receives the number of paths appended. */ +bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, + size_t* count_out); +/* Convert one basis-directory path to the receive-root-relative protection + prefix the delete walker uses (NULL when it lies outside the root). Exposed + for unit tests of the root-of-"/" and normalization edge cases. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path); + +#endif diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index 3abbda1..a9e02ab 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -20,2701 +20,16 @@ #include "delay_updates.h" #include "delta.h" #include "file.h" +#include "file_receive.h" #include "format.h" #include "identity.h" +#include "incremental_check.h" #include "log.h" #include "metadata.h" #include "protocol.h" #include "utils.h" #include "xattr.h" -#define MAX_SERVER_DELETE_COUNT 100000U -#define MAX_FILE_DATA_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE -/* Retained cost of one delete-manifest entry beyond its path bytes: the - ArrayList pointer slot plus an approximate malloc header/rounding for the - heap copy. Charged against MAX_MANIFEST_BYTES so a frame full of tiny paths - cannot retain far more than the byte budget (B5). */ -#define MANIFEST_ENTRY_OVERHEAD (sizeof(char*) + 16) - -bool file_save_to_disk(const char* root_directory, const File* file, const Config* config) { - return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL) != FILE_SAVE_ERROR; -} - -FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, - const Config* config) { - return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL); -} - -/* --delay-updates receiver path: write the file into a private staging tree - below the receive root instead of its final destination, and remember it so - it can be atomically renamed into place only once the whole transfer has - succeeded. Existence/update policies (--existing/--ignore-existing/--update) - are decided against the FINAL destination path at stage time so the run - decides exactly what an immediate (non-delayed) run would decide; the staged - file is then never re-checked at publication. Backups are deferred to - publication so the final destination is untouched until the transfer ends. */ -static FileSaveResult file_stage_delayed_update(const char* root_directory, - const char* destination_path, const File* file, - Config* config) { - if (!config) - return FILE_SAVE_ERROR; - bool sparse = config->preserve_sparse; - FileAttrPolicy policy = file_attr_policy_from_config(config); - - if (config->existing && !file_path_exists_secure(destination_path)) - return FILE_SAVE_SKIPPED; - if (config->ignore_existing && file_path_exists_secure(destination_path)) - return FILE_SAVE_SKIPPED; - if (config->update && file_destination_is_newer_secure(destination_path, file->metadata)) - return FILE_SAVE_SKIPPED; - - FileMetadata adjusted_metadata; - const FileMetadata* metadata = file->metadata; - if (metadata && config->chmod_spec && *config->chmod_spec) { - adjusted_metadata = *metadata; - if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) - return FILE_SAVE_ERROR; - metadata = &adjusted_metadata; - } - - if (!config->delay_context) { - config->delay_context = delay_updates_context_create(root_directory); - if (!config->delay_context) - return FILE_SAVE_ERROR; - } - DelayUpdatesContext* context = config->delay_context; - if (!delay_updates_prepare(context)) - return FILE_SAVE_ERROR; - - char* staged_path = path_cat(context->staging_root, file->path); - if (!staged_path) - return FILE_SAVE_ERROR; - - /* The staged location is brand new (stale leftovers from a prior crash were - wiped by prepare), so the plain atomic temp+rename engine installs the - complete file there. --temp-dir scratch is deliberately not layered on - top of the delay-updates staging tree. A --link-dest basis file is hard - linked into the staging tree (so publication's rename keeps the link). */ - bool ok; - if (file->basis_link) { - ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size, - config->preallocate, metadata, policy, config->use_fsync, NULL); - } else if (file->basis_copy) { - /* --copy-dest basis hit: stream the basis into the staging tree (bounded - buffers, so an over-limit basis still stages). */ - ok = file_copy_basis_stream_attrs(staged_path, file->basis_copy, file->data->size, - config->preallocate, metadata, policy, config->update, - config->use_fsync, file->xattrs, config->fake_super, NULL); - } else { - ok = - file_to_disk_secure_attrs(staged_path, file->data->data, file->data->size, false, sparse, - config->preallocate, metadata, policy, false, false, - config->use_fsync, file->xattrs, config->fake_super, false, NULL); - } - if (!ok) { - free(staged_path); - return FILE_SAVE_ERROR; - } - - if (!delay_updates_record(context, staged_path, destination_path, file->path)) { - unlink(staged_path); - free(staged_path); - return FILE_SAVE_ERROR; - } - free(staged_path); - return FILE_SAVE_WRITTEN; -} - -/* Read the whole content of a confined regular file (used to fall back to a - byte-identical copy when a hard-link sibling's link() fails). Symlink-safe - (parent resolved via file_open_secure_parent + O_NOFOLLOW). A zero-length - file yields *out_size 0 and *out_buf NULL as a SUCCESS. Returns false only - on a real error/read failure, setting *source_absent to true when the reason - was that the path does not exist (ENOENT/ENOTDIR), so the caller can decide - between an abort and a graceful skip. */ -static bool hardlink_read_source(const char* path, void** out_buf, unsigned long long* out_size, - bool* source_absent) { - *out_buf = NULL; - *out_size = 0; - *source_absent = false; - if (!path) - return false; - char* leaf = NULL; - int parent_fd = file_open_secure_parent(path, &leaf, false); - if (parent_fd < 0) { - *source_absent = errno == ENOENT || errno == ENOTDIR; - return false; - } - /* O_NONBLOCK is a no-op for a regular file but makes openat() fail/succeed - immediately for a client-planted FIFO instead of blocking the receive - thread forever; the post-open S_ISREG gate below is the actual type check. */ - int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - int saved_errno = errno; - free(leaf); - close(parent_fd); - if (fd < 0) { - *source_absent = saved_errno == ENOENT || saved_errno == ENOTDIR; - return false; - } - struct stat st; - if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode)) { - close(fd); - return false; - } - unsigned long long size = (unsigned long long)st.st_size; - if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) { - close(fd); - return false; - } - if (size == 0) { - close(fd); - return true; - } - void* buf = protocol_alloc((size_t)size); - if (!buf) { - close(fd); - return false; - } - size_t got = 0; - while (got < (size_t)size) { - ssize_t n = read(fd, (char*)buf + got, (size_t)size - got); - if (n <= 0) { - free(buf); - close(fd); - return false; - } - got += (size_t)n; - } - close(fd); - *out_buf = buf; - *out_size = size; - return true; -} - -/* The group's first member's installed file is absent, but its destination - path was validated (a sibling is only ever processed after its group's first - member). When the sibling's OWN destination already exists it should be - left alone -- a clean skip -- rather than aborting the whole transfer (the - asymmetric --existing case: the first member was skipped because its - destination was missing, while the sibling already has one). Only when the - sibling's destination is missing too is this a genuine failure to - link/copy, which aborts. */ -static FileSaveResult hardlink_sibling_absent_first(const char* destination_path) { - if (destination_path && file_path_exists_secure(destination_path)) - return FILE_SAVE_SKIPPED; - return FILE_SAVE_ERROR; -} - -/* Install a --hard-links/-H sibling: the destination entry is atomically - replaced (temp + rename) with a hard link to the group's first member. The - first member is guaranteed already installed at `hardlink_target` under the - root because -H relies on the receiver's single-FIFO-writer pipeline (one - receive thread, one write thread, FIFO queue => wire order == write order) - plus the sender's forced sequential scan, so a sibling is always processed - after its group's first member. When link() fails (different filesystem, - filesystem refuses links) a byte-identical copy of the first member is - written instead, so the result is never partial or corrupt. With - --delay-updates the sibling is staged as a hard link to the first member's - STAGED file (publication's renames preserve the shared inode). The final - --existing/--ignore-existing/--update policies are decided against the final - destination like every normal write. */ -static FileSaveResult file_save_hardlink_sibling(const char* root_directory, const File* file, - const Config* config, bool* created) { - Config* cfg = (Config*)config; - if (!root_directory || !file || !file->path || !file->hardlink_target) - return FILE_SAVE_ERROR; - char* destination_path = path_cat(root_directory, file->path); - if (!destination_path) - return FILE_SAVE_ERROR; - bool existed = file_path_exists_secure(destination_path); - - if (cfg->existing && !file_path_exists_secure(destination_path)) { - free(destination_path); - return FILE_SAVE_SKIPPED; - } - if (cfg->ignore_existing && file_path_exists_secure(destination_path)) { - free(destination_path); - return FILE_SAVE_SKIPPED; - } - if (cfg->update && file_destination_is_newer_secure(destination_path, file->metadata)) { - free(destination_path); - return FILE_SAVE_SKIPPED; - } - - bool preallocate = cfg && cfg->preallocate; - FileAttrPolicy policy = file_attr_policy_from_config(cfg); - bool use_fsync = cfg && cfg->use_fsync; - - if (cfg->delay_updates) { - if (!cfg->delay_context) { - cfg->delay_context = delay_updates_context_create(root_directory); - if (!cfg->delay_context) { - free(destination_path); - return FILE_SAVE_ERROR; - } - } - if (!delay_updates_prepare(cfg->delay_context)) { - free(destination_path); - return FILE_SAVE_ERROR; - } - char* staged_first = path_cat(cfg->delay_context->staging_root, file->hardlink_target); - char* staged_sibling = path_cat(cfg->delay_context->staging_root, file->path); - if (!staged_first || !staged_sibling) { - free(staged_first); - free(staged_sibling); - free(destination_path); - return FILE_SAVE_ERROR; - } - void* content = NULL; - unsigned long long content_size = 0; - bool source_absent = false; - if (!hardlink_read_source(staged_first, &content, &content_size, &source_absent)) { - FileSaveResult absent_result = - source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; - free(staged_first); - free(staged_sibling); - free(destination_path); - return absent_result; - } - FileXattrList* sibling_xattrs = - cfg->use_xattrs ? xattr_capture_path(staged_first, cfg->preserve_acls) : NULL; - bool ok = file_to_disk_secure_link_attrs(staged_sibling, staged_first, content, content_size, - preallocate, file->metadata, policy, use_fsync, - sibling_xattrs, cfg ? cfg->fake_super : false, NULL); - xattr_list_free(sibling_xattrs); - free(content); - if (ok) - ok = delay_updates_record(cfg->delay_context, staged_sibling, destination_path, file->path); - if (!ok) - unlink(staged_sibling); - free(staged_first); - free(staged_sibling); - free(destination_path); - if (ok && created && !existed) - *created = true; - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; - } - - char* first_disk = path_cat(root_directory, file->hardlink_target); - if (!first_disk) { - free(destination_path); - return FILE_SAVE_ERROR; - } - void* content = NULL; - unsigned long long content_size = 0; - bool source_absent = false; - if (!hardlink_read_source(first_disk, &content, &content_size, &source_absent)) { - FileSaveResult absent_result = - source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; - free(first_disk); - free(destination_path); - return absent_result; - } - /* Resolve a relative --temp-dir under the destination root, exactly as the - * primary save path does; an absolute or `..`-escaping value is rejected. */ - char* resolved_temp = NULL; - if (cfg->temp_dir) { - if (cfg->temp_dir[0] == '/' || has_path_traversal(cfg->temp_dir)) { - free(content); - free(first_disk); - free(destination_path); - return FILE_SAVE_ERROR; - } - resolved_temp = path_cat(root_directory, cfg->temp_dir); - if (!resolved_temp) { - free(content); - free(first_disk); - free(destination_path); - return FILE_SAVE_ERROR; - } - } - FileXattrList* sibling_xattrs = - cfg->use_xattrs ? xattr_capture_path(first_disk, cfg->preserve_acls) : NULL; - bool ok = file_to_disk_secure_link_attrs( - destination_path, first_disk, content, content_size, preallocate, file->metadata, policy, - use_fsync, sibling_xattrs, cfg ? cfg->fake_super : false, resolved_temp); - xattr_list_free(sibling_xattrs); - free(resolved_temp); - free(content); - free(first_disk); - free(destination_path); - if (ok && created && !existed) - *created = true; - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; -} - -/* Validate a transmitted special rdev against the node kind implied by `mode`'s - * S_IFMT bits. Char/block devices require a legal major/minor pair (non-negative, - * range-checked); a non-device special (FIFO/socket) must carry an empty rdev. - * Used identically on the wire path and at the secure recreation site so a - * malicious/bogus rdev can never drive a dangerous node. */ -bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) { - bool is_device = S_ISCHR(mode) || S_ISBLK(mode); - if (is_device) - return major >= 0 && minor >= 0 && major <= 0xffff && minor <= 0x00ffffff; - /* A non-device entry must actually be a special (FIFO/socket) and carry no - rdev; a regular/dir mode is never a valid special node. */ - return (S_ISFIFO(mode) || S_ISSOCK(mode)) && major == 0 && minor == 0; -} - -/* ---- Device/special node RECREATION (--devices/--specials), receiver side ---- - * - * Privilege gating: making a real device node requires CAP_MKNOD (root); making - * a FIFO works unprivileged (mkfifo). When the receiver lacks the capability, - * mknodat() fails with EPERM and the entry is SKIPPED with a warning -- the - * whole transfer must NOT abort just because the environment cannot make the - * node. CI runs non-root, so device creation is expected to skip there and - * only a FIFO is honestly assertable unprivileged. - * - * Confinement: the parent directory is opened fd-relative below the receive - * root (file_open_secure_parent: O_NOFOLLOW, no "..", root-checked) and the - * node is created with mknodat()/mkfifoat(), so it can never be placed outside - * the confined root and never follows a symlink. - * - * rdev validation: a malicious/bogus rdev (negative, out-of-range) is rejected - * here as well as on the wire (file_receive_special / chunk_deserialize), and a - * non-device entry must carry an empty rdev. - */ -static FileSaveResult file_save_special_to_disk(const char* root_directory, const File* file, - const Config* config, bool* created) { - /* The empty-path and structural checks stay unconditional; the redundant - ".." list-path re-check is skipped under --trust-sender exactly like the - receive layer (confinement is deferred to the secure parent walk below, - which is never disabled). */ - if (!root_directory || !file || !file->path || file->path[0] == '\0' || - (!file_get_trust_sender() && has_path_traversal(file->path)) || !file->metadata) - return FILE_SAVE_ERROR; - - mode_t mode = file->metadata->mode; - bool is_char = S_ISCHR(mode); - bool is_blk = S_ISBLK(mode); - bool is_fifo = S_ISFIFO(mode); - bool is_sock = S_ISSOCK(mode); - if (!is_char && !is_blk && !is_fifo && !is_sock) { - log_message(LOG_LEVEL_ERROR, "Special node has no device/FIFO/socket mode"); - return FILE_SAVE_ERROR; - } - if (is_char || is_blk) { - if (!config || !config->preserve_devices) - return FILE_SAVE_SKIPPED; - /* --super / --no-super (P7 Wave E): char/block device-node creation is a - super-user activity. --no-super forbids it even for a root receiver; - AUTO and --super attempt it (an unprivileged attempt is refused by the - kernel and skipped). The helper is evaluated against THIS config's mode - so the policy does not depend on a prior identity_set_active(). Pure - FIFO creation is unprivileged and deliberately NOT gated here. */ - if (!privilege_super_mode_permitted(config->super_mode)) { - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "skipping %s: super-user device-node creation is not permitted on this receiver", - escaped_path ? escaped_path : ""); - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - } else if (is_fifo || is_sock) { - /* FIFOs and unix sockets are recreated by --specials. mknod(S_IFSOCK) - works unprivileged on Linux (the node carries no live socket), so unlike - a socket bound to a live fd it can be materialized. */ - if (!config || !config->preserve_specials) - return FILE_SAVE_SKIPPED; - } - /* Defense-in-depth rdev/type validation (also done on the wire path). */ - if (!file_special_rdev_valid(file->rdev_major, file->rdev_minor, mode)) { - log_message(LOG_LEVEL_ERROR, "Rejected out-of-range device rdev %d:%d", file->rdev_major, - file->rdev_minor); - return FILE_SAVE_ERROR; - } - - char* destination = path_cat(root_directory, file->path); - if (!destination) - return FILE_SAVE_ERROR; - bool existed = file_path_exists_secure(destination); - char* leaf = NULL; - int parent_fd = file_open_secure_parent(destination, &leaf, true); - if (parent_fd < 0) { - free(destination); - return FILE_SAVE_ERROR; - } - - /* --existing / --ignore-existing / --update decide against the node that - would be replaced, mirroring the regular-file path. */ - if (config->existing && !file_path_exists_secure(destination)) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - if (config->ignore_existing && file_path_exists_secure(destination)) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - if (config->update && file_destination_is_newer_secure(destination, file->metadata)) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - - dev_t rdev = 0; - mode_t create_mode; - if (is_char) { - create_mode = S_IFCHR; - rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); - } else if (is_blk) { - create_mode = S_IFBLK; - rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); - } else if (is_sock) { - create_mode = S_IFSOCK; - } else { - create_mode = S_IFIFO; - } - const char* node_kind = (is_char || is_blk) ? "device" : (is_fifo ? "FIFO" : "socket"); - /* Under -p/--perms rsync copies the source's permission and special bits; a - * kernel that denies setuid/setgid/sticky reports the failure rather than - * having them masked here. Without -p the node is created like any other new - * entry: source_mode & 0777 & ~umask. When super-user activities are - * forbidden, the special bits are stripped even under -p (they are - * super-user activities just like device-node creation). */ - mode_t perms = config->preserve_perms ? (mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777)) - : (mode & 0777 & ~(mode_t)file_process_umask()); - if (!privilege_super_mode_permitted(config->super_mode)) - perms &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); - - int rc = is_fifo ? mkfifoat(parent_fd, leaf, perms) - : mknodat(parent_fd, leaf, create_mode | perms, rdev); - if (rc != 0) { - if (errno == EEXIST) { - /* An entry already exists: only skip when it already is a matching node; - never replace an existing directory or unrelated entry with the node. */ - struct stat st; - if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0 && - ((is_char && S_ISCHR(st.st_mode)) || (is_blk && S_ISBLK(st.st_mode)) || - (is_fifo && S_ISFIFO(st.st_mode)) || (is_sock && S_ISSOCK(st.st_mode)))) { - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "refusing to replace existing entry with %s: %s (skipped)", - node_kind, escaped_path ? escaped_path : ""); - free(escaped_path); - } else if (errno == EPERM || errno == EACCES) { - /* Missing CAP_MKNOD / parent write permission: the environment cannot - create the node, so skip instead of failing the whole run. */ - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "skipping %s: cannot create %s node (%s)\n" - " --devices/--specials node creation needs privilege (CAP_MKNOD)", - escaped_path ? escaped_path : "", node_kind, strerror(errno)); - free(escaped_path); - } else { - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "failed to create %s %s: %s (skipped)", node_kind, - escaped_path ? escaped_path : "", strerror(errno)); - free(escaped_path); - } - close(parent_fd); - free(leaf); - free(destination); - return FILE_SAVE_SKIPPED; - } - - /* Apply times on the fresh node (utimensat, no-follow) per the negotiated - * per-attribute policy: mtime only under -t, atime only under -U. The slot - * not requested stays UTIME_OMIT so it is left untouched. */ - FileAttrPolicy policy = file_attr_policy_from_config(config); - if (policy.times || (policy.atimes && file->metadata->atime_valid)) { - struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, - {.tv_sec = 0, .tv_nsec = UTIME_OMIT}}; - if (policy.times) { - times[1].tv_sec = file->metadata->mtime_sec; - times[1].tv_nsec = file->metadata->mtime_nsec; - } - if (policy.atimes && file->metadata->atime_valid) { - times[0].tv_sec = file->metadata->atime_sec; - times[0].tv_nsec = file->metadata->atime_nsec; - } - utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW); - } - /* P7 Wave E: apply the negotiated ownership to the node ITSELF. A FIFO is - created unprivileged, but --copy-as and explicit identity policies own - every entry (a char/block node path is already privilege-gated above). The - no-follow helper changes the node's own ownership without dereferencing it; - it is a no-op unless an identity policy is active. */ - bool owner_ok = true; - if (identity_active_enabled()) - owner_ok = identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, - (int32_t)file->metadata->gid); - close(parent_fd); - free(leaf); - free(destination); - /* A failed required --copy-as ownership marks the node as failed; every other - * identity policy stays best-effort. */ - if (owner_ok && created && !existed) - *created = true; - return owner_ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; -} - -/* --write-devices (receiver): write the received data directly into an EXISTING - * device node on the destination instead of creating a regular file. The node - * must already exist and be a char/block device (the device itself is opened and - * followed); it is confined to the receive root via file_open_secure_parent. - * Dangerous by nature, so deliberately restricted: a missing/non-device - * destination, or a write failure, is SKIPPED with a warning rather than - * allowed. On environments without device access the run still succeeds (the - * entry is skipped), never aborts. */ -static FileSaveResult file_save_write_device(const char* root_directory, const File* file) { - if (!root_directory || !file || !file->path || file->path[0] == '\0' || - (!file_get_trust_sender() && has_path_traversal(file->path))) - return FILE_SAVE_ERROR; - if (!file->data) - return FILE_SAVE_ERROR; - char* destination = path_cat(root_directory, file->path); - if (!destination) - return FILE_SAVE_ERROR; - char* leaf = NULL; - int parent_fd = file_open_secure_parent(destination, &leaf, false); - if (parent_fd < 0) { - free(destination); - return FILE_SAVE_SKIPPED; - } - /* O_NONBLOCK: a pre-existing FIFO at the target would otherwise block the - receive thread forever on open(2). With it the open only succeeds for a - readerless FIFO with O_RDWR (which the device fstat gate rejects anyway) - or fails with ENXIO/EAGAIN, both treated as a normal skip below. */ - int fd = openat(parent_fd, leaf, O_WRONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - int saved_errno = errno; - free(leaf); - close(parent_fd); - if (fd < 0) { - free(destination); - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - const char* shown_path = escaped_path ? escaped_path : ""; - if (saved_errno == ENXIO || saved_errno == EAGAIN) { - /* A FIFO with no reader / an unreadable special: skip like every other - unusable write-devices target instead of blocking or failing. */ - log_message(LOG_LEVEL_WARNING, "write-devices: %s not writable (%s); skipped", shown_path, - strerror(saved_errno)); - } else { - log_message(LOG_LEVEL_WARNING, "write-devices: cannot open %s (%s); skipped", shown_path, - strerror(saved_errno)); - } - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - struct stat st; - if (fstat(fd, &st) != 0 || !(S_ISCHR(st.st_mode) || S_ISBLK(st.st_mode))) { - close(fd); - free(destination); - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "write-devices: %s is not a device node; skipped", - escaped_path ? escaped_path : ""); - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - bool ok = true; - if (file->data->size > 0) { - size_t total = (size_t)file->data->size; - size_t written = 0; - while (written < total) { - ssize_t n = write(fd, (char*)file->data->data + written, total - written); - if (n <= 0) { - ok = false; - break; - } - written += (size_t)n; - } - } - close(fd); - free(destination); - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_SKIPPED; -} - -FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, - const Config* config, bool* created, - unsigned* created_dirs) { - if (created) - *created = false; - if (created_dirs) - *created_dirs = 0; - /* Central no-mutation guard: a server-contacting --dry-run (or a local batch - apply that somehow carries dry_run) must never touch the destination, no - matter which caller reached this primitive. The per-caller guards remain, - but this is the last line of defense for every save path. Report SKIPPED - so a --remove-source-files sender correctly keeps its source. */ - if (config && config->dry_run) - return FILE_SAVE_SKIPPED; - /* Backups are incompatible with ignore-existing: moving the entry first - would make a concurrent no-replace commit overwrite its old name. */ - bool backup_enabled = config && config->backup && !config->ignore_existing; - bool inplace = config && config->inplace; - bool sparse = config && config->preserve_sparse; - FileAttrPolicy policy = file_attr_policy_from_config(config); - const char* backup_suffix = (config && config->suffix) ? config->suffix : "~"; - const char* backup_dir = (config && config->backup_dir) ? config->backup_dir : NULL; - const char* partial_dir = (config && config->partial_dir) ? config->partial_dir : NULL; - const char* temp_dir = (config && config->temp_dir) ? config->temp_dir : NULL; - bool use_partial_root = partial_dir && config && config->partial; - char *confined_backup = NULL, *confined_partial = NULL, *disk_path = NULL; - char* destination_path = NULL; - char *backup_path = NULL, *parent_copy = NULL; - - if (!file || !file->path || !file->data || - (file->data->size != 0 && !file->data->data && !file->basis_link && !file->basis_copy) || - (!file_get_trust_sender() && has_path_traversal(file->path)) || - (backup_enabled && - (!backup_suffix || backup_suffix[0] == '\0' || strchr(backup_suffix, '/') != NULL || - strcmp(backup_suffix, ".") == 0 || strcmp(backup_suffix, "..") == 0))) { - log_message(LOG_LEVEL_ERROR, "Invalid file or path received"); - return FILE_SAVE_ERROR; - } - - /* P7 Wave D #1: a STATUS_DIR_TIMES entry is RECORD-ONLY. The scanner - captures every traversed directory -- including empty ones whose parents - were never created by a child write and directories pruned by - -m/--prune-empty-dirs. Creating them here would resurrect empty - directories (an -a behavior change) and could abort the whole transfer on a - pre-existing regular file/symlink at the mirror path. Short-circuit before - any device/write-devices/directory branch and report it as skipped so the - sink still accumulates its metadata for the deferred DirTimeList - application, but create nothing. */ - if (file->dir_time_only) - return FILE_SAVE_SKIPPED; - - /* Device/special node (--devices/--specials): recreate the node instead of - writing content (privilege-gated, confined, rdev-validated). */ - if (file->is_special) - return file_save_special_to_disk(root_directory, file, config, created); - /* --write-devices: write straight into an existing device node. Writing - into a device is a super-user activity, so --no-super must suppress it just - like device-node creation; the default AUTO/--super attempt it (the wide - open below keeps its own confinement and best-effort skip semantics). */ - if (config && config->write_devices) { - if (!privilege_super_mode_permitted(config->super_mode)) { - char* escaped_path = output_escape(file->path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "write-devices: %s skipped: super-user activities are not permitted on this " - "receiver", - escaped_path ? escaped_path : "(null)"); - free(escaped_path); - return FILE_SAVE_SKIPPED; - } - return file_save_write_device(root_directory, file); - } - - /* Explicit directory entries (--dirs) carry an empty payload; the entry is - created as a directory under the receive root, applying the same secure - mkdir-parent semantics as regular writes. Directories are created - immediately (they are never staged by --delay-updates, matching rsync, - where directory creation is not delayed). */ - if (file->is_dir) { - if (file->path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(file->path))) { - log_message(LOG_LEVEL_ERROR, "Invalid directory path received"); - return FILE_SAVE_ERROR; - } - char* dir_path = path_cat(root_directory, file->path); - if (!dir_path) - return FILE_SAVE_ERROR; - bool dir_existed = file_path_exists_secure(dir_path); - bool ok = file_ensure_directory_secure(dir_path); - /* P7 Wave E: apply the negotiated ownership to the directory ITSELF (not - just the files inside it). --copy-as and every explicit identity policy - own every entry, so a directory must not keep the receiver's owner while - its children get the policy owner. Applied no-follow on the confined - parent fd after the mkdir; identity_apply_ownership_link() is itself a - no-op unless an identity policy is active. */ - if (ok && file->metadata && identity_active_enabled()) { - char* leaf = NULL; - int parent_fd = file_open_secure_parent(dir_path, &leaf, false); - if (parent_fd >= 0) { - if (!identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, - (int32_t)file->metadata->gid)) - ok = false; - close(parent_fd); - } else if (identity_copy_as_active()) { - /* The directory exists (ok) but its required --copy-as ownership could - not be applied because the confined parent could not be opened. */ - ok = false; - } - free(leaf); - } - free(dir_path); - if (ok && created && !dir_existed) - *created = true; - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; - } - - /* Symlink entry. (The process-wide --keep-dirlinks policy is set once by the - connection handler from the negotiated config, before any receiver/writer - threads start, so it is stable throughout this walk.) */ - - if (file->is_symlink) { - if (!file->symlink_target || file->path[0] == '\0' || - (!file_get_trust_sender() && has_path_traversal(file->path))) { - log_message(LOG_LEVEL_ERROR, "Invalid symlink entry received"); - return FILE_SAVE_ERROR; - } - char* link_path = path_cat(root_directory, file->path); - if (!link_path) - return FILE_SAVE_ERROR; - bool link_existed = file_path_exists_secure(link_path); - /* The link value is stored verbatim (rsync -l parity: absolute and - ".."-bearing targets are preserved; the scanner's --safe-links / - --copy-unsafe-links decide which links are sent at all). --munge-links - is a RECEIVER-side rewrite: the stored target is prefixed with - /rsyncd-munged/, making the link unusable while the referenced directory - does not exist -- exactly as rsync's receiver munges. Only the link's - own placement path is confined below the receive root. */ - bool munge = config && config->munge_links; - char* target = str_dup(file->symlink_target); - bool ok = target != NULL; - if (ok && munge) { - char* munged = file_symlink_munge(target); - free(target); - target = munged; - ok = target != NULL; - } - if (!ok) { - free(target); - free(link_path); - return FILE_SAVE_SKIPPED; - } - char* parent = str_dup(link_path); - if (parent) { - /* Propagate a failed --copy-as ownership of the parent directory this - creates; every other failure mode stays best-effort as before. */ - ok = file_ensure_directory_secure(dirname(parent)); - free(parent); - } - if (ok) - ok = file_symlink_at_secure(link_path, target); - free(target); - /* P7 Wave D: apply the symlink's own metadata with no-follow primitives - (utimensat/lchown/fchmodat AT_SYMLINK_NOFOLLOW). -J/--omit-link-times - suppresses the timestamps; ownership stays gated by the identity policy. - A symlink has no children, so this can be applied immediately. */ - if (ok && config && config->use_metadata) { - FileAttrPolicy link_policy = file_attr_policy_from_config(config); - ok = file_restore_symlink_metadata(link_path, file->metadata, link_policy, - config->omit_link_times); - } - if (ok && created && !link_existed) - *created = true; - free(link_path); - return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; - } - - /* --hard-links/-H sibling: a later member of a link group arrives with no - payload and is installed as a hard link to (or, on link() failure, a - byte-identical copy of) the group's first member. Handled entirely here, - before the normal data-write paths (which would create an empty file). */ - if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { - return file_save_hardlink_sibling(root_directory, file, config, created); - } - - /* These options arrive from the client. --backup-dir, --partial-dir and - --temp-dir are names below the server root, never independent filesystem - roots: an absolute or `..`-escaping value is rejected outright (rsync's - daemon confines temp-dir to the module the same way). A relative temp dir - is resolved under the receive root below; if that resolution still lands on - a different filesystem than the destination the install falls back to a - non-atomic copy (see file_to_disk_secure_impl), never an abort. */ - if ((backup_dir && (backup_dir[0] == '/' || has_path_traversal(backup_dir))) || - (partial_dir && (partial_dir[0] == '/' || has_path_traversal(partial_dir))) || - (temp_dir && (temp_dir[0] == '/' || has_path_traversal(temp_dir)))) - return FILE_SAVE_ERROR; - if (backup_dir && !(confined_backup = path_cat(root_directory, backup_dir))) - return FILE_SAVE_ERROR; - if (partial_dir && !(confined_partial = path_cat(root_directory, partial_dir))) { - free(confined_backup); - return FILE_SAVE_ERROR; - } - - const char* actual_root = use_partial_root ? confined_partial : root_directory; - destination_path = path_cat(root_directory, file->path); - disk_path = path_cat(actual_root, file->path); - if (destination_path == NULL || disk_path == NULL) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_ERROR; - } - /* Snapshot the final destination's existence BEFORE any backup/force/partial - step can move or remove it, so the receiver can report rsync's - `Number of created files` (protocol 2.28.0). */ - bool dest_existed = file_path_exists_secure(destination_path); - - /* --delay-updates diverts the whole write into the staging tree; the rest of - this function is the immediate-install path. */ - if (config && config->delay_updates) { - FileSaveResult result = - file_stage_delayed_update(root_directory, destination_path, file, (Config*)config); - if (result == FILE_SAVE_WRITTEN && created && !dest_existed) - *created = true; - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return result; - } - - /* --existing checks the final destination, not a temporary partial path. */ - if (config && config->existing && !file_path_exists_secure(destination_path)) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_SKIPPED; - } - - /* --ignore-existing checks the final destination before partial files or - overwrite policies can modify it. */ - if (config && config->ignore_existing) { - bool exists = file_path_exists_secure(destination_path); - if (exists) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_SKIPPED; - } - } - - /* --update is receiver-side policy: never replace a newer destination. - In partial-dir mode the entry that would be replaced is the real - destination, not the temporary partial file. The secure stat does not - require read permission on the destination. */ - const char* update_target = use_partial_root ? destination_path : disk_path; - if (config && config->update && file_destination_is_newer_secure(update_target, file->metadata)) { - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_SKIPPED; - } - - /* --force (rsync semantics): an incoming regular file may replace a - destination DIRECTORY by removing that (possibly non-empty, symlink-safe) - tree first, so the atomic temp+rename below can install the file. Only the - immediate-install path does this: a --delay-updates run stages into its own - tree and is unaffected here (its publication renames over regular files - only). The blocking directory is removed only after the --update / - --existing / --ignore-existing decisions above, which see it as an existing - destination entry. */ - if (config && config->force_delete && !file->is_dir && - file_directory_exists_secure(destination_path)) { - if (!file_remove_tree_secure(destination_path)) - goto fail; - } - - if (backup_enabled) { - /* Back up the entry that the incoming write will replace. When writing - through a partial dir the pre-existing destination file is the one to - preserve; any stale partial file is overwritten without a backup. */ - const char* replace_target = use_partial_root ? destination_path : disk_path; - struct stat backup_stat; - if (file_stat_secure(replace_target, &backup_stat)) { - if (backup_dir) { - backup_path = path_cat(confined_backup, file->path); - } else { - size_t path_len = strlen(replace_target); - size_t suffix_len = strlen(backup_suffix); - if (path_len > SIZE_MAX - suffix_len - 1) - goto fail; - backup_path = malloc(path_len + suffix_len + 1); - if (backup_path) { - memcpy(backup_path, replace_target, path_len); - memcpy(backup_path + path_len, backup_suffix, suffix_len + 1); - } - } - if (!backup_path) - goto fail; - parent_copy = str_dup(backup_path); - if (!parent_copy || !file_ensure_directory_secure(dirname(parent_copy))) - goto fail; - free(parent_copy); - parent_copy = NULL; - if (!file_rename_secure(replace_target, backup_path)) - goto fail; - free(backup_path); - backup_path = NULL; - } - } - - FileMetadata adjusted_metadata; - const FileMetadata* metadata = file->metadata; - if (metadata && config && config->chmod_spec && *config->chmod_spec) { - adjusted_metadata = *metadata; - if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) - goto fail; - metadata = &adjusted_metadata; - } - - /* A configured --temp-dir sends the temporary working copy to a scratch - directory; the engine then atomically renames the completed file into the - final destination directory. A relative temp dir is resolved under the - receive root and must already exist (an absolute or `..`-escaping value was - rejected above); the engine falls back to a non-atomic copy on EXDEV. The - partial-dir flow already keeps its working copy in a separate directory and - --inplace writes directly, so neither diverts through the scratch dir - (matching rsync, where --inplace/--partial-dir supersede --temp-dir). */ - char* confined_temp = NULL; - bool use_temp_dir = temp_dir != NULL && !inplace && !use_partial_root; - if (use_temp_dir) { - confined_temp = path_cat(root_directory, temp_dir); - if (!confined_temp) - goto fail; - /* A user-supplied trailing slash would leave the scratch path ending in - "/", which has no final component to create/open. Normalize it away. */ - size_t temp_len = strlen(confined_temp); - while (temp_len > 1 && confined_temp[temp_len - 1] == '/') - confined_temp[--temp_len] = '\0'; - } - /* A --link-dest basis hit installs an atomic hard link (with a byte-copy - fallback); --inplace and the update/no-replace write variants do not - apply to a fresh hard link, whose inode attributes already match. The - existing/ignore-existing/update/backup preamble above has already made the - policy decision. */ - bool ok; - char* count_floor = file_transfer_root_floor(config); - if (config && file->basis_link) { - ok = file_to_disk_secure_link_attrs_counted( - disk_path, file->basis_link, file->data->data, file->data->size, config->preallocate, - metadata, policy, config->use_fsync, file->xattrs, config->fake_super, confined_temp, - created_dirs, count_floor); - } else if (config && file->basis_copy) { - /* --copy-dest: stream the basis bytes through a bounded buffer so a basis - larger than any whole-file bound still materializes. The source - metadata was transmitted with the check frame. */ - ok = file_copy_basis_stream_attrs( - disk_path, file->basis_copy, file->data->size, config->preallocate, metadata, policy, - config->update, config->use_fsync, file->xattrs, config->fake_super, confined_temp); - } else { - /* The plain no-replace / update / with-fsync engines, plus per-file xattr - (-X/-A) and --fake-super application on the written fd. */ - ok = file_to_disk_secure_attrs_counted( - disk_path, file->data->data, file->data->size, inplace, sparse, - config && config->preallocate, metadata, policy, config && config->update, - config && config->ignore_existing, config && config->use_fsync, file->xattrs, - config ? config->fake_super : false, config ? config->partial : false, confined_temp, - created_dirs, count_floor); - } - free(count_floor); - free(confined_temp); - confined_temp = NULL; - if (!ok) - goto fail; - - /* --partial --partial-dir writes the complete file under the partial dir so - interrupted transfers leave a resumable copy there. Once the file is - fully written it must be atomically installed at the real destination; - otherwise completed transfers would linger under the partial dir. */ - if (use_partial_root) { - if (!file_rename_secure(disk_path, destination_path)) - goto fail; - } - - free(parent_copy); - free(backup_path); - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - if (created && !dest_existed) - *created = true; - return FILE_SAVE_WRITTEN; - -fail: - free(parent_copy); - free(backup_path); - free(confined_backup); - free(confined_partial); - free(destination_path); - free(disk_path); - return FILE_SAVE_ERROR; -} - -void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, - unsigned created_dirs) { - if (!stats || !file) - return; - /* A basis-dir hit (--link-dest/--copy-dest) materializes bytes the sender - * never transferred. rsync reports no literal data and no created entry for - * such a file, and does not count the parent directories it creates only to - * hold it, so exclude the whole entry from the receiver tallies. */ - bool basis_sourced = file->basis_link != NULL || file->basis_copy != NULL; - if (basis_sourced) - return; - bool is_sibling = file->link_group != 0 && !file->link_first; - if (!file->is_dir && !file->is_symlink && !file->is_special && !is_sibling) { - unsigned long long literal = file->literal_bytes; - if (literal == 0 && file->matched_bytes == 0) - literal = file->data ? file->data->size : 0; - stats->literal_bytes += literal; - } - stats->created_dir += created_dirs; - if (!created) - return; - if (file->is_dir) - stats->created_dir++; - else if (file->is_symlink) - stats->created_link++; - else if (file->is_special) - stats->created_special++; - else - stats->created_reg++; -} - -/* Receive a file's xattr block (when the config enables xattr transport) and - * attach it to `file`. Returns false on a malformed/oversized frame. */ -static bool receive_file_xattrs(File* file, int fd, const Config* config) { - if (!config->use_xattrs) - return true; - int xok = 0; - FileXattrList* list = xattr_receive(fd, &xok, config->preserve_acls); - if (!xok) { - xattr_list_free(list); - return false; - } - file->xattrs = list; - return true; -} - -static File* receive_delta_file(int fd, const Config* config, const char* check_path, - void* old_data, unsigned long long old_size, bool* failed) { - if (!old_data) { - free(old_data); /* defensive: old_data is always non-NULL today */ - *failed = true; - return NULL; - } - - DeltaSignature* sig = delta_signature_create_seeded(old_data, old_size, config->delta_block_size, - (uint32_t)config->checksum_seed); - if (!sig) { - free(old_data); - *failed = true; - return NULL; - } - - Data* sig_data = delta_signature_serialize(sig); - if (!sig_data) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - bool sig_sent = send_status(fd, STATUS_DELTA_SIGNATURE) && send_data(fd, sig_data); - data_destroy(sig_data); - - if (!sig_sent) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - Status resp; - if (!receive_status(fd, &resp)) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - if (resp == STATUS_DELTA_DATA) { - Data* delta_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (!delta_data) { - delta_signature_destroy(sig); - free(old_data); - *failed = true; - return NULL; - } - - Data* raw_delta = delta_data; - if (config->use_compression && - !compression_should_skip_with_suffixes( - check_path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count : -1)) { - ProtocolSession* owner = delta_data->owner; - raw_delta = data_decompress_limited(delta_data, MAX_RECEIVE_WHOLE_FILE_SIZE); - data_destroy(delta_data); - if (!raw_delta) { - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - /* Charge the decompressed delta to the connection budget (the paired - wire buffer's charge was just released). */ - if (!data_charge_session(raw_delta, owner, raw_delta->size)) { - data_destroy(raw_delta); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - } - - Delta* delta = delta_deserialize(raw_delta); - data_destroy(raw_delta); - if (!delta) { - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - - uint64_t new_size = delta->new_file_size; - if (new_size > MAX_RECEIVE_WHOLE_FILE_SIZE || new_size > SIZE_MAX) { - delta_destroy(delta); - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - /* Wire-stats tally: bytes taken straight from the basis file (matched - delta blocks) and bytes shipped literally (protocol 2.28.0). Computed - before the delta is destroyed. */ - unsigned long long matched = 0; - unsigned long long literal = 0; - for (uint32_t k = 0; k < delta->instruction_count; k++) { - if (delta->instructions[k].type == DELTA_INSTR_BLOCK_MATCH) - matched += delta->instructions[k].match.length; - else if (delta->instructions[k].type == DELTA_INSTR_LITERAL) - literal += delta->instructions[k].literal.length; - } - void* new_data = delta_apply(old_data, old_size, delta, config->delta_block_size); - delta_destroy(delta); - - if (!new_data) { - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - - File* file = file_create(check_path); - if (!file) { - free(new_data); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - file->matched_bytes = matched; - file->literal_bytes = literal; - - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - free(new_data); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - } - if (!receive_file_xattrs(file, fd, config)) { - file_destroy(file); - free(new_data); - free(old_data); - delta_signature_destroy(sig); - *failed = true; - return NULL; - } - - Data* replacement = data_create(new_data, (size_t)new_size); - if (replacement == NULL) { - file_destroy(file); - free(old_data); - delta_signature_destroy(sig); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - data_destroy(file->data); - file->data = replacement; - - free(old_data); - delta_signature_destroy(sig); - return file; - } - - if (resp == STATUS_NEXT) { - delta_signature_destroy(sig); - free(old_data); - - File* file = file_create(check_path); - if (!file) { - *failed = true; - return NULL; - } - - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - *failed = true; - return NULL; - } - } - if (!receive_file_xattrs(file, fd, config)) { - file_destroy(file); - *failed = true; - return NULL; - } - - Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (file_data == NULL) { - file_destroy(file); - *failed = true; - return NULL; - } - - if (config->use_compression && - !compression_should_skip_with_suffixes( - file->path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count : -1)) { - Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); - ProtocolSession* owner = file_data->owner; - data_destroy(file_data); - if (uncompressed == NULL) { - file_destroy(file); - *failed = true; - return NULL; - } - if (!data_charge_session(uncompressed, owner, uncompressed->size)) { - data_destroy(uncompressed); - file_destroy(file); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - if (uncompressed->size > MAX_FILE_DATA_SIZE) { - data_destroy(uncompressed); - file_destroy(file); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; - } - file_data = uncompressed; - } - - data_destroy(file->data); - file->data = file_data; - return file; - } - - delta_signature_destroy(sig); - free(old_data); - send_status(fd, STATUS_ERROR); - *failed = true; - return NULL; -} - -/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ---- - * The receiver consults the ordered basis-dir list only when the destination - * entry is NOT already up to date. By default an "exact match" is rsync's - * metadata quick-check: an equal size and an equal mtime (unless --size-only). - * The FastSync-only --verify-basis additionally requires an equal whole-file - * content digest, so a hard link / local copy is only then made from - * byte-verified content. */ - -typedef struct BasisMatch { - bool hit; - BasisDestType type; - char* basis_path; /* owned absolute path of the matched basis file */ - struct stat st; /* fstat() of the matched basis file */ -} BasisMatch; - -static void basis_match_free(BasisMatch* match) { - if (!match) - return; - free(match->basis_path); - match->basis_path = NULL; - match->hit = false; - match->type = BASIS_DEST_NONE; -} - -/* Open `path` (via the secure, root-confined primitives) and require it to be - a regular file of exactly `expected_size` bytes. Returns an open read-only - descriptor and its fstat on success. */ -static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd, - struct stat* out_st) { - char* leaf = NULL; - int parent_fd = file_open_secure_parent(path, &leaf, false); - if (parent_fd < 0) - return false; - /* O_NONBLOCK: a client-planted FIFO must not block the receiver's openat() - forever; the fstat()/S_ISREG gate below rejects it immediately. */ - int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - free(leaf); - close(parent_fd); - if (fd < 0) - return false; - struct stat st; - if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) || - (unsigned long long)st.st_size != expected_size) { - close(fd); - return false; - } - *out_fd = fd; - *out_st = st; - return true; -} - -/* --ignore-times forces every file to be updated, so no basis hit is ever - declared (matching rsync, where -I prevents link-dest from linking). */ -bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, - long check_mtime_nsec) { - if (config->size_only) - return true; - long mtime_nsec = 0; -#ifdef __linux__ - mtime_nsec = st->st_mtim.tv_nsec; -#endif - return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec, - config->modify_window); -} - -/* True when a basis hit must be confirmed by a whole-file content digest - (--verify-basis). False is the rsync-parity default: the metadata - quick-check alone decides a hit. */ -bool file_basis_content_required(const Config* config) { - return config != NULL && config->verify_basis; -} - -/* Probe one candidate basis file: open it (confined, O_NOFOLLOW) and apply - rsync's metadata quick-check; under --verify-basis also hash its bytes and - require the sender's digest. On a hit record `candidate` in `out` and return - true. The caller retains ownership of `candidate`. */ -static bool basis_match_probe(const Config* config, const char* candidate, - unsigned long long check_size, time_t check_mtime, - long check_mtime_nsec, const uint8_t* check_digest, - size_t check_digest_len, BasisDestType type, BasisMatch* out) { - int fd; - struct stat st; - if (!basis_open_regular(candidate, check_size, &fd, &st)) - return false; - bool hit = false; - if (file_basis_quick_match(config, &st, check_mtime, check_mtime_nsec)) { - hit = true; - if (file_basis_content_required(config)) { - uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN]; - size_t basis_len = 0; - bool hashed = checksum_digest_fd((ChecksumAlgo)config->checksum_algo, config->checksum_seed, - fd, basis_digest, sizeof(basis_digest), &basis_len); - hit = hashed && basis_len == check_digest_len && check_digest_len > 0 && - memcmp(basis_digest, check_digest, check_digest_len) == 0; - } - } - close(fd); - if (!hit) - return false; - char* owned = str_dup(candidate); - if (!owned) - return false; - out->hit = true; - out->type = type; - out->basis_path = owned; - out->st = st; - return true; -} - -/* Search the basis-dir list in command-line order and return the first match. - By default (no --verify-basis) rsync's metadata quick-check is sufficient: - basis_open_regular has already required an equal size, and - file_basis_quick_match applies rsync's mtime (or --size-only) rule. - --verify-basis additionally requires the basis bytes' whole-file digest to - equal the sender's, restoring FastSync's historical content equality; that - digest is computed by streaming the open basis descriptor, so an arbitrarily - large basis is verified without buffering it. A copy/link install re-reads - the basis from its path in bounded buffers, so no content buffer is kept. - - `hash_content` gates content READS under --verify-basis: a server-contacting - --dry-run passes false because hashing a basis against a client-supplied - digest would be a 1-bit content oracle. Without --verify-basis a dry-run can - still confirm the metadata-only hit without reading any basis bytes, matching - rsync's read-only quick-check. - - Path resolution (rsync 3.4.1 parity): rsync resolves a relative - --compare-dest/--copy-dest/--link-dest DIR against the destination directory - (the receiver's cwd) and appends the file's TRANSFER-RELATIVE name, e.g. - `--compare-dest=basis` with `rsync src/ dst/` probes `dst/basis/`. - FastSync's receive root IS the destination directory, but its default transfer - mirrors the absolute source path below that root, so check_path carries the - source-root scaffolding rsync would not append. Recover rsync's spelling with - utils_strip_transfer_root for a relative DIR; under -R/--files-from the wire - path is already transfer-relative, so it is used as-is. A relative DIR also - probes the historical mirror-appended spelling as a fallback, so existing - FastSync-laid-out snapshot trees keep resolving. An absolute DIR is used - verbatim and keeps appending the destination-relative check_path (FastSync's - mirrored layout). Every candidate stays confined to the authorized root by - file_open_secure_parent. */ -static bool basis_match_find(const Config* config, const char* check_path, - unsigned long long check_size, time_t check_mtime, - long check_mtime_nsec, const uint8_t* check_digest, - size_t check_digest_len, bool hash_content, BasisMatch* out) { - memset(out, 0, sizeof(*out)); - if (!config || !config_has_basis(config) || config->ignore_times) - return false; - /* --verify-basis needs the basis content; a content-blind (dry-run) pass can - never confirm it and must not read the file, so decline without touching - the basis bytes. */ - if (file_basis_content_required(config) && !hash_content) - return false; - const char* transfer_rel = check_path; - if (!config->relative && config->files_from_set == NULL) - transfer_rel = utils_strip_transfer_root(check_path, config->send_directory); - for (int i = 0; i < config->basis_count; i++) { - const BasisDest* entry = &config->basis_dirs[i]; - /* An absolute basis path is used verbatim (rsync semantics); a relative one - is resolved below the receive root. Both remain subject to the receiver's - authorized-root confinement inside file_open_secure_parent. */ - bool absolute = entry->path[0] == '/'; - char* basis_dir = - absolute ? str_dup(entry->path) : path_cat(config->receive_root_directory, entry->path); - if (!basis_dir) - continue; - const char* names[2]; - int name_count = 0; - if (absolute) - names[name_count++] = check_path; - else - names[name_count++] = transfer_rel; - if (!absolute && strcmp(transfer_rel, check_path) != 0) - names[name_count++] = check_path; /* historical mirror-appended spelling */ - bool found = false; - for (int n = 0; n < name_count && !found; n++) { - char* candidate = path_cat(basis_dir, names[n]); - if (!candidate) - continue; - found = basis_match_probe(config, candidate, check_size, check_mtime, check_mtime_nsec, - check_digest, check_digest_len, entry->type, out); - free(candidate); - } - free(basis_dir); - if (found) - return true; - } - return false; -} - -/* --------------------------------------------------------------------------- - * -y/--fuzzy similar-file delta basis. - * - * When a file must be transferred and the destination holds no usable content - * at the exact path (the destination file is absent, or is outside the delta - * engine's size bounds), --fuzzy lets the receiver reuse an EXISTING regular - * file in the SAME destination directory as the delta basis, so the sender - * transmits only the differences instead of the whole file. This is the - * rsync "find a similar file to use as a basis for a transfer" case (e.g. a - * file recreated under a new name whose old-named sibling is still present). - * - * The delta handshake is unchanged and receiver-driven, so the sender never - * learns the basis was a different file and needs no new protocol. Byte - * exactness never depends on which bytes the basis holds: the delta protocol - * only references basis blocks whose Adler-32 + xxHash32 checksums match the - * source, delta_apply validates every reference against the basis size, and a - * basis that shares nothing simply makes the sender reply STATUS_NEXT (full - * transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a - * file. - * - * Similarity heuristic (rsync 3.4.1 parity, util1.c fuzzy_distance / - * find_filename_suffix + generator.c find_fuzzy): - * * candidates are the target's sibling entries in its destination - * directory, opened through the confined root (file_open_secure_parent + - * openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never - * followed and nothing outside the destination root is ever read; - * * dotfiles, directories, the target's own name, and the .fastsync-stage / - * temp scratch names are never candidates; - * * size gate = rsync's, NOT the ordinary delta engine's bounds: any - * non-empty regular sibling up to the receiver's whole-file buffer cap is - * eligible, regardless of the 16 KiB delta minimum or the 10x delta size - * ratio (rsync's find_fuzzy has no delta-size gate at all). The delta - * engine consumes the fuzzy basis through the same signature handshake - * whether or not it is inside delta_should_attempt's window; - * * first pass = an exact size+mtime match wins regardless of name (rsync's - * "fuzzy size/modtime match"); - * * otherwise the winner minimizes rsync's weighted Levenshtein distance - * (substitution ± byte difference, insertion UNIT+byte, 16.16 fixed point) - * plus ten times the suffix distance, accepted only when <= 25*UNIT; the - * tie-break (smallest size gap, then lexical name) keeps the result - * deterministic across filesystem readdir order (rsync leaves equal - * distances to its file-list order). - * ------------------------------------------------------------------------- */ - -/* A directory scan is linear in the number of entries; the fuzzy search stops - * after this many so a pathological huge directory cannot stall a transfer. - * The cap bounds the readdir() ITERATIONS, not the per-entry work: every - * entry that survives the (cheap) size and pre-name gates still runs an - * edit-distance DP, so the per-entry DP cost is separately bounded below by - * pre-pruning on the name length gap and the absent-character bound, and by - * trimming the common prefix/suffix before the DP runs on the middles only. */ -#define FUZZY_MAX_DIRECTORY_SCAN 4096 -/* Names longer than this never take part in fuzzy matching: the edit-distance - * DP below is O(len^2), so over-long names are bounded out of the search. */ -#define FUZZY_NAME_LIMIT 192 - -typedef struct { - char name[FUZZY_NAME_LIMIT + 1]; - unsigned long long size; - uint32_t distance; - unsigned long long size_gap; -} FuzzyCandidate; - -/* rsync's fuzzy distance is a weighted Levenshtein variant in 16.16 fixed point - * (util1.c fuzzy_distance): a substitution costs UNIT +/- the byte difference - * and an insertion costs UNIT + the inserted byte, so similar names score low. - * The search keeps only distances <= 25*UNIT. Ported verbatim for parity. */ -#define FUZZY_DIST_UNIT (1u << 16) -#define FUZZY_DIST_REJECT (0xFFFFu * FUZZY_DIST_UNIT + 1) -#define FUZZY_DIST_LIMIT (25u * FUZZY_DIST_UNIT) - -static uint32_t fuzzy_distance(const char* s1, unsigned len1, const char* s2, unsigned len2, - uint32_t upperlimit, uint32_t* scratch) { - if ((len1 > len2 ? len1 - len2 : len2 - len1) * FUZZY_DIST_UNIT > upperlimit) - return FUZZY_DIST_REJECT; - if (!len1 || !len2) { - if (!len1) { - s1 = s2; - len1 = len2; - } - uint32_t cost = 0; - for (unsigned i = 0; i < len1; i++) - cost += (uint8_t)s1[i]; - return (uint32_t)len1 * FUZZY_DIST_UNIT + cost; - } - uint32_t* a = scratch; - for (unsigned i2 = 0; i2 < len2; i2++) - a[i2] = (i2 + 1) * FUZZY_DIST_UNIT; - for (unsigned i1 = 0; i1 < len1; i1++) { - uint32_t diag = i1 * FUZZY_DIST_UNIT; - uint32_t above = (i1 + 1) * FUZZY_DIST_UNIT; - for (unsigned i2 = 0; i2 < len2; i2++) { - uint32_t left = a[i2]; - int32_t cost = (int32_t)(uint8_t)s1[i1] - (int32_t)(uint8_t)s2[i2]; - if (cost != 0) - cost = cost < 0 ? (int32_t)(FUZZY_DIST_UNIT - (uint32_t)(-cost)) - : (int32_t)(FUZZY_DIST_UNIT + (uint32_t)cost); - uint32_t diag_inc = diag + (uint32_t)cost; - uint32_t left_inc = left + FUZZY_DIST_UNIT + (uint8_t)s1[i1]; - uint32_t above_inc = above + FUZZY_DIST_UNIT + (uint8_t)s2[i2]; - a[i2] = above = left < above ? (left_inc < diag_inc ? left_inc : diag_inc) - : (above_inc < diag_inc ? above_inc : diag_inc); - diag = left; - } - } - return a[len2 - 1]; -} - -/* rsync's find_filename_suffix (util1.c): return the last significant filename - * suffix (its dot included). Leading dots are not a suffix; a trailing "~" is - * ignored; .bak/.old/.orig and a "~/" backup marker are skipped. */ -static const char* fuzzy_find_suffix(const char* fn, int fn_len, int* len_ptr) { - const char* suf; - const char* s; - bool had_tilde; - - while (fn_len && *fn == '.') { - fn++; - fn_len--; - } - if (fn_len > 1 && fn[fn_len - 1] == '~') { - fn_len--; - had_tilde = true; - } else { - had_tilde = false; - } - suf = ""; - *len_ptr = 0; - for (s = fn + fn_len; fn_len > 1;) { - int s_len; - while (--s != fn && *s != '.') { - } - if (s == fn) - break; - s_len = fn_len - (int)(s - fn); - fn_len = (int)(s - fn); - if (s_len == 4) { - if (strcmp(s + 1, "bak") == 0 || strcmp(s + 1, "old") == 0) - continue; - } else if (s_len == 5) { - if (strcmp(s + 1, "orig") == 0) - continue; - } else if (s_len > 2 && had_tilde && s[1] == '~' && isdigit((unsigned char)s[2])) { - continue; - } - *len_ptr = s_len; - suf = s; - if (s_len == 1) - break; - for (s++, s_len--; s_len > 0; s++, s_len--) { - if (!isdigit((unsigned char)*s)) - return suf; - } - s = suf; - } - return suf; -} - -/* Deterministic ordering of two fuzzy candidates with equal rsync distance: - * smallest size gap, then the lexical basename (rsync itself takes the last - * equal-distance candidate in file-list order). */ -static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) { - if (!best->name[0]) - return true; - if (cand->distance != best->distance) - return cand->distance < best->distance; - if (cand->size_gap != best->size_gap) - return cand->size_gap < best->size_gap; - return strcmp(cand->name, best->name) < 0; -} - -/* Search the destination directory that will contain `check_path` for a - * similar regular file usable as a --fuzzy delta basis and return its full - * content in a malloc'd (protocol_alloc) buffer. Returns NULL (with *out_size - * = 0) when no candidate qualifies, which means the caller performs the normal - * whole-file transfer. */ -static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path, - unsigned long long check_size, time_t check_mtime, - long check_mtime_nsec, unsigned long long* out_size) { - *out_size = 0; - if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || - !check_path || check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) - return NULL; - - char* full_path = path_cat(config->receive_root_directory, check_path); - if (!full_path) - return NULL; - char* leaf = NULL; - int dir_fd = file_open_secure_parent(full_path, &leaf, false); - if (dir_fd < 0 || !leaf) { - free(leaf); - free(full_path); - return NULL; - } - size_t target_len = strlen(leaf); - /* A target basename longer than FUZZY_NAME_LIMIT can never pass the name gate - (every candidate name is bounded by the same limit), so skip the scan. */ - if (target_len > FUZZY_NAME_LIMIT) { - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - - int scanfd = dup(dir_fd); - if (scanfd < 0) { - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - DIR* dir = fdopendir(scanfd); - if (!dir) { - close(scanfd); - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - - /* The weighted-distance scratch row is allocated once per scan (not once per - candidate). */ - uint32_t* dist_scratch = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(uint32_t)); - if (!dist_scratch) { - closedir(dir); - close(dir_fd); - free(leaf); - free(full_path); - return NULL; - } - int fname_suf_len = 0; - const char* fname_suf = fuzzy_find_suffix(leaf, (int)target_len, &fname_suf_len); - - FuzzyCandidate best; - memset(&best, 0, sizeof(best)); - uint32_t lowest_dist = FUZZY_DIST_LIMIT; - /* rsync's fuzzy search runs an exact size+mtime pass before the name-distance - pass; such a candidate is almost certainly the same content and wins - regardless of how dissimilar its name is. The first one (directory order, - deterministic) is kept. */ - FuzzyCandidate exact; - memset(&exact, 0, sizeof(exact)); - const struct dirent* entry; - size_t scanned = 0; - /* readdir() yields entries in filesystem-dependent order, so the SET of - candidates seen is order-dependent; the winner is still deterministic - because every candidate is compared with the total ordering in - fuzzy_candidate_better (acceptable for a heuristic). */ - while (scanned < FUZZY_MAX_DIRECTORY_SCAN && (entry = readdir(dir)) != NULL) { - scanned++; - const char* name = entry->d_name; - size_t name_len = strlen(name); - if (name[0] == '.' || name_len == 0 || name_len > FUZZY_NAME_LIMIT || strcmp(name, leaf) == 0) - continue; - struct stat st; - if (fstatat(dir_fd, name, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISREG(st.st_mode)) - continue; - unsigned long long cand_size = (unsigned long long)st.st_size; - if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE) - continue; - long cand_nsec = 0; -#ifdef __linux__ - cand_nsec = st.st_mtim.tv_nsec; -#endif - if (!exact.name[0] && cand_size == check_size && - metadata_mtime_matches(st.st_mtime, cand_nsec, check_mtime, check_mtime_nsec, - config->modify_window)) { - memcpy(exact.name, name, name_len + 1); - exact.size = cand_size; - exact.size_gap = 0; - continue; - } - /* rsync's name-distance pass: a weighted Levenshtein distance over the full - basenames, plus ten times the same distance over the filename suffixes, - accepted only when it does not exceed the running lowest distance. */ - int name_suf_len = 0; - const char* name_suf = fuzzy_find_suffix(name, (int)name_len, &name_suf_len); - uint32_t distance = fuzzy_distance(name, (unsigned)name_len, leaf, (unsigned)target_len, - lowest_dist, dist_scratch); - if (distance < 0xFFFF0000U) - distance += fuzzy_distance(name_suf, (unsigned)name_suf_len, fname_suf, - (unsigned)fname_suf_len, 0xFFFF0000U, dist_scratch) * - 10; - if (distance > lowest_dist) - continue; - lowest_dist = distance; - FuzzyCandidate cand; - memcpy(cand.name, name, name_len + 1); - cand.size = cand_size; - cand.distance = distance; - cand.size_gap = cand_size > check_size ? cand_size - check_size : check_size - cand_size; - if (fuzzy_candidate_better(&cand, &best)) - best = cand; - } - closedir(dir); - free(leaf); - free(dist_scratch); - - /* Prefer the exact size+mtime candidate over any name-distance winner. */ - if (exact.name[0]) - best = exact; - - void* basis = NULL; - if (best.name[0]) { - /* O_NONBLOCK: a name raced to a FIFO between the fstatat gate and this open - would otherwise block the receive thread forever on open(2); with it the - open fails (ENXIO) and the fstat/S_ISREG gate below would reject it too. - A regular file opened with O_NONBLOCK is unaffected. */ - int fd = openat(dir_fd, best.name, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); - if (fd >= 0) { - struct stat st; - if (fstat(fd, &st) == 0 && S_ISREG(st.st_mode) && - (unsigned long long)st.st_size == best.size && best.size <= SIZE_MAX) { - basis = protocol_alloc((size_t)best.size); - if (basis) { - size_t got = 0; - while (got < (size_t)best.size) { - ssize_t n = read(fd, (char*)basis + got, (size_t)best.size - got); - if (n <= 0) { - free(basis); - basis = NULL; - break; - } - got += (size_t)n; - } - } - } - close(fd); - } - } - close(dir_fd); - free(full_path); - if (basis) - *out_size = best.size; - return basis; -} - -/* Read the remainder of a full-file transfer after the receiver has already - * sent STATUS_NEXT: receive the metadata frame (when enabled) followed by the - * data frame, and return an owned File. Shared by the plain full-transfer path - * and the --append-verify prefix-mismatch fallback (a clean full transfer - * instead of a corrupt prefix+tail blend). */ -static File* receive_full_file(int fd, const Config* config, const char* path) { - File* file = file_create(path); - if (!file) - return NULL; - if (config->use_metadata) { - int meta_ok = 1; - file->metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) { - file_destroy(file); - return NULL; - } - } - if (!receive_file_xattrs(file, fd, config)) { - file_destroy(file); - return NULL; - } - Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (file_data == NULL) { - file_destroy(file); - return NULL; - } - if (config->use_compression && - !compression_should_skip_with_suffixes(file->path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count - : -1)) { - Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); - ProtocolSession* owner = file_data->owner; - data_destroy(file_data); - if (uncompressed == NULL) { - file_destroy(file); - return NULL; - } - if (!data_charge_session(uncompressed, owner, uncompressed->size)) { - data_destroy(uncompressed); - file_destroy(file); - return NULL; - } - if (uncompressed->size > MAX_FILE_DATA_SIZE) { - data_destroy(uncompressed); - file_destroy(file); - return NULL; - } - file_data = uncompressed; - } - data_destroy(file->data); - file->data = file_data; - return file; -} - -/* --------------------------------------------------------------------------- - * receive_incremental_check() decomposition. - * - * The per-file STATUS_CHECK fast path is split into the small helpers below, - * called in order by a short linear orchestrator (receive_incremental_check_ex). - * Each helper owns one decision: request validation, secure destination open, - * metadata-only skip, server-contacting --dry-run no-mutation short-circuit, - * alternate-basis match, --append tail resume, block delta, --fuzzy basis, and - * the final "send the whole file" fallback. Every protocol send/receive and - * every resource cleanup is preserved exactly; the non-dry-run wire is - * byte-for-byte unchanged. receive_incremental_check_ex additionally exposes a - * `would_transfer` out-param for the dry-run caller; the 3-arg - * receive_incremental_check wrapper passes NULL. - * ------------------------------------------------------------------------- */ - -/* Owned state threaded through the helpers below. */ -typedef struct { - int fd; - const Config* config; - char* check_path; /* received destination-relative path */ - char* full_path; /* receive-root-prefixed destination path */ - unsigned long long check_size; - long long check_mtime; - long long check_mtime_nsec; - uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN]; - size_t check_digest_len; - /* Source metadata carried alongside the check frame whenever a basis dir is - configured (rsync keeps the whole file list; FastSync's sender-driven - incremental path otherwise never transmits metadata for a SKIPPED file). - A basis materialization applies these SOURCE attributes instead of the - basis inode's, matching rsync's "copy then fix attributes". */ - FileMetadata* source_metadata; - bool dest_exists; /* any destination entry exists (lstat succeeded) */ - bool has_old_file; - int old_fd; - struct stat old_st; - unsigned long long old_size; - void* old_data; /* snapshot of the existing destination, or NULL */ -} IncrementalCheckState; - -typedef enum { - INCREMENTAL_CONTINUE, /* proceed to the next helper */ - INCREMENTAL_ERROR, /* protocol/validation failure: return NULL */ - INCREMENTAL_SKIP, /* up to date: *skipped = true, return NULL */ - INCREMENTAL_DRY_RUN, /* --dry-run resolved: flags set, return NULL */ - INCREMENTAL_FILE, /* a File* was produced (out_file) */ -} IncrementalCheckOutcome; - -static void incremental_check_state_init(IncrementalCheckState* state, int fd, - const Config* config) { - memset(state, 0, sizeof(*state)); - state->fd = fd; - state->config = config; - state->old_fd = -1; -} - -/* Release every resource the helpers may have acquired. Idempotent, so it is - safe on every exit path exactly the way the original inline cleanup was. */ -static void incremental_check_state_cleanup(IncrementalCheckState* state) { - free(state->old_data); - state->old_data = NULL; - if (state->old_fd >= 0) - close(state->old_fd); - state->old_fd = -1; - file_metadata_destroy(state->source_metadata); - state->source_metadata = NULL; - free(state->full_path); - state->full_path = NULL; - free(state->check_path); - state->check_path = NULL; -} - -/* Receive and validate the STATUS_CHECK request frame: path, size, mtime, - nanosecond mtime, and (when negotiated) the source digest. */ -static IncrementalCheckOutcome incremental_check_receive_request(IncrementalCheckState* state) { - int fd = state->fd; - const Config* config = state->config; - char* check_path = receive_wire_str(fd); - if (check_path == NULL) - return INCREMENTAL_ERROR; - state->check_path = check_path; - - if (!receive_n_data(fd, &state->check_size, sizeof(state->check_size)) || - !receive_n_data(fd, &state->check_mtime, sizeof(state->check_mtime))) - return INCREMENTAL_ERROR; - if (!receive_n_data(fd, &state->check_mtime_nsec, sizeof(state->check_mtime_nsec)) || - state->check_mtime_nsec < 0 || state->check_mtime_nsec >= 1000000000LL) { - send_error_detail(fd, "invalid check mtime nanoseconds"); - return INCREMENTAL_ERROR; - } - if ((config->checksum || config->verify_basis)) { - uint8_t wire_len; - if (!receive_n_data(fd, &wire_len, sizeof(wire_len)) || wire_len == 0 || - wire_len > CHECKSUM_MAX_DIGEST_LEN || - wire_len != checksum_digest_len((ChecksumAlgo)config->checksum_algo)) { - send_error_detail(fd, "invalid check digest length"); - return INCREMENTAL_ERROR; - } - state->check_digest_len = wire_len; - if (!receive_n_data(fd, state->check_digest, state->check_digest_len)) - return INCREMENTAL_ERROR; - } - /* The sender transmits the source metadata with every basis-configured check - so a basis hit can be materialized with the SOURCE's attributes (rsync - copies/copies-then-fixes; the receiver would otherwise only have the basis - inode's stat). The block is symmetric and consumed unconditionally here, - whether or not this file ends up as a basis hit. */ - if (config_has_basis(config) && config->use_metadata) { - int meta_ok = 1; - state->source_metadata = metadata_receive(fd, &meta_ok); - if (!meta_ok) - return INCREMENTAL_ERROR; - } - - /* A basis-configured run may materialize a file larger than the whole-file - payload bound: a basis hit is streamed from the basis path (bounded - buffers), so the check size is not itself an allocation. Every other - path (delta/append/full) still applies MAX_RECEIVE_WHOLE_FILE_SIZE, and a - miss simply falls through to the normal transfer with its own bound. */ - if (!config_has_basis(config) && state->check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) { - send_error_detail(fd, "check size exceeds receiver limit"); - return INCREMENTAL_ERROR; - } - - if (check_path[0] == '\0' || has_path_traversal(check_path)) { - char* escaped_path = output_escape(check_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_ERROR, "Invalid received check path: %s", - escaped_path ? escaped_path : ""); - free(escaped_path); - return INCREMENTAL_ERROR; - } - return INCREMENTAL_CONTINUE; -} - -/* Open the existing destination entry once, confined below the receive root, - and record its stat. */ -static IncrementalCheckOutcome incremental_check_open_destination(IncrementalCheckState* state) { - char* full_path = path_cat(state->config->receive_root_directory, state->check_path); - if (!full_path) { - send_error_detail(state->fd, "could not build destination path"); - return INCREMENTAL_ERROR; - } - state->full_path = full_path; - - char* leaf = NULL; - int parent_fd = file_open_secure_parent(full_path, &leaf, false); - if (parent_fd >= 0) { - struct stat dest_st; - if (fstatat(parent_fd, leaf, &dest_st, AT_SYMLINK_NOFOLLOW) == 0) - state->dest_exists = true; - /* O_NONBLOCK: an existing FIFO at the destination must not block this - openat(); the S_ISREG gate below rejects the non-regular entry. */ - state->old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); - free(leaf); - close(parent_fd); - state->has_old_file = state->old_fd >= 0 && fstat(state->old_fd, &state->old_st) == 0 && - S_ISREG(state->old_st.st_mode); - } - if (!state->has_old_file && state->old_fd >= 0) { - close(state->old_fd); - state->old_fd = -1; - } - state->old_size = state->has_old_file ? (unsigned long long)state->old_st.st_size : 0; - return INCREMENTAL_CONTINUE; -} - -/* Output parity (protocol 2.23.0): when the wire config asked for it, report a - snapshot of the pre-transfer destination entry BEFORE the ordinary verdict so - the sender can render rsync-accurate -i/--out-format columns. A missing - destination is reported explicitly (existed=false) rather than omitted, so - the sender can distinguish "new" from "unknown". */ -static IncrementalCheckOutcome incremental_check_report_dest_info(IncrementalCheckState* state) { - if (!state->config->report_dest_info) - return INCREMENTAL_CONTINUE; - OutputDestState info; - memset(&info, 0, sizeof(info)); - info.known = true; - info.existed = state->has_old_file; - if (state->has_old_file) { - info.size = (unsigned long long)state->old_st.st_size; - info.mtime_sec = (long long)state->old_st.st_mtime; -#ifdef __linux__ - info.mtime_nsec = state->old_st.st_mtim.tv_nsec; -#endif - info.mode = (uint32_t)state->old_st.st_mode; - info.uid = (int32_t)state->old_st.st_uid; - info.gid = (int32_t)state->old_st.st_gid; - } - if (!send_status(state->fd, STATUS_DEST_INFO) || !format_dest_state_send(state->fd, &info)) - return INCREMENTAL_ERROR; - return INCREMENTAL_CONTINUE; -} - -/* --ignore-existing short-circuit. The receiver must answer "skip" (STATUS_OK) - BEFORE the sender transmits any payload, otherwise the whole file crosses the - wire only to be discarded at write time. rsync skips an existing destination - entry regardless of its content or type, so the reply depends only on the - lstat existence probe; the ordinary --ignore-existing checks inside - file_receive remain as defense-in-depth for the frame types that have no - per-file check (directories/symlinks/specials/hard-links). */ -static IncrementalCheckOutcome -incremental_check_ignore_existing(const IncrementalCheckState* state) { - if (!state->config->ignore_existing || !state->dest_exists) - return INCREMENTAL_CONTINUE; - if (!send_status(state->fd, STATUS_OK)) - return INCREMENTAL_ERROR; - return INCREMENTAL_SKIP; -} - -/* Metadata for a materialized basis hit: prefer the SOURCE metadata the sender - transmitted with the check frame (rsync copies then fixes the destination to - the source's attributes); fall back to the basis inode's own stat when - metadata was not negotiated. Consumes state->source_metadata on success. */ -static FileMetadata* basis_take_metadata(IncrementalCheckState* state, - const struct stat* basis_st) { - if (state->source_metadata) { - FileMetadata* meta = state->source_metadata; - state->source_metadata = NULL; - return meta; - } - return file_metadata_create(NULL, basis_st, false, false); -} - -/* --link-dest relink of an already up-to-date destination. rsync hard-links a - destination entry to a matching basis even when the entry is already correct, - so a run over an existing tree still maximizes sharing with the basis. Only a - link-dest basis triggers this (copy-dest/compare-dest leave an up-to-date - destination untouched, matching rsync). The ordinary basis path further down - handles every not-up-to-date case, so this helper only adds the relink that - the quick-skip would otherwise short-circuit. */ -static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalCheckState* state, - File** out_file) { - const Config* config = state->config; - if (!config_has_basis(config) || config->ignore_times || config->dry_run) - return INCREMENTAL_CONTINUE; - if (!state->has_old_file) - return INCREMENTAL_CONTINUE; - BasisMatch basis; - basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, - true, &basis); - /* Only a link-dest hit relinks; a copy-dest/compare-dest hit (or a miss) lets - the up-to-date check below keep the existing destination. */ - if (!basis.hit || basis.type != BASIS_DEST_LINK) { - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; - } - /* Already the basis inode: nothing to do, leave the destination alone. */ - if (basis.st.st_dev == state->old_st.st_dev && basis.st.st_ino == state->old_st.st_ino) { - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; - } - File* materialized = file_create(state->check_path); - if (materialized) { - data_destroy(materialized->data); - materialized->data = data_create_reserve((size_t)state->check_size); - if (!materialized->data) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - materialized->metadata = basis_take_metadata(state, &basis.st); - materialized->skip = true; - materialized->basis_link = basis.basis_path; - basis.basis_path = NULL; - if (!materialized->metadata) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - if (!send_status(state->fd, STATUS_OK)) { - basis_match_free(&basis); - file_destroy(materialized); - return INCREMENTAL_ERROR; - } - basis_match_free(&basis); - *out_file = materialized; - return INCREMENTAL_FILE; - } - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; -} - -/* Metadata-only (and, when --checksum forces it, content) up-to-date decision. - Loads the old contents only when a checksum comparison or delta needs them. */ -static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckState* state, - bool* out_try_delta) { - int fd = state->fd; - const Config* config = state->config; - bool has_old_file = state->has_old_file; - unsigned long long old_size = state->old_size; - struct stat st = state->old_st; - - bool size_equal = has_old_file && old_size == state->check_size; - bool match_by_metadata = false; - if (size_equal && !config->ignore_times && !config->size_only) { - long long old_mtime_nsec = 0; -#ifdef __linux__ - old_mtime_nsec = st.st_mtim.tv_nsec; -#endif - match_by_metadata = - metadata_mtime_matches(st.st_mtime, old_mtime_nsec, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, config->modify_window); - } - - bool try_delta = config->use_delta && !config->whole_file && has_old_file && - delta_should_attempt(old_size, state->check_size, config->delta_max_file_size); - bool checksum_needs_read = size_equal && !config->ignore_times && config->checksum; - /* --dry-run must never read the destination file's CONTENTS: a client could - otherwise use `--dry-run --checksum` against a read-only module as a - 1-bit content oracle (hash match / mismatch) and force arbitrary reads. - Decide from metadata alone; when metadata is inconclusive (checksum or - delta would have required the body) report would-transfer. The real - (non-dry-run) behavior below is unchanged. */ - bool need_old_data = !config->dry_run && (checksum_needs_read || try_delta); - if (config->dry_run) - try_delta = false; - *out_try_delta = try_delta; - - if (need_old_data && has_old_file && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && - old_size <= SIZE_MAX) { - state->old_data = protocol_alloc((size_t)old_size); - if (state->old_data) { - size_t got = 0; - while (got < (size_t)old_size) { - ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); - if (n <= 0) { - free(state->old_data); - state->old_data = NULL; - break; - } - got += (size_t)n; - } - } - } - - bool match = false; - if (config->dry_run) { - /* Metadata-only decision: a size match plus a matching mtime is treated as - up to date; --checksum/--delta cannot be verified without reading, so an - otherwise inconclusive comparison is a would-transfer. */ - match = size_equal && !config->ignore_times && (config->size_only || match_by_metadata); - } else if (checksum_needs_read) { - uint8_t old_digest[CHECKSUM_MAX_DIGEST_LEN]; - size_t old_len = 0; - bool hashed = checksum_digest((ChecksumAlgo)config->checksum_algo, config->checksum_seed, - old_size == 0 ? "" : state->old_data, (size_t)old_size, - old_digest, sizeof(old_digest), &old_len); - match = hashed && old_len == state->check_digest_len && state->check_digest_len > 0 && - memcmp(old_digest, state->check_digest, state->check_digest_len) == 0; - } else if (size_equal && !config->ignore_times) { - match = config->size_only || match_by_metadata; - } - - if (match) { - if (!send_status(fd, STATUS_OK)) - return INCREMENTAL_ERROR; - return INCREMENTAL_SKIP; - } - return INCREMENTAL_CONTINUE; -} - -/* Server-contacting --dry-run no-mutation short-circuit. Runs after the - quick-skip decision and before any path that could touch the destination. - When dry_run is set and the file is not already up to date the receiver must - materialize nothing (no basis link/copy, no append/delta/full transfer) and - the sender must send no data, so answer STATUS_DRY_RUN_TRANSFER and stop. - - The basis lookup is content-blind: under the default metadata quick-check a - hit needs no basis bytes and is honored here just as in a real run; under - --verify-basis a real run hashes the basis against the client-supplied digest, - which in a dry-run is a 1-bit content oracle, so no basis bytes may be read - and an otherwise-matching entry is reported as would-transfer. Everything - read here (the destination file's metadata, basis candidates' metadata) is - read-only. */ -static IncrementalCheckOutcome incremental_check_dry_run_shortcut(IncrementalCheckState* state, - bool* skipped, - bool* would_transfer) { - const Config* config = state->config; - if (!config->dry_run) - return INCREMENTAL_CONTINUE; - - bool skip_via_compare = false; - if (config_has_basis(config) && !config->ignore_times) { - BasisMatch basis; - /* hash_content=false: a dry-run must not read or hash the basis file, so - under --verify-basis no compare-dest hit can be confirmed and an - otherwise-matching file is reported as would-transfer. Without - --verify-basis the metadata quick-check confirms it without touching any - basis bytes. */ - basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, - false, &basis); - if (basis.hit && basis.type == BASIS_DEST_COMPARE && !state->has_old_file) - skip_via_compare = true; - basis_match_free(&basis); - } - Status reply = skip_via_compare ? STATUS_OK : STATUS_DRY_RUN_TRANSFER; - if (!send_status(state->fd, reply)) - return INCREMENTAL_ERROR; - if (skip_via_compare) - *skipped = true; - else if (would_transfer) - *would_transfer = true; - return INCREMENTAL_DRY_RUN; -} - -/* Alternate basis directories (--compare-dest/--copy-dest/--link-dest): a hit - either suppresses the transfer (compare-dest) or materializes the file from - the basis without a data frame. */ -static IncrementalCheckOutcome incremental_check_try_basis(IncrementalCheckState* state, - File** out_file) { - int fd = state->fd; - const Config* config = state->config; - if (!config_has_basis(config)) - return INCREMENTAL_CONTINUE; - - BasisMatch basis; - basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, - (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, - true, &basis); - if (basis.hit) { - if (basis.type == BASIS_DEST_COMPARE) { - basis_match_free(&basis); - if (!state->has_old_file) { - if (!send_status(fd, STATUS_OK)) - return INCREMENTAL_ERROR; - return INCREMENTAL_SKIP; - } - } else { - /* Copy/link installs source their bytes from the basis PATH at install - time (bounded buffers), so no whole-file content buffer is needed here - even for an over-limit basis. */ - File* materialized = file_create(state->check_path); - if (materialized) { - data_destroy(materialized->data); - materialized->data = data_create_reserve((size_t)state->check_size); - if (!materialized->data) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - materialized->metadata = basis_take_metadata(state, &basis.st); - materialized->skip = true; /* receiver must not ack this as a data file */ - if (basis.type == BASIS_DEST_LINK) - materialized->basis_link = basis.basis_path; - else - materialized->basis_copy = basis.basis_path; - basis.basis_path = NULL; - if (!materialized->metadata) { - file_destroy(materialized); - materialized = NULL; - } - } - if (materialized) { - if (!send_status(fd, STATUS_OK)) { - basis_match_free(&basis); - file_destroy(materialized); - return INCREMENTAL_ERROR; - } - basis_match_free(&basis); - *out_file = materialized; - return INCREMENTAL_FILE; - } - /* Materialization setup failed: fall through to the normal transfer. */ - } - } - basis_match_free(&basis); - return INCREMENTAL_CONTINUE; -} - -/* --append / --append-verify tail resume: when the destination is a SHORTER - file in an append mode, negotiate the resume offset and receive only the - tail. Produces the reconstructed file, or falls through to delta/full. */ -static IncrementalCheckOutcome incremental_check_try_append_resume(IncrementalCheckState* state, - File** out_file) { - int fd = state->fd; - const Config* config = state->config; - const char* check_path = state->check_path; - unsigned long long old_size = state->old_size; - unsigned long long check_size = state->check_size; - - bool append_resume = (config->append || config->append_verify) && state->has_old_file && - append_resume_eligible(old_size, check_size); - if (!append_resume) - return INCREMENTAL_CONTINUE; - - /* Ensure the retained prefix (== the whole, shorter destination file) is in - memory; it is needed both to rebuild the full file and, for - --append-verify, to checksum it. A load failure is not fatal: the resume is - simply not possible and we fall through to the other paths. */ - if (state->old_data == NULL && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && - old_size <= SIZE_MAX) { - state->old_data = protocol_alloc((size_t)old_size); - if (state->old_data) { - size_t got = 0; - while (got < (size_t)old_size) { - ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); - if (n <= 0) { - free(state->old_data); - state->old_data = NULL; - break; - } - got += (size_t)n; - } - } - } - if (state->old_data == NULL && old_size != 0) - return INCREMENTAL_CONTINUE; - - if (!send_status(fd, STATUS_APPEND) || !send_n_data(fd, &old_size, sizeof(old_size))) - return INCREMENTAL_ERROR; - bool verify = config->append_verify; - bool full_fallback = false; - if (verify) { - Status sig_status; - if (!receive_status(fd, &sig_status)) - return INCREMENTAL_ERROR; - if (sig_status != STATUS_APPEND_SIG) { - send_status(fd, STATUS_ERROR); - return INCREMENTAL_ERROR; - } - uint64_t src_prefix_hash; - if (!receive_n_data(fd, &src_prefix_hash, sizeof(src_prefix_hash))) - return INCREMENTAL_ERROR; - /* Compare the retained prefix against the source prefix. A mismatch must - never be silently appended to: fall back to a full transfer so the result - is a byte-identical source copy. */ - uint64_t dst_prefix_hash = - old_size == 0 ? delta_xxhash64("", 0) : delta_xxhash64(state->old_data, (size_t)old_size); - if (dst_prefix_hash == src_prefix_hash) { - if (!send_status(fd, STATUS_APPEND_OK)) - return INCREMENTAL_ERROR; - } else { - if (!send_status(fd, STATUS_NEXT)) - return INCREMENTAL_ERROR; - full_fallback = true; - } - } - - if (full_fallback) { - /* Retained prefix differed: receive the sender's full transfer. */ - free(state->old_data); - state->old_data = NULL; - if (state->old_fd >= 0) { - close(state->old_fd); - state->old_fd = -1; - } - *out_file = receive_full_file(fd, config, check_path); - return INCREMENTAL_FILE; - } - - /* Receive the tail (STATUS_APPEND_DATA + metadata + tail bytes). */ - Status tail_status; - if (!receive_status(fd, &tail_status)) - return INCREMENTAL_ERROR; - if (tail_status != STATUS_APPEND_DATA) { - send_status(fd, STATUS_ERROR); - return INCREMENTAL_ERROR; - } - FileMetadata* meta = NULL; - FileXattrList* append_xattrs = NULL; - if (config->use_metadata) { - int meta_ok = 1; - meta = metadata_receive(fd, &meta_ok); - if (!meta_ok) - return INCREMENTAL_ERROR; - } - if (config->use_xattrs) { - int xok = 0; - append_xattrs = xattr_receive(fd, &xok, config->preserve_acls); - if (!xok) { - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - } - Data* tail = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); - if (tail == NULL) { - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (config->use_compression && - !compression_should_skip_with_suffixes(check_path, config->skip_compress_suffixes, - config->skip_compress_set ? config->skip_compress_count - : -1)) { - Data* uncompressed = data_decompress_limited(tail, MAX_RECEIVE_WHOLE_FILE_SIZE); - ProtocolSession* owner = tail->owner; - data_destroy(tail); - if (uncompressed == NULL) { - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (!data_charge_session(uncompressed, owner, uncompressed->size)) { - data_destroy(uncompressed); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (uncompressed->size > MAX_FILE_DATA_SIZE) { - data_destroy(uncompressed); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - tail = uncompressed; - } - /* The tail must complete the file exactly; anything else is a protocol - violation (never a truncated or overrun file). */ - unsigned long long expected_tail; - if (!append_tail_length(old_size, check_size, &expected_tail) || - tail->size != (size_t)expected_tail) { - send_status(fd, STATUS_ERROR); - data_destroy(tail); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - size_t full_size = (size_t)check_size; - void* full = protocol_alloc(full_size ? full_size : 1); - if (!full) { - data_destroy(tail); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - if (old_size > 0 && state->old_data) - memcpy(full, state->old_data, (size_t)old_size); - if (tail->size > 0) - memcpy((char*)full + old_size, tail->data, tail->size); - data_destroy(tail); - free(state->old_data); - state->old_data = NULL; - - File* file = file_create(check_path); - if (!file) { - free(full); - xattr_list_free(append_xattrs); - return INCREMENTAL_ERROR; - } - file->metadata = meta; - file->xattrs = append_xattrs; - append_xattrs = NULL; - data_destroy(file->data); - file->data = data_create(full, full_size); - if (!file->data) { /* data_create already freed full on failure */ - file_destroy(file); - return INCREMENTAL_ERROR; - } - *out_file = file; - return INCREMENTAL_FILE; -} - -/* Block delta transfer against the existing destination content. */ -static IncrementalCheckOutcome incremental_check_try_delta(IncrementalCheckState* state, - bool try_delta, File** out_file) { - if (try_delta && state->old_data != NULL) { - bool delta_failed = false; - File* delta_file = receive_delta_file(state->fd, state->config, state->check_path, - state->old_data, state->old_size, &delta_failed); - state->old_data = NULL; /* receive_delta_file consumes the snapshot on every path */ - if (delta_file) { - *out_file = delta_file; - return INCREMENTAL_FILE; - } - if (delta_failed) - return INCREMENTAL_ERROR; - } - free(state->old_data); - state->old_data = NULL; - return INCREMENTAL_CONTINUE; -} - -/* -y/--fuzzy similar-file delta basis. Reaching this point means the file - must be transferred and the destination's own content could not serve as a - delta basis; try an existing similar-named sibling in the same directory. */ -static IncrementalCheckOutcome incremental_check_try_fuzzy(IncrementalCheckState* state, - File** out_file) { - const Config* config = state->config; - if (!config->fuzzy || !config->use_delta) - return INCREMENTAL_CONTINUE; - unsigned long long fuzzy_size = 0; - void* fuzzy_basis = fuzzy_basis_find_and_load(config, state->check_path, state->check_size, - (time_t)state->check_mtime, - (long)state->check_mtime_nsec, &fuzzy_size); - if (fuzzy_basis != NULL) { - bool fuzzy_failed = false; - File* fuzzy_file = receive_delta_file(state->fd, config, state->check_path, fuzzy_basis, - fuzzy_size, &fuzzy_failed); - fuzzy_basis = NULL; /* receive_delta_file consumes the buffer on every path */ - if (fuzzy_file) { - *out_file = fuzzy_file; - return INCREMENTAL_FILE; - } - if (fuzzy_failed) - return INCREMENTAL_ERROR; - } - free(fuzzy_basis); - return INCREMENTAL_CONTINUE; -} - -/* Final fallback: tell the sender to transmit the whole file and receive it. */ -static File* incremental_check_receive_full(IncrementalCheckState* state) { - if (!send_status(state->fd, STATUS_NEXT)) - return NULL; - if (state->old_fd >= 0) { - close(state->old_fd); - state->old_fd = -1; - } - return receive_full_file(state->fd, state->config, state->check_path); -} - -/* Core implementation. `would_transfer` (may be NULL) is set true only on the - * server-contacting --dry-run path, when the file is not up to date and the - * receiver answered STATUS_DRY_RUN_TRANSFER; the caller then knows no File is - * returned and nothing was stored. */ -File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, - bool* would_transfer) { - if (would_transfer) - *would_transfer = false; - if (!config || !skipped) { - send_status(fd, STATUS_ERROR); - return NULL; - } - *skipped = false; - - IncrementalCheckState state; - incremental_check_state_init(&state, fd, config); - - File* result = NULL; - bool try_delta = false; - IncrementalCheckOutcome outcome; - - outcome = incremental_check_receive_request(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - - outcome = incremental_check_open_destination(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - - outcome = incremental_check_report_dest_info(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - - /* --ignore-existing must answer before any data is requested; it takes - precedence over the metadata up-to-date check below. */ - outcome = incremental_check_ignore_existing(&state); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_SKIP) { - *skipped = true; - goto done; - } - - /* A --link-dest hit relinks even an already up-to-date destination before the - quick-skip can suppress it (rsync parity). */ - outcome = incremental_check_link_dest_relink(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_quick_skip(&state, &try_delta); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_SKIP) { - *skipped = true; - goto done; - } - - /* Dry-run resolves here (no mutation) or falls through to the normal path. */ - outcome = incremental_check_dry_run_shortcut(&state, skipped, would_transfer); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome != INCREMENTAL_CONTINUE) - goto done; - - outcome = incremental_check_try_basis(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_SKIP) { - *skipped = true; - goto done; - } - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_try_append_resume(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_try_delta(&state, try_delta, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - outcome = incremental_check_try_fuzzy(&state, &result); - if (outcome == INCREMENTAL_ERROR) - goto done; - if (outcome == INCREMENTAL_FILE) - goto done; - - result = incremental_check_receive_full(&state); - -done: - incremental_check_state_cleanup(&state); - return result; -} - -File* receive_incremental_check(int fd, const Config* config, bool* skipped) { - return receive_incremental_check_ex(fd, config, skipped, NULL); -} - File* file_receive(const Config* config, int file_descriptor) { char* path = receive_wire_str(file_descriptor); if (path == NULL) @@ -3235,601 +550,3 @@ File* file_receive_special(int file_descriptor) { file->rdev_minor = minor; return file; } - -/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already - been consumed): a keep-set entry count followed by that many - destination-relative paths, then a protected-prefix count followed by that - many destination-relative prefixes, then a missing-args count followed by that - many destination-relative delete paths, then (protocol 2.23.0) a - synchronized-directory count followed by that many destination-relative - directory paths (the receive root is the "." sentinel). The frame is - self-delimiting (the counts are authoritative), so the caller decides what to - do next and continues reading the following STATUS_* frame. Every section is - validated identically: an entry must be non-empty, relative and traversal-free - and the aggregate length across ALL sections is capped by MAX_MANIFEST_BYTES - (so the missing-args deletion requests are confined like the rest of the - manifest). Returns an owned DeleteManifest, or NULL after sending STATUS_ERROR - when the frame is malformed (bad count, empty/absolute path, path traversal, - or an aggregate size beyond MAX_MANIFEST_BYTES). */ -static bool receive_manifest_section(int fd, ArrayList* list, size_t* manifest_bytes, - size_t* manifest_entries) { - int count; - if (!receive_int(fd, &count)) { - send_status(fd, STATUS_ERROR); - return false; - } - if (count < 0 || count > MAX_MANIFEST_ENTRIES || - (size_t)count > MAX_MANIFEST_ENTRIES - *manifest_entries) { - send_status(fd, STATUS_ERROR); - return false; - } - for (int i = 0; i < count; i++) { - char* s = receive_wire_str(fd); - size_t entry_size = s ? strlen(s) + MANIFEST_ENTRY_OVERHEAD : 0; - if (!s || s[0] == '\0' || s[0] == '/' || has_path_traversal(s) || - entry_size > MAX_MANIFEST_BYTES - *manifest_bytes || - (*manifest_bytes += entry_size) > MAX_MANIFEST_BYTES || !array_list_add(list, s)) { - free(s); - send_status(fd, STATUS_ERROR); - return false; - } - } - *manifest_entries += (size_t)count; - return true; -} - -DeleteManifest* receive_manifest_entries(int fd) { - DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest)); - if (!manifest) { - send_status(fd, STATUS_ERROR); - return NULL; - } - manifest->keeps = array_list_create(free); - manifest->protected = array_list_create(free); - manifest->missing = array_list_create(free); - manifest->dirs = array_list_create(free); - if (!manifest->keeps || !manifest->protected || !manifest->missing || !manifest->dirs) { - delete_manifest_free(manifest); - send_status(fd, STATUS_ERROR); - return NULL; - } - size_t manifest_bytes = 0; - size_t manifest_entries = 0; - if (!receive_manifest_section(fd, manifest->keeps, &manifest_bytes, &manifest_entries) || - !receive_manifest_section(fd, manifest->protected, &manifest_bytes, &manifest_entries) || - !receive_manifest_section(fd, manifest->missing, &manifest_bytes, &manifest_entries) || - !receive_manifest_section(fd, manifest->dirs, &manifest_bytes, &manifest_entries)) { - delete_manifest_free(manifest); - return NULL; - } - return manifest; -} - -void delete_manifest_free(DeleteManifest* manifest) { - if (!manifest) - return; - array_list_delete(manifest->keeps); - array_list_delete(manifest->protected); - array_list_delete(manifest->missing); - array_list_delete(manifest->dirs); - free(manifest); -} - -/* Shared --max-delete budget for one receiver-side deletion commit. Both the - --delete-missing-args exact-path removals and the ordinary extras walk draw - from the same tally, matching rsync (whose --max-delete counts every deleted - file or directory). `max_delete` is SIZE_MAX for an unlimited budget. */ -typedef struct { - size_t max_delete; - size_t deleted; - size_t skipped; - bool limit_hit; -} DeleteBudgetState; - -/* Build the delete-walk protection prefix for one basis directory. The walker - compares paths relative to the receive root, so a relative entry is already - in the right form; an absolute entry that lies below the root is converted to - its root-relative form, and one outside the root returns NULL (the walk - cannot reach it, and it is not protected data beneath the root). Exposed so - tests can exercise the root-of-"/" child mapping directly. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path) { - if (!path) - return NULL; - if (path[0] != '/') - return str_dup(path); - const char* root = config->receive_root_directory; - if (!root || root[0] != '/') - return NULL; - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(path, root, root_len) != 0) - return NULL; - if (root_len == 1) { - /* `root` is "/" (the only single-character absolute root): every absolute - path is below it, and the child relative form is everything after the - leading '/'. */ - if (path[1] == '\0') - return NULL; /* identical to the root, not a child */ - return str_dup(path + 1); - } - if (path[root_len] != '/') - return NULL; /* identical or a sibling sharing a name prefix */ - return str_dup(path + root_len + 1); -} - -/* Remove every destination entry under the receive root that is not in the - keep-set, bounded by the shared budget (a smaller client --max-delete=NUM - replaces the server hard bound; rsync deletes up to the bound and skips the - rest). With --delay-updates the not-yet-published staging directory is a - direct child of the receive root and must not be treated as a set of extras; - the manifest's protected prefixes (paths excluded on the source), the - size-pruned prefixes (--max-size/--min-size, always protected) and the - alternate basis directories are never destination content and are skipped at - any depth. Returns true unless a traversal/unlink error aborted the walk; - the budget's limit_hit/skipped fields report a cap-stopped run. */ -static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest* manifest, - DeleteBudgetState* budget, DeletePathObserver observer, - void* observer_context) { - if (!config || !manifest || !manifest->keeps) - return false; - fprintf(stderr, "Deleting files not in manifest...\n"); - /* Protected entries: - - the --delay-updates staging name, protected only as a DIRECT child of the - receive root (a nested destination directory that happens to be named - .fastsync-stage is ordinary content); - - alternate basis directories (--compare-dest / --copy-dest / --link-dest) - at any depth: they are extra comparison snapshots the user pointed at, - not destination content, and deleting them would destroy the very files a - --link-dest run just linked into place; - - the sender-side protected prefixes (source paths excluded by filters and - paths pruned by --max-size/--min-size), at any depth, so their destination - mirror survives --delete unless --delete-excluded opts back into removing - the filter-excluded ones (size-pruned entries are always protected). */ - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* An absolute basis outside the receive root is unreachable by this walk, - so it contributes no protection prefix (and no slot). */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } - /* Clamp rather than subtract: an accounting bug where deleted already exceeds - max_delete must never underflow into an effectively unlimited budget. */ - size_t remaining; - if (budget->max_delete == SIZE_MAX) - remaining = SIZE_MAX; - else if (budget->deleted >= budget->max_delete) - remaining = 0; - else - remaining = budget->max_delete - budget->deleted; - size_t deleted = 0; - size_t skipped = 0; - DeleteWalkResult result = delete_extras_limited_observed( - config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips, used, - config->protect_rules, &deleted, &skipped, observer, observer_context); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); - budget->deleted += deleted; - budget->skipped += skipped; - if (result == DELETE_WALK_LIMIT_REACHED) { - budget->limit_hit = true; - return true; - } - if (result != DELETE_WALK_OK) { - log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files"); - return false; - } - return true; -} - -static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest, - DeleteBudgetState* budget) { - return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL); -} - -/* Prefixes every observed path with a fixed subtree root, so a nested walk - (a recursively removed missing-arg directory) reports receive-root-relative - names like the rest of the delete output. */ -typedef struct { - DeletePathObserver inner; - void* inner_context; - const char* prefix; -} PrefixedDeleteObserver; - -static void prefixed_delete_observer(void* context, const char* rel) { - PrefixedDeleteObserver* prefixed = context; - if (!prefixed->inner || !rel) - return; - char* joined = path_cat((char*)prefixed->prefix, rel); - if (joined) { - prefixed->inner(prefixed->inner_context, joined); - free(joined); - } -} - -/* --delete-missing-args exact-path deletions: each destination mirror in - manifest->missing is an explicit user request, so it is removed even when the - ordinary extras walk (with its protected prefixes) would leave it alone. The - --delay-updates staging directory and basis snapshots are receiver artifacts - and stay protected exactly as in the extras walker. A regular file or - symlink is unlinked, an empty directory removed, and a NON-empty directory is - removed recursively only when --delete or --force is in effect (rsync parity: - the man page says a non-empty directory mirror is only deleted with --force - or --delete); otherwise it is left with a warning and the run continues. A - mirror that does not exist is a no-op. Each removal draws from the shared - --max-delete budget: once it is exhausted the remaining requests are skipped - and counted. Returns false only on a genuine error (a confinement failure on - a validated path or an I/O error), which fails the run. */ -static bool delete_missing_args_budgeted_observed(const Config* config, DeleteManifest* manifest, - DeleteBudgetState* budget, - DeletePathObserver observer, - void* observer_context) { - if (!config || !manifest) - return false; - if (!manifest->missing || manifest->missing->size == 0) - return true; - fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } - bool ok = true; - for (int i = 0; i < manifest->missing->size; i++) { - const char* rel = (const char*)manifest->missing->items[i]; - if (!rel || *rel == '\0' || *rel == '/' || has_path_traversal(rel)) { - /* Defensive only: receive_manifest_entries already validated every - section identically, so a controlled peer never reaches this branch. */ - log_message(LOG_LEVEL_ERROR, "invalid missing-args delete path"); - ok = false; - continue; - } - bool at_root = strchr(rel, '/') == NULL; - if (path_under_skip_prefix(rel, at_root, skips, used)) { - char* escaped = output_escape(rel, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "missing-args path '%s' is protected (staging directory or basis snapshot); " - "not deleting", - escaped ? escaped : ""); - free(escaped); - continue; - } - char* full = path_cat(config->receive_root_directory, rel); - if (!full) { - ok = false; - continue; - } - char* leaf = NULL; - int parent_fd = file_open_secure_parent(full, &leaf, false); - if (parent_fd < 0) { - /* The mirror's parent directory may itself not exist on the destination - (a deeper missing entry whose leading directories were never created). - That is a no-op -- there is nothing to delete -- matching - file_remove_tree_secure's absent-path handling; only a genuine I/O - error (EACCES, a symlink loop, ...) fails the run. */ - bool absent = errno == ENOENT || errno == ENOTDIR; - free(full); - free(leaf); - if (!absent) - ok = false; - continue; - } - struct stat st; - if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) { - /* Already absent: nothing to delete (a no-op, not a deletion). */ - if (errno != ENOENT) - ok = false; - close(parent_fd); - free(leaf); - free(full); - continue; - } - /* An entry that exists is one deletion: skip it (and count it) when the - shared --max-delete budget is already exhausted. */ - if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - close(parent_fd); - free(leaf); - free(full); - continue; - } - bool removed = false; - if (S_ISDIR(st.st_mode)) { - if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) { - removed = true; - } else if (errno == ENOTEMPTY || errno == EEXIST) { - close(parent_fd); - parent_fd = -1; - free(leaf); - leaf = NULL; - if (config->use_delete || config->force_delete) { - /* Remove the contents entry-by-entry through the budgeted extras - walker so every deleted file/dir counts toward --max-delete (rsync - parity); the now-empty directory itself costs one more. A run that - hits the cap leaves the remaining entries in place. */ - ArrayList* no_keeps = array_list_create(free); - /* Never let an accounting slip (deleted > max_delete) underflow the - remaining budget into SIZE_MAX, which would grant unlimited - deletions. */ - size_t remaining = - budget->deleted >= budget->max_delete ? 0 : budget->max_delete - budget->deleted; - size_t contents_deleted = 0; - size_t contents_skipped = 0; - PrefixedDeleteObserver nested = {observer, observer_context, rel}; - DeleteWalkResult walk = - no_keeps ? delete_extras_limited_observed(full, no_keeps, NULL, remaining, NULL, 0, - NULL, &contents_deleted, &contents_skipped, - observer ? prefixed_delete_observer : NULL, - observer ? &nested : NULL) - : DELETE_WALK_ERROR; - if (no_keeps) - array_list_delete(no_keeps); - budget->deleted += contents_deleted; - budget->skipped += contents_skipped; - if (walk == DELETE_WALK_LIMIT_REACHED) { - budget->limit_hit = true; - } else if (walk != DELETE_WALK_OK) { - ok = false; - } else if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - } else if (file_remove_tree_secure(full)) { - /* The shared `if (removed)` tail charges this directory exactly - once; counting it here too would consume two budget units. */ - removed = true; - } else { - ok = false; - } - } else { - char* escaped = output_escape(rel, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, - "missing-args destination '%s' is a non-empty directory; use --force or " - "--delete to remove it", - escaped ? escaped : ""); - free(escaped); - } - } else if (errno != ENOENT) { - ok = false; - } - } else { - if (unlinkat(parent_fd, leaf, 0) == 0) { - removed = true; - } else if (errno != ENOENT) { - ok = false; - } - } - if (removed) { - budget->deleted++; - if (observer) - observer(observer_context, rel); - char* escaped = output_escape(rel, log_get_8_bit_output()); - fprintf(stderr, " Deleted: %s\n", escaped ? escaped : ""); - free(escaped); - } - if (parent_fd >= 0) - close(parent_fd); - free(leaf); - free(full); - if (!ok) - break; - } - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); - return ok; -} - -/* Public wrappers used outside the commit path (and by unit tests): no - --max-delete budget. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out) { - if (count_out) - *count_out = 0; - if (!config || !manifest || !manifest->keeps || !out) - return false; - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* Normalize exactly like the real commit path: a relative entry is - already root-relative, an absolute one inside the receive root is - converted, and one outside contributes no protection prefix. */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } - bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs, - skips, used, config->protect_rules, out, count_out); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); - return ok; -} - -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) { - DeleteBudgetState budget = { - .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; - return delete_extras_budgeted(config, manifest, &budget); -} - -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) { - DeleteBudgetState budget = { - .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; - return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL); -} - -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, size_t* skipped, - bool* limit_hit) { - return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted, - skipped, limit_hit, NULL, NULL); -} - -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context) { - DeleteBudgetState budget = { - .max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; - bool ok = - delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context); - if (deleted) - *deleted = budget.deleted; - if (skipped) - *skipped = budget.skipped; - if (limit_hit) - *limit_hit = budget.limit_hit; - return ok; -} - -/* Commit every deletion family the manifest carries. The --delete-missing-args - exact-path deletions run FIRST: they are explicit user requests and must not - be blocked by the extras walker's filter-exclusion protection (a protected - leftover inside a missing-argument directory must not make that user-requested - removal fail). The ordinary extras walk then runs when --delete is active. - Both draw from one --max-delete budget; the result reports a cap-stopped - (partial) commit distinctly so the client can exit 25 like rsync. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest) { - return manifest_delete_all_counted(config, manifest, NULL); -} - -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, - size_t* deleted) { - return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL); -} - -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, - void* observer_context) { - if (deleted) - *deleted = 0; - if (!config || !manifest) - return DELETE_COMMIT_ERROR; - /* Central no-mutation guard: a dry-run never deletes. No manifest is sent on - the dry-run path, but a hostile/buggy peer could; treat it as a no-op so - the receiver can never remove anything. */ - if (config->dry_run) - return DELETE_COMMIT_OK; - /* A client --max-delete=NUM smaller than the server's hard bound replaces it - for this run; both still bound the commit. */ - bool user_limited = - config->max_delete >= 0 && (size_t)config->max_delete < MAX_SERVER_DELETE_COUNT; - DeleteBudgetState budget = {.max_delete = user_limited ? (size_t)config->max_delete - : MAX_SERVER_DELETE_COUNT, - .deleted = 0, - .skipped = 0, - .limit_hit = false}; - if (config->delete_missing_args && - !delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context)) - return DELETE_COMMIT_ERROR; - if (config->use_delete && - !delete_extras_budgeted_observed(config, manifest, &budget, observer, observer_context)) - return DELETE_COMMIT_ERROR; - if (deleted) - *deleted = budget.deleted; - if (budget.limit_hit) { - if (user_limited) { - log_message(LOG_LEVEL_ERROR, "Deletions stopped due to --max-delete limit (%zu skipped)", - budget.skipped); - } else { - log_message(LOG_LEVEL_ERROR, - "Deletions stopped due to the server deletion limit of %u (%zu skipped)", - (unsigned)MAX_SERVER_DELETE_COUNT, budget.skipped); - } - return DELETE_COMMIT_LIMIT_REACHED; - } - return DELETE_COMMIT_OK; -} diff --git a/src/shared/file_receive.h b/src/shared/file_receive.h index e316000..c55fc4c 100644 --- a/src/shared/file_receive.h +++ b/src/shared/file_receive.h @@ -2,11 +2,19 @@ #define FILE_RECEIVE_H #include "config.h" +#include "delete_commit.h" +#include "file_save.h" #include "file_types.h" +#include "incremental_check.h" #include "utils.h" #include -/* Server-side file receive/save path. */ +/* Server-side file receive/save path. + * + * This header is the public facade for the file_receive module family: the + * wire receive dispatch (this file) plus the save-to-disk (file_save.h), the + * incremental check (incremental_check.h) and the delete-commit + * (delete_commit.h) modules. */ /* Cumulative caps for the deferred directory-time accumulator. The sender may * legitimately split a large tree across repeated STATUS_DIR_TIMES frames, so a @@ -23,25 +31,6 @@ File* file_receive_dir_time(int file_descriptor, const Config* config); File* file_receive_hardlink(int file_descriptor); File* file_receive_symlink(int file_descriptor, const Config* config); File* file_receive_special(int file_descriptor); -bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode); -/* Testable basis quick-check / verification policy. file_basis_quick_match is - * rsync's metadata quick-check for a basis candidate (equal size is required - * separately by the caller; this adds the --size-only / mtime / --modify-window - * leg). file_basis_content_required reports whether a hit must ALSO be - * confirmed by a whole-file content digest (--verify-basis; false is the - * default rsync-parity behavior). */ -bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, - long check_mtime_nsec); -bool file_basis_content_required(const Config* config); - -File* receive_incremental_check(int fd, const Config* config, bool* skipped); -/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set - * true only on the server-contacting --dry-run path when the file is not up to - * date: the receiver has already sent STATUS_DRY_RUN_TRANSFER and returns NULL - * without storing anything. On that path `*skipped` is true for an up-to-date - * (STATUS_OK) file and both flags are false for a genuine error. */ -File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, - bool* would_transfer); /* P7 Wave D directory-time accumulator. The receiver collects the metadata of * every directory it creates/receives (STATUS_MKDIR with metadata and/or the @@ -85,126 +74,4 @@ bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetad void dir_metadata_list_apply(const DirTimeList* list, const char* root_directory, const Config* config); -/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative - paths the sender transferred/keeps) plus `protected`, destination-relative - prefixes the sender asks the receiver never to delete (paths excluded on the - source, protected at any depth). When --delete-excluded is given the sender - transmits an empty protected list so excluded destination mirrors are treated - as ordinary extras. With --delete-missing-args a third section (`missing`) - carries the destination mirrors of explicitly-listed source entries that do - not exist: each is an exact deletion request, independent of the ordinary - extras walk (never blocked by the protected prefixes) and processed when the - manifest is committed. */ -typedef struct DeleteManifest { - ArrayList* keeps; - ArrayList* protected; - ArrayList* missing; - /* Destination-relative paths of the directories the sender synchronized for - this run. The extras walker only removes entries directly inside one of - these (the receive root is the "." sentinel); `--files-from` runs therefore - leave untransmitted directories and the unlisted parts of listed ones - alone, matching rsync's "delete only in synchronized directories". */ - ArrayList* dirs; -} DeleteManifest; - -void delete_manifest_free(DeleteManifest* manifest); -/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then - protected count + protected prefixes, then missing count + missing paths, - then synchronized-directory count + directory paths (self-delimiting; the - leading STATUS_MANIFEST code has been consumed). Returns an owned - DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */ -DeleteManifest* receive_manifest_entries(int fd); -/* Remove destination entries under config->receive_root_directory that are not - in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and - protected-prefix skips). `--max-delete` and `--force` are honored here. The - caller decides WHEN to run it based on the negotiated delete timing. Returns - false (and the transfer fails) when the deletion cannot be committed. */ -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); -/* --delete-missing-args exact-path deletions: remove each destination mirror - in `manifest->missing` (never blocked by the protected prefixes, staging dir - and basis dirs excluded). A regular file/symlink is unlinked; an empty - directory is removed; a NON-empty directory is removed recursively only when - --delete or --force is in effect, otherwise it is left with a warning (rsync - parity). A missing path is a no-op. Returns false only on a genuine - confinement or I/O error (the run then fails); tolerated per-path cases are - reported and skipped. */ -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest); -/* Budgeted form of manifest_delete_missing_args for the per-directory delete - session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited) - and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set - when the budget stopped the pass with entries left over. Returns false only - on a genuine deletion error. */ -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, size_t* skipped, - bool* limit_hit); -/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may - be NULL) is invoked for every destination-relative path truly removed. */ -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context); -/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's - partial --max-delete result: the budget allowed some deletions and the rest - were skipped (the run still stores all file data but the client exits 25). */ -typedef enum { - DELETE_COMMIT_OK = 0, - DELETE_COMMIT_LIMIT_REACHED, - DELETE_COMMIT_ERROR -} DeleteCommitResult; - -/* Run every deletion family the manifest carries: the --delete-missing-args - exact-path deletions first (user requests are not blocked by exclusion - protection), then the ordinary extras walk when --delete is active. Both - share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to - do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget - stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest); -/* Like manifest_delete_all, but reports how many destination entries the commit - removed (for the end-of-transfer wire stats). `deleted` may be NULL. */ -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, - size_t* deleted); -/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL) - is invoked for every destination-relative path truly removed. */ -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, - void* observer_context); - -/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as - the delete pass would and append (strdup'd) destination-relative paths that - WOULD be removed to `out`, without touching disk. Uses the same staging-dir, - basis-dir and protected-prefix skips as the real commit. Returns true on a - clean walk; `*count_out` receives the number of paths appended. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out); -/* Convert one basis-directory path to the receive-root-relative protection - prefix the delete walker uses (NULL when it lies outside the root). Exposed - for unit tests of the root-of-"/" and normalization edge cases. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path); - -/* Outcome of a single file_save_to_disk operation. The receiver needs to - distinguish "written" from "skipped" so --remove-source-files can be told - which sources were actually stored. */ -typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2 } FileSaveResult; - -FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, - const Config* config); -/* Protocol 2.28.0 variant: also reports through `created` (when non-NULL) - * whether the destination entry did not exist before this save, and through - * `created_dirs` how many parent directories the confined walk created, so the - * receiver can build rsync's `Number of created files` breakdown. The plain - * file_save_to_disk_full() is this with both out-params NULL. */ -FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, - const Config* config, bool* created, - unsigned* created_dirs); -bool file_save_to_disk(const char* root_directory, const File* file, const Config* config); - -/* Protocol 2.28.0 receiver counter accumulator: fold one successfully saved - * entry into `stats`, adding its receiver-observed literal bytes and, when - * `created`, the matching created-by-type counter (regular file / symlink / - * special) plus `created_dirs` implicitly-created parent directories. - * Non-first hardlink siblings contribute no literal bytes. */ -void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, - unsigned created_dirs); - #endif diff --git a/src/shared/file_save.c b/src/shared/file_save.c new file mode 100644 index 0000000..98000ae --- /dev/null +++ b/src/shared/file_save.c @@ -0,0 +1,1164 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "array_list.h" +#include "charset.h" +#include "chmod.h" +#include "chunk.h" +#include "compression.h" +#include "config.h" +#include "data.h" +#include "delay_updates.h" +#include "delta.h" +#include "file.h" +#include "file_save.h" +#include "format.h" +#include "identity.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include "xattr.h" + +bool file_save_to_disk(const char* root_directory, const File* file, const Config* config) { + return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL) != FILE_SAVE_ERROR; +} + +FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, + const Config* config) { + return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL); +} + +/* --delay-updates receiver path: write the file into a private staging tree + below the receive root instead of its final destination, and remember it so + it can be atomically renamed into place only once the whole transfer has + succeeded. Existence/update policies (--existing/--ignore-existing/--update) + are decided against the FINAL destination path at stage time so the run + decides exactly what an immediate (non-delayed) run would decide; the staged + file is then never re-checked at publication. Backups are deferred to + publication so the final destination is untouched until the transfer ends. */ +static FileSaveResult file_stage_delayed_update(const char* root_directory, + const char* destination_path, const File* file, + Config* config) { + if (!config) + return FILE_SAVE_ERROR; + bool sparse = config->preserve_sparse; + FileAttrPolicy policy = file_attr_policy_from_config(config); + + if (config->existing && !file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + if (config->ignore_existing && file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + if (config->update && file_destination_is_newer_secure(destination_path, file->metadata)) + return FILE_SAVE_SKIPPED; + + FileMetadata adjusted_metadata; + const FileMetadata* metadata = file->metadata; + if (metadata && config->chmod_spec && *config->chmod_spec) { + adjusted_metadata = *metadata; + if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) + return FILE_SAVE_ERROR; + metadata = &adjusted_metadata; + } + + if (!config->delay_context) { + config->delay_context = delay_updates_context_create(root_directory); + if (!config->delay_context) + return FILE_SAVE_ERROR; + } + DelayUpdatesContext* context = config->delay_context; + if (!delay_updates_prepare(context)) + return FILE_SAVE_ERROR; + + char* staged_path = path_cat(context->staging_root, file->path); + if (!staged_path) + return FILE_SAVE_ERROR; + + /* The staged location is brand new (stale leftovers from a prior crash were + wiped by prepare), so the plain atomic temp+rename engine installs the + complete file there. --temp-dir scratch is deliberately not layered on + top of the delay-updates staging tree. A --link-dest basis file is hard + linked into the staging tree (so publication's rename keeps the link). */ + bool ok; + if (file->basis_link) { + ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size, + config->preallocate, metadata, policy, config->use_fsync, NULL); + } else if (file->basis_copy) { + /* --copy-dest basis hit: stream the basis into the staging tree (bounded + buffers, so an over-limit basis still stages). */ + ok = file_copy_basis_stream_attrs(staged_path, file->basis_copy, file->data->size, + config->preallocate, metadata, policy, config->update, + config->use_fsync, file->xattrs, config->fake_super, NULL); + } else { + ok = + file_to_disk_secure_attrs(staged_path, file->data->data, file->data->size, false, sparse, + config->preallocate, metadata, policy, false, false, + config->use_fsync, file->xattrs, config->fake_super, false, NULL); + } + if (!ok) { + free(staged_path); + return FILE_SAVE_ERROR; + } + + if (!delay_updates_record(context, staged_path, destination_path, file->path)) { + unlink(staged_path); + free(staged_path); + return FILE_SAVE_ERROR; + } + free(staged_path); + return FILE_SAVE_WRITTEN; +} + +/* Read the whole content of a confined regular file (used to fall back to a + byte-identical copy when a hard-link sibling's link() fails). Symlink-safe + (parent resolved via file_open_secure_parent + O_NOFOLLOW). A zero-length + file yields *out_size 0 and *out_buf NULL as a SUCCESS. Returns false only + on a real error/read failure, setting *source_absent to true when the reason + was that the path does not exist (ENOENT/ENOTDIR), so the caller can decide + between an abort and a graceful skip. */ +static bool hardlink_read_source(const char* path, void** out_buf, unsigned long long* out_size, + bool* source_absent) { + *out_buf = NULL; + *out_size = 0; + *source_absent = false; + if (!path) + return false; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) { + *source_absent = errno == ENOENT || errno == ENOTDIR; + return false; + } + /* O_NONBLOCK is a no-op for a regular file but makes openat() fail/succeed + immediately for a client-planted FIFO instead of blocking the receive + thread forever; the post-open S_ISREG gate below is the actual type check. */ + int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + int saved_errno = errno; + free(leaf); + close(parent_fd); + if (fd < 0) { + *source_absent = saved_errno == ENOENT || saved_errno == ENOTDIR; + return false; + } + struct stat st; + if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode)) { + close(fd); + return false; + } + unsigned long long size = (unsigned long long)st.st_size; + if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX) { + close(fd); + return false; + } + if (size == 0) { + close(fd); + return true; + } + void* buf = protocol_alloc((size_t)size); + if (!buf) { + close(fd); + return false; + } + size_t got = 0; + while (got < (size_t)size) { + ssize_t n = read(fd, (char*)buf + got, (size_t)size - got); + if (n <= 0) { + free(buf); + close(fd); + return false; + } + got += (size_t)n; + } + close(fd); + *out_buf = buf; + *out_size = size; + return true; +} + +/* The group's first member's installed file is absent, but its destination + path was validated (a sibling is only ever processed after its group's first + member). When the sibling's OWN destination already exists it should be + left alone -- a clean skip -- rather than aborting the whole transfer (the + asymmetric --existing case: the first member was skipped because its + destination was missing, while the sibling already has one). Only when the + sibling's destination is missing too is this a genuine failure to + link/copy, which aborts. */ +static FileSaveResult hardlink_sibling_absent_first(const char* destination_path) { + if (destination_path && file_path_exists_secure(destination_path)) + return FILE_SAVE_SKIPPED; + return FILE_SAVE_ERROR; +} + +/* Install a --hard-links/-H sibling: the destination entry is atomically + replaced (temp + rename) with a hard link to the group's first member. The + first member is guaranteed already installed at `hardlink_target` under the + root because -H relies on the receiver's single-FIFO-writer pipeline (one + receive thread, one write thread, FIFO queue => wire order == write order) + plus the sender's forced sequential scan, so a sibling is always processed + after its group's first member. When link() fails (different filesystem, + filesystem refuses links) a byte-identical copy of the first member is + written instead, so the result is never partial or corrupt. With + --delay-updates the sibling is staged as a hard link to the first member's + STAGED file (publication's renames preserve the shared inode). The final + --existing/--ignore-existing/--update policies are decided against the final + destination like every normal write. */ +static FileSaveResult file_save_hardlink_sibling(const char* root_directory, const File* file, + const Config* config, bool* created) { + Config* cfg = (Config*)config; + if (!root_directory || !file || !file->path || !file->hardlink_target) + return FILE_SAVE_ERROR; + char* destination_path = path_cat(root_directory, file->path); + if (!destination_path) + return FILE_SAVE_ERROR; + bool existed = file_path_exists_secure(destination_path); + + if (cfg->existing && !file_path_exists_secure(destination_path)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + if (cfg->ignore_existing && file_path_exists_secure(destination_path)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + if (cfg->update && file_destination_is_newer_secure(destination_path, file->metadata)) { + free(destination_path); + return FILE_SAVE_SKIPPED; + } + + bool preallocate = cfg && cfg->preallocate; + FileAttrPolicy policy = file_attr_policy_from_config(cfg); + bool use_fsync = cfg && cfg->use_fsync; + + if (cfg->delay_updates) { + if (!cfg->delay_context) { + cfg->delay_context = delay_updates_context_create(root_directory); + if (!cfg->delay_context) { + free(destination_path); + return FILE_SAVE_ERROR; + } + } + if (!delay_updates_prepare(cfg->delay_context)) { + free(destination_path); + return FILE_SAVE_ERROR; + } + char* staged_first = path_cat(cfg->delay_context->staging_root, file->hardlink_target); + char* staged_sibling = path_cat(cfg->delay_context->staging_root, file->path); + if (!staged_first || !staged_sibling) { + free(staged_first); + free(staged_sibling); + free(destination_path); + return FILE_SAVE_ERROR; + } + void* content = NULL; + unsigned long long content_size = 0; + bool source_absent = false; + if (!hardlink_read_source(staged_first, &content, &content_size, &source_absent)) { + FileSaveResult absent_result = + source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; + free(staged_first); + free(staged_sibling); + free(destination_path); + return absent_result; + } + FileXattrList* sibling_xattrs = + cfg->use_xattrs ? xattr_capture_path(staged_first, cfg->preserve_acls) : NULL; + bool ok = file_to_disk_secure_link_attrs(staged_sibling, staged_first, content, content_size, + preallocate, file->metadata, policy, use_fsync, + sibling_xattrs, cfg ? cfg->fake_super : false, NULL); + xattr_list_free(sibling_xattrs); + free(content); + if (ok) + ok = delay_updates_record(cfg->delay_context, staged_sibling, destination_path, file->path); + if (!ok) + unlink(staged_sibling); + free(staged_first); + free(staged_sibling); + free(destination_path); + if (ok && created && !existed) + *created = true; + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; + } + + char* first_disk = path_cat(root_directory, file->hardlink_target); + if (!first_disk) { + free(destination_path); + return FILE_SAVE_ERROR; + } + void* content = NULL; + unsigned long long content_size = 0; + bool source_absent = false; + if (!hardlink_read_source(first_disk, &content, &content_size, &source_absent)) { + FileSaveResult absent_result = + source_absent ? hardlink_sibling_absent_first(destination_path) : FILE_SAVE_ERROR; + free(first_disk); + free(destination_path); + return absent_result; + } + /* Resolve a relative --temp-dir under the destination root, exactly as the + * primary save path does; an absolute or `..`-escaping value is rejected. */ + char* resolved_temp = NULL; + if (cfg->temp_dir) { + if (cfg->temp_dir[0] == '/' || has_path_traversal(cfg->temp_dir)) { + free(content); + free(first_disk); + free(destination_path); + return FILE_SAVE_ERROR; + } + resolved_temp = path_cat(root_directory, cfg->temp_dir); + if (!resolved_temp) { + free(content); + free(first_disk); + free(destination_path); + return FILE_SAVE_ERROR; + } + } + FileXattrList* sibling_xattrs = + cfg->use_xattrs ? xattr_capture_path(first_disk, cfg->preserve_acls) : NULL; + bool ok = file_to_disk_secure_link_attrs( + destination_path, first_disk, content, content_size, preallocate, file->metadata, policy, + use_fsync, sibling_xattrs, cfg ? cfg->fake_super : false, resolved_temp); + xattr_list_free(sibling_xattrs); + free(resolved_temp); + free(content); + free(first_disk); + free(destination_path); + if (ok && created && !existed) + *created = true; + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* Validate a transmitted special rdev against the node kind implied by `mode`'s + * S_IFMT bits. Char/block devices require a legal major/minor pair (non-negative, + * range-checked); a non-device special (FIFO/socket) must carry an empty rdev. + * Used identically on the wire path and at the secure recreation site so a + * malicious/bogus rdev can never drive a dangerous node. */ +bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) { + bool is_device = S_ISCHR(mode) || S_ISBLK(mode); + if (is_device) + return major >= 0 && minor >= 0 && major <= 0xffff && minor <= 0x00ffffff; + /* A non-device entry must actually be a special (FIFO/socket) and carry no + rdev; a regular/dir mode is never a valid special node. */ + return (S_ISFIFO(mode) || S_ISSOCK(mode)) && major == 0 && minor == 0; +} + +/* ---- Device/special node RECREATION (--devices/--specials), receiver side ---- + * + * Privilege gating: making a real device node requires CAP_MKNOD (root); making + * a FIFO works unprivileged (mkfifo). When the receiver lacks the capability, + * mknodat() fails with EPERM and the entry is SKIPPED with a warning -- the + * whole transfer must NOT abort just because the environment cannot make the + * node. CI runs non-root, so device creation is expected to skip there and + * only a FIFO is honestly assertable unprivileged. + * + * Confinement: the parent directory is opened fd-relative below the receive + * root (file_open_secure_parent: O_NOFOLLOW, no "..", root-checked) and the + * node is created with mknodat()/mkfifoat(), so it can never be placed outside + * the confined root and never follows a symlink. + * + * rdev validation: a malicious/bogus rdev (negative, out-of-range) is rejected + * here as well as on the wire (file_receive_special / chunk_deserialize), and a + * non-device entry must carry an empty rdev. + */ +static FileSaveResult file_save_special_to_disk(const char* root_directory, const File* file, + const Config* config, bool* created) { + /* The empty-path and structural checks stay unconditional; the redundant + ".." list-path re-check is skipped under --trust-sender exactly like the + receive layer (confinement is deferred to the secure parent walk below, + which is never disabled). */ + if (!root_directory || !file || !file->path || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path)) || !file->metadata) + return FILE_SAVE_ERROR; + + mode_t mode = file->metadata->mode; + bool is_char = S_ISCHR(mode); + bool is_blk = S_ISBLK(mode); + bool is_fifo = S_ISFIFO(mode); + bool is_sock = S_ISSOCK(mode); + if (!is_char && !is_blk && !is_fifo && !is_sock) { + log_message(LOG_LEVEL_ERROR, "Special node has no device/FIFO/socket mode"); + return FILE_SAVE_ERROR; + } + if (is_char || is_blk) { + if (!config || !config->preserve_devices) + return FILE_SAVE_SKIPPED; + /* --super / --no-super (P7 Wave E): char/block device-node creation is a + super-user activity. --no-super forbids it even for a root receiver; + AUTO and --super attempt it (an unprivileged attempt is refused by the + kernel and skipped). The helper is evaluated against THIS config's mode + so the policy does not depend on a prior identity_set_active(). Pure + FIFO creation is unprivileged and deliberately NOT gated here. */ + if (!privilege_super_mode_permitted(config->super_mode)) { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "skipping %s: super-user device-node creation is not permitted on this receiver", + escaped_path ? escaped_path : ""); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + } else if (is_fifo || is_sock) { + /* FIFOs and unix sockets are recreated by --specials. mknod(S_IFSOCK) + works unprivileged on Linux (the node carries no live socket), so unlike + a socket bound to a live fd it can be materialized. */ + if (!config || !config->preserve_specials) + return FILE_SAVE_SKIPPED; + } + /* Defense-in-depth rdev/type validation (also done on the wire path). */ + if (!file_special_rdev_valid(file->rdev_major, file->rdev_minor, mode)) { + log_message(LOG_LEVEL_ERROR, "Rejected out-of-range device rdev %d:%d", file->rdev_major, + file->rdev_minor); + return FILE_SAVE_ERROR; + } + + char* destination = path_cat(root_directory, file->path); + if (!destination) + return FILE_SAVE_ERROR; + bool existed = file_path_exists_secure(destination); + char* leaf = NULL; + int parent_fd = file_open_secure_parent(destination, &leaf, true); + if (parent_fd < 0) { + free(destination); + return FILE_SAVE_ERROR; + } + + /* --existing / --ignore-existing / --update decide against the node that + would be replaced, mirroring the regular-file path. */ + if (config->existing && !file_path_exists_secure(destination)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + if (config->ignore_existing && file_path_exists_secure(destination)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + if (config->update && file_destination_is_newer_secure(destination, file->metadata)) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + + dev_t rdev = 0; + mode_t create_mode; + if (is_char) { + create_mode = S_IFCHR; + rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); + } else if (is_blk) { + create_mode = S_IFBLK; + rdev = makedev((unsigned)file->rdev_major, (unsigned)file->rdev_minor); + } else if (is_sock) { + create_mode = S_IFSOCK; + } else { + create_mode = S_IFIFO; + } + const char* node_kind = (is_char || is_blk) ? "device" : (is_fifo ? "FIFO" : "socket"); + /* Under -p/--perms rsync copies the source's permission and special bits; a + * kernel that denies setuid/setgid/sticky reports the failure rather than + * having them masked here. Without -p the node is created like any other new + * entry: source_mode & 0777 & ~umask. When super-user activities are + * forbidden, the special bits are stripped even under -p (they are + * super-user activities just like device-node creation). */ + mode_t perms = config->preserve_perms ? (mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777)) + : (mode & 0777 & ~(mode_t)file_process_umask()); + if (!privilege_super_mode_permitted(config->super_mode)) + perms &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); + + int rc = is_fifo ? mkfifoat(parent_fd, leaf, perms) + : mknodat(parent_fd, leaf, create_mode | perms, rdev); + if (rc != 0) { + if (errno == EEXIST) { + /* An entry already exists: only skip when it already is a matching node; + never replace an existing directory or unrelated entry with the node. */ + struct stat st; + if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0 && + ((is_char && S_ISCHR(st.st_mode)) || (is_blk && S_ISBLK(st.st_mode)) || + (is_fifo && S_ISFIFO(st.st_mode)) || (is_sock && S_ISSOCK(st.st_mode)))) { + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "refusing to replace existing entry with %s: %s (skipped)", + node_kind, escaped_path ? escaped_path : ""); + free(escaped_path); + } else if (errno == EPERM || errno == EACCES) { + /* Missing CAP_MKNOD / parent write permission: the environment cannot + create the node, so skip instead of failing the whole run. */ + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "skipping %s: cannot create %s node (%s)\n" + " --devices/--specials node creation needs privilege (CAP_MKNOD)", + escaped_path ? escaped_path : "", node_kind, strerror(errno)); + free(escaped_path); + } else { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "failed to create %s %s: %s (skipped)", node_kind, + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_SKIPPED; + } + + /* Apply times on the fresh node (utimensat, no-follow) per the negotiated + * per-attribute policy: mtime only under -t, atime only under -U. The slot + * not requested stays UTIME_OMIT so it is left untouched. */ + FileAttrPolicy policy = file_attr_policy_from_config(config); + if (policy.times || (policy.atimes && file->metadata->atime_valid)) { + struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = 0, .tv_nsec = UTIME_OMIT}}; + if (policy.times) { + times[1].tv_sec = file->metadata->mtime_sec; + times[1].tv_nsec = file->metadata->mtime_nsec; + } + if (policy.atimes && file->metadata->atime_valid) { + times[0].tv_sec = file->metadata->atime_sec; + times[0].tv_nsec = file->metadata->atime_nsec; + } + utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW); + } + /* P7 Wave E: apply the negotiated ownership to the node ITSELF. A FIFO is + created unprivileged, but --copy-as and explicit identity policies own + every entry (a char/block node path is already privilege-gated above). The + no-follow helper changes the node's own ownership without dereferencing it; + it is a no-op unless an identity policy is active. */ + bool owner_ok = true; + if (identity_active_enabled()) + owner_ok = identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, + (int32_t)file->metadata->gid); + close(parent_fd); + free(leaf); + free(destination); + /* A failed required --copy-as ownership marks the node as failed; every other + * identity policy stays best-effort. */ + if (owner_ok && created && !existed) + *created = true; + return owner_ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* --write-devices (receiver): write the received data directly into an EXISTING + * device node on the destination instead of creating a regular file. The node + * must already exist and be a char/block device (the device itself is opened and + * followed); it is confined to the receive root via file_open_secure_parent. + * Dangerous by nature, so deliberately restricted: a missing/non-device + * destination, or a write failure, is SKIPPED with a warning rather than + * allowed. On environments without device access the run still succeeds (the + * entry is skipped), never aborts. */ +static FileSaveResult file_save_write_device(const char* root_directory, const File* file) { + if (!root_directory || !file || !file->path || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path))) + return FILE_SAVE_ERROR; + if (!file->data) + return FILE_SAVE_ERROR; + char* destination = path_cat(root_directory, file->path); + if (!destination) + return FILE_SAVE_ERROR; + char* leaf = NULL; + int parent_fd = file_open_secure_parent(destination, &leaf, false); + if (parent_fd < 0) { + free(destination); + return FILE_SAVE_SKIPPED; + } + /* O_NONBLOCK: a pre-existing FIFO at the target would otherwise block the + receive thread forever on open(2). With it the open only succeeds for a + readerless FIFO with O_RDWR (which the device fstat gate rejects anyway) + or fails with ENXIO/EAGAIN, both treated as a normal skip below. */ + int fd = openat(parent_fd, leaf, O_WRONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + int saved_errno = errno; + free(leaf); + close(parent_fd); + if (fd < 0) { + free(destination); + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + const char* shown_path = escaped_path ? escaped_path : ""; + if (saved_errno == ENXIO || saved_errno == EAGAIN) { + /* A FIFO with no reader / an unreadable special: skip like every other + unusable write-devices target instead of blocking or failing. */ + log_message(LOG_LEVEL_WARNING, "write-devices: %s not writable (%s); skipped", shown_path, + strerror(saved_errno)); + } else { + log_message(LOG_LEVEL_WARNING, "write-devices: cannot open %s (%s); skipped", shown_path, + strerror(saved_errno)); + } + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + struct stat st; + if (fstat(fd, &st) != 0 || !(S_ISCHR(st.st_mode) || S_ISBLK(st.st_mode))) { + close(fd); + free(destination); + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "write-devices: %s is not a device node; skipped", + escaped_path ? escaped_path : ""); + free(escaped_path); + return FILE_SAVE_SKIPPED; + } + bool ok = true; + if (file->data->size > 0) { + size_t total = (size_t)file->data->size; + size_t written = 0; + while (written < total) { + ssize_t n = write(fd, (char*)file->data->data + written, total - written); + if (n <= 0) { + ok = false; + break; + } + written += (size_t)n; + } + } + close(fd); + free(destination); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_SKIPPED; +} + +/* ---- file_save_to_disk_full_ex decomposition ---- + * + * The regular-file install path is split into small static helpers that share + * one FileSavePlan (the owned path-set) and route every exit through a single + * cleanup epilogue so the path-set is released exactly once. Each helper owns + * one decision: request validation, special/device dispatch, directory and + * symlink creation, path resolution, the --existing/--ignore-existing/--update/ + * --force/--backup pre-write policies, and the data install. The ordering of + * every check and every protocol/write operation is unchanged. + */ + +/* Owned state for one regular-file install. The path-set pointers are owned by + * the plan and freed together by file_save_plan_dispose(). */ +typedef struct { + const char* root_directory; + const File* file; + const Config* config; + bool backup_enabled; + bool inplace; + bool sparse; + FileAttrPolicy policy; + const char* backup_suffix; + const char* backup_dir; + const char* partial_dir; + const char* temp_dir; + bool use_partial_root; + bool dest_existed; + char* confined_backup; + char* confined_partial; + char* disk_path; + char* destination_path; + char* backup_path; + char* parent_copy; + char* confined_temp; +} FileSavePlan; + +static void file_save_plan_init(FileSavePlan* plan, const char* root_directory, const File* file, + const Config* config) { + memset(plan, 0, sizeof(*plan)); + plan->root_directory = root_directory; + plan->file = file; + plan->config = config; + plan->backup_enabled = config && config->backup && !config->ignore_existing; + plan->inplace = config && config->inplace; + plan->sparse = config && config->preserve_sparse; + plan->policy = file_attr_policy_from_config(config); + plan->backup_suffix = (config && config->suffix) ? config->suffix : "~"; + plan->backup_dir = (config && config->backup_dir) ? config->backup_dir : NULL; + plan->partial_dir = (config && config->partial_dir) ? config->partial_dir : NULL; + plan->temp_dir = (config && config->temp_dir) ? config->temp_dir : NULL; + plan->use_partial_root = plan->partial_dir && config && config->partial; +} + +/* Single cleanup epilogue: release the whole owned path-set exactly once. */ +static void file_save_plan_dispose(FileSavePlan* plan) { + free(plan->parent_copy); + free(plan->backup_path); + free(plan->confined_backup); + free(plan->confined_partial); + free(plan->destination_path); + free(plan->disk_path); + free(plan->confined_temp); +} + +/* Structural validation of the received entry. */ +static bool file_save_validate(const FileSavePlan* plan) { + const File* file = plan->file; + const char* backup_suffix = plan->backup_suffix; + return file && file->path && file->data && + (file->data->size == 0 || file->data->data || file->basis_link || file->basis_copy) && + (file_get_trust_sender() || !has_path_traversal(file->path)) && + (!plan->backup_enabled || + (backup_suffix && backup_suffix[0] != '\0' && strchr(backup_suffix, '/') == NULL && + strcmp(backup_suffix, ".") != 0 && strcmp(backup_suffix, "..") != 0)); +} + +/* Explicit directory entry (--dirs). */ +static FileSaveResult file_save_directory_to_disk(const FileSavePlan* plan, bool* created) { + const File* file = plan->file; + if (file->path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(file->path))) { + log_message(LOG_LEVEL_ERROR, "Invalid directory path received"); + return FILE_SAVE_ERROR; + } + char* dir_path = path_cat(plan->root_directory, file->path); + if (!dir_path) + return FILE_SAVE_ERROR; + bool dir_existed = file_path_exists_secure(dir_path); + bool ok = file_ensure_directory_secure(dir_path); + /* P7 Wave E: apply the negotiated ownership to the directory ITSELF (not + just the files inside it). --copy-as and every explicit identity policy + own every entry, so a directory must not keep the receiver's owner while + its children get the policy owner. Applied no-follow on the confined + parent fd after the mkdir; identity_apply_ownership_link() is itself a + no-op unless an identity policy is active. */ + if (ok && file->metadata && identity_active_enabled()) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(dir_path, &leaf, false); + if (parent_fd >= 0) { + if (!identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, + (int32_t)file->metadata->gid)) + ok = false; + close(parent_fd); + } else if (identity_copy_as_active()) { + /* The directory exists (ok) but its required --copy-as ownership could + not be applied because the confined parent could not be opened. */ + ok = false; + } + free(leaf); + } + free(dir_path); + if (ok && created && !dir_existed) + *created = true; + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* Symlink entry. (The process-wide --keep-dirlinks policy is set once by the + connection handler from the negotiated config, before any receiver/writer + threads start, so it is stable throughout this walk.) */ +static FileSaveResult file_save_symlink_to_disk(const FileSavePlan* plan, bool* created) { + const File* file = plan->file; + const Config* config = plan->config; + if (!file->symlink_target || file->path[0] == '\0' || + (!file_get_trust_sender() && has_path_traversal(file->path))) { + log_message(LOG_LEVEL_ERROR, "Invalid symlink entry received"); + return FILE_SAVE_ERROR; + } + char* link_path = path_cat(plan->root_directory, file->path); + if (!link_path) + return FILE_SAVE_ERROR; + bool link_existed = file_path_exists_secure(link_path); + /* The link value is stored verbatim (rsync -l parity: absolute and + ".."-bearing targets are preserved; the scanner's --safe-links / + --copy-unsafe-links decide which links are sent at all). --munge-links + is a RECEIVER-side rewrite: the stored target is prefixed with + /rsyncd-munged/, making the link unusable while the referenced directory + does not exist -- exactly as rsync's receiver munges. Only the link's + own placement path is confined below the receive root. */ + bool munge = config && config->munge_links; + char* target = str_dup(file->symlink_target); + bool ok = target != NULL; + if (ok && munge) { + char* munged = file_symlink_munge(target); + free(target); + target = munged; + ok = target != NULL; + } + if (!ok) { + free(target); + free(link_path); + return FILE_SAVE_SKIPPED; + } + char* parent = str_dup(link_path); + if (parent) { + /* Propagate a failed --copy-as ownership of the parent directory this + creates; every other failure mode stays best-effort as before. */ + ok = file_ensure_directory_secure(dirname(parent)); + free(parent); + } + if (ok) + ok = file_symlink_at_secure(link_path, target); + free(target); + /* P7 Wave D: apply the symlink's own metadata with no-follow primitives + (utimensat/lchown/fchmodat AT_SYMLINK_NOFOLLOW). -J/--omit-link-times + suppresses the timestamps; ownership stays gated by the identity policy. + A symlink has no children, so this can be applied immediately. */ + if (ok && config && config->use_metadata) { + FileAttrPolicy link_policy = file_attr_policy_from_config(config); + ok = file_restore_symlink_metadata(link_path, file->metadata, link_policy, + config->omit_link_times); + } + if (ok && created && !link_existed) + *created = true; + free(link_path); + return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR; +} + +/* Device/special node (--devices/--specials) and --write-devices dispatch. */ +static bool file_save_try_special_dispatch(const FileSavePlan* plan, bool* created, + FileSaveResult* out) { + const File* file = plan->file; + const Config* config = plan->config; + /* Device/special node (--devices/--specials): recreate the node instead of + writing content (privilege-gated, confined, rdev-validated). */ + if (file->is_special) { + *out = file_save_special_to_disk(plan->root_directory, file, config, created); + return true; + } + /* --write-devices: write straight into an existing device node. Writing + into a device is a super-user activity, so --no-super must suppress it just + like device-node creation; the default AUTO/--super attempt it (the wide + open below keeps its own confinement and best-effort skip semantics). */ + if (config && config->write_devices) { + if (!privilege_super_mode_permitted(config->super_mode)) { + char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, + "write-devices: %s skipped: super-user activities are not permitted on this " + "receiver", + escaped_path ? escaped_path : "(null)"); + free(escaped_path); + *out = FILE_SAVE_SKIPPED; + return true; + } + *out = file_save_write_device(plan->root_directory, file); + return true; + } + return false; +} + +/* Resolve and confine the backup/partial/temp directories and the destination + and disk paths. Returns false on an invalid/escaping option or an + allocation failure (the caller routes to the cleanup epilogue). */ +static bool file_save_resolve_paths(FileSavePlan* plan) { + /* These options arrive from the client. --backup-dir, --partial-dir and + --temp-dir are names below the server root, never independent filesystem + roots: an absolute or `..`-escaping value is rejected outright (rsync's + daemon confines temp-dir to the module the same way). A relative temp dir + is resolved under the receive root below; if that resolution still lands on + a different filesystem than the destination the install falls back to a + non-atomic copy (see file_to_disk_secure_impl), never an abort. */ + if ((plan->backup_dir && (plan->backup_dir[0] == '/' || has_path_traversal(plan->backup_dir))) || + (plan->partial_dir && + (plan->partial_dir[0] == '/' || has_path_traversal(plan->partial_dir))) || + (plan->temp_dir && (plan->temp_dir[0] == '/' || has_path_traversal(plan->temp_dir)))) + return false; + if (plan->backup_dir && + !(plan->confined_backup = path_cat(plan->root_directory, plan->backup_dir))) + return false; + if (plan->partial_dir && + !(plan->confined_partial = path_cat(plan->root_directory, plan->partial_dir))) + return false; + + const char* actual_root = plan->use_partial_root ? plan->confined_partial : plan->root_directory; + plan->destination_path = path_cat(plan->root_directory, plan->file->path); + plan->disk_path = path_cat(actual_root, plan->file->path); + if (plan->destination_path == NULL || plan->disk_path == NULL) + return false; + /* Snapshot the final destination's existence BEFORE any backup/force/partial + step can move or remove it, so the receiver can report rsync's + `Number of created files` (protocol 2.28.0). */ + plan->dest_existed = file_path_exists_secure(plan->destination_path); + return true; +} + +typedef enum { + FILE_SAVE_POLICY_CONTINUE, /* proceed to the install */ + FILE_SAVE_POLICY_SKIP, /* --existing/--ignore-existing/--update skip */ + FILE_SAVE_POLICY_ERROR, /* --force/--backup failure */ +} FileSavePolicyOutcome; + +/* The immediate-install pre-write policies: --existing, --ignore-existing, + --update, --force and --backup, in that order. */ +static FileSavePolicyOutcome file_save_apply_prewrite_policies(FileSavePlan* plan) { + const Config* config = plan->config; + const File* file = plan->file; + + /* --existing checks the final destination, not a temporary partial path. */ + if (config && config->existing && !file_path_exists_secure(plan->destination_path)) + return FILE_SAVE_POLICY_SKIP; + + /* --ignore-existing checks the final destination before partial files or + overwrite policies can modify it. */ + if (config && config->ignore_existing) { + bool exists = file_path_exists_secure(plan->destination_path); + if (exists) + return FILE_SAVE_POLICY_SKIP; + } + + /* --update is receiver-side policy: never replace a newer destination. + In partial-dir mode the entry that would be replaced is the real + destination, not the temporary partial file. The secure stat does not + require read permission on the destination. */ + const char* update_target = plan->use_partial_root ? plan->destination_path : plan->disk_path; + if (config && config->update && file_destination_is_newer_secure(update_target, file->metadata)) + return FILE_SAVE_POLICY_SKIP; + + /* --force (rsync semantics): an incoming regular file may replace a + destination DIRECTORY by removing that (possibly non-empty, symlink-safe) + tree first, so the atomic temp+rename below can install the file. Only the + immediate-install path does this: a --delay-updates run stages into its own + tree and is unaffected here (its publication renames over regular files + only). The blocking directory is removed only after the --update / + --existing / --ignore-existing decisions above, which see it as an existing + destination entry. */ + if (config && config->force_delete && !file->is_dir && + file_directory_exists_secure(plan->destination_path)) { + if (!file_remove_tree_secure(plan->destination_path)) + return FILE_SAVE_POLICY_ERROR; + } + + if (plan->backup_enabled) { + /* Back up the entry that the incoming write will replace. When writing + through a partial dir the pre-existing destination file is the one to + preserve; any stale partial file is overwritten without a backup. */ + const char* replace_target = plan->use_partial_root ? plan->destination_path : plan->disk_path; + struct stat backup_stat; + if (file_stat_secure(replace_target, &backup_stat)) { + if (plan->backup_dir) { + plan->backup_path = path_cat(plan->confined_backup, file->path); + } else { + size_t path_len = strlen(replace_target); + size_t suffix_len = strlen(plan->backup_suffix); + if (path_len > SIZE_MAX - suffix_len - 1) + return FILE_SAVE_POLICY_ERROR; + plan->backup_path = malloc(path_len + suffix_len + 1); + if (plan->backup_path) { + memcpy(plan->backup_path, replace_target, path_len); + memcpy(plan->backup_path + path_len, plan->backup_suffix, suffix_len + 1); + } + } + if (!plan->backup_path) + return FILE_SAVE_POLICY_ERROR; + plan->parent_copy = str_dup(plan->backup_path); + if (!plan->parent_copy || !file_ensure_directory_secure(dirname(plan->parent_copy))) + return FILE_SAVE_POLICY_ERROR; + free(plan->parent_copy); + plan->parent_copy = NULL; + if (!file_rename_secure(replace_target, plan->backup_path)) + return FILE_SAVE_POLICY_ERROR; + free(plan->backup_path); + plan->backup_path = NULL; + } + } + return FILE_SAVE_POLICY_CONTINUE; +} + +/* Install the file data into the destination (or staging/partial path): + --link-dest hard link, --copy-dest streamed copy, or the plain atomic + temp+rename engine with per-file xattr/--fake-super application. */ +static bool file_save_install_data(FileSavePlan* plan, const FileMetadata* metadata, + unsigned* created_dirs) { + const File* file = plan->file; + const Config* config = plan->config; + char* count_floor = file_transfer_root_floor(config); + bool ok; + if (config && file->basis_link) { + ok = file_to_disk_secure_link_attrs_counted( + plan->disk_path, file->basis_link, file->data->data, file->data->size, config->preallocate, + metadata, plan->policy, config->use_fsync, file->xattrs, config->fake_super, + plan->confined_temp, created_dirs, count_floor); + } else if (config && file->basis_copy) { + /* --copy-dest: stream the basis bytes through a bounded buffer so a basis + larger than any whole-file bound still materializes. The source + metadata was transmitted with the check frame. */ + ok = file_copy_basis_stream_attrs(plan->disk_path, file->basis_copy, file->data->size, + config->preallocate, metadata, plan->policy, config->update, + config->use_fsync, file->xattrs, config->fake_super, + plan->confined_temp); + } else { + /* The plain no-replace / update / with-fsync engines, plus per-file xattr + (-X/-A) and --fake-super application on the written fd. */ + ok = file_to_disk_secure_attrs_counted( + plan->disk_path, file->data->data, file->data->size, plan->inplace, plan->sparse, + config && config->preallocate, metadata, plan->policy, config && config->update, + config && config->ignore_existing, config && config->use_fsync, file->xattrs, + config ? config->fake_super : false, config ? config->partial : false, plan->confined_temp, + created_dirs, count_floor); + } + free(count_floor); + return ok; +} + +FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, + const Config* config, bool* created, + unsigned* created_dirs) { + if (created) + *created = false; + if (created_dirs) + *created_dirs = 0; + /* Central no-mutation guard: a server-contacting --dry-run (or a local batch + apply that somehow carries dry_run) must never touch the destination, no + matter which caller reached this primitive. The per-caller guards remain, + but this is the last line of defense for every save path. Report SKIPPED + so a --remove-source-files sender correctly keeps its source. */ + if (config && config->dry_run) + return FILE_SAVE_SKIPPED; + + FileSavePlan plan; + file_save_plan_init(&plan, root_directory, file, config); + FileSaveResult result = FILE_SAVE_ERROR; + + if (!file_save_validate(&plan)) { + log_message(LOG_LEVEL_ERROR, "Invalid file or path received"); + goto out; + } + + /* P7 Wave D #1: a STATUS_DIR_TIMES entry is RECORD-ONLY. The scanner + captures every traversed directory -- including empty ones whose parents + were never created by a child write and directories pruned by + -m/--prune-empty-dirs. Creating them here would resurrect empty + directories (an -a behavior change) and could abort the whole transfer on a + pre-existing regular file/symlink at the mirror path. Short-circuit before + any device/write-devices/directory branch and report it as skipped so the + sink still accumulates its metadata for the deferred DirTimeList + application, but create nothing. */ + if (file->dir_time_only) { + result = FILE_SAVE_SKIPPED; + goto out; + } + + FileSaveResult dispatched; + if (file_save_try_special_dispatch(&plan, created, &dispatched)) { + result = dispatched; + goto out; + } + + /* Explicit directory entries (--dirs) carry an empty payload; the entry is + created as a directory under the receive root, applying the same secure + mkdir-parent semantics as regular writes. Directories are created + immediately (they are never staged by --delay-updates, matching rsync, + where directory creation is not delayed). */ + if (file->is_dir) { + result = file_save_directory_to_disk(&plan, created); + goto out; + } + + if (file->is_symlink) { + result = file_save_symlink_to_disk(&plan, created); + goto out; + } + + /* --hard-links/-H sibling: a later member of a link group arrives with no + payload and is installed as a hard link to (or, on link() failure, a + byte-identical copy of) the group's first member. Handled entirely here, + before the normal data-write paths (which would create an empty file). */ + if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) { + result = file_save_hardlink_sibling(root_directory, file, config, created); + goto out; + } + + if (!file_save_resolve_paths(&plan)) + goto out; + + /* --delay-updates diverts the whole write into the staging tree; the rest of + this function is the immediate-install path. */ + if (config && config->delay_updates) { + FileSaveResult staged = + file_stage_delayed_update(root_directory, plan.destination_path, file, (Config*)config); + if (staged == FILE_SAVE_WRITTEN && created && !plan.dest_existed) + *created = true; + result = staged; + goto out; + } + + FileSavePolicyOutcome policy_outcome = file_save_apply_prewrite_policies(&plan); + if (policy_outcome == FILE_SAVE_POLICY_SKIP) { + result = FILE_SAVE_SKIPPED; + goto out; + } + if (policy_outcome == FILE_SAVE_POLICY_ERROR) + goto out; + + FileMetadata adjusted_metadata; + const FileMetadata* metadata = file->metadata; + if (metadata && config && config->chmod_spec && *config->chmod_spec) { + adjusted_metadata = *metadata; + if (!chmod_apply(adjusted_metadata.mode, config->chmod_spec, &adjusted_metadata.mode)) + goto out; + metadata = &adjusted_metadata; + } + + /* A configured --temp-dir sends the temporary working copy to a scratch + directory; the engine then atomically renames the completed file into the + final destination directory. A relative temp dir is resolved under the + receive root and must already exist (an absolute or `..`-escaping value was + rejected above); the engine falls back to a non-atomic copy on EXDEV. The + partial-dir flow already keeps its working copy in a separate directory and + --inplace writes directly, so neither diverts through the scratch dir + (matching rsync, where --inplace/--partial-dir supersede --temp-dir). */ + bool use_temp_dir = plan.temp_dir != NULL && !plan.inplace && !plan.use_partial_root; + if (use_temp_dir) { + plan.confined_temp = path_cat(root_directory, plan.temp_dir); + if (!plan.confined_temp) + goto out; + /* A user-supplied trailing slash would leave the scratch path ending in + "/", which has no final component to create/open. Normalize it away. */ + size_t temp_len = strlen(plan.confined_temp); + while (temp_len > 1 && plan.confined_temp[temp_len - 1] == '/') + plan.confined_temp[--temp_len] = '\0'; + } + + /* A --link-dest basis hit installs an atomic hard link (with a byte-copy + fallback); --inplace and the update/no-replace write variants do not + apply to a fresh hard link, whose inode attributes already match. The + existing/ignore-existing/update/backup preamble above has already made the + policy decision. */ + if (!file_save_install_data(&plan, metadata, created_dirs)) + goto out; + + /* --partial --partial-dir writes the complete file under the partial dir so + interrupted transfers leave a resumable copy there. Once the file is + fully written it must be atomically installed at the real destination; + otherwise completed transfers would linger under the partial dir. */ + if (plan.use_partial_root) { + if (!file_rename_secure(plan.disk_path, plan.destination_path)) + goto out; + } + + if (created && !plan.dest_existed) + *created = true; + result = FILE_SAVE_WRITTEN; + +out: + file_save_plan_dispose(&plan); + return result; +} + +void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, + unsigned created_dirs) { + if (!stats || !file) + return; + /* A basis-dir hit (--link-dest/--copy-dest) materializes bytes the sender + * never transferred. rsync reports no literal data and no created entry for + * such a file, and does not count the parent directories it creates only to + * hold it, so exclude the whole entry from the receiver tallies. */ + bool basis_sourced = file->basis_link != NULL || file->basis_copy != NULL; + if (basis_sourced) + return; + bool is_sibling = file->link_group != 0 && !file->link_first; + if (!file->is_dir && !file->is_symlink && !file->is_special && !is_sibling) { + unsigned long long literal = file->literal_bytes; + if (literal == 0 && file->matched_bytes == 0) + literal = file->data ? file->data->size : 0; + stats->literal_bytes += literal; + } + stats->created_dir += created_dirs; + if (!created) + return; + if (file->is_dir) + stats->created_dir++; + else if (file->is_symlink) + stats->created_link++; + else if (file->is_special) + stats->created_special++; + else + stats->created_reg++; +} diff --git a/src/shared/file_save.h b/src/shared/file_save.h new file mode 100644 index 0000000..d995e5f --- /dev/null +++ b/src/shared/file_save.h @@ -0,0 +1,40 @@ +#ifndef FILE_SAVE_H +#define FILE_SAVE_H + +#include "config.h" +#include "file_types.h" +#include "format.h" +#include + +/* Save-to-disk module: regular-file/symlink/hardlink/special install, xattr + * application, --fake-super and the --delay-updates staging path. These + * declarations are re-exported by the file_receive.h facade. */ + +/* Outcome of a single file_save_to_disk operation. The receiver needs to + distinguish "written" from "skipped" so --remove-source-files can be told + which sources were actually stored. */ +typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2 } FileSaveResult; + +bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode); + +FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file, + const Config* config); +/* Protocol 2.28.0 variant: also reports through `created` (when non-NULL) + * whether the destination entry did not exist before this save, and through + * `created_dirs` how many parent directories the confined walk created, so the + * receiver can build rsync's `Number of created files` breakdown. The plain + * file_save_to_disk_full() is this with both out-params NULL. */ +FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file, + const Config* config, bool* created, + unsigned* created_dirs); +bool file_save_to_disk(const char* root_directory, const File* file, const Config* config); + +/* Protocol 2.28.0 receiver counter accumulator: fold one successfully saved + * entry into `stats`, adding its receiver-observed literal bytes and, when + * `created`, the matching created-by-type counter (regular file / symlink / + * special) plus `created_dirs` implicitly-created parent directories. + * Non-first hardlink siblings contribute no literal bytes. */ +void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created, + unsigned created_dirs); + +#endif diff --git a/src/shared/incremental_check.c b/src/shared/incremental_check.c new file mode 100644 index 0000000..98d4eb0 --- /dev/null +++ b/src/shared/incremental_check.c @@ -0,0 +1,1676 @@ +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include "array_list.h" +#include "charset.h" +#include "chmod.h" +#include "chunk.h" +#include "compression.h" +#include "config.h" +#include "data.h" +#include "delay_updates.h" +#include "delta.h" +#include "file.h" +#include "format.h" +#include "identity.h" +#include "incremental_check.h" +#include "log.h" +#include "metadata.h" +#include "protocol.h" +#include "utils.h" +#include "xattr.h" + +/* Receive a file's xattr block (when the config enables xattr transport) and + * attach it to `file`. Returns false on a malformed/oversized frame. */ +bool receive_file_xattrs(File* file, int fd, const Config* config) { + if (!config->use_xattrs) + return true; + int xok = 0; + FileXattrList* list = xattr_receive(fd, &xok, config->preserve_acls); + if (!xok) { + xattr_list_free(list); + return false; + } + file->xattrs = list; + return true; +} + +static File* receive_delta_file(int fd, const Config* config, const char* check_path, + void* old_data, unsigned long long old_size, bool* failed) { + if (!old_data) { + free(old_data); /* defensive: old_data is always non-NULL today */ + *failed = true; + return NULL; + } + + DeltaSignature* sig = delta_signature_create_seeded(old_data, old_size, config->delta_block_size, + (uint32_t)config->checksum_seed); + if (!sig) { + free(old_data); + *failed = true; + return NULL; + } + + Data* sig_data = delta_signature_serialize(sig); + if (!sig_data) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + bool sig_sent = send_status(fd, STATUS_DELTA_SIGNATURE) && send_data(fd, sig_data); + data_destroy(sig_data); + + if (!sig_sent) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + Status resp; + if (!receive_status(fd, &resp)) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + if (resp == STATUS_DELTA_DATA) { + Data* delta_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (!delta_data) { + delta_signature_destroy(sig); + free(old_data); + *failed = true; + return NULL; + } + + Data* raw_delta = delta_data; + if (config->use_compression && + !compression_should_skip_with_suffixes( + check_path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count : -1)) { + ProtocolSession* owner = delta_data->owner; + raw_delta = data_decompress_limited(delta_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + data_destroy(delta_data); + if (!raw_delta) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + /* Charge the decompressed delta to the connection budget (the paired + wire buffer's charge was just released). */ + if (!data_charge_session(raw_delta, owner, raw_delta->size)) { + data_destroy(raw_delta); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + } + + Delta* delta = delta_deserialize(raw_delta); + data_destroy(raw_delta); + if (!delta) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + uint64_t new_size = delta->new_file_size; + if (new_size > MAX_RECEIVE_WHOLE_FILE_SIZE || new_size > SIZE_MAX) { + delta_destroy(delta); + free(old_data); + delta_signature_destroy(sig); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + /* Wire-stats tally: bytes taken straight from the basis file (matched + delta blocks) and bytes shipped literally (protocol 2.28.0). Computed + before the delta is destroyed. */ + unsigned long long matched = 0; + unsigned long long literal = 0; + for (uint32_t k = 0; k < delta->instruction_count; k++) { + if (delta->instructions[k].type == DELTA_INSTR_BLOCK_MATCH) + matched += delta->instructions[k].match.length; + else if (delta->instructions[k].type == DELTA_INSTR_LITERAL) + literal += delta->instructions[k].literal.length; + } + void* new_data = delta_apply(old_data, old_size, delta, config->delta_block_size); + delta_destroy(delta); + + if (!new_data) { + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + File* file = file_create(check_path); + if (!file) { + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + file->matched_bytes = matched; + file->literal_bytes = literal; + + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + free(new_data); + free(old_data); + delta_signature_destroy(sig); + *failed = true; + return NULL; + } + + Data* replacement = data_create(new_data, (size_t)new_size); + if (replacement == NULL) { + file_destroy(file); + free(old_data); + delta_signature_destroy(sig); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + data_destroy(file->data); + file->data = replacement; + + free(old_data); + delta_signature_destroy(sig); + return file; + } + + if (resp == STATUS_NEXT) { + delta_signature_destroy(sig); + free(old_data); + + File* file = file_create(check_path); + if (!file) { + *failed = true; + return NULL; + } + + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + *failed = true; + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + *failed = true; + return NULL; + } + + Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (file_data == NULL) { + file_destroy(file); + *failed = true; + return NULL; + } + + if (config->use_compression && + !compression_should_skip_with_suffixes( + file->path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count : -1)) { + Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + ProtocolSession* owner = file_data->owner; + data_destroy(file_data); + if (uncompressed == NULL) { + file_destroy(file); + *failed = true; + return NULL; + } + if (!data_charge_session(uncompressed, owner, uncompressed->size)) { + data_destroy(uncompressed); + file_destroy(file); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + file_destroy(file); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; + } + file_data = uncompressed; + } + + data_destroy(file->data); + file->data = file_data; + return file; + } + + delta_signature_destroy(sig); + free(old_data); + send_status(fd, STATUS_ERROR); + *failed = true; + return NULL; +} + +/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ---- + * The receiver consults the ordered basis-dir list only when the destination + * entry is NOT already up to date. By default an "exact match" is rsync's + * metadata quick-check: an equal size and an equal mtime (unless --size-only). + * The FastSync-only --verify-basis additionally requires an equal whole-file + * content digest, so a hard link / local copy is only then made from + * byte-verified content. */ + +typedef struct BasisMatch { + bool hit; + BasisDestType type; + char* basis_path; /* owned absolute path of the matched basis file */ + struct stat st; /* fstat() of the matched basis file */ +} BasisMatch; + +static void basis_match_free(BasisMatch* match) { + if (!match) + return; + free(match->basis_path); + match->basis_path = NULL; + match->hit = false; + match->type = BASIS_DEST_NONE; +} + +/* Open `path` (via the secure, root-confined primitives) and require it to be + a regular file of exactly `expected_size` bytes. Returns an open read-only + descriptor and its fstat on success. */ +static bool basis_open_regular(const char* path, unsigned long long expected_size, int* out_fd, + struct stat* out_st) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(path, &leaf, false); + if (parent_fd < 0) + return false; + /* O_NONBLOCK: a client-planted FIFO must not block the receiver's openat() + forever; the fstat()/S_ISREG gate below rejects it immediately. */ + int fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + free(leaf); + close(parent_fd); + if (fd < 0) + return false; + struct stat st; + if (fstat(fd, &st) != 0 || !S_ISREG(st.st_mode) || + (unsigned long long)st.st_size != expected_size) { + close(fd); + return false; + } + *out_fd = fd; + *out_st = st; + return true; +} + +/* --ignore-times forces every file to be updated, so no basis hit is ever + declared (matching rsync, where -I prevents link-dest from linking). */ +bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, + long check_mtime_nsec) { + if (config->size_only) + return true; + long mtime_nsec = 0; +#ifdef __linux__ + mtime_nsec = st->st_mtim.tv_nsec; +#endif + return metadata_mtime_matches(st->st_mtime, mtime_nsec, check_mtime, check_mtime_nsec, + config->modify_window); +} + +/* True when a basis hit must be confirmed by a whole-file content digest + (--verify-basis). False is the rsync-parity default: the metadata + quick-check alone decides a hit. */ +bool file_basis_content_required(const Config* config) { + return config != NULL && config->verify_basis; +} + +/* Probe one candidate basis file: open it (confined, O_NOFOLLOW) and apply + rsync's metadata quick-check; under --verify-basis also hash its bytes and + require the sender's digest. On a hit record `candidate` in `out` and return + true. The caller retains ownership of `candidate`. */ +static bool basis_match_probe(const Config* config, const char* candidate, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, const uint8_t* check_digest, + size_t check_digest_len, BasisDestType type, BasisMatch* out) { + int fd; + struct stat st; + if (!basis_open_regular(candidate, check_size, &fd, &st)) + return false; + bool hit = false; + if (file_basis_quick_match(config, &st, check_mtime, check_mtime_nsec)) { + hit = true; + if (file_basis_content_required(config)) { + uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t basis_len = 0; + bool hashed = checksum_digest_fd((ChecksumAlgo)config->checksum_algo, config->checksum_seed, + fd, basis_digest, sizeof(basis_digest), &basis_len); + hit = hashed && basis_len == check_digest_len && check_digest_len > 0 && + memcmp(basis_digest, check_digest, check_digest_len) == 0; + } + } + close(fd); + if (!hit) + return false; + char* owned = str_dup(candidate); + if (!owned) + return false; + out->hit = true; + out->type = type; + out->basis_path = owned; + out->st = st; + return true; +} + +/* Search the basis-dir list in command-line order and return the first match. + By default (no --verify-basis) rsync's metadata quick-check is sufficient: + basis_open_regular has already required an equal size, and + file_basis_quick_match applies rsync's mtime (or --size-only) rule. + --verify-basis additionally requires the basis bytes' whole-file digest to + equal the sender's, restoring FastSync's historical content equality; that + digest is computed by streaming the open basis descriptor, so an arbitrarily + large basis is verified without buffering it. A copy/link install re-reads + the basis from its path in bounded buffers, so no content buffer is kept. + + `hash_content` gates content READS under --verify-basis: a server-contacting + --dry-run passes false because hashing a basis against a client-supplied + digest would be a 1-bit content oracle. Without --verify-basis a dry-run can + still confirm the metadata-only hit without reading any basis bytes, matching + rsync's read-only quick-check. + + Path resolution (rsync 3.4.1 parity): rsync resolves a relative + --compare-dest/--copy-dest/--link-dest DIR against the destination directory + (the receiver's cwd) and appends the file's TRANSFER-RELATIVE name, e.g. + `--compare-dest=basis` with `rsync src/ dst/` probes `dst/basis/`. + FastSync's receive root IS the destination directory, but its default transfer + mirrors the absolute source path below that root, so check_path carries the + source-root scaffolding rsync would not append. Recover rsync's spelling with + utils_strip_transfer_root for a relative DIR; under -R/--files-from the wire + path is already transfer-relative, so it is used as-is. A relative DIR also + probes the historical mirror-appended spelling as a fallback, so existing + FastSync-laid-out snapshot trees keep resolving. An absolute DIR is used + verbatim and keeps appending the destination-relative check_path (FastSync's + mirrored layout). Every candidate stays confined to the authorized root by + file_open_secure_parent. */ +static bool basis_match_find(const Config* config, const char* check_path, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, const uint8_t* check_digest, + size_t check_digest_len, bool hash_content, BasisMatch* out) { + memset(out, 0, sizeof(*out)); + if (!config || !config_has_basis(config) || config->ignore_times) + return false; + /* --verify-basis needs the basis content; a content-blind (dry-run) pass can + never confirm it and must not read the file, so decline without touching + the basis bytes. */ + if (file_basis_content_required(config) && !hash_content) + return false; + const char* transfer_rel = check_path; + if (!config->relative && config->files_from_set == NULL) + transfer_rel = utils_strip_transfer_root(check_path, config->send_directory); + for (int i = 0; i < config->basis_count; i++) { + const BasisDest* entry = &config->basis_dirs[i]; + /* An absolute basis path is used verbatim (rsync semantics); a relative one + is resolved below the receive root. Both remain subject to the receiver's + authorized-root confinement inside file_open_secure_parent. */ + bool absolute = entry->path[0] == '/'; + char* basis_dir = + absolute ? str_dup(entry->path) : path_cat(config->receive_root_directory, entry->path); + if (!basis_dir) + continue; + const char* names[2]; + int name_count = 0; + if (absolute) + names[name_count++] = check_path; + else + names[name_count++] = transfer_rel; + if (!absolute && strcmp(transfer_rel, check_path) != 0) + names[name_count++] = check_path; /* historical mirror-appended spelling */ + bool found = false; + for (int n = 0; n < name_count && !found; n++) { + char* candidate = path_cat(basis_dir, names[n]); + if (!candidate) + continue; + found = basis_match_probe(config, candidate, check_size, check_mtime, check_mtime_nsec, + check_digest, check_digest_len, entry->type, out); + free(candidate); + } + free(basis_dir); + if (found) + return true; + } + return false; +} + +/* --------------------------------------------------------------------------- + * -y/--fuzzy similar-file delta basis. + * + * When a file must be transferred and the destination holds no usable content + * at the exact path (the destination file is absent, or is outside the delta + * engine's size bounds), --fuzzy lets the receiver reuse an EXISTING regular + * file in the SAME destination directory as the delta basis, so the sender + * transmits only the differences instead of the whole file. This is the + * rsync "find a similar file to use as a basis for a transfer" case (e.g. a + * file recreated under a new name whose old-named sibling is still present). + * + * The delta handshake is unchanged and receiver-driven, so the sender never + * learns the basis was a different file and needs no new protocol. Byte + * exactness never depends on which bytes the basis holds: the delta protocol + * only references basis blocks whose Adler-32 + xxHash32 checksums match the + * source, delta_apply validates every reference against the basis size, and a + * basis that shares nothing simply makes the sender reply STATUS_NEXT (full + * transfer). A fuzzy basis can therefore waste bandwidth but never corrupt a + * file. + * + * Similarity heuristic (rsync 3.4.1 parity, util1.c fuzzy_distance / + * find_filename_suffix + generator.c find_fuzzy): + * * candidates are the target's sibling entries in its destination + * directory, opened through the confined root (file_open_secure_parent + + * openat O_NOFOLLOW, fstatat AT_SYMLINK_NOFOLLOW) -- symlinks are never + * followed and nothing outside the destination root is ever read; + * * dotfiles, directories, the target's own name, and the .fastsync-stage / + * temp scratch names are never candidates; + * * size gate = rsync's, NOT the ordinary delta engine's bounds: any + * non-empty regular sibling up to the receiver's whole-file buffer cap is + * eligible, regardless of the 16 KiB delta minimum or the 10x delta size + * ratio (rsync's find_fuzzy has no delta-size gate at all). The delta + * engine consumes the fuzzy basis through the same signature handshake + * whether or not it is inside delta_should_attempt's window; + * * first pass = an exact size+mtime match wins regardless of name (rsync's + * "fuzzy size/modtime match"); + * * otherwise the winner minimizes rsync's weighted Levenshtein distance + * (substitution ± byte difference, insertion UNIT+byte, 16.16 fixed point) + * plus ten times the suffix distance, accepted only when <= 25*UNIT; the + * tie-break (smallest size gap, then lexical name) keeps the result + * deterministic across filesystem readdir order (rsync leaves equal + * distances to its file-list order). + * ------------------------------------------------------------------------- */ + +/* A directory scan is linear in the number of entries; the fuzzy search stops + * after this many so a pathological huge directory cannot stall a transfer. + * The cap bounds the readdir() ITERATIONS, not the per-entry work: every + * entry that survives the (cheap) size and pre-name gates still runs an + * edit-distance DP, so the per-entry DP cost is separately bounded below by + * pre-pruning on the name length gap and the absent-character bound, and by + * trimming the common prefix/suffix before the DP runs on the middles only. */ +#define FUZZY_MAX_DIRECTORY_SCAN 4096 +/* Names longer than this never take part in fuzzy matching: the edit-distance + * DP below is O(len^2), so over-long names are bounded out of the search. */ +#define FUZZY_NAME_LIMIT 192 + +typedef struct { + char name[FUZZY_NAME_LIMIT + 1]; + unsigned long long size; + uint32_t distance; + unsigned long long size_gap; +} FuzzyCandidate; + +/* rsync's fuzzy distance is a weighted Levenshtein variant in 16.16 fixed point + * (util1.c fuzzy_distance): a substitution costs UNIT +/- the byte difference + * and an insertion costs UNIT + the inserted byte, so similar names score low. + * The search keeps only distances <= 25*UNIT. Ported verbatim for parity. */ +#define FUZZY_DIST_UNIT (1u << 16) +#define FUZZY_DIST_REJECT (0xFFFFu * FUZZY_DIST_UNIT + 1) +#define FUZZY_DIST_LIMIT (25u * FUZZY_DIST_UNIT) + +static uint32_t fuzzy_distance(const char* s1, unsigned len1, const char* s2, unsigned len2, + uint32_t upperlimit, uint32_t* scratch) { + if ((len1 > len2 ? len1 - len2 : len2 - len1) * FUZZY_DIST_UNIT > upperlimit) + return FUZZY_DIST_REJECT; + if (!len1 || !len2) { + if (!len1) { + s1 = s2; + len1 = len2; + } + uint32_t cost = 0; + for (unsigned i = 0; i < len1; i++) + cost += (uint8_t)s1[i]; + return (uint32_t)len1 * FUZZY_DIST_UNIT + cost; + } + uint32_t* a = scratch; + for (unsigned i2 = 0; i2 < len2; i2++) + a[i2] = (i2 + 1) * FUZZY_DIST_UNIT; + for (unsigned i1 = 0; i1 < len1; i1++) { + uint32_t diag = i1 * FUZZY_DIST_UNIT; + uint32_t above = (i1 + 1) * FUZZY_DIST_UNIT; + for (unsigned i2 = 0; i2 < len2; i2++) { + uint32_t left = a[i2]; + int32_t cost = (int32_t)(uint8_t)s1[i1] - (int32_t)(uint8_t)s2[i2]; + if (cost != 0) + cost = cost < 0 ? (int32_t)(FUZZY_DIST_UNIT - (uint32_t)(-cost)) + : (int32_t)(FUZZY_DIST_UNIT + (uint32_t)cost); + uint32_t diag_inc = diag + (uint32_t)cost; + uint32_t left_inc = left + FUZZY_DIST_UNIT + (uint8_t)s1[i1]; + uint32_t above_inc = above + FUZZY_DIST_UNIT + (uint8_t)s2[i2]; + a[i2] = above = left < above ? (left_inc < diag_inc ? left_inc : diag_inc) + : (above_inc < diag_inc ? above_inc : diag_inc); + diag = left; + } + } + return a[len2 - 1]; +} + +/* rsync's find_filename_suffix (util1.c): return the last significant filename + * suffix (its dot included). Leading dots are not a suffix; a trailing "~" is + * ignored; .bak/.old/.orig and a "~/" backup marker are skipped. */ +static const char* fuzzy_find_suffix(const char* fn, int fn_len, int* len_ptr) { + const char* suf; + const char* s; + bool had_tilde; + + while (fn_len && *fn == '.') { + fn++; + fn_len--; + } + if (fn_len > 1 && fn[fn_len - 1] == '~') { + fn_len--; + had_tilde = true; + } else { + had_tilde = false; + } + suf = ""; + *len_ptr = 0; + for (s = fn + fn_len; fn_len > 1;) { + int s_len; + while (--s != fn && *s != '.') { + } + if (s == fn) + break; + s_len = fn_len - (int)(s - fn); + fn_len = (int)(s - fn); + if (s_len == 4) { + if (strcmp(s + 1, "bak") == 0 || strcmp(s + 1, "old") == 0) + continue; + } else if (s_len == 5) { + if (strcmp(s + 1, "orig") == 0) + continue; + } else if (s_len > 2 && had_tilde && s[1] == '~' && isdigit((unsigned char)s[2])) { + continue; + } + *len_ptr = s_len; + suf = s; + if (s_len == 1) + break; + for (s++, s_len--; s_len > 0; s++, s_len--) { + if (!isdigit((unsigned char)*s)) + return suf; + } + s = suf; + } + return suf; +} + +/* Deterministic ordering of two fuzzy candidates with equal rsync distance: + * smallest size gap, then the lexical basename (rsync itself takes the last + * equal-distance candidate in file-list order). */ +static bool fuzzy_candidate_better(const FuzzyCandidate* cand, const FuzzyCandidate* best) { + if (!best->name[0]) + return true; + if (cand->distance != best->distance) + return cand->distance < best->distance; + if (cand->size_gap != best->size_gap) + return cand->size_gap < best->size_gap; + return strcmp(cand->name, best->name) < 0; +} + +/* Search the destination directory that will contain `check_path` for a + * similar regular file usable as a --fuzzy delta basis and return its full + * content in a malloc'd (protocol_alloc) buffer. Returns NULL (with *out_size + * = 0) when no candidate qualifies, which means the caller performs the normal + * whole-file transfer. */ +static void* fuzzy_basis_find_and_load(const Config* config, const char* check_path, + unsigned long long check_size, time_t check_mtime, + long check_mtime_nsec, unsigned long long* out_size) { + *out_size = 0; + if (!config || !config->receive_root_directory || !config->fuzzy || !config->use_delta || + !check_path || check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) + return NULL; + + char* full_path = path_cat(config->receive_root_directory, check_path); + if (!full_path) + return NULL; + char* leaf = NULL; + int dir_fd = file_open_secure_parent(full_path, &leaf, false); + if (dir_fd < 0 || !leaf) { + free(leaf); + free(full_path); + return NULL; + } + size_t target_len = strlen(leaf); + /* A target basename longer than FUZZY_NAME_LIMIT can never pass the name gate + (every candidate name is bounded by the same limit), so skip the scan. */ + if (target_len > FUZZY_NAME_LIMIT) { + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + + int scanfd = dup(dir_fd); + if (scanfd < 0) { + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + + /* The weighted-distance scratch row is allocated once per scan (not once per + candidate). */ + uint32_t* dist_scratch = malloc((FUZZY_NAME_LIMIT + 1) * sizeof(uint32_t)); + if (!dist_scratch) { + closedir(dir); + close(dir_fd); + free(leaf); + free(full_path); + return NULL; + } + int fname_suf_len = 0; + const char* fname_suf = fuzzy_find_suffix(leaf, (int)target_len, &fname_suf_len); + + FuzzyCandidate best; + memset(&best, 0, sizeof(best)); + uint32_t lowest_dist = FUZZY_DIST_LIMIT; + /* rsync's fuzzy search runs an exact size+mtime pass before the name-distance + pass; such a candidate is almost certainly the same content and wins + regardless of how dissimilar its name is. The first one (directory order, + deterministic) is kept. */ + FuzzyCandidate exact; + memset(&exact, 0, sizeof(exact)); + const struct dirent* entry; + size_t scanned = 0; + /* readdir() yields entries in filesystem-dependent order, so the SET of + candidates seen is order-dependent; the winner is still deterministic + because every candidate is compared with the total ordering in + fuzzy_candidate_better (acceptable for a heuristic). */ + while (scanned < FUZZY_MAX_DIRECTORY_SCAN && (entry = readdir(dir)) != NULL) { + scanned++; + const char* name = entry->d_name; + size_t name_len = strlen(name); + if (name[0] == '.' || name_len == 0 || name_len > FUZZY_NAME_LIMIT || strcmp(name, leaf) == 0) + continue; + struct stat st; + if (fstatat(dir_fd, name, &st, AT_SYMLINK_NOFOLLOW) != 0 || !S_ISREG(st.st_mode)) + continue; + unsigned long long cand_size = (unsigned long long)st.st_size; + if (cand_size == 0 || cand_size > MAX_RECEIVE_WHOLE_FILE_SIZE) + continue; + long cand_nsec = 0; +#ifdef __linux__ + cand_nsec = st.st_mtim.tv_nsec; +#endif + if (!exact.name[0] && cand_size == check_size && + metadata_mtime_matches(st.st_mtime, cand_nsec, check_mtime, check_mtime_nsec, + config->modify_window)) { + memcpy(exact.name, name, name_len + 1); + exact.size = cand_size; + exact.size_gap = 0; + continue; + } + /* rsync's name-distance pass: a weighted Levenshtein distance over the full + basenames, plus ten times the same distance over the filename suffixes, + accepted only when it does not exceed the running lowest distance. */ + int name_suf_len = 0; + const char* name_suf = fuzzy_find_suffix(name, (int)name_len, &name_suf_len); + uint32_t distance = fuzzy_distance(name, (unsigned)name_len, leaf, (unsigned)target_len, + lowest_dist, dist_scratch); + if (distance < 0xFFFF0000U) + distance += fuzzy_distance(name_suf, (unsigned)name_suf_len, fname_suf, + (unsigned)fname_suf_len, 0xFFFF0000U, dist_scratch) * + 10; + if (distance > lowest_dist) + continue; + lowest_dist = distance; + FuzzyCandidate cand; + memcpy(cand.name, name, name_len + 1); + cand.size = cand_size; + cand.distance = distance; + cand.size_gap = cand_size > check_size ? cand_size - check_size : check_size - cand_size; + if (fuzzy_candidate_better(&cand, &best)) + best = cand; + } + closedir(dir); + free(leaf); + free(dist_scratch); + + /* Prefer the exact size+mtime candidate over any name-distance winner. */ + if (exact.name[0]) + best = exact; + + void* basis = NULL; + if (best.name[0]) { + /* O_NONBLOCK: a name raced to a FIFO between the fstatat gate and this open + would otherwise block the receive thread forever on open(2); with it the + open fails (ENXIO) and the fstat/S_ISREG gate below would reject it too. + A regular file opened with O_NONBLOCK is unaffected. */ + int fd = openat(dir_fd, best.name, O_RDONLY | O_NOFOLLOW | O_NONBLOCK | O_CLOEXEC); + if (fd >= 0) { + struct stat st; + if (fstat(fd, &st) == 0 && S_ISREG(st.st_mode) && + (unsigned long long)st.st_size == best.size && best.size <= SIZE_MAX) { + basis = protocol_alloc((size_t)best.size); + if (basis) { + size_t got = 0; + while (got < (size_t)best.size) { + ssize_t n = read(fd, (char*)basis + got, (size_t)best.size - got); + if (n <= 0) { + free(basis); + basis = NULL; + break; + } + got += (size_t)n; + } + } + } + close(fd); + } + } + close(dir_fd); + free(full_path); + if (basis) + *out_size = best.size; + return basis; +} + +/* Read the remainder of a full-file transfer after the receiver has already + * sent STATUS_NEXT: receive the metadata frame (when enabled) followed by the + * data frame, and return an owned File. Shared by the plain full-transfer path + * and the --append-verify prefix-mismatch fallback (a clean full transfer + * instead of a corrupt prefix+tail blend). */ +static File* receive_full_file(int fd, const Config* config, const char* path) { + File* file = file_create(path); + if (!file) + return NULL; + if (config->use_metadata) { + int meta_ok = 1; + file->metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) { + file_destroy(file); + return NULL; + } + } + if (!receive_file_xattrs(file, fd, config)) { + file_destroy(file); + return NULL; + } + Data* file_data = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (file_data == NULL) { + file_destroy(file); + return NULL; + } + if (config->use_compression && + !compression_should_skip_with_suffixes(file->path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count + : -1)) { + Data* uncompressed = data_decompress_limited(file_data, MAX_RECEIVE_WHOLE_FILE_SIZE); + ProtocolSession* owner = file_data->owner; + data_destroy(file_data); + if (uncompressed == NULL) { + file_destroy(file); + return NULL; + } + if (!data_charge_session(uncompressed, owner, uncompressed->size)) { + data_destroy(uncompressed); + file_destroy(file); + return NULL; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + file_destroy(file); + return NULL; + } + file_data = uncompressed; + } + data_destroy(file->data); + file->data = file_data; + return file; +} + +/* --------------------------------------------------------------------------- + * receive_incremental_check() decomposition. + * + * The per-file STATUS_CHECK fast path is split into the small helpers below, + * called in order by a short linear orchestrator (receive_incremental_check_ex). + * Each helper owns one decision: request validation, secure destination open, + * metadata-only skip, server-contacting --dry-run no-mutation short-circuit, + * alternate-basis match, --append tail resume, block delta, --fuzzy basis, and + * the final "send the whole file" fallback. Every protocol send/receive and + * every resource cleanup is preserved exactly; the non-dry-run wire is + * byte-for-byte unchanged. receive_incremental_check_ex additionally exposes a + * `would_transfer` out-param for the dry-run caller; the 3-arg + * receive_incremental_check wrapper passes NULL. + * ------------------------------------------------------------------------- */ + +/* Owned state threaded through the helpers below. */ +typedef struct { + int fd; + const Config* config; + char* check_path; /* received destination-relative path */ + char* full_path; /* receive-root-prefixed destination path */ + unsigned long long check_size; + long long check_mtime; + long long check_mtime_nsec; + uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t check_digest_len; + /* Source metadata carried alongside the check frame whenever a basis dir is + configured (rsync keeps the whole file list; FastSync's sender-driven + incremental path otherwise never transmits metadata for a SKIPPED file). + A basis materialization applies these SOURCE attributes instead of the + basis inode's, matching rsync's "copy then fix attributes". */ + FileMetadata* source_metadata; + bool dest_exists; /* any destination entry exists (lstat succeeded) */ + bool has_old_file; + int old_fd; + struct stat old_st; + unsigned long long old_size; + void* old_data; /* snapshot of the existing destination, or NULL */ +} IncrementalCheckState; + +typedef enum { + INCREMENTAL_CONTINUE, /* proceed to the next helper */ + INCREMENTAL_ERROR, /* protocol/validation failure: return NULL */ + INCREMENTAL_SKIP, /* up to date: *skipped = true, return NULL */ + INCREMENTAL_DRY_RUN, /* --dry-run resolved: flags set, return NULL */ + INCREMENTAL_FILE, /* a File* was produced (out_file) */ +} IncrementalCheckOutcome; + +static void incremental_check_state_init(IncrementalCheckState* state, int fd, + const Config* config) { + memset(state, 0, sizeof(*state)); + state->fd = fd; + state->config = config; + state->old_fd = -1; +} + +/* Release every resource the helpers may have acquired. Idempotent, so it is + safe on every exit path exactly the way the original inline cleanup was. */ +static void incremental_check_state_cleanup(IncrementalCheckState* state) { + free(state->old_data); + state->old_data = NULL; + if (state->old_fd >= 0) + close(state->old_fd); + state->old_fd = -1; + file_metadata_destroy(state->source_metadata); + state->source_metadata = NULL; + free(state->full_path); + state->full_path = NULL; + free(state->check_path); + state->check_path = NULL; +} + +/* Receive and validate the STATUS_CHECK request frame: path, size, mtime, + nanosecond mtime, and (when negotiated) the source digest. */ +static IncrementalCheckOutcome incremental_check_receive_request(IncrementalCheckState* state) { + int fd = state->fd; + const Config* config = state->config; + char* check_path = receive_wire_str(fd); + if (check_path == NULL) + return INCREMENTAL_ERROR; + state->check_path = check_path; + + if (!receive_n_data(fd, &state->check_size, sizeof(state->check_size)) || + !receive_n_data(fd, &state->check_mtime, sizeof(state->check_mtime))) + return INCREMENTAL_ERROR; + if (!receive_n_data(fd, &state->check_mtime_nsec, sizeof(state->check_mtime_nsec)) || + state->check_mtime_nsec < 0 || state->check_mtime_nsec >= 1000000000LL) { + send_error_detail(fd, "invalid check mtime nanoseconds"); + return INCREMENTAL_ERROR; + } + if ((config->checksum || config->verify_basis)) { + uint8_t wire_len; + if (!receive_n_data(fd, &wire_len, sizeof(wire_len)) || wire_len == 0 || + wire_len > CHECKSUM_MAX_DIGEST_LEN || + wire_len != checksum_digest_len((ChecksumAlgo)config->checksum_algo)) { + send_error_detail(fd, "invalid check digest length"); + return INCREMENTAL_ERROR; + } + state->check_digest_len = wire_len; + if (!receive_n_data(fd, state->check_digest, state->check_digest_len)) + return INCREMENTAL_ERROR; + } + /* The sender transmits the source metadata with every basis-configured check + so a basis hit can be materialized with the SOURCE's attributes (rsync + copies/copies-then-fixes; the receiver would otherwise only have the basis + inode's stat). The block is symmetric and consumed unconditionally here, + whether or not this file ends up as a basis hit. */ + if (config_has_basis(config) && config->use_metadata) { + int meta_ok = 1; + state->source_metadata = metadata_receive(fd, &meta_ok); + if (!meta_ok) + return INCREMENTAL_ERROR; + } + + /* A basis-configured run may materialize a file larger than the whole-file + payload bound: a basis hit is streamed from the basis path (bounded + buffers), so the check size is not itself an allocation. Every other + path (delta/append/full) still applies MAX_RECEIVE_WHOLE_FILE_SIZE, and a + miss simply falls through to the normal transfer with its own bound. */ + if (!config_has_basis(config) && state->check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) { + send_error_detail(fd, "check size exceeds receiver limit"); + return INCREMENTAL_ERROR; + } + + if (check_path[0] == '\0' || has_path_traversal(check_path)) { + char* escaped_path = output_escape(check_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_ERROR, "Invalid received check path: %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + return INCREMENTAL_ERROR; + } + return INCREMENTAL_CONTINUE; +} + +/* Open the existing destination entry once, confined below the receive root, + and record its stat. */ +static IncrementalCheckOutcome incremental_check_open_destination(IncrementalCheckState* state) { + char* full_path = path_cat(state->config->receive_root_directory, state->check_path); + if (!full_path) { + send_error_detail(state->fd, "could not build destination path"); + return INCREMENTAL_ERROR; + } + state->full_path = full_path; + + char* leaf = NULL; + int parent_fd = file_open_secure_parent(full_path, &leaf, false); + if (parent_fd >= 0) { + struct stat dest_st; + if (fstatat(parent_fd, leaf, &dest_st, AT_SYMLINK_NOFOLLOW) == 0) + state->dest_exists = true; + /* O_NONBLOCK: an existing FIFO at the destination must not block this + openat(); the S_ISREG gate below rejects the non-regular entry. */ + state->old_fd = openat(parent_fd, leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK); + free(leaf); + close(parent_fd); + state->has_old_file = state->old_fd >= 0 && fstat(state->old_fd, &state->old_st) == 0 && + S_ISREG(state->old_st.st_mode); + } + if (!state->has_old_file && state->old_fd >= 0) { + close(state->old_fd); + state->old_fd = -1; + } + state->old_size = state->has_old_file ? (unsigned long long)state->old_st.st_size : 0; + return INCREMENTAL_CONTINUE; +} + +/* Output parity (protocol 2.23.0): when the wire config asked for it, report a + snapshot of the pre-transfer destination entry BEFORE the ordinary verdict so + the sender can render rsync-accurate -i/--out-format columns. A missing + destination is reported explicitly (existed=false) rather than omitted, so + the sender can distinguish "new" from "unknown". */ +static IncrementalCheckOutcome incremental_check_report_dest_info(IncrementalCheckState* state) { + if (!state->config->report_dest_info) + return INCREMENTAL_CONTINUE; + OutputDestState info; + memset(&info, 0, sizeof(info)); + info.known = true; + info.existed = state->has_old_file; + if (state->has_old_file) { + info.size = (unsigned long long)state->old_st.st_size; + info.mtime_sec = (long long)state->old_st.st_mtime; +#ifdef __linux__ + info.mtime_nsec = state->old_st.st_mtim.tv_nsec; +#endif + info.mode = (uint32_t)state->old_st.st_mode; + info.uid = (int32_t)state->old_st.st_uid; + info.gid = (int32_t)state->old_st.st_gid; + } + if (!send_status(state->fd, STATUS_DEST_INFO) || !format_dest_state_send(state->fd, &info)) + return INCREMENTAL_ERROR; + return INCREMENTAL_CONTINUE; +} + +/* --ignore-existing short-circuit. The receiver must answer "skip" (STATUS_OK) + BEFORE the sender transmits any payload, otherwise the whole file crosses the + wire only to be discarded at write time. rsync skips an existing destination + entry regardless of its content or type, so the reply depends only on the + lstat existence probe; the ordinary --ignore-existing checks inside + file_receive remain as defense-in-depth for the frame types that have no + per-file check (directories/symlinks/specials/hard-links). */ +static IncrementalCheckOutcome +incremental_check_ignore_existing(const IncrementalCheckState* state) { + if (!state->config->ignore_existing || !state->dest_exists) + return INCREMENTAL_CONTINUE; + if (!send_status(state->fd, STATUS_OK)) + return INCREMENTAL_ERROR; + return INCREMENTAL_SKIP; +} + +/* Metadata for a materialized basis hit: prefer the SOURCE metadata the sender + transmitted with the check frame (rsync copies then fixes the destination to + the source's attributes); fall back to the basis inode's own stat when + metadata was not negotiated. Consumes state->source_metadata on success. */ +static FileMetadata* basis_take_metadata(IncrementalCheckState* state, + const struct stat* basis_st) { + if (state->source_metadata) { + FileMetadata* meta = state->source_metadata; + state->source_metadata = NULL; + return meta; + } + return file_metadata_create(NULL, basis_st, false, false); +} + +/* --link-dest relink of an already up-to-date destination. rsync hard-links a + destination entry to a matching basis even when the entry is already correct, + so a run over an existing tree still maximizes sharing with the basis. Only a + link-dest basis triggers this (copy-dest/compare-dest leave an up-to-date + destination untouched, matching rsync). The ordinary basis path further down + handles every not-up-to-date case, so this helper only adds the relink that + the quick-skip would otherwise short-circuit. */ +static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalCheckState* state, + File** out_file) { + const Config* config = state->config; + if (!config_has_basis(config) || config->ignore_times || config->dry_run) + return INCREMENTAL_CONTINUE; + if (!state->has_old_file) + return INCREMENTAL_CONTINUE; + BasisMatch basis; + basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, + true, &basis); + /* Only a link-dest hit relinks; a copy-dest/compare-dest hit (or a miss) lets + the up-to-date check below keep the existing destination. */ + if (!basis.hit || basis.type != BASIS_DEST_LINK) { + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; + } + /* Already the basis inode: nothing to do, leave the destination alone. */ + if (basis.st.st_dev == state->old_st.st_dev && basis.st.st_ino == state->old_st.st_ino) { + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; + } + File* materialized = file_create(state->check_path); + if (materialized) { + data_destroy(materialized->data); + materialized->data = data_create_reserve((size_t)state->check_size); + if (!materialized->data) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + materialized->metadata = basis_take_metadata(state, &basis.st); + materialized->skip = true; + materialized->basis_link = basis.basis_path; + basis.basis_path = NULL; + if (!materialized->metadata) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + if (!send_status(state->fd, STATUS_OK)) { + basis_match_free(&basis); + file_destroy(materialized); + return INCREMENTAL_ERROR; + } + basis_match_free(&basis); + *out_file = materialized; + return INCREMENTAL_FILE; + } + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; +} + +/* Metadata-only (and, when --checksum forces it, content) up-to-date decision. + Loads the old contents only when a checksum comparison or delta needs them. */ +static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckState* state, + bool* out_try_delta) { + int fd = state->fd; + const Config* config = state->config; + bool has_old_file = state->has_old_file; + unsigned long long old_size = state->old_size; + struct stat st = state->old_st; + + bool size_equal = has_old_file && old_size == state->check_size; + bool match_by_metadata = false; + if (size_equal && !config->ignore_times && !config->size_only) { + long long old_mtime_nsec = 0; +#ifdef __linux__ + old_mtime_nsec = st.st_mtim.tv_nsec; +#endif + match_by_metadata = + metadata_mtime_matches(st.st_mtime, old_mtime_nsec, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, config->modify_window); + } + + bool try_delta = config->use_delta && !config->whole_file && has_old_file && + delta_should_attempt(old_size, state->check_size, config->delta_max_file_size); + bool checksum_needs_read = size_equal && !config->ignore_times && config->checksum; + /* --dry-run must never read the destination file's CONTENTS: a client could + otherwise use `--dry-run --checksum` against a read-only module as a + 1-bit content oracle (hash match / mismatch) and force arbitrary reads. + Decide from metadata alone; when metadata is inconclusive (checksum or + delta would have required the body) report would-transfer. The real + (non-dry-run) behavior below is unchanged. */ + bool need_old_data = !config->dry_run && (checksum_needs_read || try_delta); + if (config->dry_run) + try_delta = false; + *out_try_delta = try_delta; + + if (need_old_data && has_old_file && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && + old_size <= SIZE_MAX) { + state->old_data = protocol_alloc((size_t)old_size); + if (state->old_data) { + size_t got = 0; + while (got < (size_t)old_size) { + ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); + if (n <= 0) { + free(state->old_data); + state->old_data = NULL; + break; + } + got += (size_t)n; + } + } + } + + bool match = false; + if (config->dry_run) { + /* Metadata-only decision: a size match plus a matching mtime is treated as + up to date; --checksum/--delta cannot be verified without reading, so an + otherwise inconclusive comparison is a would-transfer. */ + match = size_equal && !config->ignore_times && (config->size_only || match_by_metadata); + } else if (checksum_needs_read) { + uint8_t old_digest[CHECKSUM_MAX_DIGEST_LEN]; + size_t old_len = 0; + bool hashed = checksum_digest((ChecksumAlgo)config->checksum_algo, config->checksum_seed, + old_size == 0 ? "" : state->old_data, (size_t)old_size, + old_digest, sizeof(old_digest), &old_len); + match = hashed && old_len == state->check_digest_len && state->check_digest_len > 0 && + memcmp(old_digest, state->check_digest, state->check_digest_len) == 0; + } else if (size_equal && !config->ignore_times) { + match = config->size_only || match_by_metadata; + } + + if (match) { + if (!send_status(fd, STATUS_OK)) + return INCREMENTAL_ERROR; + return INCREMENTAL_SKIP; + } + return INCREMENTAL_CONTINUE; +} + +/* Server-contacting --dry-run no-mutation short-circuit. Runs after the + quick-skip decision and before any path that could touch the destination. + When dry_run is set and the file is not already up to date the receiver must + materialize nothing (no basis link/copy, no append/delta/full transfer) and + the sender must send no data, so answer STATUS_DRY_RUN_TRANSFER and stop. + + The basis lookup is content-blind: under the default metadata quick-check a + hit needs no basis bytes and is honored here just as in a real run; under + --verify-basis a real run hashes the basis against the client-supplied digest, + which in a dry-run is a 1-bit content oracle, so no basis bytes may be read + and an otherwise-matching entry is reported as would-transfer. Everything + read here (the destination file's metadata, basis candidates' metadata) is + read-only. */ +static IncrementalCheckOutcome incremental_check_dry_run_shortcut(IncrementalCheckState* state, + bool* skipped, + bool* would_transfer) { + const Config* config = state->config; + if (!config->dry_run) + return INCREMENTAL_CONTINUE; + + bool skip_via_compare = false; + if (config_has_basis(config) && !config->ignore_times) { + BasisMatch basis; + /* hash_content=false: a dry-run must not read or hash the basis file, so + under --verify-basis no compare-dest hit can be confirmed and an + otherwise-matching file is reported as would-transfer. Without + --verify-basis the metadata quick-check confirms it without touching any + basis bytes. */ + basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, + false, &basis); + if (basis.hit && basis.type == BASIS_DEST_COMPARE && !state->has_old_file) + skip_via_compare = true; + basis_match_free(&basis); + } + Status reply = skip_via_compare ? STATUS_OK : STATUS_DRY_RUN_TRANSFER; + if (!send_status(state->fd, reply)) + return INCREMENTAL_ERROR; + if (skip_via_compare) + *skipped = true; + else if (would_transfer) + *would_transfer = true; + return INCREMENTAL_DRY_RUN; +} + +/* Alternate basis directories (--compare-dest/--copy-dest/--link-dest): a hit + either suppresses the transfer (compare-dest) or materializes the file from + the basis without a data frame. */ +static IncrementalCheckOutcome incremental_check_try_basis(IncrementalCheckState* state, + File** out_file) { + int fd = state->fd; + const Config* config = state->config; + if (!config_has_basis(config)) + return INCREMENTAL_CONTINUE; + + BasisMatch basis; + basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime, + (long)state->check_mtime_nsec, state->check_digest, state->check_digest_len, + true, &basis); + if (basis.hit) { + if (basis.type == BASIS_DEST_COMPARE) { + basis_match_free(&basis); + if (!state->has_old_file) { + if (!send_status(fd, STATUS_OK)) + return INCREMENTAL_ERROR; + return INCREMENTAL_SKIP; + } + } else { + /* Copy/link installs source their bytes from the basis PATH at install + time (bounded buffers), so no whole-file content buffer is needed here + even for an over-limit basis. */ + File* materialized = file_create(state->check_path); + if (materialized) { + data_destroy(materialized->data); + materialized->data = data_create_reserve((size_t)state->check_size); + if (!materialized->data) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + materialized->metadata = basis_take_metadata(state, &basis.st); + materialized->skip = true; /* receiver must not ack this as a data file */ + if (basis.type == BASIS_DEST_LINK) + materialized->basis_link = basis.basis_path; + else + materialized->basis_copy = basis.basis_path; + basis.basis_path = NULL; + if (!materialized->metadata) { + file_destroy(materialized); + materialized = NULL; + } + } + if (materialized) { + if (!send_status(fd, STATUS_OK)) { + basis_match_free(&basis); + file_destroy(materialized); + return INCREMENTAL_ERROR; + } + basis_match_free(&basis); + *out_file = materialized; + return INCREMENTAL_FILE; + } + /* Materialization setup failed: fall through to the normal transfer. */ + } + } + basis_match_free(&basis); + return INCREMENTAL_CONTINUE; +} + +/* --append / --append-verify tail resume: when the destination is a SHORTER + file in an append mode, negotiate the resume offset and receive only the + tail. Produces the reconstructed file, or falls through to delta/full. */ +static IncrementalCheckOutcome incremental_check_try_append_resume(IncrementalCheckState* state, + File** out_file) { + int fd = state->fd; + const Config* config = state->config; + const char* check_path = state->check_path; + unsigned long long old_size = state->old_size; + unsigned long long check_size = state->check_size; + + bool append_resume = (config->append || config->append_verify) && state->has_old_file && + append_resume_eligible(old_size, check_size); + if (!append_resume) + return INCREMENTAL_CONTINUE; + + /* Ensure the retained prefix (== the whole, shorter destination file) is in + memory; it is needed both to rebuild the full file and, for + --append-verify, to checksum it. A load failure is not fatal: the resume is + simply not possible and we fall through to the other paths. */ + if (state->old_data == NULL && old_size > 0 && old_size <= MAX_RECEIVE_WHOLE_FILE_SIZE && + old_size <= SIZE_MAX) { + state->old_data = protocol_alloc((size_t)old_size); + if (state->old_data) { + size_t got = 0; + while (got < (size_t)old_size) { + ssize_t n = read(state->old_fd, (char*)state->old_data + got, (size_t)old_size - got); + if (n <= 0) { + free(state->old_data); + state->old_data = NULL; + break; + } + got += (size_t)n; + } + } + } + if (state->old_data == NULL && old_size != 0) + return INCREMENTAL_CONTINUE; + + if (!send_status(fd, STATUS_APPEND) || !send_n_data(fd, &old_size, sizeof(old_size))) + return INCREMENTAL_ERROR; + bool verify = config->append_verify; + bool full_fallback = false; + if (verify) { + Status sig_status; + if (!receive_status(fd, &sig_status)) + return INCREMENTAL_ERROR; + if (sig_status != STATUS_APPEND_SIG) { + send_status(fd, STATUS_ERROR); + return INCREMENTAL_ERROR; + } + uint64_t src_prefix_hash; + if (!receive_n_data(fd, &src_prefix_hash, sizeof(src_prefix_hash))) + return INCREMENTAL_ERROR; + /* Compare the retained prefix against the source prefix. A mismatch must + never be silently appended to: fall back to a full transfer so the result + is a byte-identical source copy. */ + uint64_t dst_prefix_hash = + old_size == 0 ? delta_xxhash64("", 0) : delta_xxhash64(state->old_data, (size_t)old_size); + if (dst_prefix_hash == src_prefix_hash) { + if (!send_status(fd, STATUS_APPEND_OK)) + return INCREMENTAL_ERROR; + } else { + if (!send_status(fd, STATUS_NEXT)) + return INCREMENTAL_ERROR; + full_fallback = true; + } + } + + if (full_fallback) { + /* Retained prefix differed: receive the sender's full transfer. */ + free(state->old_data); + state->old_data = NULL; + if (state->old_fd >= 0) { + close(state->old_fd); + state->old_fd = -1; + } + *out_file = receive_full_file(fd, config, check_path); + return INCREMENTAL_FILE; + } + + /* Receive the tail (STATUS_APPEND_DATA + metadata + tail bytes). */ + Status tail_status; + if (!receive_status(fd, &tail_status)) + return INCREMENTAL_ERROR; + if (tail_status != STATUS_APPEND_DATA) { + send_status(fd, STATUS_ERROR); + return INCREMENTAL_ERROR; + } + FileMetadata* meta = NULL; + FileXattrList* append_xattrs = NULL; + if (config->use_metadata) { + int meta_ok = 1; + meta = metadata_receive(fd, &meta_ok); + if (!meta_ok) + return INCREMENTAL_ERROR; + } + if (config->use_xattrs) { + int xok = 0; + append_xattrs = xattr_receive(fd, &xok, config->preserve_acls); + if (!xok) { + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + } + Data* tail = receive_data_limited(fd, MAX_RECEIVE_WHOLE_FILE_SIZE); + if (tail == NULL) { + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (config->use_compression && + !compression_should_skip_with_suffixes(check_path, config->skip_compress_suffixes, + config->skip_compress_set ? config->skip_compress_count + : -1)) { + Data* uncompressed = data_decompress_limited(tail, MAX_RECEIVE_WHOLE_FILE_SIZE); + ProtocolSession* owner = tail->owner; + data_destroy(tail); + if (uncompressed == NULL) { + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (!data_charge_session(uncompressed, owner, uncompressed->size)) { + data_destroy(uncompressed); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (uncompressed->size > MAX_FILE_DATA_SIZE) { + data_destroy(uncompressed); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + tail = uncompressed; + } + /* The tail must complete the file exactly; anything else is a protocol + violation (never a truncated or overrun file). */ + unsigned long long expected_tail; + if (!append_tail_length(old_size, check_size, &expected_tail) || + tail->size != (size_t)expected_tail) { + send_status(fd, STATUS_ERROR); + data_destroy(tail); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + size_t full_size = (size_t)check_size; + void* full = protocol_alloc(full_size ? full_size : 1); + if (!full) { + data_destroy(tail); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + if (old_size > 0 && state->old_data) + memcpy(full, state->old_data, (size_t)old_size); + if (tail->size > 0) + memcpy((char*)full + old_size, tail->data, tail->size); + data_destroy(tail); + free(state->old_data); + state->old_data = NULL; + + File* file = file_create(check_path); + if (!file) { + free(full); + xattr_list_free(append_xattrs); + return INCREMENTAL_ERROR; + } + file->metadata = meta; + file->xattrs = append_xattrs; + append_xattrs = NULL; + data_destroy(file->data); + file->data = data_create(full, full_size); + if (!file->data) { /* data_create already freed full on failure */ + file_destroy(file); + return INCREMENTAL_ERROR; + } + *out_file = file; + return INCREMENTAL_FILE; +} + +/* Block delta transfer against the existing destination content. */ +static IncrementalCheckOutcome incremental_check_try_delta(IncrementalCheckState* state, + bool try_delta, File** out_file) { + if (try_delta && state->old_data != NULL) { + bool delta_failed = false; + File* delta_file = receive_delta_file(state->fd, state->config, state->check_path, + state->old_data, state->old_size, &delta_failed); + state->old_data = NULL; /* receive_delta_file consumes the snapshot on every path */ + if (delta_file) { + *out_file = delta_file; + return INCREMENTAL_FILE; + } + if (delta_failed) + return INCREMENTAL_ERROR; + } + free(state->old_data); + state->old_data = NULL; + return INCREMENTAL_CONTINUE; +} + +/* -y/--fuzzy similar-file delta basis. Reaching this point means the file + must be transferred and the destination's own content could not serve as a + delta basis; try an existing similar-named sibling in the same directory. */ +static IncrementalCheckOutcome incremental_check_try_fuzzy(IncrementalCheckState* state, + File** out_file) { + const Config* config = state->config; + if (!config->fuzzy || !config->use_delta) + return INCREMENTAL_CONTINUE; + unsigned long long fuzzy_size = 0; + void* fuzzy_basis = fuzzy_basis_find_and_load(config, state->check_path, state->check_size, + (time_t)state->check_mtime, + (long)state->check_mtime_nsec, &fuzzy_size); + if (fuzzy_basis != NULL) { + bool fuzzy_failed = false; + File* fuzzy_file = receive_delta_file(state->fd, config, state->check_path, fuzzy_basis, + fuzzy_size, &fuzzy_failed); + fuzzy_basis = NULL; /* receive_delta_file consumes the buffer on every path */ + if (fuzzy_file) { + *out_file = fuzzy_file; + return INCREMENTAL_FILE; + } + if (fuzzy_failed) + return INCREMENTAL_ERROR; + } + free(fuzzy_basis); + return INCREMENTAL_CONTINUE; +} + +/* Final fallback: tell the sender to transmit the whole file and receive it. */ +static File* incremental_check_receive_full(IncrementalCheckState* state) { + if (!send_status(state->fd, STATUS_NEXT)) + return NULL; + if (state->old_fd >= 0) { + close(state->old_fd); + state->old_fd = -1; + } + return receive_full_file(state->fd, state->config, state->check_path); +} + +/* Core implementation. `would_transfer` (may be NULL) is set true only on the + * server-contacting --dry-run path, when the file is not up to date and the + * receiver answered STATUS_DRY_RUN_TRANSFER; the caller then knows no File is + * returned and nothing was stored. */ +File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, + bool* would_transfer) { + if (would_transfer) + *would_transfer = false; + if (!config || !skipped) { + send_status(fd, STATUS_ERROR); + return NULL; + } + *skipped = false; + + IncrementalCheckState state; + incremental_check_state_init(&state, fd, config); + + File* result = NULL; + bool try_delta = false; + IncrementalCheckOutcome outcome; + + outcome = incremental_check_receive_request(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + + outcome = incremental_check_open_destination(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + + outcome = incremental_check_report_dest_info(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + + /* --ignore-existing must answer before any data is requested; it takes + precedence over the metadata up-to-date check below. */ + outcome = incremental_check_ignore_existing(&state); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_SKIP) { + *skipped = true; + goto done; + } + + /* A --link-dest hit relinks even an already up-to-date destination before the + quick-skip can suppress it (rsync parity). */ + outcome = incremental_check_link_dest_relink(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_quick_skip(&state, &try_delta); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_SKIP) { + *skipped = true; + goto done; + } + + /* Dry-run resolves here (no mutation) or falls through to the normal path. */ + outcome = incremental_check_dry_run_shortcut(&state, skipped, would_transfer); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome != INCREMENTAL_CONTINUE) + goto done; + + outcome = incremental_check_try_basis(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_SKIP) { + *skipped = true; + goto done; + } + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_try_append_resume(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_try_delta(&state, try_delta, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + outcome = incremental_check_try_fuzzy(&state, &result); + if (outcome == INCREMENTAL_ERROR) + goto done; + if (outcome == INCREMENTAL_FILE) + goto done; + + result = incremental_check_receive_full(&state); + +done: + incremental_check_state_cleanup(&state); + return result; +} + +File* receive_incremental_check(int fd, const Config* config, bool* skipped) { + return receive_incremental_check_ex(fd, config, skipped, NULL); +} diff --git a/src/shared/incremental_check.h b/src/shared/incremental_check.h new file mode 100644 index 0000000..862673d --- /dev/null +++ b/src/shared/incremental_check.h @@ -0,0 +1,41 @@ +#ifndef INCREMENTAL_CHECK_H +#define INCREMENTAL_CHECK_H + +#include "config.h" +#include "file_types.h" +#include "protocol.h" +#include + +/* Incremental-check module: the per-file STATUS_CHECK state machine, the + * incremental delta / alternate-basis / fuzzy matching helpers and the shared + * xattr receive helper. These declarations are re-exported by the + * file_receive.h facade. */ + +/* Whole-file payload bound shared by the plain receive path and the + * incremental check paths. */ +#define MAX_FILE_DATA_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE + +/* Receive a file's xattr block (when the config enables xattr transport) and + * attach it to `file`. Returns false on a malformed/oversized frame. */ +bool receive_file_xattrs(File* file, int fd, const Config* config); + +File* receive_incremental_check(int fd, const Config* config, bool* skipped); +/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set + * true only on the server-contacting --dry-run path when the file is not up to + * date: the receiver has already sent STATUS_DRY_RUN_TRANSFER and returns NULL + * without storing anything. On that path `*skipped` is true for an up-to-date + * (STATUS_OK) file and both flags are false for a genuine error. */ +File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped, + bool* would_transfer); + +/* Testable basis quick-check / verification policy. file_basis_quick_match is + * rsync's metadata quick-check for a basis candidate (equal size is required + * separately by the caller; this adds the --size-only / mtime / --modify-window + * leg). file_basis_content_required reports whether a hit must ALSO be + * confirmed by a whole-file content digest (--verify-basis; false is the + * default rsync-parity behavior). */ +bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime, + long check_mtime_nsec); +bool file_basis_content_required(const Config* config); + +#endif -- 2.54.0 From 707bb659e80e4fb5c2da16b8163086b4fa6ce88e Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:46:17 +0200 Subject: [PATCH 37/68] refactor(server): decompose handler into phases --- src/server/server.c | 455 ++++++++++++++++++++++++++------------------ 1 file changed, 269 insertions(+), 186 deletions(-) diff --git a/src/server/server.c b/src/server/server.c index e3a0a35..5978d71 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -705,69 +705,85 @@ static const char* server_module_gate(const Config* config, void* context) { return module_gate_install_root(config, module); } -void handler(int file_descriptor) { - SSL* ssl = io_get_ssl(); +/* Per-connection state threaded through the handler phase helpers below. The + * fields are a faithful split of the former handler() locals: the protocol + * session, the config-frame gate context, the accepted config, the optional + * multithreaded pipeline context and the teardown bookkeeping all live here so + * the single `done` epilogue in handler() can release them exactly as before. */ +typedef struct ServerSession { + int fd; + SSL* ssl; ProtocolSession session; - protocol_session_init(&session, file_descriptor, file_descriptor); - protocol_session_set_ssl(&session, ssl); - protocol_session_bind(&session); ModuleGateContext gate_ctx; - gate_ctx.ssl = ssl; - gate_ctx.fd = file_descriptor; - gate_ctx.super_mode_override = -1; - gate_ctx.has_peer_ip = false; - gate_ctx.peer_ip[0] = '\0'; - gate_ctx.is_local = false; - /* All teardown state starts empty so the single `done` epilogue is safe to - * reach from any error path (including before the config frame arrives). */ - Config* config = NULL; - PipelineContextReceiver* context = NULL; - char* joined_destination = NULL; - bool charset_ready = false; - config = config_receive_with_validate(file_descriptor, server_module_gate, &gate_ctx); - if (config == NULL) { + Config* config; + PipelineContextReceiver* context; + char* joined_destination; + bool charset_ready; +} ServerSession; + +/* Phase 1 -- config receipt + validation. Receives the client config frame + * through the module gate, applies the super-mode override the gate recorded + * exactly once, and installs the per-connection protocol/compression state. + * Returns false when the config frame was refused (the gate has already + * answered the client); the caller jumps to the shared `done` epilogue. */ +static bool server_accept_config(ServerSession* state) { + state->config = config_receive_with_validate(state->fd, server_module_gate, &state->gate_ctx); + if (state->config == NULL) { log_message(LOG_LEVEL_ERROR, "Failed to receive config"); - goto done; + return false; } /* Apply the super-mode veto the gate decided on (operator --no-super, or a * daemon module without the `client owner = yes` opt-in) exactly once, so * every downstream gate (identity_apply_ownership via privilege_super_permitted, * device-node creation) sees SUPER_MODE_OFF. The gate never mutated the * received config. */ - if (gate_ctx.super_mode_override != -1) - config->super_mode = (SuperMode)gate_ctx.super_mode_override; + if (state->gate_ctx.super_mode_override != -1) + state->config->super_mode = (SuperMode)state->gate_ctx.super_mode_override; /* Install the codec this connection negotiated before the receiver/writer * threads start (the server forks per connection, so the process-global * codec is private to this session). */ - compression_set_algo((CompressionAlgo)config->compression_algo); + compression_set_algo((CompressionAlgo)state->config->compression_algo); /* If the client requested ownership but the effective super mode forbids it * (operator --no-super, a privileged standalone receiver's secure default, or * a daemon module without `client owner = yes`), say so ONCE per connection so * a successful -a/-o/-g transfer is not mistaken for preserved ownership. */ - if (config->super_mode == SUPER_MODE_OFF && identity_ownership_requested(config)) + if (state->config->super_mode == SUPER_MODE_OFF && identity_ownership_requested(state->config)) log_message(LOG_LEVEL_WARNING, "requested ownership will NOT be applied: super-user activities are disabled " "for this connection (operator veto, or module without `client owner = yes`)"); - protocol_set_8_bit_output(config->eight_bit_output); + protocol_set_8_bit_output(state->config->eight_bit_output); /* Server-side per-message protocol deadline for every frame from here on. * `timeout` is not serialized, so this is the server's own config (the server * has no --timeout CLI and defaults it to 0). A client's --timeout tightens * only that client's own protocol I/O; the server floors its own deadline at * SERVER_IO_TIMEOUT_SEC so a silent peer can never hold a session slot * forever (the socket layer gets the same floor at startup). */ - protocol_session_set_io_timeout(&session, protocol_server_io_timeout_sec(config->timeout)); + protocol_session_set_io_timeout(&state->session, + protocol_server_io_timeout_sec(state->config->timeout)); + return true; +} + +/* Phase 2 -- security gates. The ORDER here is load-bearing and must not be + * merged or reordered: transport/authentication (plaintext refusal, TLS + * client-CN verification), then daemon-root confinement (absolute-destination + * rejection, traversal + within-authorized-root), then delete/force + * authorization -- exactly the sequence the former handler() used. Returns + * false after logging the matching rejection; the caller jumps to the shared + * `done` epilogue. */ +static bool server_apply_security_gates(ServerSession* state) { + Config* config = state->config; const char* authorized_root = utils_get_authorized_root_path(); if (!authorized_root) { log_message(LOG_LEVEL_ERROR, "No server-side destination root configured"); - goto done; + return false; } - if (!allow_unauthenticated && ssl == NULL) { + if (!allow_unauthenticated && state->ssl == NULL) { log_message(LOG_LEVEL_ERROR, "Rejected unauthenticated plaintext connection"); - goto done; + return false; } - if (ssl && required_client_cn && !tls_client_identity_allowed(ssl)) { + if (state->ssl && required_client_cn && !tls_client_identity_allowed(state->ssl)) { log_message(LOG_LEVEL_ERROR, "Rejected TLS client with unauthorized identity"); - goto done; + return false; } /* Daemon mode: the module's root is the authorized root (installed by server_module_gate), and the client's destination is a MODULE-RELATIVE @@ -777,27 +793,27 @@ void handler(int file_descriptor) { if (g_daemon_conf && config->receive_root_directory && config->receive_root_directory[0] == '/') { log_message(LOG_LEVEL_ERROR, "Rejected absolute daemon destination (must be relative to the " "selected module root)"); - goto done; + return false; } char* destination = config->receive_root_directory; if (destination && destination[0] != '/') - joined_destination = path_cat(authorized_root, destination); - if (joined_destination) - destination = joined_destination; + state->joined_destination = path_cat(authorized_root, destination); + if (state->joined_destination) + destination = state->joined_destination; if (!destination || has_path_traversal(destination) || !path_is_within_root(authorized_root, destination)) { log_message(LOG_LEVEL_ERROR, "Rejected destination outside authorized root"); - free(joined_destination); - joined_destination = NULL; - goto done; + free(state->joined_destination); + state->joined_destination = NULL; + return false; } - if (joined_destination) { + if (state->joined_destination) { free(config->receive_root_directory); - config->receive_root_directory = joined_destination; - joined_destination = NULL; + config->receive_root_directory = state->joined_destination; + state->joined_destination = NULL; } if (!config->receive_root_directory) { - goto done; + return false; } config->use_delete = config->use_delete && allow_delete; /* --force (receiver-side) is deletion authority too: it lets an incoming @@ -807,6 +823,18 @@ void handler(int file_descriptor) { * --delete-missing-args, so a client cannot use --force to bypass the delete * policy. */ config->force_delete = config->force_delete && allow_delete; + return true; +} + +/* Phase 3 -- session preparation. Installs the negotiated conversion, applies + * the remaining deletion policy, materializes the destination root (--mkpath), + * creates the --delay-updates staging tree, snapshots the identity policy, and + * publishes the --keep-dirlinks/--trust-sender globals and the daemon MOTD. + * All of it must happen before any receiver/writer thread is spawned. Returns + * false after logging the matching failure; the caller jumps to the shared + * `done` epilogue. */ +static bool server_prepare_session(ServerSession* state) { + Config* config = state->config; /* --iconv (protocol 2.16.0): install the receiver-side wire->local conversion now that the client's full CONVERT_SPEC has been received and validated, before any received file name is decoded. The server's own --iconv (if @@ -817,14 +845,14 @@ void handler(int file_descriptor) { if (!charset_wire_init_receiver(config->iconv_spec, server_iconv_spec)) { log_message(LOG_LEVEL_ERROR, "--iconv: unsupported charset conversion requested (LOCAL[,REMOTE])"); - goto done; + return false; } - charset_ready = true; + state->charset_ready = true; } /* --delete-missing-args deletes destination mirrors receiver-side, so it is - deletion and stays gated by the same --allow-delete server policy. When - the server policy is off the flag is inert (the missing entries are still - skipped via its implied --ignore-missing-args, but nothing is deleted). */ + * deletion and stays gated by the same --allow-delete server policy. When + * the server policy is off the flag is inert (the missing entries are still + * skipped via its implied --ignore-missing-args, but nothing is deleted). */ config->delete_missing_args = config->delete_missing_args && allow_delete; /* --mkpath: create the destination root (and its missing leading components) * before anything else; without it the root must pre-exist. The precondition @@ -839,7 +867,7 @@ void handler(int file_descriptor) { log_message(LOG_LEVEL_ERROR, "destination root is not available: %s", escaped_root ? escaped_root : ""); free(escaped_root); - goto done; + return false; } /* A --delay-updates transfer stages under a private 0700 directory inside the receive root. Create it up front (wiping leftovers of any previously @@ -849,7 +877,7 @@ void handler(int file_descriptor) { config->delay_context = delay_updates_context_create(config->receive_root_directory); if (!config->delay_context || !delay_updates_prepare(config->delay_context)) { log_message(LOG_LEVEL_ERROR, "Failed to initialize --delay-updates staging area"); - goto done; + return false; } } /* Preserve the negotiated identity policy for the fd-relative ownership @@ -859,7 +887,7 @@ void handler(int file_descriptor) { rather than silently applying the wrong ownership policy. */ if (!identity_set_active(config)) { log_message(LOG_LEVEL_ERROR, "Failed to activate identity policy"); - goto done; + return false; } /* Persist the negotiated --keep-dirlinks policy once, here at config-accept, before any multithreaded receiver/writer threads are spawned, so the @@ -887,140 +915,195 @@ void handler(int file_descriptor) { Wave C note in config.h). */ if (g_daemon_conf) { char* motd = motd_read_file(g_daemon_conf->global.motd_file); - if (!motd_send(file_descriptor, motd ? motd : "")) { + if (!motd_send(state->fd, motd ? motd : "")) { free(motd); log_message(LOG_LEVEL_ERROR, "Failed to send daemon MOTD"); - goto done; + return false; } free(motd); } - if (config->use_multithreading) { - Queue* q = queue_create(100, file_destroy); - if (q == NULL) - goto done; - context = pipeline_context_receiver_create(config, q, file_descriptor, ssl); - if (context == NULL) { - queue_destroy(q); - goto done; - } - protocol_session_set_max_alloc(&context->session, config->max_alloc); - protocol_session_set_io_timeout(&context->session, - protocol_server_io_timeout_sec(config->timeout)); - atomic_store(&context->session.total_allocated_bytes, - atomic_load(&session.total_allocated_bytes)); - pipeline_context_receiver_set_queue_byte_limit(context, RECEIVER_QUEUE_MAX_BYTES); - thrd_t receiver = {0}; - thrd_t writer = {0}; - bool receiver_created = thrd_create(&receiver, receive_thread, context) == thrd_success; - bool writer_created = false; - if (receiver_created) - writer_created = thrd_create(&writer, write_thread, context) == thrd_success; - if (!receiver_created || !writer_created) { - log_perror("Error creating Threads"); - if (receiver_created) { - mtx_lock(&context->mutex); - atomic_store(&context->cancelled, true); - cnd_broadcast(&context->condition_not_full); - cnd_broadcast(&context->condition_not_empty); - mtx_unlock(&context->mutex); - /* Unblock a worker parked in socket I/O without closing the fd (the - * child owns the single close). shutdown() only affects sockets; for - * the --stdio pipe the receiver's per-message poll timeout still - * bounds the join, so do nothing there rather than close a descriptor - * another thread may still be using. */ - struct stat fd_stat; - if (fstat(file_descriptor, &fd_stat) == 0 && S_ISSOCK(fd_stat.st_mode)) - shutdown(file_descriptor, SHUT_RDWR); - thrd_join(receiver, NULL); - } - if (writer_created) - thrd_join(writer, NULL); - goto done; - } - int receiver_result; - int writer_result; - thrd_join(receiver, &receiver_result); - thrd_join(writer, &writer_result); - bool transfer_ok = receiver_result == thrd_success && writer_result == thrd_success; - if (transfer_ok && !config->dry_run) { - /* Commit-style (late) deletion: receive_thread handed the keep-set - manifest here instead of deleting while write_thread might still be - draining, so by now every file is on disk and the whole transfer is - known to have succeeded. Remove the extras before publishing a - --delay-updates run; the walker skips the staging directory. A - server-contacting --dry-run deletes nothing (no manifest is sent). */ - if (context->deferred_manifest) { - size_t deleted = 0; - DeletePathObserver observer = config->report_deletes ? receiver_record_deleted_path : NULL; - DeleteCommitResult deletion = manifest_delete_all_observed( - config, context->deferred_manifest, &deleted, observer, (void*)context->deleted_paths); - context->stats.deleted_files += deleted; - if (deletion == DELETE_COMMIT_ERROR) { - transfer_ok = false; - } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { - /* The transfer still succeeds; the terminal frame reports the capped - deletion so the sender exits 25 like rsync. */ - context->delete_limit_reached = true; - } - delete_manifest_free(context->deferred_manifest); - context->deferred_manifest = NULL; - } - /* --delete-delay: receive_thread snapshotted each plan's extras as it - arrived; with the disk writer drained, commit the deferred removals. - --delete-during already applied its plans on the receive thread. */ - if (context->deferred_plans) { - /* Defence in depth (the enclosing block already excludes dry-run): a - -n run never commits a deletion. */ - if (config->report_deletes) - delete_plan_session_set_delete_observer( - context->deferred_plans, receiver_record_deleted_path, (void*)context->deleted_paths); - DeleteCommitResult deletion = - config->dry_run ? DELETE_COMMIT_OK - : delete_plan_session_commit(context->deferred_plans, config); - context->stats.deleted_files += delete_plan_session_deleted(context->deferred_plans); - if (deletion == DELETE_COMMIT_ERROR) { - transfer_ok = false; - } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { - context->delete_limit_reached = true; - } - delete_plan_session_destroy(context->deferred_plans); - context->deferred_plans = NULL; - } - } - if (transfer_ok && !config->dry_run) { - /* --delay-updates: receive_thread has finished the whole protocol stream - (including manifest/delete handling) and write_thread has drained its - queue, so every staged file is complete. Publish atomically before the - success/outcome frame so a --remove-source-files sender only learns of - files that were actually installed. */ - if (config->delay_updates && config->delay_context && - !delay_updates_publish(config->delay_context, config)) { - transfer_ok = false; - } - /* P7 Wave D: all writers have joined and the late deletion (and - --delay-updates publication) has committed above, so it is finally safe - to stamp directory times; a directory's mtime must not be clobbered by - its children or by an extra removal. */ - if (transfer_ok) - dir_metadata_list_apply(&context->dir_times, config->receive_root_directory, config); - } - if (transfer_ok) { - Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK; - /* Emit the optional wire-stats record first (protocol 2.25.0), then the - success/outcome frame, exactly like the single-threaded receiver. */ - if (!receiver_send_stats_frame(file_descriptor, config, &context->stats, - context->would_delete, context->deleted_paths) || - !receiver_send_final_success(file_descriptor, config, &context->outcomes, final_status)) - transfer_ok = false; - } else { - send_error_detail(file_descriptor, "transfer failed on receiver"); - } - if (!transfer_ok) - log_message(LOG_LEVEL_ERROR, "Transfer failed"); - } else { - if (receiver_receive_files(config, file_descriptor) != 0) - log_message(LOG_LEVEL_ERROR, "Transfer failed"); + return true; +} + +/* Phase 4a -- transfer via the multithreaded receiver. Spawns the receive/write + * thread pair, joins them, then commits the late deletion, --delay-updates + * publication and directory times before emitting the terminal stats/success + * frame. On any failure the helper just returns; the caller's `done` epilogue + * releases the pipeline context (which owns the config and queue) exactly as the + * former inline code did. */ +static void server_run_mt_receiver(ServerSession* state) { + Config* config = state->config; + Queue* q = queue_create(100, file_destroy); + if (q == NULL) + return; + state->context = pipeline_context_receiver_create(config, q, state->fd, state->ssl); + if (state->context == NULL) { + queue_destroy(q); + return; } + protocol_session_set_max_alloc(&state->context->session, config->max_alloc); + protocol_session_set_io_timeout(&state->context->session, + protocol_server_io_timeout_sec(config->timeout)); + atomic_store(&state->context->session.total_allocated_bytes, + atomic_load(&state->session.total_allocated_bytes)); + pipeline_context_receiver_set_queue_byte_limit(state->context, RECEIVER_QUEUE_MAX_BYTES); + thrd_t receiver = {0}; + thrd_t writer = {0}; + bool receiver_created = thrd_create(&receiver, receive_thread, state->context) == thrd_success; + bool writer_created = false; + if (receiver_created) + writer_created = thrd_create(&writer, write_thread, state->context) == thrd_success; + if (!receiver_created || !writer_created) { + log_perror("Error creating Threads"); + if (receiver_created) { + mtx_lock(&state->context->mutex); + atomic_store(&state->context->cancelled, true); + cnd_broadcast(&state->context->condition_not_full); + cnd_broadcast(&state->context->condition_not_empty); + mtx_unlock(&state->context->mutex); + /* Unblock a worker parked in socket I/O without closing the fd (the + * child owns the single close). shutdown() only affects sockets; for + * the --stdio pipe the receiver's per-message poll timeout still + * bounds the join, so do nothing there rather than close a descriptor + * another thread may still be using. */ + struct stat fd_stat; + if (fstat(state->fd, &fd_stat) == 0 && S_ISSOCK(fd_stat.st_mode)) + shutdown(state->fd, SHUT_RDWR); + thrd_join(receiver, NULL); + } + if (writer_created) + thrd_join(writer, NULL); + return; + } + int receiver_result; + int writer_result; + thrd_join(receiver, &receiver_result); + thrd_join(writer, &writer_result); + bool transfer_ok = receiver_result == thrd_success && writer_result == thrd_success; + PipelineContextReceiver* context = state->context; + if (transfer_ok && !config->dry_run) { + /* Commit-style (late) deletion: receive_thread handed the keep-set + manifest here instead of deleting while write_thread might still be + draining, so by now every file is on disk and the whole transfer is + known to have succeeded. Remove the extras before publishing a + --delay-updates run; the walker skips the staging directory. A + server-contacting --dry-run deletes nothing (no manifest is sent). */ + if (context->deferred_manifest) { + size_t deleted = 0; + DeletePathObserver observer = config->report_deletes ? receiver_record_deleted_path : NULL; + DeleteCommitResult deletion = manifest_delete_all_observed( + config, context->deferred_manifest, &deleted, observer, (void*)context->deleted_paths); + context->stats.deleted_files += deleted; + if (deletion == DELETE_COMMIT_ERROR) { + transfer_ok = false; + } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { + /* The transfer still succeeds; the terminal frame reports the capped + deletion so the sender exits 25 like rsync. */ + context->delete_limit_reached = true; + } + delete_manifest_free(context->deferred_manifest); + context->deferred_manifest = NULL; + } + /* --delete-delay: receive_thread snapshotted each plan's extras as it + arrived; with the disk writer drained, commit the deferred removals. + --delete-during already applied its plans on the receive thread. */ + if (context->deferred_plans) { + /* Defence in depth (the enclosing block already excludes dry-run): a + -n run never commits a deletion. */ + if (config->report_deletes) + delete_plan_session_set_delete_observer( + context->deferred_plans, receiver_record_deleted_path, (void*)context->deleted_paths); + DeleteCommitResult deletion = + config->dry_run ? DELETE_COMMIT_OK + : delete_plan_session_commit(context->deferred_plans, config); + context->stats.deleted_files += delete_plan_session_deleted(context->deferred_plans); + if (deletion == DELETE_COMMIT_ERROR) { + transfer_ok = false; + } else if (deletion == DELETE_COMMIT_LIMIT_REACHED) { + context->delete_limit_reached = true; + } + delete_plan_session_destroy(context->deferred_plans); + context->deferred_plans = NULL; + } + } + if (transfer_ok && !config->dry_run) { + /* --delay-updates: receive_thread has finished the whole protocol stream + (including manifest/delete handling) and write_thread has drained its + queue, so every staged file is complete. Publish atomically before the + success/outcome frame so a --remove-source-files sender only learns of + files that were actually installed. */ + if (config->delay_updates && config->delay_context && + !delay_updates_publish(config->delay_context, config)) { + transfer_ok = false; + } + /* P7 Wave D: all writers have joined and the late deletion (and + --delay-updates publication) has committed above, so it is finally safe + to stamp directory times; a directory's mtime must not be clobbered by + its children or by an extra removal. */ + if (transfer_ok) + dir_metadata_list_apply(&context->dir_times, config->receive_root_directory, config); + } + if (transfer_ok) { + Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK; + /* Emit the optional wire-stats record first (protocol 2.25.0), then the + success/outcome frame, exactly like the single-threaded receiver. */ + if (!receiver_send_stats_frame(state->fd, config, &context->stats, context->would_delete, + context->deleted_paths) || + !receiver_send_final_success(state->fd, config, &context->outcomes, final_status)) + transfer_ok = false; + } else { + send_error_detail(state->fd, "transfer failed on receiver"); + } + if (!transfer_ok) + log_message(LOG_LEVEL_ERROR, "Transfer failed"); +} + +/* Phase 4b -- transfer via the single-threaded receiver. Failure is logged + * exactly as before; the caller's `done` epilogue then releases the config. */ +static void server_run_st_receiver(ServerSession* state) { + if (receiver_receive_files(state->config, state->fd) != 0) + log_message(LOG_LEVEL_ERROR, "Transfer failed"); +} + +/* Phase 4 dispatch -- choose the receiver implementation the config asks for. + * Both helpers own their success/failure logging; the caller falls through to + * the shared `done` epilogue either way. */ +static void server_run_transfer(ServerSession* state) { + if (state->config->use_multithreading) + server_run_mt_receiver(state); + else + server_run_st_receiver(state); +} + +void handler(int file_descriptor) { + /* Single per-connection state; every phase helper below advances it and + * returns false on a logged failure. All teardown state starts empty so the + * single `done` epilogue is safe to reach from any error path (including + * before the config frame arrives). */ + ServerSession state; + state.fd = file_descriptor; + state.ssl = io_get_ssl(); + protocol_session_init(&state.session, file_descriptor, file_descriptor); + protocol_session_set_ssl(&state.session, state.ssl); + protocol_session_bind(&state.session); + state.gate_ctx.ssl = state.ssl; + state.gate_ctx.fd = file_descriptor; + state.gate_ctx.super_mode_override = -1; + state.gate_ctx.has_peer_ip = false; + state.gate_ctx.peer_ip[0] = '\0'; + state.gate_ctx.is_local = false; + state.config = NULL; + state.context = NULL; + state.joined_destination = NULL; + state.charset_ready = false; + + if (!server_accept_config(&state)) + goto done; + if (!server_apply_security_gates(&state)) + goto done; + if (!server_prepare_session(&state)) + goto done; + server_run_transfer(&state); done: /* Single cleanup epilogue: every error path jumps here, so the iconv @@ -1029,22 +1112,22 @@ done: * connection fd is deliberately NOT closed here -- the child functions own * its single close (plain_child_fn / tls_child_fn), and the --stdio call * site must leave stdin/stdout open. */ - if (charset_ready) + if (state.charset_ready) charset_wire_free(); /* The delay-updates staging tree is released by config_delete (which the branch below always reaches), so it is cleaned exactly once. */ identity_clear_active(); protocol_session_unbind(); - if (context != NULL) { + if (state.context != NULL) { /* context owns both the config and the queue it was created with. */ - pipeline_context_receiver_destroy(context); - context = NULL; - config = NULL; + pipeline_context_receiver_destroy(state.context); + state.context = NULL; + state.config = NULL; } else { - config_delete(config); - config = NULL; + config_delete(state.config); + state.config = NULL; } - free(joined_destination); + free(state.joined_destination); } #ifndef FASTSYNC_SERVER_AS_LIB -- 2.54.0 From ade9be86007720e8efbe7b8c20c8519cea1c342a Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:49:28 +0200 Subject: [PATCH 38/68] refactor(receiver): per-status dispatch and shared pending teardown --- src/server/receiver.c | 495 ++++++++++++++++++++++++++---------------- 1 file changed, 302 insertions(+), 193 deletions(-) diff --git a/src/server/receiver.c b/src/server/receiver.c index 178ca34..10c4c45 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -303,6 +303,264 @@ int receiver_process(Config* config, int file_descriptor, const ReceiverSink* si return receiver_process_pending(config, file_descriptor, sink, NULL, NULL); } +/* Per-connection state threaded through the status handlers below. The parked + keep-set / per-directory session live here so one teardown helper can release + them on every exit path. */ +typedef struct { + Config* config; + int fd; + const ReceiverSink* sink; + DeleteManifest** pending_manifest; + DeletePlanSession** pending_plans; + /* Parked keep-set for the late/commit timing. Every exit path frees it + exactly once; the only exception is the successful FINISHED handoff, which + transfers ownership to *pending_manifest (used by the -m receiver). */ + DeleteManifest* deferred_manifest; + /* Per-directory delete session for --delete-during/--delete-delay. During the + loop it applies plans inline (during) or snapshots their extras (delay); on + a successful FINISHED it is either committed here or handed to + *pending_plans so the -m caller commits after its disk writer drained. */ + DeletePlanSession* plan_session; + bool early_delete; + bool per_dir_delete; + bool delete_limit_noted; +} ReceiverPendingState; + +/* Outcome of one frame handler. NEXT reads the following status frame; FAIL + tears the connection down without a peer STATUS_ERROR; ERROR tears it down + and (when the sink owns error reporting) emits STATUS_ERROR. */ +typedef enum { + RECEIVER_STEP_NEXT, + RECEIVER_STEP_FAIL, + RECEIVER_STEP_ERROR, +} ReceiverStep; + +static ReceiverStep receiver_handle_keepalive(ReceiverPendingState* state) { + if (!send_status(state->fd, STATUS_KEEPALIVE)) + return RECEIVER_STEP_FAIL; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_abort(ReceiverPendingState* state) { + (void)state; + log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); + return RECEIVER_STEP_FAIL; +} + +static ReceiverStep receiver_handle_check(ReceiverPendingState* state) { + bool skipped = false; + bool would_transfer = false; + File* file = receive_incremental_check_ex(state->fd, state->config, &skipped, &would_transfer); + if (state->config->dry_run) { + /* Server-contacting --dry-run: the reply has already been sent + (STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and + nothing may be stored. Both flags false means a genuine protocol + error (STATUS_ERROR already sent or sent by receive_error below). */ + if (!skipped && !would_transfer) + return RECEIVER_STEP_ERROR; + } else if (!skipped && (!file || !state->sink->store_file(file, state->sink->context))) { + return RECEIVER_STEP_ERROR; + } + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_chunk(ReceiverPendingState* state) { + Chunk* chunk = receive_chunk_data(state->fd, state->config); + if (!chunk || !receiver_process_chunk(chunk, state->sink)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_check_batch(ReceiverPendingState* state) { + if (!receiver_process_batch(state->config, state->fd)) + return RECEIVER_STEP_FAIL; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_mkdir(ReceiverPendingState* state) { + File* dir = file_receive_directory(state->fd, state->config); + if (!dir || !state->sink->store_file(dir, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_dir_times(ReceiverPendingState* state) { + if (!receiver_process_dir_times(state->fd, state->config, state->sink)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_hardlink(ReceiverPendingState* state) { + File* file = file_receive_hardlink(state->fd); + if (!file || !state->sink->store_file(file, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_symlink(ReceiverPendingState* state) { + File* sym = file_receive_symlink(state->fd, state->config); + if (!sym || !state->sink->store_file(sym, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_special(ReceiverPendingState* state) { + File* file = file_receive_special(state->fd); + if (!file || !state->sink->store_file(file, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_manifest(ReceiverPendingState* state) { + Config* config = state->config; + int fd = state->fd; + const ReceiverSink* sink = state->sink; + DeleteManifest* manifest = receive_manifest_entries(fd); + if (!manifest) + return RECEIVER_STEP_FAIL; /* receive_manifest_entries already sent STATUS_ERROR */ + if (config->dry_run) { + /* Server-contacting --dry-run mutates nothing, so a keep-set manifest + is consumed and discarded. The early-delete mode still needs its ACK + so a sender blocked on the delete handshake is not left hanging. + When would-delete reporting is armed, enumerate (read-only) the + destination extras so the terminal STATUS_STATS frame can list them. */ + if (config->use_delete && sink->would_delete) { + size_t count = 0; + if (!manifest_would_delete_list(config, manifest, sink->would_delete, &count)) + log_message(LOG_LEVEL_WARNING, "dry-run: could not enumerate would-delete paths"); + } + delete_manifest_free(manifest); + if (state->early_delete && !send_status(fd, STATUS_OK)) + return RECEIVER_STEP_FAIL; + return RECEIVER_STEP_NEXT; + } + if (state->early_delete) { + /* --delete-before: the whole-tree manifest is authoritative the moment + it arrives, before any file data. Delete now and acknowledge so the + sender only starts streaming once the deletion committed (or failed). + A later transfer failure does not restore these deletions. A + --max-delete-capped commit still succeeds and the transfer proceeds; + the terminal success frame reports the cap. */ + size_t deleted = 0; + DeletePathObserver observer = + (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL; + DeleteCommitResult deletion = + (config->use_delete || config->delete_missing_args) + ? manifest_delete_all_observed(config, manifest, &deleted, observer, + (void*)sink->deleted_paths) + : DELETE_COMMIT_OK; + receiver_tally_deleted(sink, deleted); + delete_manifest_free(manifest); + if (deletion == DELETE_COMMIT_ERROR) { + send_status(fd, STATUS_ERROR); + return RECEIVER_STEP_FAIL; + } + if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit) + sink->note_delete_limit(sink->context); + if (!send_status(fd, STATUS_OK)) + return RECEIVER_STEP_FAIL; + } else if (config->use_delete || config->delete_missing_args) { + /* Plain --delete / --delete-after and the --delete-missing-args + exact-path deletions: hold the manifest and commit it only after + STATUS_FINISHED. The per-directory modes never send this frame. */ + if (state->deferred_manifest) { + log_message(LOG_LEVEL_ERROR, "Received a second delete manifest"); + delete_manifest_free(state->deferred_manifest); + state->deferred_manifest = NULL; + delete_manifest_free(manifest); + send_status(fd, STATUS_ERROR); + return RECEIVER_STEP_FAIL; + } + state->deferred_manifest = manifest; + } else { + delete_manifest_free(manifest); + } + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_delete_plan(ReceiverPendingState* state) { + Config* config = state->config; + int fd = state->fd; + const ReceiverSink* sink = state->sink; + if (!state->per_dir_delete) { + log_message(LOG_LEVEL_ERROR, "Received a per-directory delete plan without a per-dir " + "delete timing"); + send_status(fd, STATUS_ERROR); + return RECEIVER_STEP_FAIL; + } + if (!state->plan_session) { + state->plan_session = delete_plan_session_create(config); + if (state->plan_session && config->report_deletes && sink->deleted_paths) + delete_plan_session_set_delete_observer(state->plan_session, receiver_record_deleted_path, + (void*)sink->deleted_paths); + } + if (!state->plan_session || delete_plan_session_receive(state->plan_session, config, fd) != 0) + return RECEIVER_STEP_FAIL; + if (delete_plan_session_limit_reached(state->plan_session) && !state->delete_limit_noted && + sink->note_delete_limit) { + sink->note_delete_limit(sink->context); + state->delete_limit_noted = true; + } + return RECEIVER_STEP_NEXT; +} + +static ReceiverStep receiver_handle_file(ReceiverPendingState* state) { + File* file = file_receive(state->config, state->fd); + if (!file) { + log_message(LOG_LEVEL_ERROR, "Failed to receive file"); + return RECEIVER_STEP_ERROR; + } + if (!state->sink->store_file(file, state->sink->context)) + return RECEIVER_STEP_ERROR; + return RECEIVER_STEP_NEXT; +} + +/* One dispatch per admitted frame type; STATUS_NEXT (and any other + data-bearing status) falls through to the regular file receiver. */ +static ReceiverStep receiver_dispatch_status(ReceiverPendingState* state, Status status) { + switch (status) { + case STATUS_KEEPALIVE: + return receiver_handle_keepalive(state); + case STATUS_ABORT: + return receiver_handle_abort(state); + case STATUS_CHECK: + return receiver_handle_check(state); + case STATUS_CHUNK: + return receiver_handle_chunk(state); + case STATUS_CHECK_BATCH: + return receiver_handle_check_batch(state); + case STATUS_MKDIR: + return receiver_handle_mkdir(state); + case STATUS_DIR_TIMES: + return receiver_handle_dir_times(state); + case STATUS_HARDLINK: + return receiver_handle_hardlink(state); + case STATUS_SYMLINK: + return receiver_handle_symlink(state); + case STATUS_SPECIAL: + return receiver_handle_special(state); + case STATUS_MANIFEST: + return receiver_handle_manifest(state); + case STATUS_DELETE_PLAN: + return receiver_handle_delete_plan(state); + default: + return receiver_handle_file(state); + } +} + +/* Release the parked keep-set / per-directory session exactly once on every + failure exit. Never commit a deletion for a failed stream. */ +static void receiver_drop_pending(ReceiverPendingState* state) { + if (state->deferred_manifest) { + delete_manifest_free(state->deferred_manifest); + state->deferred_manifest = NULL; + } + if (state->plan_session) { + delete_plan_session_destroy(state->plan_session); + state->plan_session = NULL; + } +} + /* Runs the whole receive loop. The delete manifest may legitimately arrive either FIRST (--delete-before / --delete-during: the sender transmits the validated keep-set before any file data) or LAST (--delete-after / @@ -312,9 +570,9 @@ int receiver_process(Config* config, int file_descriptor, const ReceiverSink* si deletion has committed (or failed); in the late modes the manifest is held and the deletion is committed only after the terminal STATUS_FINISHED proves the whole transfer succeeded. A plain --delete defaults to the per-directory - delete-during plan mode (no manifest at all). See - receiver_process_pending() for how the -m receiver defers that commit until - its disk writer has drained. */ + delete-during plan mode (no manifest at all). See the per-frame handlers + above for how the -m receiver defers that commit until its disk writer has + drained. */ int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink, DeleteManifest** pending_manifest, DeletePlanSession** pending_plans) { Status status; @@ -329,166 +587,29 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver last_progress = session_start; if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink)) return -1; - bool early_delete = config_delete_timing_early(config); - bool per_dir_delete = config_delete_timing_per_dir(config); - /* Parked keep-set for the late/commit timing. Every exit path below frees it - exactly once; the only exception is the successful FINISHED handoff, which - transfers ownership to *pending_manifest (used by the -m receiver). */ - DeleteManifest* deferred_manifest = NULL; - /* Per-directory delete session for --delete-during/--delete-delay. During the - loop it applies plans inline (during) or snapshots their extras (delay); on - a successful FINISHED it is either committed here or handed to - *pending_plans so the -m caller commits after its disk writer drained. */ - DeletePlanSession* plan_session = NULL; - bool delete_limit_noted = false; + ReceiverPendingState state = { + .config = config, + .fd = file_descriptor, + .sink = sink, + .pending_manifest = pending_manifest, + .pending_plans = pending_plans, + .deferred_manifest = NULL, + .plan_session = NULL, + .early_delete = config_delete_timing_early(config), + .per_dir_delete = config_delete_timing_per_dir(config), + .delete_limit_noted = false, + }; + bool notify_peer = false; while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK || status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH || status == STATUS_MKDIR || status == STATUS_MANIFEST || status == STATUS_HARDLINK || status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES || status == STATUS_DELETE_PLAN) { - if (status == STATUS_KEEPALIVE) { - if (!send_status(file_descriptor, STATUS_KEEPALIVE)) - goto fail; - goto next_status; - } - if (status == STATUS_ABORT) { - log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); + ReceiverStep step = receiver_dispatch_status(&state, status); + if (step == RECEIVER_STEP_FAIL) goto fail; - } - if (status == STATUS_CHECK) { - bool skipped = false; - bool would_transfer = false; - File* file = receive_incremental_check_ex(file_descriptor, config, &skipped, &would_transfer); - if (config->dry_run) { - /* Server-contacting --dry-run: the reply has already been sent - (STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and - nothing may be stored. Both flags false means a genuine protocol - error (STATUS_ERROR already sent or sent by receive_error below). */ - if (!skipped && !would_transfer) - goto receive_error; - } else if (!skipped && (!file || !sink->store_file(file, sink->context))) { - goto receive_error; - } - } else if (status == STATUS_CHUNK) { - Chunk* chunk = receive_chunk_data(file_descriptor, config); - if (!chunk || !receiver_process_chunk(chunk, sink)) - goto receive_error; - } else if (status == STATUS_CHECK_BATCH) { - if (!receiver_process_batch(config, file_descriptor)) - goto fail; - goto next_status; - } else if (status == STATUS_MKDIR) { - File* dir = file_receive_directory(file_descriptor, config); - if (!dir || !sink->store_file(dir, sink->context)) - goto receive_error; - } else if (status == STATUS_DIR_TIMES) { - if (!receiver_process_dir_times(file_descriptor, config, sink)) - goto receive_error; - } else if (status == STATUS_HARDLINK) { - File* file = file_receive_hardlink(file_descriptor); - if (!file || !sink->store_file(file, sink->context)) - goto receive_error; - } else if (status == STATUS_SYMLINK) { - File* sym = file_receive_symlink(file_descriptor, config); - if (!sym || !sink->store_file(sym, sink->context)) - goto receive_error; - } else if (status == STATUS_SPECIAL) { - File* file = file_receive_special(file_descriptor); - if (!file || !sink->store_file(file, sink->context)) - goto receive_error; - } else if (status == STATUS_MANIFEST) { - DeleteManifest* manifest = receive_manifest_entries(file_descriptor); - if (!manifest) - goto fail; /* receive_manifest_entries already sent STATUS_ERROR */ - if (config->dry_run) { - /* Server-contacting --dry-run mutates nothing, so a keep-set manifest - is consumed and discarded. The early-delete mode still needs its ACK - so a sender blocked on the delete handshake is not left hanging. - When would-delete reporting is armed, enumerate (read-only) the - destination extras so the terminal STATUS_STATS frame can list them. */ - if (config->use_delete && sink->would_delete) { - size_t count = 0; - if (!manifest_would_delete_list(config, manifest, sink->would_delete, &count)) - log_message(LOG_LEVEL_WARNING, "dry-run: could not enumerate would-delete paths"); - } - delete_manifest_free(manifest); - if (early_delete && !send_status(file_descriptor, STATUS_OK)) - goto fail; - goto next_status; - } - if (early_delete) { - /* --delete-before: the whole-tree manifest is authoritative the moment - it arrives, before any file data. Delete now and acknowledge so the - sender only starts streaming once the deletion committed (or failed). - A later transfer failure does not restore these deletions. A - --max-delete-capped commit still succeeds and the transfer proceeds; - the terminal success frame reports the cap. */ - size_t deleted = 0; - DeletePathObserver observer = - (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL; - DeleteCommitResult deletion = - (config->use_delete || config->delete_missing_args) - ? manifest_delete_all_observed(config, manifest, &deleted, observer, - (void*)sink->deleted_paths) - : DELETE_COMMIT_OK; - receiver_tally_deleted(sink, deleted); - delete_manifest_free(manifest); - if (deletion == DELETE_COMMIT_ERROR) { - send_status(file_descriptor, STATUS_ERROR); - goto fail; - } - if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit) - sink->note_delete_limit(sink->context); - if (!send_status(file_descriptor, STATUS_OK)) - goto fail; - } else if (config->use_delete || config->delete_missing_args) { - /* Plain --delete / --delete-after and the --delete-missing-args - exact-path deletions: hold the manifest and commit it only after - STATUS_FINISHED. The per-directory modes never send this frame. */ - if (deferred_manifest) { - log_message(LOG_LEVEL_ERROR, "Received a second delete manifest"); - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; - delete_manifest_free(manifest); - send_status(file_descriptor, STATUS_ERROR); - goto fail; - } - deferred_manifest = manifest; - } else { - delete_manifest_free(manifest); - } - goto next_status; - } else if (status == STATUS_DELETE_PLAN) { - if (!per_dir_delete) { - log_message(LOG_LEVEL_ERROR, "Received a per-directory delete plan without a per-dir " - "delete timing"); - send_status(file_descriptor, STATUS_ERROR); - goto fail; - } - if (!plan_session) { - plan_session = delete_plan_session_create(config); - if (plan_session && config->report_deletes && sink->deleted_paths) - delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path, - (void*)sink->deleted_paths); - } - if (!plan_session || delete_plan_session_receive(plan_session, config, file_descriptor) != 0) - goto fail; - if (delete_plan_session_limit_reached(plan_session) && !delete_limit_noted && - sink->note_delete_limit) { - sink->note_delete_limit(sink->context); - delete_limit_noted = true; - } - goto next_status; - } else { - File* file = file_receive(config, file_descriptor); - if (!file) { - log_message(LOG_LEVEL_ERROR, "Failed to receive file"); - goto receive_error; - } - if (!sink->store_file(file, sink->context)) - goto receive_error; - } - next_status: + if (step == RECEIVER_STEP_ERROR) + goto receive_error; if (!receive_status(file_descriptor, &status)) goto receive_error; if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink)) @@ -507,19 +628,19 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver disk writer may still be draining; the caller commits after the writer has joined so no extra file is removed unless the transfer is known to have succeeded. */ - if (deferred_manifest) { - if (pending_manifest) { - *pending_manifest = deferred_manifest; - deferred_manifest = NULL; + if (state.deferred_manifest) { + if (state.pending_manifest) { + *state.pending_manifest = state.deferred_manifest; + state.deferred_manifest = NULL; } else { size_t deleted = 0; DeletePathObserver observer = (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL; DeleteCommitResult deletion = manifest_delete_all_observed( - config, deferred_manifest, &deleted, observer, (void*)sink->deleted_paths); + config, state.deferred_manifest, &deleted, observer, (void*)sink->deleted_paths); receiver_tally_deleted(sink, deleted); - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; + delete_manifest_free(state.deferred_manifest); + state.deferred_manifest = NULL; if (deletion == DELETE_COMMIT_ERROR) { send_status(file_descriptor, STATUS_ERROR); goto fail; @@ -533,28 +654,28 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver nothing yet and applies its decompressed snapshot here. The -m receiver hands the session to its caller instead, which commits after the disk writer drained. */ - if (plan_session) { + if (state.plan_session) { if (config->report_deletes && sink->deleted_paths) - delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path, + delete_plan_session_set_delete_observer(state.plan_session, receiver_record_deleted_path, (void*)sink->deleted_paths); - if (pending_plans) { - *pending_plans = plan_session; - plan_session = NULL; + if (state.pending_plans) { + *state.pending_plans = state.plan_session; + state.plan_session = NULL; } else if (config->dry_run) { /* Central dry-run no-op: never commit a deletion for a -n run. */ - delete_plan_session_destroy(plan_session); - plan_session = NULL; + delete_plan_session_destroy(state.plan_session); + state.plan_session = NULL; } else { - DeleteCommitResult deletion = delete_plan_session_commit(plan_session, config); - bool limit = delete_plan_session_limit_reached(plan_session); - receiver_tally_deleted(sink, delete_plan_session_deleted(plan_session)); - delete_plan_session_destroy(plan_session); - plan_session = NULL; + DeleteCommitResult deletion = delete_plan_session_commit(state.plan_session, config); + bool limit = delete_plan_session_limit_reached(state.plan_session); + receiver_tally_deleted(sink, delete_plan_session_deleted(state.plan_session)); + delete_plan_session_destroy(state.plan_session); + state.plan_session = NULL; if (deletion == DELETE_COMMIT_ERROR) { send_status(file_descriptor, STATUS_ERROR); goto fail; } - if (limit && !delete_limit_noted && sink->note_delete_limit) + if (limit && !state.delete_limit_noted && sink->note_delete_limit) sink->note_delete_limit(sink->context); } } @@ -568,26 +689,14 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver } return 0; +receive_error: + notify_peer = true; fail: /* Failure exits that must not (or already did) report a STATUS_ERROR. The parked keep-set/session is dropped: never commit a deletion for a failed stream. */ - if (deferred_manifest) { - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; - } - if (plan_session) - delete_plan_session_destroy(plan_session); - return -1; - -receive_error: - if (deferred_manifest) { - delete_manifest_free(deferred_manifest); - deferred_manifest = NULL; - } - if (plan_session) - delete_plan_session_destroy(plan_session); - if (sink->send_error) + receiver_drop_pending(&state); + if (notify_peer && sink->send_error) send_status(file_descriptor, STATUS_ERROR); return -1; } -- 2.54.0 From 934defa9659837ae935703618c46590d890a1120 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 13:52:57 +0200 Subject: [PATCH 39/68] refactor(delete): consolidate delete engine into delete.c --- CMakeLists.txt | 1 + src/shared/delete.c | 656 +++++++++++++++++++++++++++++++++++++ src/shared/delete.h | 151 +++++++++ src/shared/delete_commit.c | 193 ++--------- src/shared/delete_commit.h | 6 +- src/shared/delete_plan.c | 49 +-- src/shared/delete_plan.h | 1 + src/shared/utils.c | 635 ----------------------------------- src/shared/utils.h | 94 ------ tests/test_shared_utils.c | 1 + 10 files changed, 845 insertions(+), 942 deletions(-) create mode 100644 src/shared/delete.c create mode 100644 src/shared/delete.h diff --git a/CMakeLists.txt b/CMakeLists.txt index eb01dbb..8f5dbf2 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -99,6 +99,7 @@ set(SHARED_SRCS src/shared/daemon_limits.c src/shared/data.c src/shared/delay_updates.c + src/shared/delete.c src/shared/delete_commit.c src/shared/delete_plan.c src/shared/delta.c diff --git a/src/shared/delete.c b/src/shared/delete.c new file mode 100644 index 0000000..3e4fb62 --- /dev/null +++ b/src/shared/delete.c @@ -0,0 +1,656 @@ +#include "delete.h" + +#include "delay_updates.h" +#include "filter.h" +#include "log.h" +#include "utils.h" +#include +#include +#include +#include +#include +#include +#include +#include + +/* Build the keep-set index from the exact manifest entries only. A lookup of + `rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor + directory of kept content (the old is_dir_in_manifest predicate); the sorted + view answers "is an ancestor of kept content" without materializing any + per-component prefix copy, so the index is O(manifest size) memory. */ +static bool build_keep_index(const ArrayList* manifest, PathIndex* index) { + if (!manifest || manifest->size <= 0) + return path_index_build(index, NULL, 0); + return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size); +} + +static bool keep_is_dir(const PathIndex* index, const char* rel_path) { + return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path); +} + +static bool keep_is_file(const PathIndex* index, const char* rel_path) { + return path_index_contains(index, rel_path); +} + +/* True when child_rel is, or lies below, a protected entry. A prefix "a" + therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only + set only protect DIRECT children of the receive root (at_root); nested + directories that share such a name stay ordinary destination content. */ +bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, + int skip_count) { + for (int i = 0; i < skip_count; i++) { + if (skips[i].top_level_only && !at_root) + continue; + size_t prefix_len = strlen(skips[i].prefix); + if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 && + (child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/')) + return true; + } + return false; +} + +/* Per-run deletion budget and tallies. `max_delete` is the cap on the number + of entries the walker may remove (SIZE_MAX = unlimited); once it is reached + the remaining extras are counted in `skipped` and left in place, matching + rsync's partial --max-delete behavior. */ +typedef struct { + size_t max_delete; + size_t deleted; + size_t skipped; + bool limit_hit; +} DeleteBudget; + +/* True when direct children of the directory named by `rel` may be removed. + With no synchronization info (dirs == NULL) the whole tree is deletable; when + a dirs index is supplied only its exact entries are (the receive root is the + "." sentinel). */ +static bool is_synced_dir(const PathIndex* dirs, const char* rel) { + if (!dirs) + return true; + return path_index_contains(dirs, rel[0] == '\0' ? "." : rel); +} + +/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed + strcmp would order bytes >= 0x80 differently). */ +static int delete_name_cmp(const char* a, const char* b) { + const unsigned char* pa = (const unsigned char*)a; + const unsigned char* pb = (const unsigned char*)b; + while (*pa != '\0' && *pa == *pb) { + pa++; + pb++; + } + return (int)*pa - (int)*pb; +} + +bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, + bool* operation_ok) { + *out = NULL; + *count = 0; + if (operation_ok) + *operation_ok = true; + int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + if (scanfd < 0) + return false; + DIR* dir = fdopendir(scanfd); + if (!dir) { + close(scanfd); + return false; + } + DeleteDirEntry* entries = NULL; + size_t used = 0; + size_t capacity = 0; + bool ok = true; + const struct dirent* entry; + while ((entry = readdir(dir)) != NULL) { + if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) + continue; + struct stat st; + if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { + if (errno != ENOENT && operation_ok) + *operation_ok = false; + continue; + } + if (used == capacity) { + size_t next = capacity == 0 ? 16 : capacity * 2; + DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown)); + if (!grown) { + ok = false; + break; + } + entries = grown; + capacity = next; + } + entries[used].name = str_dup(entry->d_name); + if (!entries[used].name) { + ok = false; + break; + } + entries[used].is_dir = S_ISDIR(st.st_mode); + used++; + } + closedir(dir); + if (!ok) { + delete_dir_entries_free(entries, used); + return false; + } + *out = entries; + *count = used; + return true; +} + +void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) { + if (!entries) + return; + for (size_t i = 0; i < count; i++) + free(entries[i].name); + free(entries); +} + +/* rsync's extraneous-entry order: subdirectories before files, each group in + descending name order. */ +int delete_dir_entry_cmp_desc(const void* a, const void* b) { + const DeleteDirEntry* ea = a; + const DeleteDirEntry* eb = b; + if (ea->is_dir != eb->is_dir) + return ea->is_dir ? -1 : 1; + return -delete_name_cmp(ea->name, eb->name); +} + +/* rsync's kept-subdirectory order: plain ascending name. */ +int delete_dir_entry_cmp_asc(const void* a, const void* b) { + const DeleteDirEntry* ea = a; + const DeleteDirEntry* eb = b; + return delete_name_cmp(ea->name, eb->name); +} + +/* How the shared classification/descent walk disposes of an extra it has + identified. LIST records the destination-relative path without touching disk + (the -n/--dry-run would-delete enumeration); DELETE unlinks/rmdirs it, charges + the shared --max-delete budget and notifies the observer. Both modes classify + and traverse identically, so the dry-run enumeration and the real deletion + cannot drift. */ +typedef enum { DELETE_WALK_MODE_DELETE, DELETE_WALK_MODE_LIST } DeleteWalkMode; + +typedef struct { + DeleteWalkMode mode; + DeleteBudget* budget; /* DELETE mode */ + ArrayList* out; /* LIST mode: receives strdup'd relative paths */ + size_t* recorded; /* LIST mode */ + DeletePathObserver observer; /* DELETE mode */ + void* observer_context; /* DELETE mode */ +} DeleteWalkState; + +/* Remove the extras directly inside the directory open on `dirfd` (DELETE mode) + or record the paths that WOULD be removed (LIST mode), recursing into every + child directory so kept content below a synchronized prefix is reached. + `all_removed` reports whether every child entry was removed (so the caller may + rmdir this directory). A child directory is never removed when it is itself a + synchronized directory or holds kept content; with a dirs index supplied, + direct children of a non-synchronized directory are never extras at all (they + are left in place but still descended into). Symlinks are unlinked like any + other non-directory extra (never followed). + + Entries are processed in rsync's order (extraneous subdirectories in + descending name order, then extraneous files, then kept subdirectories in + ascending order) rather than readdir() order, so `--max-delete` leaves the + same survivors and the `--info=del`/dry-run line order matches rsync. */ +static bool delete_walk_fd(int dirfd, const char* rel_path, const PathIndex* keep, + const PathIndex* dirs, DeleteWalkState* state, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, bool parent_deletable, + bool* all_removed) { + DeleteDirEntry* entries = NULL; + size_t count = 0; + bool collect_ok = true; + if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) + return false; + bool operation_ok = collect_ok; + bool local_survives = false; + bool* shielded = calloc(count ? count : 1, sizeof(bool)); + bool* is_extra = calloc(count ? count : 1, sizeof(bool)); + if (!shielded || !is_extra) { + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); + return false; + } + /* A directory is deletable when it or ANY ancestor is synchronized; the + `parent_deletable` flag carries that down the recursion so dest-only + directories below a synchronized root are removed wholesale. */ + bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); + bool at_root = rel_path[0] == '\0'; + + /* Reproduce rsync's traversal order: extraneous subdirectories in descending + name order, then extraneous files in descending name order, and kept + subdirectories only afterwards (ascending). Sorting up front also fixes the + identity of the survivors under a partial --max-delete. */ + if (count > 1) + qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); + size_t dir_count = 0; + while (dir_count < count && entries[dir_count].is_dir) + dir_count++; + + /* Classify every entry up front (the verdict does not depend on processing + order) so the ordered passes below can act on it. */ + for (size_t i = 0; i < count; i++) { + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + /* A --delay-updates run keeps its staging directory as a direct child of + the receive root, and basis-dir snapshots live below it too. Their + contents are not manifest entries, so descending into them would delete + every staged / basis file as an "extra". Only the staging name (a + top-level-only prefix) and the basis prefixes are protected: a nested + destination directory that happens to be called .fastsync-stage is + ordinary content. */ + if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { + shielded[i] = true; + local_survives = true; + } else if (protect_rules && + filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, + FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { + /* A first-match protect rule shields the extra; for a directory the whole + subtree is shielded (rsync prunes an excluded directory), so do not + descend. */ + shielded[i] = true; + local_survives = true; + } else if (entries[i].is_dir) { + bool child_synced = dirs && path_index_contains(dirs, child_rel); + is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } else { + is_extra[i] = deletable && !keep_is_file(keep, child_rel); + if (!is_extra[i]) + local_survives = true; + } + free(child_rel); + } + + /* Pass 1: extraneous subdirectories, descending. */ + for (size_t i = 0; i < dir_count; i++) { + if (!is_extra[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!delete_walk_fd(childfd, child_rel, keep, dirs, state, skips, skip_count, protect_rules, + deletable, &child_all_removed)) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + if (child_all_removed && deletable) { + if (state->mode == DELETE_WALK_MODE_LIST) { + /* Record the directory with rsync's trailing slash. */ + size_t len = strlen(child_rel); + char* copy = malloc(len + 2); + if (!copy) { + operation_ok = false; + } else { + memcpy(copy, child_rel, len); + copy[len] = '/'; + copy[len + 1] = '\0'; + if (!array_list_add(state->out, copy)) { + free(copy); + operation_ok = false; + } else { + (*state->recorded)++; + } + } + } else if (state->budget->deleted >= state->budget->max_delete) { + state->budget->limit_hit = true; + state->budget->skipped++; + local_survives = true; + } else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) { + /* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still + holds entries the walker leaves in place (a protected excluded + prefix, a kept file the manifest protects, a symlink); rsync leaves + such a directory behind, so this is not an error. Only genuine I/O + failures abort the deletion. */ + if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) + operation_ok = false; + local_survives = true; + } else { + state->budget->deleted++; + /* rsync reports a removed directory with a trailing slash. */ + if (state->observer) { + size_t len = strlen(child_rel); + char* with_slash = malloc(len + 2); + if (with_slash) { + memcpy(with_slash, child_rel, len); + with_slash[len] = '/'; + with_slash[len + 1] = '\0'; + state->observer(state->observer_context, with_slash); + free(with_slash); + } else { + state->observer(state->observer_context, child_rel); + } + } + } + } else { + local_survives = true; + } + free(child_rel); + } + + /* Pass 2: extraneous files, descending. */ + for (size_t i = dir_count; i < count; i++) { + if (!is_extra[i]) + continue; + if (state->mode == DELETE_WALK_MODE_LIST) { + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + char* copy = str_dup(child_rel); + if (!copy || !array_list_add(state->out, copy)) { + free(copy); + operation_ok = false; + } else { + (*state->recorded)++; + } + free(child_rel); + } else if (state->budget->deleted >= state->budget->max_delete) { + state->budget->limit_hit = true; + state->budget->skipped++; + local_survives = true; + } else if (unlinkat(dirfd, entries[i].name, 0) != 0) { + if (errno != ENOENT) + operation_ok = false; + local_survives = true; + } else { + state->budget->deleted++; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (child_rel) { + if (state->observer) + state->observer(state->observer_context, child_rel); + char* escaped_path = output_escape(child_rel, log_get_8_bit_output()); + fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : ""); + free(escaped_path); + } + free(child_rel); + } + } + + /* Pass 3: kept subdirectories, ascending (rsync descends into these only + after the parent's own extras have been handled). */ + for (size_t i = dir_count; i-- > 0;) { + if (is_extra[i] || shielded[i]) + continue; + char* child_rel = path_cat((char*)rel_path, entries[i].name); + if (!child_rel) { + operation_ok = false; + continue; + } + int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + bool child_all_removed = false; + if (childfd >= 0) { + if (!delete_walk_fd(childfd, child_rel, keep, dirs, state, skips, skip_count, protect_rules, + deletable, &child_all_removed)) + operation_ok = false; + close(childfd); + } else if (errno != ENOENT) { + operation_ok = false; + } + /* A kept/synchronized directory is never removed. */ + local_survives = true; + free(child_rel); + } + + free(shielded); + free(is_extra); + delete_dir_entries_free(entries, count); + *all_removed = !local_survives; + return operation_ok; +} + +/* Open the receive root following the same authorized-root confinement the + walker uses, or dest_root directly when no authorized root is installed. */ +static int open_destination_root(const char* dest_root) { + int root_fd = utils_get_authorized_root_fd(); + if (root_fd >= 0) { + if (utils_get_authorized_root_path()) + return utils_open_authorized_destination(dest_root); + if (dest_root == NULL) + return dup(root_fd); + return -1; + } + return open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); +} + +bool delete_extras_list(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out) { + if (count_out) + *count_out = 0; + if (!manifest || !out) + return false; + PathIndex keep; + if (!build_keep_index(manifest, &keep)) + return false; + PathIndex dirs; + bool have_dirs = synced_dirs != NULL; + if (have_dirs && + !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { + path_index_free(&keep); + return false; + } + int rootfd = open_destination_root(dest_root); + if (rootfd < 0) { + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + return false; + } + bool all_removed = false; + size_t recorded = 0; + DeleteWalkState state = {.mode = DELETE_WALK_MODE_LIST, + .budget = NULL, + .out = out, + .recorded = &recorded, + .observer = NULL, + .observer_context = NULL}; + bool ok = delete_walk_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &state, skips, skip_count, + protect_rules, false, &all_removed); + if (close(rootfd) != 0) + ok = false; + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + if (count_out) + *count_out = recorded; + return ok; +} + +DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, + size_t* deleted_out, size_t* skipped_out, + DeletePathObserver observer, + void* observer_context) { + if (deleted_out) + *deleted_out = 0; + if (skipped_out) + *skipped_out = 0; + if (!manifest) + return DELETE_WALK_ERROR; + /* Index the keep-set (and the synchronized-dir set, when supplied) once so + membership is answered in O(path length) instead of scanning every entry + for every destination entry. */ + PathIndex keep; + if (!build_keep_index(manifest, &keep)) + return DELETE_WALK_ERROR; + PathIndex dirs; + bool have_dirs = synced_dirs != NULL; + if (have_dirs && + !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { + path_index_free(&keep); + return DELETE_WALK_ERROR; + } + int rootfd = open_destination_root(dest_root); + if (rootfd < 0) { + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + return DELETE_WALK_ERROR; + } + DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; + bool all_removed = false; + DeleteWalkState state = {.mode = DELETE_WALK_MODE_DELETE, + .budget = &budget, + .out = NULL, + .recorded = NULL, + .observer = observer, + .observer_context = observer_context}; + bool ok = delete_walk_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &state, skips, skip_count, + protect_rules, false, &all_removed); + if (close(rootfd) != 0) + ok = false; + path_index_free(&keep); + if (have_dirs) + path_index_free(&dirs); + if (deleted_out) + *deleted_out = budget.deleted; + if (skipped_out) + *skipped_out = budget.skipped; + if (!ok) + return DELETE_WALK_ERROR; + return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK; +} + +DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, size_t* deleted_out, + size_t* skipped_out) { + return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips, + skip_count, protect_rules, deleted_out, skipped_out, NULL, + NULL); +} + +bool delete_extras(const char* dest_root, const ArrayList* manifest) { + return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) == + DELETE_WALK_OK; +} + +/* Build the delete-walk protection prefix for one basis directory. The walker + compares paths relative to the receive root, so a relative entry is already + in the right form; an absolute entry that lies below the root is converted to + its root-relative form, and one outside the root returns NULL (the walk + cannot reach it, and it is not protected data beneath the root). Exposed so + tests can exercise the root-of-"/" child mapping directly. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path) { + if (!path) + return NULL; + if (path[0] != '/') + return str_dup(path); + const char* root = config->receive_root_directory; + if (!root || root[0] != '/') + return NULL; + size_t root_len = strlen(root); + while (root_len > 1 && root[root_len - 1] == '/') + root_len--; + if (strncmp(path, root, root_len) != 0) + return NULL; + if (root_len == 1) { + /* `root` is "/" (the only single-character absolute root): every absolute + path is below it, and the child relative form is everything after the + leading '/'. */ + if (path[1] == '\0') + return NULL; /* identical to the root, not a child */ + return str_dup(path + 1); + } + if (path[root_len] != '/') + return NULL; /* identical or a sibling sharing a name prefix */ + return str_dup(path + root_len + 1); +} + +bool delete_skips_build(const Config* config, const ArrayList* protected_paths, + const ArrayList* size_skipped, bool basis_root_relative, + DeleteSkipSet* out) { + if (!out) + return false; + out->entries = NULL; + out->owned_prefixes = NULL; + out->count = 0; + out->owned_count = 0; + if (!config) + return false; + int protected_count = protected_paths ? protected_paths->size : 0; + int size_skipped_count = size_skipped ? size_skipped->size : 0; + int count = + (config->delay_updates ? 1 : 0) + config->basis_count + protected_count + size_skipped_count; + if (count == 0) + return true; + out->entries = calloc((size_t)count, sizeof(DeleteSkipEntry)); + if (!out->entries) + return false; + if (basis_root_relative && config->basis_count > 0) { + out->owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); + if (!out->owned_prefixes) { + free(out->entries); + out->entries = NULL; + return false; + } + out->owned_count = config->basis_count; + } + int idx = 0; + if (config->delay_updates) { + out->entries[idx].prefix = DELAY_UPDATES_STAGING_DIR; + out->entries[idx].top_level_only = true; + idx++; + } + for (int i = 0; i < config->basis_count; i++) { + const char* prefix = config->basis_dirs[i].path; + if (basis_root_relative) { + /* An absolute basis outside the receive root is unreachable by this walk, + so it contributes no protection prefix (and no slot). */ + char* relative = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + if (!relative) + continue; + out->owned_prefixes[i] = relative; + prefix = relative; + } + out->entries[idx].prefix = prefix; + out->entries[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < protected_count; i++) { + out->entries[idx].prefix = (const char*)protected_paths->items[i]; + out->entries[idx].top_level_only = false; + idx++; + } + for (int i = 0; i < size_skipped_count; i++) { + out->entries[idx].prefix = (const char*)size_skipped->items[i]; + out->entries[idx].top_level_only = false; + idx++; + } + out->count = idx; + return true; +} + +void delete_skips_free(DeleteSkipSet* set) { + if (!set) + return; + if (set->owned_prefixes) { + for (int i = 0; i < set->owned_count; i++) + free(set->owned_prefixes[i]); + } + free(set->owned_prefixes); + free(set->entries); + set->entries = NULL; + set->owned_prefixes = NULL; + set->count = 0; + set->owned_count = 0; +} diff --git a/src/shared/delete.h b/src/shared/delete.h new file mode 100644 index 0000000..235db8c --- /dev/null +++ b/src/shared/delete.h @@ -0,0 +1,151 @@ +#ifndef DELETE_H +#define DELETE_H + +#include "array_list.h" +#include "config.h" +#include +#include + +/* Delete engine. + * + * This module owns destination-relative delete traversal: the ordered directory + * walker that reproduces rsync's extraneous-entry order, the skip-prefix + * protection set shared by every delete pass, and the read-only enumeration + * that mirrors the walker for -n/--dry-run. The budgeted manifest commit + * (delete_commit.c) and the per-directory delete plans (delete_plan.c) are + * built on the primitives exported here. */ + +/* Result of a bounded extra-file deletion run. */ +typedef enum { + /* Every extra entry was removed (or there were none). */ + DELETE_WALK_OK = 0, + /* The numeric cap for this run was reached before every extra was removed. + The walker removed exactly the entries the cap allowed and skipped (without + removing) the rest, matching rsync's partial --max-delete behavior. */ + DELETE_WALK_LIMIT_REACHED, + /* A traversal or unlink failure aborted the deletion (partial removal is + possible, mirroring the delete pass). */ + DELETE_WALK_ERROR +} DeleteWalkResult; + +/* One protected entry for the delete walker. When top_level_only is true the + prefix is skipped only as a DIRECT child of dest_root (the --delay-updates + staging directory, which must not hide genuine extras inside a nested + destination directory that happens to share the staging name); otherwise the + prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest + basis trees, and the sender-side protected filter-excluded prefixes, which + are never destination content). */ +typedef struct { + const char* prefix; + bool top_level_only; +} DeleteSkipEntry; + +/* A built skip-prefix set. `entries`/`count` are what path_under_skip_prefix() + consumes. `owned_prefixes` holds any prefix strings the builder had to + allocate (root-relative basis-dir conversions); it is NULL when every prefix + is borrowed from the config or the caller's lists. Release with + delete_skips_free(). */ +typedef struct { + DeleteSkipEntry* entries; + char** owned_prefixes; + int count; + int owned_count; +} DeleteSkipSet; + +/* True when child_rel is, or lies below, one of the protected entries (a prefix + "a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect + only DIRECT children of the destination root, i.e. child_rel has no '/'). */ +bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, + int skip_count); + +/* One destination-directory entry collected up front so the delete walkers can + reproduce rsync's traversal order instead of readdir() order. rsync processes + a directory's extraneous subdirectories first (descending name, depth-first), + then its extraneous files (descending name), and only afterwards descends into + its kept subdirectories (ascending name). */ +typedef struct { + char* name; + bool is_dir; +} DeleteDirEntry; +/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."), + stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of + *count entries whose names the caller frees with delete_dir_entries_free(). + Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is + skipped, any other stat failure is reported through *operation_ok while the + walk continues. */ +bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok); +void delete_dir_entries_free(DeleteDirEntry* entries, size_t count); +/* Sort comparators: `_desc` orders subdirectories before files and each group by + descending name (rsync's extraneous-entry order); `_asc` orders plain ascending + name (rsync's kept-subdirectory order). */ +int delete_dir_entry_cmp_desc(const void* a, const void* b); +int delete_dir_entry_cmp_asc(const void* a, const void* b); + +/* Remove files/dirs/symlinks under dest_root that are not listed in manifest + without ever descending into a protected prefix (see DeleteSkipEntry). When + `synced_dirs` is non-NULL, extras are only removed directly inside a directory + whose destination-relative path is an exact entry in that list (the receive + root is the "." sentinel); directories outside the synchronized set are still + descended into so kept content below a listed directory is preserved, but + nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of + treating the whole destination tree as deletable. `max_delete` caps the + number of removed entries (SIZE_MAX = unlimited): the walker removes up to the + cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained. + `deleted_out`/`skipped_out` optionally receive the number of entries removed + and the number skipped because of the cap. */ +DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, size_t* deleted_out, + size_t* skipped_out); + +/* Optional per-deletion observer: called for each destination-relative path + actually removed (a file, symlink, or directory), in removal order, so the + receiver can stream rsync's `--info=del`/`--info=remove` lines. */ +typedef void (*DeletePathObserver)(void* context, const char* rel_path); + +/* `delete_extras_limited_observed` is delete_extras_limited with an optional + * observer; the observer is invoked only for entries truly removed. When + * `protect_rules` is non-NULL its receiver-side verdict is evaluated for every + * candidate extra: a first-match PROTECT leaves the entry (and, for a + * directory, its whole subtree) in place, while RISK/NONE fall through to the + * ordinary skip-prefix/keep-set logic. */ +DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, size_t max_delete, + const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, + size_t* deleted_out, size_t* skipped_out, + DeletePathObserver observer, + void* observer_context); +/* Read-only companion to delete_extras_limited: walk the destination exactly as + the delete pass would and APPEND (strdup'd) destination-relative paths that + WOULD be removed, without touching disk. Used for -n/--dry-run --delete + would-delete reporting. Returns true on a clean walk; the caller owns the + strings appended to `out` and receives their count in *count_out. */ +bool delete_extras_list(const char* dest_root, const ArrayList* manifest, + const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, + const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out); +bool delete_extras(const char* dest_root, const ArrayList* manifest); + +/* Build the delete walk's skip-prefix set from the config's --delay-updates + staging directory, its --compare-dest/--copy-dest/--link-dest basis dirs, and + the caller-supplied protection lists, in that order. `protected_paths` and + `size_skipped` are borrowed (may be NULL); every entry in them is protected at + any depth. The staging directory is protected only as a DIRECT child of the + receive root. `basis_root_relative` selects how a basis path becomes a + prefix: true converts an absolute path under the receive root to its + root-relative form (the whole-tree commit walk; an unreachable path + contributes no slot), false keeps the configured path verbatim (the + per-directory plan walk). On success the caller releases `*out` with + delete_skips_free(); returns false on allocation failure. */ +bool delete_skips_build(const Config* config, const ArrayList* protected_paths, + const ArrayList* size_skipped, bool basis_root_relative, + DeleteSkipSet* out); +void delete_skips_free(DeleteSkipSet* set); + +/* Convert one basis-directory path to the receive-root-relative protection + prefix the delete walker uses (NULL when it lies outside the root). Exposed + for unit tests of the root-of-"/" and normalization edge cases. */ +char* file_receive_basis_delete_relative(const Config* config, const char* path); + +#endif diff --git a/src/shared/delete_commit.c b/src/shared/delete_commit.c index a522dd8..07f72db 100644 --- a/src/shared/delete_commit.c +++ b/src/shared/delete_commit.c @@ -126,38 +126,6 @@ typedef struct { bool limit_hit; } DeleteBudgetState; -/* Build the delete-walk protection prefix for one basis directory. The walker - compares paths relative to the receive root, so a relative entry is already - in the right form; an absolute entry that lies below the root is converted to - its root-relative form, and one outside the root returns NULL (the walk - cannot reach it, and it is not protected data beneath the root). Exposed so - tests can exercise the root-of-"/" child mapping directly. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path) { - if (!path) - return NULL; - if (path[0] != '/') - return str_dup(path); - const char* root = config->receive_root_directory; - if (!root || root[0] != '/') - return NULL; - size_t root_len = strlen(root); - while (root_len > 1 && root[root_len - 1] == '/') - root_len--; - if (strncmp(path, root, root_len) != 0) - return NULL; - if (root_len == 1) { - /* `root` is "/" (the only single-character absolute root): every absolute - path is below it, and the child relative form is everything after the - leading '/'. */ - if (path[1] == '\0') - return NULL; /* identical to the root, not a child */ - return str_dup(path + 1); - } - if (path[root_len] != '/') - return NULL; /* identical or a sibling sharing a name prefix */ - return str_dup(path + root_len + 1); -} - /* Remove every destination entry under the receive root that is not in the keep-set, bounded by the shared budget (a smaller client --max-delete=NUM replaces the server hard bound; rsync deletes up to the bound and skips the @@ -174,55 +142,13 @@ static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest if (!config || !manifest || !manifest->keeps) return false; fprintf(stderr, "Deleting files not in manifest...\n"); - /* Protected entries: - - the --delay-updates staging name, protected only as a DIRECT child of the - receive root (a nested destination directory that happens to be named - .fastsync-stage is ordinary content); - - alternate basis directories (--compare-dest / --copy-dest / --link-dest) - at any depth: they are extra comparison snapshots the user pointed at, - not destination content, and deleting them would destroy the very files a - --link-dest run just linked into place; - - the sender-side protected prefixes (source paths excluded by filters and - paths pruned by --max-size/--min-size), at any depth, so their destination - mirror survives --delete unless --delete-excluded opts back into removing - the filter-excluded ones (size-pruned entries are always protected). */ - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* An absolute basis outside the receive root is unreachable by this walk, - so it contributes no protection prefix (and no slot). */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } + /* Protected entries: the --delay-updates staging name (only as a DIRECT child + of the receive root), the alternate basis directories and the sender-side + protected prefixes (filter-excluded and size-pruned source mirrors), all at + any depth. See delete_skips_build(). */ + DeleteSkipSet skips; + if (!delete_skips_build(config, manifest->protected, NULL, true, &skips)) + return false; /* Clamp rather than subtract: an accounting bug where deleted already exceeds max_delete must never underflow into an effectively unlimited budget. */ size_t remaining; @@ -235,14 +161,9 @@ static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest size_t deleted = 0; size_t skipped = 0; DeleteWalkResult result = delete_extras_limited_observed( - config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips, used, - config->protect_rules, &deleted, &skipped, observer, observer_context); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); + config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips.entries, + skips.count, config->protect_rules, &deleted, &skipped, observer, observer_context); + delete_skips_free(&skips); budget->deleted += deleted; budget->skipped += skipped; if (result == DELETE_WALK_LIMIT_REACHED) { @@ -303,35 +224,12 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa if (!manifest->missing || manifest->missing->size == 0) return true; fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n"); - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count; - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } + /* The staging directory and basis snapshots stay protected exactly as in the + extras walker (the missing-args path overrides the ordinary protected + prefixes, so those are not passed here). */ + DeleteSkipSet skips; + if (!delete_skips_build(config, NULL, NULL, true, &skips)) + return false; bool ok = true; for (int i = 0; i < manifest->missing->size; i++) { const char* rel = (const char*)manifest->missing->items[i]; @@ -343,7 +241,7 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa continue; } bool at_root = strchr(rel, '/') == NULL; - if (path_under_skip_prefix(rel, at_root, skips, used)) { + if (path_under_skip_prefix(rel, at_root, skips.entries, skips.count)) { char* escaped = output_escape(rel, log_get_8_bit_output()); log_message(LOG_LEVEL_WARNING, "missing-args path '%s' is protected (staging directory or basis snapshot); " @@ -472,12 +370,7 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa if (!ok) break; } - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); + delete_skips_free(&skips); return ok; } @@ -489,52 +382,12 @@ bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, *count_out = 0; if (!config || !manifest || !manifest->keeps || !out) return false; - int skip_count = (config->delay_updates ? 1 : 0) + config->basis_count + - (manifest->protected ? manifest->protected->size : 0); - DeleteSkipEntry* skips = NULL; - char** owned_prefixes = NULL; - int used = 0; - if (skip_count > 0) { - skips = calloc((size_t)skip_count, sizeof(DeleteSkipEntry)); - owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*)); - if (!skips || (config->basis_count > 0 && !owned_prefixes)) { - free(skips); - free(owned_prefixes); - return false; - } - int idx = 0; - if (config->delay_updates) { - skips[idx].prefix = DELAY_UPDATES_STAGING_DIR; - skips[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - /* Normalize exactly like the real commit path: a relative entry is - already root-relative, an absolute one inside the receive root is - converted, and one outside contributes no protection prefix. */ - char* prefix = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); - if (!prefix) - continue; - owned_prefixes[i] = prefix; - skips[idx].prefix = prefix; - skips[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < manifest->protected->size; i++) { - skips[idx].prefix = (const char*)manifest->protected->items[i]; - skips[idx].top_level_only = false; - idx++; - } - used = idx; - } + DeleteSkipSet skips; + if (!delete_skips_build(config, manifest->protected, NULL, true, &skips)) + return false; bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs, - skips, used, config->protect_rules, out, count_out); - if (owned_prefixes) { - for (int i = 0; i < config->basis_count; i++) - free(owned_prefixes[i]); - } - free(owned_prefixes); - free(skips); + skips.entries, skips.count, config->protect_rules, out, count_out); + delete_skips_free(&skips); return ok; } diff --git a/src/shared/delete_commit.h b/src/shared/delete_commit.h index c2d374b..16f00ae 100644 --- a/src/shared/delete_commit.h +++ b/src/shared/delete_commit.h @@ -3,7 +3,7 @@ #include "array_list.h" #include "config.h" -#include "utils.h" +#include "delete.h" #include /* Delete-commit module: delete-manifest receive plus the budgeted extras and @@ -102,9 +102,5 @@ DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteMani clean walk; `*count_out` receives the number of paths appended. */ bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, size_t* count_out); -/* Convert one basis-directory path to the receive-root-relative protection - prefix the delete walker uses (NULL when it lies outside the root). Exposed - for unit tests of the root-of-"/" and normalization edge cases. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path); #endif diff --git a/src/shared/delete_plan.c b/src/shared/delete_plan.c index 58eb52e..f69213b 100644 --- a/src/shared/delete_plan.c +++ b/src/shared/delete_plan.c @@ -2,6 +2,7 @@ #include "charset.h" #include "delay_updates.h" +#include "delete.h" #include "file.h" #include "log.h" #include "utils.h" @@ -601,9 +602,8 @@ static int open_plan_dir(const Config* config, const char* dir) { return fd; } -typedef struct PlanSkips { - DeleteSkipEntry* entries; - int count; +typedef struct { + DeleteSkipSet set; /* Receiver-side delete-protection rules received on the config frame (NULL when the sender sent none). Evaluated per extra so a protect/risk rule is honored under --delete-during/--delete-delay exactly like the whole-tree @@ -613,39 +613,12 @@ typedef struct PlanSkips { static bool build_plan_skips(const Config* config, const DeletePlanSession* session, PlanSkips* out) { - out->entries = NULL; - out->count = 0; out->protect_rules = config->protect_rules; - int count = (config->delay_updates ? 1 : 0) + config->basis_count + - session->protected_prefixes->size + session->size_skipped->size; - if (count == 0) - return true; - out->entries = calloc((size_t)count, sizeof(DeleteSkipEntry)); - if (!out->entries) - return false; - int idx = 0; - if (config->delay_updates) { - out->entries[idx].prefix = DELAY_UPDATES_STAGING_DIR; - out->entries[idx].top_level_only = true; - idx++; - } - for (int i = 0; i < config->basis_count; i++) { - out->entries[idx].prefix = config->basis_dirs[i].path; - out->entries[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < session->protected_prefixes->size; i++) { - out->entries[idx].prefix = (const char*)session->protected_prefixes->items[i]; - out->entries[idx].top_level_only = false; - idx++; - } - for (int i = 0; i < session->size_skipped->size; i++) { - out->entries[idx].prefix = (const char*)session->size_skipped->items[i]; - out->entries[idx].top_level_only = false; - idx++; - } - out->count = idx; - return true; + /* The per-directory plan walk keeps each basis path verbatim (it does not + convert an absolute under-root path to its root-relative form, unlike the + whole-tree commit walk). */ + return delete_skips_build(config, session->protected_prefixes, session->size_skipped, false, + &out->set); } static bool budget_available(const DeletePlanSession* session) { @@ -796,7 +769,7 @@ static bool process_children(int dirfd, const char* dir_rel, const ArrayList* ke operation_ok = false; continue; } - if (path_under_skip_prefix(child_rel, at_root, skips->entries, skips->count)) { + if (path_under_skip_prefix(child_rel, at_root, skips->set.entries, skips->set.count)) { shielded[i] = true; local_survives = true; free(child_rel); @@ -883,7 +856,7 @@ static bool apply_plan_dir(DeletePlanSession* session, const Config* config, con bool survives = false; bool ok = process_children(dirfd, dir, dirs, files, strcmp(dir, ".") == 0, false, &skips, session, &survives); - free(skips.entries); + delete_skips_free(&skips.set); close(dirfd); if (!ok) log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files"); @@ -1024,7 +997,7 @@ static bool apply_deferred_path(DeletePlanSession* session, const Config* config } bool survives = false; bool ok = process_children(dirfd, rel, NULL, NULL, false, true, &skips, session, &survives); - free(skips.entries); + delete_skips_free(&skips.set); close(dirfd); if (!ok) { close(parent_fd); diff --git a/src/shared/delete_plan.h b/src/shared/delete_plan.h index 93929c8..9472d65 100644 --- a/src/shared/delete_plan.h +++ b/src/shared/delete_plan.h @@ -3,6 +3,7 @@ #include "array_list.h" #include "config.h" +#include "delete.h" #include "file_receive.h" #include "protocol.h" #include "utils.h" diff --git a/src/shared/utils.c b/src/shared/utils.c index 947a4d9..f758e4e 100644 --- a/src/shared/utils.c +++ b/src/shared/utils.c @@ -625,641 +625,6 @@ bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_si return written >= 0 && (size_t)written < buffer_size; } -/* Build the keep-set index from the exact manifest entries only. A lookup of - `rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor - directory of kept content (the old is_dir_in_manifest predicate); the sorted - view answers "is an ancestor of kept content" without materializing any - per-component prefix copy, so the index is O(manifest size) memory. */ -static bool build_keep_index(const ArrayList* manifest, PathIndex* index) { - if (!manifest || manifest->size <= 0) - return path_index_build(index, NULL, 0); - return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size); -} - -static bool keep_is_dir(const PathIndex* index, const char* rel_path) { - return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path); -} - -static bool keep_is_file(const PathIndex* index, const char* rel_path) { - return path_index_contains(index, rel_path); -} - -/* True when child_rel is, or lies below, a protected entry. A prefix "a" - therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only - set only protect DIRECT children of the receive root (at_root); nested - directories that share such a name stay ordinary destination content. */ -bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, - int skip_count) { - for (int i = 0; i < skip_count; i++) { - if (skips[i].top_level_only && !at_root) - continue; - size_t prefix_len = strlen(skips[i].prefix); - if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 && - (child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/')) - return true; - } - return false; -} - -/* Per-run deletion budget and tallies. `max_delete` is the cap on the number - of entries the walker may remove (SIZE_MAX = unlimited); once it is reached - the remaining extras are counted in `skipped` and left in place, matching - rsync's partial --max-delete behavior. */ -typedef struct { - size_t max_delete; - size_t deleted; - size_t skipped; - bool limit_hit; -} DeleteBudget; - -/* True when direct children of the directory named by `rel` may be removed. - With no synchronization info (dirs == NULL) the whole tree is deletable; when - a dirs index is supplied only its exact entries are (the receive root is the - "." sentinel). */ -static bool is_synced_dir(const PathIndex* dirs, const char* rel) { - if (!dirs) - return true; - return path_index_contains(dirs, rel[0] == '\0' ? "." : rel); -} - -/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed - strcmp would order bytes >= 0x80 differently). */ -static int delete_name_cmp(const char* a, const char* b) { - const unsigned char* pa = (const unsigned char*)a; - const unsigned char* pb = (const unsigned char*)b; - while (*pa != '\0' && *pa == *pb) { - pa++; - pb++; - } - return (int)*pa - (int)*pb; -} - -bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, - bool* operation_ok) { - *out = NULL; - *count = 0; - if (operation_ok) - *operation_ok = true; - int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - if (scanfd < 0) - return false; - DIR* dir = fdopendir(scanfd); - if (!dir) { - close(scanfd); - return false; - } - DeleteDirEntry* entries = NULL; - size_t used = 0; - size_t capacity = 0; - bool ok = true; - const struct dirent* entry; - while ((entry = readdir(dir)) != NULL) { - if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0) - continue; - struct stat st; - if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) { - if (errno != ENOENT && operation_ok) - *operation_ok = false; - continue; - } - if (used == capacity) { - size_t next = capacity == 0 ? 16 : capacity * 2; - DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown)); - if (!grown) { - ok = false; - break; - } - entries = grown; - capacity = next; - } - entries[used].name = str_dup(entry->d_name); - if (!entries[used].name) { - ok = false; - break; - } - entries[used].is_dir = S_ISDIR(st.st_mode); - used++; - } - closedir(dir); - if (!ok) { - delete_dir_entries_free(entries, used); - return false; - } - *out = entries; - *count = used; - return true; -} - -void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) { - if (!entries) - return; - for (size_t i = 0; i < count; i++) - free(entries[i].name); - free(entries); -} - -/* rsync's extraneous-entry order: subdirectories before files, each group in - descending name order. */ -int delete_dir_entry_cmp_desc(const void* a, const void* b) { - const DeleteDirEntry* ea = a; - const DeleteDirEntry* eb = b; - if (ea->is_dir != eb->is_dir) - return ea->is_dir ? -1 : 1; - return -delete_name_cmp(ea->name, eb->name); -} - -/* rsync's kept-subdirectory order: plain ascending name. */ -int delete_dir_entry_cmp_asc(const void* a, const void* b) { - const DeleteDirEntry* ea = a; - const DeleteDirEntry* eb = b; - return delete_name_cmp(ea->name, eb->name); -} - -/* Remove the extras directly inside the directory open on `dirfd`, recursing - into every child directory so kept content below a synchronized prefix is - reached. `all_removed` reports whether every child entry was removed (so the - caller may rmdir this directory). A child directory is never removed when it - is itself a synchronized directory or holds kept content; with a dirs index - supplied, direct children of a non-synchronized directory are never extras at - all (they are left in place but still descended into). Symlinks are unlinked - like any other non-directory extra (never followed). - - Entries are processed in rsync's order (extraneous subdirectories in - descending name order, then extraneous files, then kept subdirectories in - ascending order) rather than readdir() order, so `--max-delete` leaves the - same survivors and the `--info=del`/dry-run line order matches rsync. */ -static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, - const PathIndex* dirs, DeleteBudget* budget, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, bool parent_deletable, - bool* all_removed, DeletePathObserver observer, - void* observer_context) { - DeleteDirEntry* entries = NULL; - size_t count = 0; - bool collect_ok = true; - if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) - return false; - bool operation_ok = collect_ok; - bool local_survives = false; - bool* shielded = calloc(count ? count : 1, sizeof(bool)); - bool* is_extra = calloc(count ? count : 1, sizeof(bool)); - if (!shielded || !is_extra) { - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - return false; - } - /* A directory is deletable when it or ANY ancestor is synchronized; the - `parent_deletable` flag carries that down the recursion so dest-only - directories below a synchronized root are removed wholesale. */ - bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); - bool at_root = rel_path[0] == '\0'; - - /* Reproduce rsync's traversal order: extraneous subdirectories in descending - name order, then extraneous files in descending name order, and kept - subdirectories only afterwards (ascending). Sorting up front also fixes the - identity of the survivors under a partial --max-delete. */ - if (count > 1) - qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); - size_t dir_count = 0; - while (dir_count < count && entries[dir_count].is_dir) - dir_count++; - - /* Classify every entry up front (the verdict does not depend on processing - order) so the ordered passes below can act on it. */ - for (size_t i = 0; i < count; i++) { - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - /* A --delay-updates run keeps its staging directory as a direct child of - the receive root, and basis-dir snapshots live below it too. Their - contents are not manifest entries, so descending into them would delete - every staged / basis file as an "extra". Only the staging name (a - top-level-only prefix) and the basis prefixes are protected: a nested - destination directory that happens to be called .fastsync-stage is - ordinary content. */ - if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { - shielded[i] = true; - local_survives = true; - } else if (protect_rules && - filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, - FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { - /* A first-match protect rule shields the extra; for a directory the whole - subtree is shielded (rsync prunes an excluded directory), so do not - descend. */ - shielded[i] = true; - local_survives = true; - } else if (entries[i].is_dir) { - bool child_synced = dirs && path_index_contains(dirs, child_rel); - is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } else { - is_extra[i] = deletable && !keep_is_file(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } - free(child_rel); - } - - /* Pass 1: extraneous subdirectories, descending. */ - for (size_t i = 0; i < dir_count; i++) { - if (!is_extra[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, - protect_rules, deletable, &child_all_removed, observer, - observer_context)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - if (child_all_removed && deletable) { - if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - local_survives = true; - } else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) { - /* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still - holds entries the walker leaves in place (a protected excluded - prefix, a kept file the manifest protects, a symlink); rsync leaves - such a directory behind, so this is not an error. Only genuine I/O - failures abort the deletion. */ - if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) - operation_ok = false; - local_survives = true; - } else { - budget->deleted++; - /* rsync reports a removed directory with a trailing slash. */ - if (observer) { - size_t len = strlen(child_rel); - char* with_slash = malloc(len + 2); - if (with_slash) { - memcpy(with_slash, child_rel, len); - with_slash[len] = '/'; - with_slash[len + 1] = '\0'; - observer(observer_context, with_slash); - free(with_slash); - } else { - observer(observer_context, child_rel); - } - } - } - } else { - local_survives = true; - } - free(child_rel); - } - - /* Pass 2: extraneous files, descending. */ - for (size_t i = dir_count; i < count; i++) { - if (!is_extra[i]) - continue; - if (budget->deleted >= budget->max_delete) { - budget->limit_hit = true; - budget->skipped++; - local_survives = true; - } else if (unlinkat(dirfd, entries[i].name, 0) != 0) { - if (errno != ENOENT) - operation_ok = false; - local_survives = true; - } else { - budget->deleted++; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (child_rel) { - if (observer) - observer(observer_context, child_rel); - char* escaped_path = output_escape(child_rel, log_get_8_bit_output()); - fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : ""); - free(escaped_path); - } - free(child_rel); - } - } - - /* Pass 3: kept subdirectories, ascending (rsync descends into these only - after the parent's own extras have been handled). */ - for (size_t i = dir_count; i-- > 0;) { - if (is_extra[i] || shielded[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, - protect_rules, deletable, &child_all_removed, observer, - observer_context)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - /* A kept/synchronized directory is never removed. */ - local_survives = true; - free(child_rel); - } - - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - *all_removed = !local_survives; - return operation_ok; -} - -/* Read-only mirror of delete_extras_fd: records the paths that WOULD be removed - without unlinking anything. A child directory is reported after its own - reportable children (depth-first), matching the delete pass's ordering. */ -static bool list_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, - const PathIndex* dirs, ArrayList* out, size_t* recorded, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, bool parent_deletable, - bool* all_removed) { - DeleteDirEntry* entries = NULL; - size_t count = 0; - bool collect_ok = true; - if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok)) - return false; - bool operation_ok = collect_ok; - bool local_survives = false; - bool* shielded = calloc(count ? count : 1, sizeof(bool)); - bool* is_extra = calloc(count ? count : 1, sizeof(bool)); - if (!shielded || !is_extra) { - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - return false; - } - bool deletable = parent_deletable || is_synced_dir(dirs, rel_path); - bool at_root = rel_path[0] == '\0'; - - /* Mirror the delete walk's rsync order (extraneous subdirectories descending, - then extraneous files descending, then kept subdirectories ascending). */ - if (count > 1) - qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc); - size_t dir_count = 0; - while (dir_count < count && entries[dir_count].is_dir) - dir_count++; - - for (size_t i = 0; i < count; i++) { - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) { - shielded[i] = true; - local_survives = true; - } else if (protect_rules && - filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir, - FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) { - /* Mirror the delete walk: a protected entry is never reported as a - would-delete and a protected directory's subtree is not enumerated. */ - shielded[i] = true; - local_survives = true; - } else if (entries[i].is_dir) { - bool child_synced = dirs && path_index_contains(dirs, child_rel); - is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } else { - is_extra[i] = deletable && !keep_is_file(keep, child_rel); - if (!is_extra[i]) - local_survives = true; - } - free(child_rel); - } - - /* Pass 1: extraneous subdirectories, descending (recorded after contents). */ - for (size_t i = 0; i < dir_count; i++) { - if (!is_extra[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count, - protect_rules, deletable, &child_all_removed)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - if (child_all_removed && deletable) { - size_t len = strlen(child_rel); - char* copy = malloc(len + 2); - if (!copy) { - operation_ok = false; - } else { - memcpy(copy, child_rel, len); - copy[len] = '/'; - copy[len + 1] = '\0'; - if (!array_list_add(out, copy)) { - free(copy); - operation_ok = false; - } else { - (*recorded)++; - } - } - } else { - local_survives = true; - } - free(child_rel); - } - - /* Pass 2: extraneous files, descending. */ - for (size_t i = dir_count; i < count; i++) { - if (!is_extra[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - char* copy = str_dup(child_rel); - if (!copy || !array_list_add(out, copy)) { - free(copy); - operation_ok = false; - } else { - (*recorded)++; - } - free(child_rel); - } - - /* Pass 3: kept subdirectories, ascending. */ - for (size_t i = dir_count; i-- > 0;) { - if (is_extra[i] || shielded[i]) - continue; - char* child_rel = path_cat((char*)rel_path, entries[i].name); - if (!child_rel) { - operation_ok = false; - continue; - } - int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - bool child_all_removed = false; - if (childfd >= 0) { - if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count, - protect_rules, deletable, &child_all_removed)) - operation_ok = false; - close(childfd); - } else if (errno != ENOENT) { - operation_ok = false; - } - local_survives = true; - free(child_rel); - } - - free(shielded); - free(is_extra); - delete_dir_entries_free(entries, count); - *all_removed = !local_survives; - return operation_ok; -} - -bool delete_extras_list(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out) { - if (count_out) - *count_out = 0; - if (!manifest || !out) - return false; - PathIndex keep; - if (!build_keep_index(manifest, &keep)) - return false; - PathIndex dirs; - bool have_dirs = synced_dirs != NULL; - if (have_dirs && - !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { - path_index_free(&keep); - return false; - } - int rootfd; - int root_fd = utils_get_authorized_root_fd(); - if (root_fd >= 0) { - if (utils_get_authorized_root_path()) - rootfd = utils_open_authorized_destination(dest_root); - else if (dest_root == NULL) - rootfd = dup(root_fd); - else - rootfd = -1; - } else { - rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - } - if (rootfd < 0) { - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - return false; - } - bool all_removed = false; - size_t recorded = 0; - bool ok = list_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, out, &recorded, skips, - skip_count, protect_rules, false, &all_removed); - if (close(rootfd) != 0) - ok = false; - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - if (count_out) - *count_out = recorded; - return ok; -} - -DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, - size_t* deleted_out, size_t* skipped_out, - DeletePathObserver observer, - void* observer_context) { - if (deleted_out) - *deleted_out = 0; - if (skipped_out) - *skipped_out = 0; - if (!manifest) - return DELETE_WALK_ERROR; - /* Index the keep-set (and the synchronized-dir set, when supplied) once so - membership is answered in O(path length) instead of scanning every entry - for every destination entry. */ - PathIndex keep; - if (!build_keep_index(manifest, &keep)) - return DELETE_WALK_ERROR; - PathIndex dirs; - bool have_dirs = synced_dirs != NULL; - if (have_dirs && - !path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) { - path_index_free(&keep); - return DELETE_WALK_ERROR; - } - int rootfd; - int root_fd = utils_get_authorized_root_fd(); - if (root_fd >= 0) { - if (utils_get_authorized_root_path()) - rootfd = utils_open_authorized_destination(dest_root); - else if (dest_root == NULL) - rootfd = dup(root_fd); - else - rootfd = -1; - } else { - rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); - } - if (rootfd < 0) { - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - return DELETE_WALK_ERROR; - } - DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; - bool all_removed = false; - bool ok = - delete_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &budget, skips, skip_count, - protect_rules, false, &all_removed, observer, observer_context); - if (close(rootfd) != 0) - ok = false; - path_index_free(&keep); - if (have_dirs) - path_index_free(&dirs); - if (deleted_out) - *deleted_out = budget.deleted; - if (skipped_out) - *skipped_out = budget.skipped; - if (!ok) - return DELETE_WALK_ERROR; - return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK; -} - -DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, size_t* deleted_out, - size_t* skipped_out) { - return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips, - skip_count, protect_rules, deleted_out, skipped_out, NULL, - NULL); -} - -bool delete_extras(const char* dest_root, const ArrayList* manifest) { - return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) == - DELETE_WALK_OK; -} - bool has_path_traversal(const char* path) { if (!path) return true; diff --git a/src/shared/utils.h b/src/shared/utils.h index aedb9ce..492200f 100644 --- a/src/shared/utils.h +++ b/src/shared/utils.h @@ -106,101 +106,7 @@ int env_choice_first(const char* env_name, int (*resolve)(const char*), bool* sp ssize_t utils_getdelim_bounded(FILE* stream, char** line, size_t* cap, int delim, size_t max_len); char* path_cat(const char* path1, const char* path2); bool glob_match(const char* pattern, const char* str); -/* Result of a bounded extra-file deletion run. */ -typedef enum { - /* Every extra entry was removed (or there were none). */ - DELETE_WALK_OK = 0, - /* The numeric cap for this run was reached before every extra was removed. - The walker removed exactly the entries the cap allowed and skipped (without - removing) the rest, matching rsync's partial --max-delete behavior. */ - DELETE_WALK_LIMIT_REACHED, - /* A traversal or unlink failure aborted the deletion (partial removal is - possible, mirroring the delete pass). */ - DELETE_WALK_ERROR -} DeleteWalkResult; -/* One protected entry for the delete walker. When top_level_only is true the - prefix is skipped only as a DIRECT child of dest_root (the --delay-updates - staging directory, which must not hide genuine extras inside a nested - destination directory that happens to share the staging name); otherwise the - prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest - basis trees, and the sender-side protected filter-excluded prefixes, which - are never destination content). */ -typedef struct { - const char* prefix; - bool top_level_only; -} DeleteSkipEntry; -/* True when child_rel is, or lies below, one of the protected entries (a prefix - "a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect - only DIRECT children of the destination root, i.e. child_rel has no '/'). */ -bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips, - int skip_count); -/* One destination-directory entry collected up front so the delete walkers can - reproduce rsync's traversal order instead of readdir() order. rsync processes - a directory's extraneous subdirectories first (descending name, depth-first), - then its extraneous files (descending name), and only afterwards descends into - its kept subdirectories (ascending name). */ -typedef struct { - char* name; - bool is_dir; -} DeleteDirEntry; -/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."), - stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of - *count entries whose names the caller frees with delete_dir_entries_free(). - Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is - skipped, any other stat failure is reported through *operation_ok while the - walk continues. */ -bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok); -void delete_dir_entries_free(DeleteDirEntry* entries, size_t count); -/* Sort comparators: `_desc` orders subdirectories before files and each group by - descending name (rsync's extraneous-entry order); `_asc` orders plain ascending - name (rsync's kept-subdirectory order). */ -int delete_dir_entry_cmp_desc(const void* a, const void* b); -int delete_dir_entry_cmp_asc(const void* a, const void* b); -/* Remove files/dirs/symlinks under dest_root that are not listed in manifest - without ever descending into a protected prefix (see DeleteSkipEntry). When - `synced_dirs` is non-NULL, extras are only removed directly inside a directory - whose destination-relative path is an exact entry in that list (the receive - root is the "." sentinel); directories outside the synchronized set are still - descended into so kept content below a listed directory is preserved, but - nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of - treating the whole destination tree as deletable. `max_delete` caps the - number of removed entries (SIZE_MAX = unlimited): the walker removes up to the - cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained. - `deleted_out`/`skipped_out` optionally receive the number of entries removed - and the number skipped because of the cap. */ -DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, size_t* deleted_out, - size_t* skipped_out); -/* Optional per-deletion observer: called for each destination-relative path - actually removed (a file, symlink, or directory), in removal order, so the - receiver can stream rsync's `--info=del`/`--info=remove` lines. */ -typedef void (*DeletePathObserver)(void* context, const char* rel_path); - -/* `delete_extras_limited_observed` is delete_extras_limited with an optional - * observer; the observer is invoked only for entries truly removed. When - * `protect_rules` is non-NULL its receiver-side verdict is evaluated for every - * candidate extra: a first-match PROTECT leaves the entry (and, for a - * directory, its whole subtree) in place, while RISK/NONE fall through to the - * ordinary skip-prefix/keep-set logic. */ -DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, size_t max_delete, - const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, - size_t* deleted_out, size_t* skipped_out, - DeletePathObserver observer, - void* observer_context); -/* Read-only companion to delete_extras_limited: walk the destination exactly as - the delete pass would and APPEND (strdup'd) destination-relative paths that - WOULD be removed, without touching disk. Used for -n/--dry-run --delete - would-delete reporting. Returns true on a clean walk; the caller owns the - strings appended to `out` and receives their count in *count_out. */ -bool delete_extras_list(const char* dest_root, const ArrayList* manifest, - const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count, - const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out); -bool delete_extras(const char* dest_root, const ArrayList* manifest); /* Open the existing destination directory at `dest_root`, confined to the authorized root with an O_NOFOLLOW component walk (the same confinement the deletion walker uses for its root). Returns a new fd the caller owns, or -1 diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c index b822f1e..5fafa7e 100644 --- a/tests/test_shared_utils.c +++ b/tests/test_shared_utils.c @@ -1,4 +1,5 @@ #include "test_shared_utils.h" +#include "delete.h" #include "utils.h" #include "protocol.h" #include "test_utils.h" -- 2.54.0 From b549887138a889783b15872dbdc1bb916693db7e Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:03:10 +0200 Subject: [PATCH 40/68] refactor(config): group CLI-parse state; drop old_args field --- src/client/change_list.c | 2 +- src/client/client_cli.c | 65 +++++++++++++------------- src/client/client_manifest.c | 2 +- src/client/client_send.c | 4 +- src/shared/config.c | 23 +++++----- src/shared/config.h | 88 ++++++++++++++++++++---------------- tests/test_change_list.c | 2 +- tests/test_client_cli.c | 45 +++++++++--------- tests/test_config.c | 8 ++-- 9 files changed, 126 insertions(+), 113 deletions(-) diff --git a/src/client/change_list.c b/src/client/change_list.c index 9f3d728..a2c5096 100644 --- a/src/client/change_list.c +++ b/src/client/change_list.c @@ -245,7 +245,7 @@ static char* change_render_name_uptodate(const ChangeEvent* event) { * resolves to xxh128, so an explicit selection and the default both render the * selected algorithm's digest. */ static ChecksumAlgo out_format_checksum_algo(const Config* config) { - return (ChecksumAlgo)config->checksum_transfer_algo; + return (ChecksumAlgo)config->cli.checksum_transfer_algo; } /* Render a digest as rsync's sum_as_hex: xxh128 prints the HIGH 64-bit half diff --git a/src/client/client_cli.c b/src/client/client_cli.c index fb0411b..12f0c6d 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -170,7 +170,7 @@ static int set_positive_int_option(int* dest, const char* value, const char* opt * name is a hard error with rsync's exit code 4, never a silent no-op. */ static int set_compression_choice(Config* config, const char* value) { if (!value) { - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } int algo; @@ -178,7 +178,7 @@ static int set_compression_choice(Config* config, const char* value) { algo = compression_choice_resolve(); if (algo < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_COMPRESS_LIST names no supported compression algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } } else { @@ -189,7 +189,7 @@ static int set_compression_choice(Config* config, const char* value) { "--compress-choice '%s' is not a supported algorithm; FastSync supports zstd, " "lz4, zlib, zlibx, none or auto", value); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } const char* canonical = compression_algo_name((CompressionAlgo)algo); @@ -226,7 +226,7 @@ static int resolve_checksum_name(const char* name, size_t len, int* out) { * resolves to FastSync's negotiated default (xxh128). */ static int set_checksum_choice(Config* config, const char* value) { if (!value) { - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } const char* comma = strchr(value, ','); @@ -244,7 +244,7 @@ static int set_checksum_choice(Config* config, const char* value) { "--checksum-choice '%s' is invalid; FastSync supports xxh64 (or xxhash), xxh128, " "xxh3, md5, md4, sha1, none or auto, optionally as 'transfer,pre-transfer'", value); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } int negotiated = -1; @@ -252,7 +252,7 @@ static int set_checksum_choice(Config* config, const char* value) { negotiated = checksum_choice_resolve(); if (negotiated < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_CHECKSUM_LIST names no supported checksum algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } } @@ -264,8 +264,8 @@ static int set_checksum_choice(Config* config, const char* value) { pre = negotiated; config->checksum_algo = pre; - config->checksum_transfer_algo = transfer; - config->checksum_choice_set = true; + config->cli.checksum_transfer_algo = transfer; + config->cli.checksum_choice_set = true; /* rsync: "none" for the transfer checksum forces --whole-file. */ if (transfer == (int)CHECKSUM_ALGO_NONE) config->whole_file = true; @@ -963,7 +963,10 @@ static const OptionEntry OPTION_TABLE[] = { * faithful no-op (accepted silently, never consumes an argument). */ {"--recursive", "-r", OPT_NOOP, 0}, {"--update", "-u", OPT_FLAG, offsetof(Config, update)}, - {"--old-args", NULL, OPT_FLAG, offsetof(Config, old_args)}, + /* rsync's --old-args: accepted for CLI compatibility as a documented no-op + * (the remote server path is always safely quoted; see usage.c). It is + * recognized but stores no Config field. */ + {"--old-args", NULL, OPT_NOOP, 0}, {"--rsh", "-e", OPT_STRING, offsetof(Config, rsh_command)}, {"--blocking-io", NULL, OPT_FLAG, offsetof(Config, blocking_io)}, {"--links", "-l", OPT_FLAG, offsetof(Config, follow_symlinks)}, @@ -1196,12 +1199,12 @@ static int apply_negation(Config* config, const char* arg) { config->preserve_times = false; config->preserve_owner = false; config->preserve_group = false; - config->metadata_explicitly_disabled = true; + config->cli.metadata_explicitly_disabled = true; /* --no-preserve is an explicit opt-out of the whole bundle: record it so * the --incremental/--delta auto-preserve in cli_finalize_config does not * silently re-enable perms/times. */ - config->preserve_perms_explicit_off = true; - config->preserve_times_explicit_off = true; + config->cli.preserve_perms_explicit_off = true; + config->cli.preserve_times_explicit_off = true; return 0; } *(bool*)((char*)config + entry->offset) = false; @@ -1209,9 +1212,9 @@ static int apply_negation(Config* config, const char* arg) { * auto-preserve the OTHER attribute without undoing this one. A later * -p/-t sets the attribute directly; this flag only gates the implication. */ if (entry->offset == offsetof(Config, preserve_perms)) - config->preserve_perms_explicit_off = true; + config->cli.preserve_perms_explicit_off = true; else if (entry->offset == offsetof(Config, preserve_times)) - config->preserve_times_explicit_off = true; + config->cli.preserve_times_explicit_off = true; return 0; } @@ -1440,7 +1443,7 @@ static bool cli_handle_range_time_options(CliParseCtx* ctx) { ctx->exit_code = -1; return true; } - config->stop_at_set = true; + config->cli.stop_at_set = true; return true; } if (strcmp(arg, "--stop-at") == 0) { @@ -1455,7 +1458,7 @@ static bool cli_handle_range_time_options(CliParseCtx* ctx) { ctx->exit_code = -1; return true; } - config->stop_at_set = true; + config->cli.stop_at_set = true; return true; } const char* threads_prefix = "--compress-threads="; @@ -1526,7 +1529,7 @@ static bool cli_handle_table_option(CliParseCtx* ctx) { return true; } if (entry->offset == offsetof(Config, compression_level)) - config->compression_level_set = true; + config->cli.compression_level_set = true; if (entry->offset == offsetof(Config, chmod_spec)) { mode_t ignored; if (!chmod_apply(0, config->chmod_spec, &ignored)) { @@ -1539,7 +1542,7 @@ static bool cli_handle_table_option(CliParseCtx* ctx) { defaults to 127.0.0.1, so a value check cannot distinguish it). Used by --dry-run to route an explicit remote target to the server. */ if (entry->offset == offsetof(Config, server_host)) - config->server_host_set = true; + config->cli.server_host_set = true; } } else if (apply_table_option(config, entry, NULL) != 0) { ctx->exit_code = -1; @@ -1813,7 +1816,7 @@ static bool cli_handle_transfer_flags(CliParseCtx* ctx) { return true; } config->compression_level = (int)level; - config->compression_level_set = true; + config->cli.compression_level_set = true; log_info_message(LOG_INFO_MISC, "Set Compression level to %ld", level); ctx->i++; } @@ -1877,7 +1880,7 @@ static int set_server_port_option(Config* config, const char* value, const char* return -1; } config->server_port = port; - config->server_port_set = true; + config->cli.server_port_set = true; return 0; } @@ -2648,7 +2651,7 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool int resolved = compression_choice_resolve(); if (resolved < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_COMPRESS_LIST names no supported compression algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } config->compression_algo = resolved; @@ -2659,7 +2662,7 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool * clamped to the codec's range, otherwise the codec's own default is used. */ if (config->use_compression) { CompressionAlgo algo = (CompressionAlgo)config->compression_algo; - config->compression_level = config->compression_level_set + config->compression_level = config->cli.compression_level_set ? compression_clamp_level(algo, config->compression_level) : compression_default_level(algo); log_debug_message(LOG_DEBUG_UTIL, "Client compression: %s (level %d)", @@ -2668,22 +2671,22 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool /* The negotiated checksum is always resolved (rsync negotiates one for the * delta strong sum even without --checksum): RSYNC_CHECKSUM_LIST first, then * the compiled-in order. An explicit --checksum-choice already set it. */ - if (!config->checksum_choice_set) { + if (!config->cli.checksum_choice_set) { int resolved = checksum_choice_resolve(); if (resolved < 0) { log_message(LOG_LEVEL_ERROR, "RSYNC_CHECKSUM_LIST names no supported checksum algorithm"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } config->checksum_algo = resolved; - config->checksum_transfer_algo = resolved; + config->cli.checksum_transfer_algo = resolved; } /* rsync parity: "none" as the pre-transfer checksum cannot be combined with * --checksum (exit 4). The check runs here because --checksum may appear on * either side of --checksum-choice. */ if (config->checksum && config->checksum_algo == (int)CHECKSUM_ALGO_NONE) { log_message(LOG_LEVEL_ERROR, "Invalid checksum-choice for --checksum: none"); - config->cli_exit_code = 4; + config->cli.cli_exit_code = 4; return -1; } @@ -2762,11 +2765,11 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool * explicitly negated them (--no-perms/--no-times/--no-preserve). This runs * BEFORE the derived use_metadata bit so the transport frame is still sent * for the incremental/delta handshake even when both attributes were negated - * via --no-preserve (metadata_explicitly_disabled handles that opt-out). */ - if (preserve_implied && !config->metadata_explicitly_disabled) { - if (!config->preserve_perms_explicit_off) + * via --no-preserve (cli.metadata_explicitly_disabled handles that opt-out). */ + if (preserve_implied && !config->cli.metadata_explicitly_disabled) { + if (!config->cli.preserve_perms_explicit_off) config->preserve_perms = true; - if (!config->preserve_times_explicit_off) + if (!config->cli.preserve_times_explicit_off) config->preserve_times = true; } @@ -3153,7 +3156,7 @@ int main(int argc, char* argv[]) { int parse_ret = parse_args(config, argc, argv, positional_args, &positional_count); if (parse_ret != 0) { if (parse_ret < 0) - exit_code = config->cli_exit_code ? config->cli_exit_code : 1; + exit_code = config->cli.cli_exit_code ? config->cli.cli_exit_code : 1; goto cleanup; } diff --git a/src/client/client_manifest.c b/src/client/client_manifest.c index d476bb3..55a3f67 100644 --- a/src/client/client_manifest.c +++ b/src/client/client_manifest.c @@ -31,7 +31,7 @@ bool dry_run_targets_server(const Config* config) { return true; if (config->module && config->module[0] != '\0') return true; - if (config->server_host_set || config->server_port_set) + if (config->cli.server_host_set || config->cli.server_port_set) return true; if (config->use_tls) return true; diff --git a/src/client/client_send.c b/src/client/client_send.c index 3b77503..2f3422b 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1578,7 +1578,7 @@ static bool send_files_run(Config* config, SendFilesState* state) { now_mono.tv_nsec = 0; } state->stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, - config->stop_at_set, config->stop_at, now_mono); + config->cli.stop_at_set, config->stop_at, now_mono); state->prepared.options.stop_condition = &state->stop; /* The early-delete pre-scan above already ran; only the data pass should feed the directory-time list (otherwise every directory would be captured @@ -1916,7 +1916,7 @@ int send_files_multithreaded(Config* config) { now_mono.tv_nsec = 0; } context->stop_condition = - stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->stop_at_set, + stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->cli.stop_at_set, config->stop_at, now_mono); bool collect_excluded = config->use_delete && !config->delete_excluded; unsigned long long pre_scan_non_dir = 0; diff --git a/src/shared/config.c b/src/shared/config.c index f4b255d..eea3974 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -20,9 +20,9 @@ static void config_set_defaults(Config* config) { config->scanner_threads = 0; - config->metadata_explicitly_disabled = false; - config->preserve_perms_explicit_off = false; - config->preserve_times_explicit_off = false; + config->cli.preserve_perms_explicit_off = false; + config->cli.preserve_times_explicit_off = false; + config->cli.metadata_explicitly_disabled = false; config->show_progress = false; config->compression_threads = 0; config->ssh_port = 22; @@ -44,8 +44,8 @@ static void config_set_defaults(Config* config) { config->tls_ca = NULL; config->server_host = str_dup("127.0.0.1"); config->server_port = 8080; - config->server_port_set = false; - config->server_host_set = false; + config->cli.server_port_set = false; + config->cli.server_host_set = false; /* rsync defaults: --timeout=0 (I/O timeouts disabled) and --contimeout=60. * A value of 0 disables the client's own deadline on both the socket layer * (tcp_set_timeouts) and the protocol layer @@ -68,10 +68,10 @@ static void config_set_defaults(Config* config) { config->human_readable = false; config->ignore_errors = false; config->ignore_missing_args = false; - config->checksum_transfer_algo = CHECKSUM_ALGO_DEFAULT; - config->cli_exit_code = 0; - config->compression_level_set = false; - config->checksum_choice_set = false; + config->cli.checksum_transfer_algo = CHECKSUM_ALGO_DEFAULT; + config->cli.cli_exit_code = 0; + config->cli.compression_level_set = false; + config->cli.checksum_choice_set = false; config->filters = NULL; config->files_from = NULL; config->files_from_set = NULL; @@ -85,7 +85,6 @@ static void config_set_defaults(Config* config) { config->rsh_command = NULL; config->blocking_io = false; config->outbuf = OUTBUF_BLOCK; - config->old_args = false; config->remote_options = NULL; config->remote_option_count = 0; config->address = NULL; @@ -101,7 +100,7 @@ static void config_set_defaults(Config* config) { config->trust_sender = false; config->stop_after_mins = 0; config->stop_at = 0; - config->stop_at_set = false; + config->cli.stop_at_set = false; config->write_batch = NULL; config->only_write_batch = NULL; config->read_batch = NULL; @@ -342,7 +341,7 @@ bool config_derived_use_metadata(const Config* config) { config->chown_uid_set || config->chown_gid_set || config->usermap_count > 0 || config->groupmap_count > 0 || config->update) return true; - return (config->use_incremental || config->use_delta) && !config->metadata_explicitly_disabled; + return (config->use_incremental || config->use_delta) && !config->cli.metadata_explicitly_disabled; } bool config_has_basis(const Config* config) { diff --git a/src/shared/config.h b/src/shared/config.h index c4d7130..3596620 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -342,22 +342,60 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF CONFIG_WIRE_CODEC_FIELDS(X) \ CONFIG_WIRE_PROTECT_FIELDS(X) +/* Client-only, CLI-parse bookkeeping (never serialized). These members exist + * only so the client command-line parser can record HOW an option was + * specified (explicitly set, explicitly negated, or a parser-requested exit + * code); no other module and no wire peer ever needs them. Grouping them in + * one nested member keeps the public Config free of client-CLI-only state. */ +typedef struct { + /* Set when the user explicitly turned an attribute off with --no-perms / + * --no-times (long or short form). --incremental/--delta historically + * auto-enabled mode and mtime preservation; these flags let + * cli_finalize_config restore that behavior while still honoring the + * explicit per-attribute negation. A later -p/-t re-enables the attribute + * directly, so the flag only prevents the incremental/delta implication, + * never a POSITIVE request. */ + bool preserve_perms_explicit_off; + bool preserve_times_explicit_off; + /* Set by --no-preserve, the explicit opt-out of the whole preservation + * bundle, so the --incremental/--delta auto-preserve implication stays off. */ + bool metadata_explicitly_disabled; + /* True when --server-port/--port was explicitly given. --dry-run uses it to + * decide whether a real server handshake was requested, so a plain local + * destination (no explicit port) keeps the existing client-side dry-run + * behavior instead of dialing the default 127.0.0.1:8080. */ + bool server_port_set; + /* True when --server-host was explicitly given, and distinct from the + * "127.0.0.1" default: --dry-run uses it to route an explicit remote target + * to the server so it reports receiver state exactly like a real run, + * instead of silently running the client-side manifest. */ + bool server_host_set; + /* Codec-negotiation CLI state. The effective pre-transfer checksum is + * Config->checksum_algo (serialized); checksum_transfer_algo is the rsync + * "transfer" half of a two-name --checksum-choice form (validated and used + * only to mirror rsync's whole-file forcing, since FastSync's per-block + * strong hash is fixed). cli_exit_code carries a parser-requested process + * exit status (rsync uses 4 for an unsupported checksum/compress algorithm) + * so main() can mirror it. */ + int checksum_transfer_algo; + int cli_exit_code; + /* "The user explicitly chose" bits. They let the per-codec default level / + * checksum list be applied only when the corresponding rsync option was + * omitted (an explicit --compress-level / --checksum-choice always wins). */ + bool compression_level_set; + bool checksum_choice_set; + /* True when --stop-at was given. */ + bool stop_at_set; +} ConfigCliParse; + typedef struct Config { /* -j/--threads=N: number of parallel scanner worker threads for the -m * pipeline. 0 (the default, also set by bare -j/--threads) means "use the * scanner's built-in default" (4). CLIENT-ONLY: it is a local scheduling * concern and is NEVER serialized into the wire config frame. */ int scanner_threads; - bool metadata_explicitly_disabled; - /* CLIENT-ONLY (never serialized; not in CONFIG_WIRE_FIELDS). Set when the - * user explicitly turned an attribute off with --no-perms / --no-times (long - * or short form). --incremental/--delta historically auto-enabled mode and - * mtime preservation; these flags let cli_finalize_config restore that - * behavior while still honoring the explicit per-attribute negation. A - * later -p/-t re-enables the attribute directly, so the flag only prevents - * the incremental/delta implication, never a POSITIVE request. */ - bool preserve_perms_explicit_off; - bool preserve_times_explicit_off; + /* Client-only CLI-parse bookkeeping (never serialized). See ConfigCliParse. */ + ConfigCliParse cli; bool show_progress; int compression_threads; int ssh_port; @@ -378,18 +416,6 @@ typedef struct Config { bool use_tls; char* server_host; int server_port; - /* True when --server-port/--port was explicitly given. CLIENT-ONLY (never - * serialized): --dry-run uses it to decide whether a real server handshake - * was requested, so a plain local destination (no explicit port) keeps the - * existing client-side dry-run behavior instead of dialing the default - * 127.0.0.1:8080. */ - bool server_port_set; - /* True when --server-host was explicitly given. CLIENT-ONLY (never - * serialized), and distinct from the "127.0.0.1" default: --dry-run uses it - * to route an explicit remote target to the server so it reports receiver - * state exactly like a real run, instead of silently running the client-side - * manifest. */ - bool server_host_set; char* tls_cert; char* tls_key; char* tls_ca; @@ -435,22 +461,6 @@ typedef struct Config { * enters the keep-set. Implied by --delete-missing-args. */ bool ignore_missing_args; - /* Codec-negotiation CLI state (all client-only, never serialized). The - * effective pre-transfer checksum is Config->checksum_algo (serialized); - * checksum_transfer_algo is the rsync "transfer" half of a two-name - * --checksum-choice form (validated and used only to mirror rsync's - * whole-file forcing, since FastSync's per-block strong hash is fixed). - * cli_exit_code carries a parser-requested process exit status (rsync uses 4 - * for an unsupported checksum/compress algorithm) so main() can mirror it. */ - int checksum_transfer_algo; - int cli_exit_code; - /* Client-only "the user explicitly chose" bits. They let the per-codec - * default level / checksum list be applied only when the corresponding - * rsync option was omitted (an explicit --compress-level / --checksum-choice - * always wins). Never serialized. */ - bool compression_level_set; - bool checksum_choice_set; - // Issue #129: Advanced file selection. These fields are CLIENT-ONLY: they are // never serialized to the wire (the receiver must not learn them). ArrayList* filters; /* --filter=RULE rule strings, in order */ @@ -486,7 +496,6 @@ typedef struct Config { /* --outbuf mode (OutbufMode): stdout/stderr buffering. Client-only launch * concern: NEVER crosses the wire. */ int outbuf; - bool old_args; /* --remote-option=OPT (Phase 5, long form only): one or more extra command-line * options to append to the REMOTE server invocation over SSH. CLIENT-ONLY: * they are composed into the remote command line by ssh_build_remote_command() @@ -556,7 +565,6 @@ typedef struct Config { * process and are NEVER serialized into the config frame. */ int stop_after_mins; /* --stop-after=MINS minutes; 0 when unset */ time_t stop_at; /* --stop-at=... absolute wall-clock deadline */ - bool stop_at_set; /* true when --stop-at was given */ /* Client-only residual-batch paths. A residual batch is a self-contained * single-file record of the whole source tree (full file images using the diff --git a/tests/test_change_list.c b/tests/test_change_list.c index f9ad8a5..ef0f16d 100644 --- a/tests/test_change_list.c +++ b/tests/test_change_list.c @@ -195,7 +195,7 @@ static void test_format_C_padding_uses_transfer_algo() { {CHECKSUM_ALGO_NONE, 2}, }; for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) { - config->checksum_transfer_algo = cases[i].algo; + config->cli.checksum_transfer_algo = cases[i].algo; char expected[64]; size_t n = 0; expected[n++] = '['; diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index d86f8f6..e86aec7 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -730,7 +730,7 @@ static void test_parse_args_port_alias() { EXPECT_EQ_INT(cfg->server_port, 9000); /* The default port is 8080; the explicit bit is what lets --dry-run tell an explicit remote target from the default and route to the server. */ - EXPECT_TRUE(cfg->server_port_set); + EXPECT_TRUE(cfg->cli.server_port_set); config_delete(cfg); cfg = config_create(); @@ -738,7 +738,7 @@ static void test_parse_args_port_alias() { positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv_inline, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->server_port, 9001); - EXPECT_TRUE(cfg->server_port_set); + EXPECT_TRUE(cfg->cli.server_port_set); config_delete(cfg); cfg = config_create(); @@ -746,7 +746,7 @@ static void test_parse_args_port_alias() { positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv_long, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->server_port, 9002); - EXPECT_TRUE(cfg->server_port_set); + EXPECT_TRUE(cfg->cli.server_port_set); config_delete(cfg); } @@ -757,18 +757,18 @@ static void test_parse_args_server_host_sets_routing_bit() { Config* cfg = config_create(); int positional_args[2]; int positional_count = 0; - EXPECT_FALSE(cfg->server_host_set); + EXPECT_FALSE(cfg->cli.server_host_set); char* argv_space[] = {"fastsync", "--server-host", "example.test", "/src", "/dst"}; EXPECT_EQ_INT(parse_args(cfg, 5, argv_space, positional_args, &positional_count), 0); EXPECT_EQ_STR(cfg->server_host, "example.test"); - EXPECT_TRUE(cfg->server_host_set); + EXPECT_TRUE(cfg->cli.server_host_set); config_delete(cfg); cfg = config_create(); char* argv_inline[] = {"fastsync", "--server-host=example.test", "/src", "/dst"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv_inline, positional_args, &positional_count), 0); - EXPECT_TRUE(cfg->server_host_set); + EXPECT_TRUE(cfg->cli.server_host_set); config_delete(cfg); } @@ -1726,7 +1726,7 @@ static void test_parse_args_no_preserve_blocks_implicit_metadata() { EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); EXPECT_FALSE(cfg->use_metadata); - EXPECT_TRUE(cfg->metadata_explicitly_disabled); + EXPECT_TRUE(cfg->cli.metadata_explicitly_disabled); config_delete(cfg); } } @@ -1777,7 +1777,7 @@ static void test_parse_args_checksum_choice_rejects_unsupported() { int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); } } @@ -1817,7 +1817,7 @@ static void test_parse_args_checksum_choice_new_algos() { positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv4, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->checksum_algo, single[i]); - EXPECT_EQ_INT(cfg->checksum_transfer_algo, single[i]); + EXPECT_EQ_INT(cfg->cli.checksum_transfer_algo, single[i]); config_delete(cfg); } @@ -1827,7 +1827,7 @@ static void test_parse_args_checksum_choice_new_algos() { char* argv5[] = {"fastsync", "--cc=sha1,md4", "/checksum/src", "/dst"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv5, positional_args, &positional_count), 0); - EXPECT_EQ_INT(cfg->checksum_transfer_algo, (int)CHECKSUM_ALGO_SHA1); + EXPECT_EQ_INT(cfg->cli.checksum_transfer_algo, (int)CHECKSUM_ALGO_SHA1); EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_MD4); config_delete(cfg); @@ -1849,14 +1849,14 @@ static void test_parse_args_checksum_none_with_checksum_rejected() { int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); cfg = config_create(); char* argv2[] = {"fastsync", "--checksum", "--cc=md5,none", "/checksum/src", "/dst"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv2, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); /* "none" as the TRANSFER checksum with a real pre-transfer checksum is @@ -1958,7 +1958,7 @@ static void test_parse_args_compress_choice_parity() { int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); } } @@ -2078,6 +2078,9 @@ static void test_parse_args_temp_dir() { config_delete(cfg); } +/* --old-args is accepted for rsync CLI compatibility as a documented no-op (the + * remote server path is always safely quoted); it stores no Config field, so + * parsing it must simply succeed and leave the positional arguments intact. */ static void test_parse_args_old_args() { Config* cfg = config_create(); char* argv[] = {"fastsync", "--old-args", "/src", "/dst"}; @@ -2085,7 +2088,7 @@ static void test_parse_args_old_args() { int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); - EXPECT_TRUE(cfg->old_args); + EXPECT_EQ_INT(positional_count, 2); config_delete(cfg); } @@ -2719,7 +2722,7 @@ static void test_parse_args_compression_env_list() { cfg = config_create(); positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); unsetenv("RSYNC_COMPRESS_LIST"); } @@ -2734,7 +2737,7 @@ static void test_parse_args_checksum_env_list() { setenv("RSYNC_CHECKSUM_LIST", "md5", 1); EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0); EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_MD5); - EXPECT_EQ_INT(cfg->checksum_transfer_algo, (int)CHECKSUM_ALGO_MD5); + EXPECT_EQ_INT(cfg->cli.checksum_transfer_algo, (int)CHECKSUM_ALGO_MD5); config_delete(cfg); /* An explicit --cc wins. */ @@ -2750,7 +2753,7 @@ static void test_parse_args_checksum_env_list() { cfg = config_create(); positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1); - EXPECT_EQ_INT(cfg->cli_exit_code, 4); + EXPECT_EQ_INT(cfg->cli.cli_exit_code, 4); config_delete(cfg); unsetenv("RSYNC_CHECKSUM_LIST"); } @@ -4385,7 +4388,7 @@ static void test_parse_args_preserve_long_form() { } /* --no-perms/--no-times/--no-owner/--no-group (long and short) clear only - * their own attribute bit; they never set metadata_explicitly_disabled. */ + * their own attribute bit; they never set cli.metadata_explicitly_disabled. */ static void test_parse_args_preserve_negations() { struct { const char* arg; @@ -4413,7 +4416,7 @@ static void test_parse_args_preserve_negations() { bool expected = all[j] != cases[i].offset; EXPECT_TRUE(*(bool*)((char*)cfg + all[j]) == expected); } - EXPECT_FALSE(cfg->metadata_explicitly_disabled); + EXPECT_FALSE(cfg->cli.metadata_explicitly_disabled); /* -a's devices/specials keep the metadata frame on. */ EXPECT_TRUE(cfg->use_metadata); config_delete(cfg); @@ -4456,7 +4459,7 @@ static void test_parse_args_no_preserve_disables_bundle() { EXPECT_FALSE(cfg->preserve_times); EXPECT_FALSE(cfg->preserve_owner); EXPECT_FALSE(cfg->preserve_group); - EXPECT_TRUE(cfg->metadata_explicitly_disabled); + EXPECT_TRUE(cfg->cli.metadata_explicitly_disabled); EXPECT_TRUE(cfg->use_incremental); EXPECT_FALSE(cfg->use_metadata); config_delete(cfg); @@ -4509,7 +4512,7 @@ static void test_parse_args_incremental_implies_preserve() { EXPECT_EQ_INT(parse_args(cfg, 5, argv4, positional_args, &positional_count), 0); EXPECT_FALSE(cfg->preserve_perms); EXPECT_FALSE(cfg->preserve_times); - EXPECT_TRUE(cfg->metadata_explicitly_disabled); + EXPECT_TRUE(cfg->cli.metadata_explicitly_disabled); EXPECT_FALSE(cfg->use_metadata); config_delete(cfg); } diff --git a/tests/test_config.c b/tests/test_config.c index a801d70..4c87b34 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -2533,15 +2533,15 @@ static void test_config_derived_use_metadata() { /* Incremental/delta imply metadata unless --no-preserve disabled it. */ c->use_incremental = true; EXPECT_TRUE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = true; + c->cli.metadata_explicitly_disabled = true; EXPECT_FALSE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = false; + c->cli.metadata_explicitly_disabled = false; c->use_incremental = false; c->use_delta = true; EXPECT_TRUE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = true; + c->cli.metadata_explicitly_disabled = true; EXPECT_FALSE(config_derived_use_metadata(c)); - c->metadata_explicitly_disabled = false; + c->cli.metadata_explicitly_disabled = false; c->use_delta = false; /* Flags that must NOT imply metadata on their own. */ -- 2.54.0 From be20e836dedd465fc178243d4dbb4dad4d88818b Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:06:26 +0200 Subject: [PATCH 41/68] fix: correct throttle legacy resolution; add EXDEV temp-dir coverage --- RSYNC_COMPAT.md | 7 ++- src/shared/file_send.c | 2 +- src/shared/protocol.c | 13 ++-- src/shared/protocol.h | 13 ++-- tests/integration/test_temp_dir_exdev.py | 80 ++++++++++++++++++++++++ tests/test_protocol.c | 45 ++++++++++++- 6 files changed, 146 insertions(+), 14 deletions(-) create mode 100644 tests/integration/test_temp_dir_exdev.py diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 7657532..decd2d5 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -1013,7 +1013,12 @@ integration tests unless it is explicitly listed as a limitation. - **`--temp-dir` is confined to the receive root on the receiver:** a relative dir resolves below it; an absolute path or one containing `..` is rejected. - An `EXDEV` install falls back to a non-atomic copy instead of aborting. + An `EXDEV` install falls back to a non-atomic copy instead of aborting. (The + confined receiver path cannot be mount-tested in the CI container — no + `CAP_SYS_ADMIN` and unprivileged user namespaces are disabled — so the + cross-filesystem fallback is exercised end-to-end through the unconfined local + `--read-batch` apply against a `/dev/shm` scratch dir, in + `tests/integration/test_temp_dir_exdev.py`.) - **Deletion scoping:** the manifest carries the synchronized directories, so the extras walk only visits their subtrees; `--files-from` subsets no longer delete untransmitted paths outside the listed directories. diff --git a/src/shared/file_send.c b/src/shared/file_send.c index b5224bd..9f43724 100644 --- a/src/shared/file_send.c +++ b/src/shared/file_send.c @@ -185,7 +185,7 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta return false; } protocol_note_bytes_written((unsigned long long)sent); - protocol_throttle_bytes((size_t)sent); + protocol_throttle_bytes(file_descriptor, (size_t)sent); } close(fd); diff --git a/src/shared/protocol.c b/src/shared/protocol.c index fa1fff2..b6f8280 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -302,11 +302,14 @@ static ProtocolSession* legacy_session(int read_fd, int write_fd) { } /* Pace an out-of-band write that bypassed protocol_send_n_data (the plaintext - * sendfile fast path). The bound/legacy session is resolved exactly as - * send_n_data resolves it, so the same token-bucket state is throttled and the - * TLS and plaintext transports share identical --bwlimit semantics. */ -void protocol_throttle_bytes(size_t bytes) { - bw_throttle_session(legacy_session(-1, -1), bytes); + * sendfile fast path). The bound/legacy session is resolved exactly as the + * preceding send_n_data(fd, ...) resolved it, so the same token-bucket state is + * throttled and the TLS and plaintext transports share identical --bwlimit + * semantics. Passing the wire fd (rather than -1) is essential: the sendfile + * send left legacy_io_session.write_fd bound to it, so resolving with -1 would + * mismatch, re-initialize the session and hand out a second first-call burst. */ +void protocol_throttle_bytes(int file_descriptor, size_t bytes) { + bw_throttle_session(legacy_session(-1, file_descriptor), bytes); } bool send_n_data(int file_descriptor, const void* data, size_t data_size) { diff --git a/src/shared/protocol.h b/src/shared/protocol.h index d13c3f7..0bb0b42 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -225,11 +225,14 @@ unsigned long long protocol_bytes_written(void); unsigned long long protocol_bytes_read(void); void protocol_note_bytes_written(unsigned long long bytes); /* Apply --bwlimit pacing to bytes written outside protocol_send_n_data (the - * plaintext zero-copy sendfile fast path). Resolves the bound/legacy session - * exactly as send_n_data does and runs the same token-bucket throttle, so the - * sendfile transport is paced identically to the buffered/TLS paths. A no-op - * when the effective session has no bandwidth limit. */ -void protocol_throttle_bytes(size_t bytes); + * plaintext zero-copy sendfile fast path). `file_descriptor` is the wire fd + * the bytes were written to, so the legacy session is resolved exactly as the + * preceding send_n_data call resolved it (the bound TLS session still wins when + * set); resolving with the same fd avoids re-initializing the legacy session + * and granting a second first-call burst. Runs the same token-bucket throttle, + * so the sendfile transport is paced identically to the buffered/TLS paths. A + * no-op when the effective session has no bandwidth limit. */ +void protocol_throttle_bytes(int file_descriptor, size_t bytes); void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd); /* Transitional bridge for helpers whose signatures still carry only an fd. */ diff --git a/tests/integration/test_temp_dir_exdev.py b/tests/integration/test_temp_dir_exdev.py new file mode 100644 index 0000000..d9236dd --- /dev/null +++ b/tests/integration/test_temp_dir_exdev.py @@ -0,0 +1,80 @@ +"""End-to-end coverage for the `--temp-dir` EXDEV (cross-filesystem) fallback. + +`file_to_disk_secure_impl` installs a completed temp file with `renameat(2)`; +when the scratch dir lives on a different filesystem the rename fails with +`EXDEV` and the engine retries with no scratch dir, writing the file directly in +the destination directory (a non-atomic copy), matching rsync. + +The daemon receiver confines `--temp-dir` to the authorized receive root, so a +genuine cross-fs scratch there would require an in-root mount point. Bind/tmpfs +mounting is not permitted in the CI container (no `CAP_SYS_ADMIN`, and +unprivileged user namespaces are disabled), so this test reaches the exact same +code path through the local `--read-batch` apply instead: it has no +authorized-root confinement, so a relative `--temp-dir` that is a symlink to a +tmpfs (`/dev/shm`) is accepted and the final install then crosses filesystems. +""" +import os +import shutil +import subprocess +import sys + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import CLIENT_CMD, get_dest_received_dir + +TMPFS = "/dev/shm" + + +def _run(args): + return subprocess.run(CLIENT_CMD + args, capture_output=True, text=True, timeout=180) + + +def _read(path): + with open(path, "rb") as fh: + return fh.read() + + +def test_read_batch_temp_dir_cross_filesystem_fallback(tmp_path): + if not os.path.isdir(TMPFS): + pytest.skip("no /dev/shm tmpfs available to force a cross-filesystem install") + + source = tmp_path / "src" + dest = tmp_path / "dst" + source.mkdir() + dest.mkdir() + files = { + "payload.bin": bytes(range(256)) * 64, + "sub/nested.txt": b"nested exdev fallback\n" * 8, + } + for rel, data in files.items(): + full = source / rel + full.parent.mkdir(parents=True, exist_ok=True) + full.write_bytes(data) + + batch = tmp_path / "tree.batch" + r = _run(["--only-write-batch", str(batch), str(source)]) + assert r.returncode == 0, (r.stdout, r.stderr) + + # A cross-filesystem scratch dir, reached through a relative --temp-dir + # symlink (the local batch apply performs no authorized-root confinement). + scratch = os.path.join(TMPFS, "fastsync_exdev_%d" % os.getpid()) + shutil.rmtree(scratch, ignore_errors=True) + os.makedirs(scratch) + os.symlink(scratch, dest / "scratch") + try: + assert os.stat(scratch).st_dev != os.stat(dest).st_dev, ( + "scratch and destination share a filesystem; EXDEV cannot be exercised" + ) + r = _run(["--read-batch", str(batch), str(dest), "--temp-dir=scratch"]) + assert r.returncode == 0, (r.stdout, r.stderr) + # The engine must report the non-atomic cross-fs fallback rather than + # silently claiming an atomic install. + assert "different filesystem" in (r.stdout + r.stderr), (r.stdout, r.stderr) + # The tree is still byte-exact and the scratch dir is left clean. + received = get_dest_received_dir(str(dest), str(source)) + for rel, data in files.items(): + assert _read(os.path.join(received, rel)) == data, f"content mismatch for {rel}" + assert os.listdir(scratch) == [], "cross-fs temp file was not cleaned up" + finally: + shutil.rmtree(scratch, ignore_errors=True) diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 483ee82..3992256 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -1,6 +1,8 @@ #include "protocol.h" #include "test_utils.h" +#include #include +#include #include #include #include @@ -715,7 +717,7 @@ static void test_protocol_throttle_bytes_paces() { struct timespec start; clock_gettime(CLOCK_MONOTONIC, &start); - protocol_throttle_bytes(150000); + protocol_throttle_bytes(-1, 150000); struct timespec now; clock_gettime(CLOCK_MONOTONIC, &now); long long elapsed_ms = @@ -736,7 +738,7 @@ static void test_protocol_throttle_bytes_unlimited() { struct timespec start; clock_gettime(CLOCK_MONOTONIC, &start); - protocol_throttle_bytes(100000000ULL); + protocol_throttle_bytes(-1, 100000000ULL); struct timespec now; clock_gettime(CLOCK_MONOTONIC, &now); long long elapsed_ms = @@ -746,6 +748,44 @@ static void test_protocol_throttle_bytes_unlimited() { protocol_session_unbind(); } +/* Regression for the plaintext sendfile path: it calls protocol_throttle_bytes() + * immediately after send_n_data(), which already bound legacy_io_session.write_fd + * to the wire fd. Resolving the throttle session with (read=-1, write=-1) + * mismatched that fd and re-initialized the legacy session, granting a *second* + * first-call burst and discarding the accumulated debt. This drives the same + * sequence and asserts the debt from send_n_data carries into the throttle. */ +static void test_protocol_throttle_bytes_legacy_same_session() { + const size_t payload = 150000; /* 1.5x the 100 KB burst at --bwlimit=1 MB/s */ + unsigned char* buffer = malloc(payload); + EXPECT_TRUE(buffer != NULL); + memset(buffer, 0, payload); + + io_set_fds(-1, -1); + io_set_bwlimit(1000000ULL); + + int fd = open("/dev/null", O_WRONLY); + EXPECT_TRUE(fd >= 0); + + struct timespec start; + clock_gettime(CLOCK_MONOTONIC, &start); + /* send_n_data() consumes the whole 100 KB burst and sleeps ~50 ms. */ + EXPECT_TRUE(send_n_data(fd, buffer, payload)); + /* The throttle must share that session, so the 150 KB is all debt and sleeps + ~150 ms (total ~200 ms). A re-initialized session would hand out a fresh + 100 KB burst and sleep only ~50 ms (total ~100 ms). */ + protocol_throttle_bytes(fd, payload); + struct timespec now; + clock_gettime(CLOCK_MONOTONIC, &now); + long long elapsed_ms = + (now.tv_sec - start.tv_sec) * 1000LL + (now.tv_nsec - start.tv_nsec) / 1000000LL; + EXPECT_TRUE(elapsed_ms >= 150); + + close(fd); + free(buffer); + io_set_bwlimit(0); + io_set_fds(-1, -1); +} + void test_protocol() { test_send_receive_n_data(); test_send_receive_n_data_zero(); @@ -778,4 +818,5 @@ void test_protocol() { test_data_create_starts_uncharged_and_unowned(); test_protocol_throttle_bytes_paces(); test_protocol_throttle_bytes_unlimited(); + test_protocol_throttle_bytes_legacy_same_session(); } -- 2.54.0 From e98729f00e5faab7cc4a55d8670e3ed4976e3345 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:24:39 +0200 Subject: [PATCH 42/68] refactor: const-correct delete-manifest API; apply clang-format --- src/client/client_send.c | 6 +++--- src/server/receiver.c | 2 +- src/shared/config.c | 3 ++- src/shared/delete_commit.c | 34 +++++++++++++++++----------------- src/shared/delete_commit.h | 27 +++++++++++++-------------- 5 files changed, 36 insertions(+), 36 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index 2f3422b..3b9d623 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1653,7 +1653,7 @@ static bool send_files_run(Config* config, SendFilesState* state) { /* Completion tail: send the late delete manifest and captured directory times, * finalize the receiver handshake, remove transferred sources and report stats. * Returns the rsync-compatible exit code. */ -static int send_files_finalize(Config* config, SendFilesState* state) { +static int send_files_finalize(const Config* config, SendFilesState* state) { Client* client = state->client; if (directory_scanner_failed(state->scanner)) return 1; @@ -1916,8 +1916,8 @@ int send_files_multithreaded(Config* config) { now_mono.tv_nsec = 0; } context->stop_condition = - stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->cli.stop_at_set, - config->stop_at, now_mono); + stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, + config->cli.stop_at_set, config->stop_at, now_mono); bool collect_excluded = config->use_delete && !config->delete_excluded; unsigned long long pre_scan_non_dir = 0; if (config->use_delete) { diff --git a/src/server/receiver.c b/src/server/receiver.c index 10c4c45..50f4e0b 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -384,7 +384,7 @@ static ReceiverStep receiver_handle_mkdir(ReceiverPendingState* state) { return RECEIVER_STEP_NEXT; } -static ReceiverStep receiver_handle_dir_times(ReceiverPendingState* state) { +static ReceiverStep receiver_handle_dir_times(const ReceiverPendingState* state) { if (!receiver_process_dir_times(state->fd, state->config, state->sink)) return RECEIVER_STEP_ERROR; return RECEIVER_STEP_NEXT; diff --git a/src/shared/config.c b/src/shared/config.c index eea3974..223daad 100644 --- a/src/shared/config.c +++ b/src/shared/config.c @@ -341,7 +341,8 @@ bool config_derived_use_metadata(const Config* config) { config->chown_uid_set || config->chown_gid_set || config->usermap_count > 0 || config->groupmap_count > 0 || config->update) return true; - return (config->use_incremental || config->use_delta) && !config->cli.metadata_explicitly_disabled; + return (config->use_incremental || config->use_delta) && + !config->cli.metadata_explicitly_disabled; } bool config_has_basis(const Config* config) { diff --git a/src/shared/delete_commit.c b/src/shared/delete_commit.c index 07f72db..0a2b66f 100644 --- a/src/shared/delete_commit.c +++ b/src/shared/delete_commit.c @@ -136,7 +136,7 @@ typedef struct { alternate basis directories are never destination content and are skipped at any depth. Returns true unless a traversal/unlink error aborted the walk; the budget's limit_hit/skipped fields report a cap-stopped run. */ -static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest* manifest, +static bool delete_extras_budgeted_observed(const Config* config, const DeleteManifest* manifest, DeleteBudgetState* budget, DeletePathObserver observer, void* observer_context) { if (!config || !manifest || !manifest->keeps) @@ -177,7 +177,7 @@ static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest return true; } -static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest, +static bool delete_extras_budgeted(const Config* config, const DeleteManifest* manifest, DeleteBudgetState* budget) { return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL); } @@ -215,7 +215,8 @@ static void prefixed_delete_observer(void* context, const char* rel) { --max-delete budget: once it is exhausted the remaining requests are skipped and counted. Returns false only on a genuine error (a confinement failure on a validated path or an I/O error), which fails the run. */ -static bool delete_missing_args_budgeted_observed(const Config* config, DeleteManifest* manifest, +static bool delete_missing_args_budgeted_observed(const Config* config, + const DeleteManifest* manifest, DeleteBudgetState* budget, DeletePathObserver observer, void* observer_context) { @@ -376,8 +377,8 @@ static bool delete_missing_args_budgeted_observed(const Config* config, DeleteMa /* Public wrappers used outside the commit path (and by unit tests): no --max-delete budget. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out) { +bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest, + ArrayList* out, size_t* count_out) { if (count_out) *count_out = 0; if (!config || !manifest || !manifest->keeps || !out) @@ -391,30 +392,28 @@ bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, return ok; } -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) { +bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest) { DeleteBudgetState budget = { .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; return delete_extras_budgeted(config, manifest, &budget); } -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) { +bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest) { DeleteBudgetState budget = { .max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false}; return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL); } -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, +bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, size_t* skipped, bool* limit_hit) { return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted, skipped, limit_hit, NULL, NULL); } -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context) { +bool manifest_delete_missing_args_limited_observed( + const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context) { DeleteBudgetState budget = { .max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false}; bool ok = @@ -435,17 +434,18 @@ bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteM removal fail). The ordinary extras walk then runs when --delete is active. Both draw from one --max-delete budget; the result reports a cap-stopped (partial) commit distinctly so the client can exit 25 like rsync. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest) { +DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest) { return manifest_delete_all_counted(config, manifest, NULL); } -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, +DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest, size_t* deleted) { return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL); } -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, +DeleteCommitResult manifest_delete_all_observed(const Config* config, + const DeleteManifest* manifest, size_t* deleted, + DeletePathObserver observer, void* observer_context) { if (deleted) *deleted = 0; diff --git a/src/shared/delete_commit.h b/src/shared/delete_commit.h index 16f00ae..ba760f1 100644 --- a/src/shared/delete_commit.h +++ b/src/shared/delete_commit.h @@ -44,7 +44,7 @@ DeleteManifest* receive_manifest_entries(int fd); protected-prefix skips). `--max-delete` and `--force` are honored here. The caller decides WHEN to run it based on the negotiated delete timing. Returns false (and the transfer fails) when the deletion cannot be committed. */ -bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); +bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest); /* --delete-missing-args exact-path deletions: remove each destination mirror in `manifest->missing` (never blocked by the protected prefixes, staging dir and basis dirs excluded). A regular file/symlink is unlinked; an empty @@ -53,22 +53,20 @@ bool manifest_delete_extras(const Config* config, DeleteManifest* manifest); parity). A missing path is a no-op. Returns false only on a genuine confinement or I/O error (the run then fails); tolerated per-path cases are reported and skipped. */ -bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest); +bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest); /* Budgeted form of manifest_delete_missing_args for the per-directory delete session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited) and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set when the budget stopped the pass with entries left over. Returns false only on a genuine deletion error. */ -bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest, +bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, size_t* skipped, bool* limit_hit); /* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may be NULL) is invoked for every destination-relative path truly removed. */ -bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest, - size_t max_delete, size_t* deleted, - size_t* skipped, bool* limit_hit, - DeletePathObserver observer, - void* observer_context); +bool manifest_delete_missing_args_limited_observed( + const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted, + size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context); /* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's partial --max-delete result: the budget allowed some deletions and the rest were skipped (the run still stores all file data but the client exits 25). */ @@ -84,15 +82,16 @@ typedef enum { share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */ -DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest); +DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest); /* Like manifest_delete_all, but reports how many destination entries the commit removed (for the end-of-transfer wire stats). `deleted` may be NULL. */ -DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest, +DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest, size_t* deleted); /* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL) is invoked for every destination-relative path truly removed. */ -DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest, - size_t* deleted, DeletePathObserver observer, +DeleteCommitResult manifest_delete_all_observed(const Config* config, + const DeleteManifest* manifest, size_t* deleted, + DeletePathObserver observer, void* observer_context); /* -n/--dry-run --delete would-delete reporting: walk the destination exactly as @@ -100,7 +99,7 @@ DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteMani WOULD be removed to `out`, without touching disk. Uses the same staging-dir, basis-dir and protected-prefix skips as the real commit. Returns true on a clean walk; `*count_out` receives the number of paths appended. */ -bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out, - size_t* count_out); +bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest, + ArrayList* out, size_t* count_out); #endif -- 2.54.0 From 1167e7970b2a60f955cb809ee795c5e2feec7df2 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 14:52:36 +0200 Subject: [PATCH 43/68] refactor(delete): rename basis helper to delete_basis_relative --- src/shared/delete.c | 4 ++-- src/shared/delete.h | 2 +- tests/test_file.c | 12 ++++++------ 3 files changed, 9 insertions(+), 9 deletions(-) diff --git a/src/shared/delete.c b/src/shared/delete.c index 3e4fb62..a47cc3b 100644 --- a/src/shared/delete.c +++ b/src/shared/delete.c @@ -550,7 +550,7 @@ bool delete_extras(const char* dest_root, const ArrayList* manifest) { its root-relative form, and one outside the root returns NULL (the walk cannot reach it, and it is not protected data beneath the root). Exposed so tests can exercise the root-of-"/" child mapping directly. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path) { +char* delete_basis_relative(const Config* config, const char* path) { if (!path) return NULL; if (path[0] != '/') @@ -616,7 +616,7 @@ bool delete_skips_build(const Config* config, const ArrayList* protected_paths, if (basis_root_relative) { /* An absolute basis outside the receive root is unreachable by this walk, so it contributes no protection prefix (and no slot). */ - char* relative = file_receive_basis_delete_relative(config, config->basis_dirs[i].path); + char* relative = delete_basis_relative(config, config->basis_dirs[i].path); if (!relative) continue; out->owned_prefixes[i] = relative; diff --git a/src/shared/delete.h b/src/shared/delete.h index 235db8c..f0921a1 100644 --- a/src/shared/delete.h +++ b/src/shared/delete.h @@ -146,6 +146,6 @@ void delete_skips_free(DeleteSkipSet* set); /* Convert one basis-directory path to the receive-root-relative protection prefix the delete walker uses (NULL when it lies outside the root). Exposed for unit tests of the root-of-"/" and normalization edge cases. */ -char* file_receive_basis_delete_relative(const Config* config, const char* path); +char* delete_basis_relative(const Config* config, const char* path); #endif diff --git a/tests/test_file.c b/tests/test_file.c index 69912f0..efb87a7 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -2255,26 +2255,26 @@ static void test_basis_delete_relative_root_slash() { EXPECT_NOT_NULL(cfg); cfg->receive_root_directory = str_dup("/"); - char* rel = file_receive_basis_delete_relative(cfg, "/a"); + char* rel = delete_basis_relative(cfg, "/a"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "a"); free(rel); - rel = file_receive_basis_delete_relative(cfg, "/a/b"); + rel = delete_basis_relative(cfg, "/a/b"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "a/b"); free(rel); /* The root itself is not a child. */ - EXPECT_NULL(file_receive_basis_delete_relative(cfg, "/")); + EXPECT_NULL(delete_basis_relative(cfg, "/")); /* A relative entry is already root-relative. */ - rel = file_receive_basis_delete_relative(cfg, "x/y"); + rel = delete_basis_relative(cfg, "x/y"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "x/y"); free(rel); /* An absolute path outside a non-"/" root is unreachable. */ free(cfg->receive_root_directory); cfg->receive_root_directory = str_dup("/root"); - EXPECT_NULL(file_receive_basis_delete_relative(cfg, "/other/a")); - rel = file_receive_basis_delete_relative(cfg, "/root/a"); + EXPECT_NULL(delete_basis_relative(cfg, "/other/a")); + rel = delete_basis_relative(cfg, "/root/a"); EXPECT_NOT_NULL(rel); EXPECT_EQ_STR(rel, "a"); free(rel); -- 2.54.0 From b7cb213c8cf10328811d92fcb80b99acf579a398 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 16:03:18 +0200 Subject: [PATCH 44/68] fix: FROM name globs for identity maps; transport fallback tests (#294, #219) --- src/shared/identity.c | 169 +++++++++++++++++++++++++++++++--- src/shared/identity.h | 10 +- tests/test_client_cli.c | 174 ++++++++++++++++++++++++++++++++++- tests/test_transport_tcp.c | 182 ++++++++++++++++++++++++++++++++++++- tests/test_transport_tls.c | 47 ++++++++++ 5 files changed, 567 insertions(+), 15 deletions(-) diff --git a/src/shared/identity.c b/src/shared/identity.c index 1f717d5..31b205b 100644 --- a/src/shared/identity.c +++ b/src/shared/identity.c @@ -3,6 +3,7 @@ #include "utils.h" #include #include +#include #include #include #include @@ -12,6 +13,8 @@ #include #include +static bool identity_id_fits_int32(unsigned long id); + /* The active identity snapshot lives in a per-process global. The TCP server * forks one child process per connection, so a connection never shares this * with another; within a connection the multithreaded receiver reads it without @@ -419,17 +422,10 @@ static int identity_parse_from(const char* token, bool is_group, int32_t* out_fr /* Not a numeric LOW-HIGH range: fall through and treat as a name (a * hyphenated account name like "wayne-smith" must still resolve). */ } - /* A sender-side name. A wildcard other than the bare '*' is matched by rsync - * against the sender's names; because FastSync transmits numeric ids only, the - * receiver cannot evaluate it, so reject rather than silently mis-match. */ - if (identity_token_has_glob(token)) { - log_message(LOG_LEVEL_ERROR, - "%smap FROM '%s': name wildcards other than '*' are not supported " - "(FastSync transmits numeric ids, so sender names are unavailable on the " - "receiver)", - is_group ? "--group" : "--user", token); - return -1; - } + /* A sender-side name. A FROM name wildcard other than the bare '*' is handled + * by identity_expand_from_glob() in the caller (it expands against the + * sender's account database at CLI-parse time), so this function only sees the + * bare '*' or a literal name here. */ int32_t id; if (identity_resolve_token(token, is_group, &id) != 0) return -1; @@ -486,6 +482,138 @@ static int identity_append_rule(IdentityMap** map, int* count, const IdentityMap return 0; } +/* True when `lo` and `hi` are adjacent ids (no overflow at INT32_MAX). */ +static bool identity_ids_adjacent(int32_t lo, int32_t hi) { + return lo < INT32_MAX && hi == lo + 1; +} + +static int identity_id_cmp(const void* a, const void* b) { + int32_t x = *(const int32_t*)a; + int32_t y = *(const int32_t*)b; + return (x > y) - (x < y); +} + +static bool identity_ids_push(int32_t** ids, size_t* count, size_t* cap, int32_t id) { + if (*count == *cap) { + size_t grown_cap = *cap ? *cap * 2 : 16; + int32_t* grown = realloc(*ids, grown_cap * sizeof(int32_t)); + if (!grown) + return false; + *ids = grown; + *cap = grown_cap; + } + (*ids)[(*count)++] = id; + return true; +} + +/* Expand a FROM name wildcard (rsync's match against sender-side account names) + * into one rule per contiguous run of matching numeric ids, all sharing the same + * TO side. FastSync transmits numeric ids only, so the wildcard must be + * resolved here -- at CLI-parse time -- against the SENDER's passwd/group + * database; the receiver has no sender names to match. Contiguous matched ids + * are collapsed into a single LOW-HIGH range (a range of adjacent ids contains + * exactly the ids it spans, so this is semantically exact). Returns 0 on + * success, -1 on an allocation failure, a wildcard that matches no sender + * account, or an expansion that would push the map past MAX_IDENTITY_MAP. */ +static int identity_expand_from_glob(Config* config, const char* glob, bool is_group, + const IdentityMap* to_rule) { + const char* optname = is_group ? "--groupmap" : "--usermap"; + size_t cap = 0; + size_t n = 0; + int32_t* ids = NULL; + bool alloc_failed = false; + + if (is_group) { + setgrent(); + struct group* gr; + while ((gr = getgrent()) != NULL) { + if (fnmatch(glob, gr->gr_name, 0) != 0) + continue; + if (!identity_id_fits_int32((unsigned long)gr->gr_gid)) + continue; + if (!identity_ids_push(&ids, &n, &cap, (int32_t)gr->gr_gid)) { + alloc_failed = true; + break; + } + } + endgrent(); + } else { + setpwent(); + struct passwd* pw; + while ((pw = getpwent()) != NULL) { + if (fnmatch(glob, pw->pw_name, 0) != 0) + continue; + if (!identity_id_fits_int32((unsigned long)pw->pw_uid)) + continue; + if (!identity_ids_push(&ids, &n, &cap, (int32_t)pw->pw_uid)) { + alloc_failed = true; + break; + } + } + endpwent(); + } + + if (alloc_failed) { + free(ids); + log_message(LOG_LEVEL_ERROR, "%s: memory allocation failed expanding FROM '%s'", optname, glob); + return -1; + } + if (n == 0) { + free(ids); + log_message(LOG_LEVEL_ERROR, "%s FROM '%s': no source account name matches the wildcard", + optname, glob); + return -1; + } + + qsort(ids, n, sizeof(int32_t), identity_id_cmp); + size_t unique = 0; + for (size_t i = 0; i < n; i++) { + if (unique == 0 || ids[unique - 1] != ids[i]) + ids[unique++] = ids[i]; + } + n = unique; + + int runs = 0; + for (size_t i = 0; i < n; i++) { + if (i == 0 || !identity_ids_adjacent(ids[i - 1], ids[i])) + runs++; + } + + IdentityMap** map = is_group ? &config->groupmap : &config->usermap; + int* count = is_group ? &config->groupmap_count : &config->usermap_count; + if (*count > MAX_IDENTITY_MAP - runs) { + log_message(LOG_LEVEL_ERROR, + "%s FROM '%s': the name wildcard expands to %d rule(s), which would exceed " + "the maximum of %d map rules", + optname, glob, runs, MAX_IDENTITY_MAP); + free(ids); + return -1; + } + + for (size_t i = 0; i < n;) { + size_t j = i; + while (j + 1 < n && identity_ids_adjacent(ids[j], ids[j + 1])) + j++; + IdentityMap rule; + rule.from = ids[i]; + rule.from_hi = ids[j]; + rule.to = to_rule->to; + rule.to_name = to_rule->to_name ? str_dup(to_rule->to_name) : NULL; + if (to_rule->to_name && !rule.to_name) { + free(ids); + return -1; + } + if (identity_append_rule(map, count, &rule) != 0) { + free(rule.to_name); + free(ids); + return -1; + } + i = j + 1; + } + free(ids); + return 0; +} + int identity_parse_map(Config* config, const char* value, bool is_group) { if (!config || !value || *value == '\0') { log_message(LOG_LEVEL_ERROR, "%smap requires a value", is_group ? "--group" : "--user"); @@ -509,6 +637,25 @@ int identity_parse_map(Config* config, const char* value, bool is_group) { char* to_token = colon + 1; IdentityMap parsed; memset(&parsed, 0, sizeof(parsed)); + /* A FROM name wildcard (anything with a glob metacharacter other than the + * bare '*') is expanded against the sender's account database here, while + * the sender's passwd/group DB is still available; the resulting numeric + * rules travel on the wire like an explicit list. The TO side is parsed + * first so every expanded rule shares it. */ + if (strcmp(from_token, "*") != 0 && identity_token_has_glob(from_token)) { + if (identity_parse_to(to_token, is_group, &parsed.to, &parsed.to_name) != 0) { + log_message(LOG_LEVEL_ERROR, "%s could not parse TO '%s' in '%s'", optname, to_token, + value); + free(list); + return -1; + } + if (identity_expand_from_glob(config, from_token, is_group, &parsed) != 0) { + free(parsed.to_name); + free(list); + return -1; + } + continue; + } if (identity_parse_from(from_token, is_group, &parsed.from, &parsed.from_hi) != 0) { log_message(LOG_LEVEL_ERROR, "%s could not resolve FROM '%s' in '%s' (a name must exist on the " diff --git a/src/shared/identity.h b/src/shared/identity.h index 327205c..a7548c2 100644 --- a/src/shared/identity.h +++ b/src/shared/identity.h @@ -25,8 +25,14 @@ /* Parse one --usermap= / --groupmap= value (comma-separated FROM:TO rules, * first match wins) into config->usermap / config->groupmap. is_group selects - * the group tables and name databases. Returns 0 on success, -1 on a - * malformed spec or an unresolvable name (never a silent no-op). */ + * the group tables and name databases. A FROM name wildcard (containing `*`, + * `?` or `[...]`, but not the bare `*`) is expanded against the SENDER's + * account database at parse time into one or more numeric id/range rules + * (contiguous ids collapse to a range) sharing the same TO, because only + * numeric ids cross the wire; the expansion is capped at MAX_IDENTITY_MAP and a + * wildcard matching no account is an error. Returns 0 on success, -1 on a + * malformed spec, an unresolvable name, an unmatched wildcard, or a map that + * would exceed MAX_IDENTITY_MAP (never a silent no-op). */ int identity_parse_map(Config* config, const char* value, bool is_group); /* Parse --chown=USER:GROUP. Supports USER:GROUP, USER (owner only), :GROUP diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index e86aec7..e6c91b8 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -10,6 +10,8 @@ #include "protocol.h" #include "test_utils.h" #include "utils.h" +#include +#include #include #include #include @@ -3537,6 +3539,172 @@ static void test_parse_args_usermap_rsync_forms() { config_delete(cfg); } +/* Independent oracle for the FROM name-glob tests: enumerate the sender's + * account database and fill `ids` with the DISTINCT ids whose name matches + * `glob`, sorted ascending. Returns the count (bounded by `max`). */ +static int cli_collect_glob_ids(const char* glob, bool is_group, int32_t* ids, int max) { + int n = 0; + if (is_group) { + setgrent(); + struct group* gr; + while ((gr = getgrent()) != NULL) { + if (fnmatch(glob, gr->gr_name, 0) != 0) + continue; + if ((unsigned long)gr->gr_gid > (unsigned long)INT32_MAX) + continue; + int32_t id = (int32_t)gr->gr_gid; + bool dup = false; + for (int i = 0; i < n; i++) + if (ids[i] == id) + dup = true; + if (!dup && n < max) + ids[n++] = id; + } + endgrent(); + } else { + setpwent(); + struct passwd* pw; + while ((pw = getpwent()) != NULL) { + if (fnmatch(glob, pw->pw_name, 0) != 0) + continue; + if ((unsigned long)pw->pw_uid > (unsigned long)INT32_MAX) + continue; + int32_t id = (int32_t)pw->pw_uid; + bool dup = false; + for (int i = 0; i < n; i++) + if (ids[i] == id) + dup = true; + if (!dup && n < max) + ids[n++] = id; + } + endpwent(); + } + for (int i = 1; i < n; i++) { + int32_t key = ids[i]; + int j = i - 1; + while (j >= 0 && ids[j] > key) { + ids[j + 1] = ids[j]; + j--; + } + ids[j + 1] = key; + } + return n; +} + +static int cli_count_runs(const int32_t* ids, int n) { + int runs = 0; + for (int i = 0; i < n; i++) { + if (i == 0 || ids[i - 1] == INT32_MAX || ids[i] != ids[i - 1] + 1) + runs++; + } + return runs; +} + +/* #294: a FROM name wildcard must expand, at CLI-parse time, against the + * sender's account database into numeric id/range rules. Prefer a prefix that + * matches >=2 DISTINCT NON-contiguous ids (exercising multi-rule expansion); if + * no such prefix exists on this host, fall back to one whose ids are contiguous + * (exercising range collapse). The expected rules are derived independently by + * enumerating the same database. */ +static void test_parse_args_identity_map_from_name_glob(bool is_group) { + int32_t ids[512]; + int chosen_n = 0; + int chosen_runs = 0; + char chosen_c = 0; + for (char c = 'a'; c <= 'z'; c++) { + const char glob[3] = {c, '*', '\0'}; + int n = cli_collect_glob_ids(glob, is_group, ids, (int)(sizeof(ids) / sizeof(ids[0]))); + if (n < 2) + continue; + int runs = cli_count_runs(ids, n); + if (runs >= 2 || chosen_c == 0) { + chosen_c = c; + chosen_n = n; + chosen_runs = runs; + } + if (runs >= 2) + break; + } + if (chosen_c == 0) + return; /* no multi-match prefix on this host (skipped, not failed) */ + + const char glob[3] = {chosen_c, '*', '\0'}; + chosen_n = cli_collect_glob_ids(glob, is_group, ids, (int)(sizeof(ids) / sizeof(ids[0]))); + chosen_runs = cli_count_runs(ids, chosen_n); + EXPECT_TRUE(chosen_n >= 2); + + char map_value[16]; + snprintf(map_value, sizeof(map_value), "%s:@0", glob); + Config* cfg = config_create(); + char* argv[] = {"fastsync", is_group ? "--groupmap" : "--usermap", map_value, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + + int got = is_group ? cfg->groupmap_count : cfg->usermap_count; + EXPECT_EQ_INT(got, chosen_runs); + const IdentityMap* map = is_group ? cfg->groupmap : cfg->usermap; + /* Every matched id is covered by some expanded rule. */ + for (int i = 0; i < chosen_n; i++) { + bool covered = false; + for (int r = 0; r < got; r++) + if (ids[i] >= map[r].from && ids[i] <= map[r].from_hi) + covered = true; + EXPECT_TRUE(covered); + } + /* Every id inside every expanded range is one the glob actually matched, so + * the range collapse cannot over-match a name that does not fit the glob. */ + for (int r = 0; r < got; r++) { + EXPECT_EQ_INT(map[r].to, 0); + for (int32_t v = map[r].from; v <= map[r].from_hi; v++) { + bool expected = false; + for (int i = 0; i < chosen_n; i++) + if (ids[i] == v) + expected = true; + EXPECT_TRUE(expected); + if (v == INT32_MAX) + break; + } + } + config_delete(cfg); +} + +static void test_parse_args_usermap_from_name_glob() { + test_parse_args_identity_map_from_name_glob(false); +} + +static void test_parse_args_groupmap_from_name_glob() { + test_parse_args_identity_map_from_name_glob(true); +} + +/* #294: an expansion that would push the map past MAX_IDENTITY_MAP must fail + * with a clear error rather than silently truncating. Prefill the map to the + * cap and then add a wildcard guaranteed to match at least the current user. */ +static void test_parse_args_identity_map_from_name_glob_over_cap() { + const struct passwd* self = getpwuid(geteuid()); + if (!self || self->pw_name[0] == '\0') + return; + char glob[8]; + snprintf(glob, sizeof(glob), "%c*", self->pw_name[0]); + + size_t need = (size_t)MAX_IDENTITY_MAP * 6 + strlen(glob) + 4 + 1; + char* value = malloc(need); + if (!value) + return; + size_t off = 0; + for (int i = 0; i < MAX_IDENTITY_MAP; i++) + off += (size_t)snprintf(value + off, need - off, "@0:@0,"); + snprintf(value + off, need - off, "%s:@0", glob); + + Config* cfg = config_create(); + char* argv[] = {"fastsync", "--usermap", value, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), -1); + config_delete(cfg); + free(value); +} + /* #294: rsync refuses to mix --chown with --usermap/--groupmap on the same * side (either order). --chown=USER conflicts with a prior --usermap; * --chown=:GROUP conflicts with a prior --groupmap; the opposite side is fine. */ @@ -3716,7 +3884,8 @@ static void test_parse_args_rejects_malformed_identity() { {"--usermap", "definitely_not_a_real_user_zzz:@1"}, {"--usermap", "0-"}, {"--usermap", "5-2:@1"}, - {"--usermap", "roo*:@1"}, + {"--usermap", "zzz_definitely_no_such_user_glob_zzz*:@1"}, + {"--groupmap", "zzz_definitely_no_such_group_glob_zzz*:@1"}, {"--groupmap", "@1"}, {"--groupmap", "no_such_group_qqq:x"}, {"--chown", "a:b:c"}, @@ -4882,6 +5051,9 @@ void test_client_cli() { test_parse_args_groupmap(); test_parse_args_usermap_name_resolution(); test_parse_args_usermap_rsync_forms(); + test_parse_args_usermap_from_name_glob(); + test_parse_args_groupmap_from_name_glob(); + test_parse_args_identity_map_from_name_glob_over_cap(); test_parse_args_identity_map_chown_conflict(); test_parse_args_chown(); test_parse_args_copy_as(); diff --git a/tests/test_transport_tcp.c b/tests/test_transport_tcp.c index 825a6c4..6dd09f5 100644 --- a/tests/test_transport_tcp.c +++ b/tests/test_transport_tcp.c @@ -3,11 +3,13 @@ #include "test_utils.h" #include "transport_tcp.h" #include +#include #include #include #include -#include #include +#include +#include /* -4/-6 map to a getaddrinfo ai_family hint: -4 -> AF_INET, -6 -> AF_INET6, * and neither -> AF_UNSPEC. Both flags together are rejected earlier (in @@ -246,6 +248,180 @@ static void test_tcp_nodelay_default_and_override() { server_delete(&s); } +/* Count the process's open descriptors via /proc/self/fd. The opendir + * descriptor is itself counted and closed before returning, so repeated calls + * are consistent and a before/after delta reflects only the code under test. */ +static int count_open_fds(void) { + DIR* dir = opendir("/proc/self/fd"); + if (!dir) + return -1; + int count = 0; + const struct dirent* ent; + while ((ent = readdir(dir)) != NULL) { + if (strcmp(ent->d_name, ".") == 0 || strcmp(ent->d_name, "..") == 0) + continue; + count++; + } + closedir(dir); + return count; +} + +/* Bind + listen on the SECOND address getaddrinfo returns for "localhost", so + * the first candidate is connection-refused and the shared connect loop must + * fall back to a later one. Returns the listener fd and its port, or -1 when + * this host does not resolve localhost to at least two addresses (the test then + * skips rather than claiming coverage it does not have). */ +static int bind_second_localhost_address(int* out_port) { + struct addrinfo hints; + memset(&hints, 0, sizeof(hints)); + hints.ai_family = AF_UNSPEC; + hints.ai_socktype = SOCK_STREAM; + struct addrinfo* res = NULL; + if (getaddrinfo("localhost", "0", &hints, &res) != 0 || !res) + return -1; + const struct addrinfo* chosen = res->ai_next; + if (!chosen) { + freeaddrinfo(res); + return -1; + } + int fd = socket(chosen->ai_family, chosen->ai_socktype, chosen->ai_protocol); + if (fd < 0) { + freeaddrinfo(res); + return -1; + } + int opt = 1; + setsockopt(fd, SOL_SOCKET, SO_REUSEADDR, &opt, sizeof(opt)); + if (bind(fd, chosen->ai_addr, chosen->ai_addrlen) != 0 || listen(fd, 1) != 0) { + close(fd); + freeaddrinfo(res); + return -1; + } + struct sockaddr_storage bound; + socklen_t bound_len = sizeof(bound); + if (getsockname(fd, (struct sockaddr*)&bound, &bound_len) != 0) { + close(fd); + freeaddrinfo(res); + return -1; + } + if (bound.ss_family == AF_INET6) + *out_port = ntohs(((struct sockaddr_in6*)&bound)->sin6_port); + else + *out_port = ntohs(((struct sockaddr_in*)&bound)->sin_port); + freeaddrinfo(res); + return fd; +} + +/* #219 AC3: when the first getaddrinfo candidate is refused, the connect loop + * must fall back to the next address and end with exactly ONE open descriptor + * (proving the failed attempt's fd was closed before the retry). */ +static void test_tcp_connect_falls_back_to_next_address() { + int port = 0; + int listener = bind_second_localhost_address(&port); + if (listener < 0) + return; /* localhost is single-address on this host: cannot exercise fallback */ + int before = count_open_fds(); + Client* c = client_create(); + EXPECT_NOT_NULL(c); + EXPECT_TRUE(client_connect(c, "localhost", port)); + EXPECT_TRUE(c->file_descriptor >= 0); + if (before >= 0) + EXPECT_EQ_INT(count_open_fds(), before + 1); + client_disconnect(c); + if (before >= 0) + EXPECT_EQ_INT(count_open_fds(), before); + client_delete(c); + close(listener); +} + +/* #219 AC3: a connect that fails on every candidate leaves at most one + * descriptor (the last failed attempt) and none after client_disconnect. */ +static void test_tcp_connect_failed_attempts_do_not_leak_fds() { + /* Reserve an ephemeral port, then close it: connecting to it must fail. */ + int probe = socket(AF_INET, SOCK_STREAM, 0); + EXPECT_TRUE(probe >= 0); + struct sockaddr_in addr; + memset(&addr, 0, sizeof(addr)); + addr.sin_family = AF_INET; + addr.sin_addr.s_addr = htonl(INADDR_LOOPBACK); + addr.sin_port = 0; + EXPECT_EQ_INT(bind(probe, (struct sockaddr*)&addr, sizeof(addr)), 0); + socklen_t addr_len = sizeof(addr); + EXPECT_EQ_INT(getsockname(probe, (struct sockaddr*)&addr, &addr_len), 0); + int port = ntohs(addr.sin_port); + close(probe); + + int before = count_open_fds(); + Client* c = client_create(); + EXPECT_NOT_NULL(c); + EXPECT_FALSE(client_connect(c, "localhost", port)); + if (before >= 0) + EXPECT_TRUE(count_open_fds() <= before + 1); + client_disconnect(c); + if (before >= 0) + EXPECT_EQ_INT(count_open_fds(), before); + client_delete(c); +} + +/* #219 AC3: the shared tcp_connect_socket_ex() (used by both the plain and TLS + * entry points) must install the --contimeout as SO_RCVTIMEO/SO_SNDTIMEO before + * connecting. Calling it directly lets us observe the pre-connect state (the + * plain wrapper later overrides the receive timeout with the IO --timeout). */ +static void test_tcp_connect_socket_ex_applies_contimeout() { + Server* s = server_create(0); + EXPECT_NOT_NULL(s); + EXPECT_EQ_INT(listen(s->file_descriptor, 1), 0); + struct sockaddr_in bound; + socklen_t bound_len = sizeof(bound); + EXPECT_EQ_INT(getsockname(s->file_descriptor, (struct sockaddr*)&bound, &bound_len), 0); + int port = ntohs(bound.sin_port); + + tcp_set_timeouts(30, 7); + Client* c = client_create(); + EXPECT_NOT_NULL(c); + TcpConnectOptions opts; + memset(&opts, 0, sizeof(opts)); + EXPECT_TRUE(tcp_connect_socket_ex(c, "127.0.0.1", port, &opts)); + struct timeval tv; + socklen_t tv_len = sizeof(tv); + EXPECT_EQ_INT(getsockopt(c->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &tv, &tv_len), 0); + EXPECT_EQ_INT((int)tv.tv_sec, 7); + tv_len = sizeof(tv); + EXPECT_EQ_INT(getsockopt(c->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &tv, &tv_len), 0); + EXPECT_EQ_INT((int)tv.tv_sec, 7); + client_disconnect(c); + client_delete(c); + tcp_set_timeouts(30, 10); + server_delete(&s); +} + +/* The plain wrapper applies the post-connect IO --timeout, which supersedes the + * contimeout installed during connect. */ +static void test_tcp_connect_post_timeout_applied() { + Server* s = server_create(0); + EXPECT_NOT_NULL(s); + EXPECT_EQ_INT(listen(s->file_descriptor, 1), 0); + struct sockaddr_in bound; + socklen_t bound_len = sizeof(bound); + EXPECT_EQ_INT(getsockname(s->file_descriptor, (struct sockaddr*)&bound, &bound_len), 0); + int port = ntohs(bound.sin_port); + + tcp_set_timeouts(5, 7); + Client* c = client_create(); + EXPECT_NOT_NULL(c); + EXPECT_TRUE(client_connect(c, "127.0.0.1", port)); + struct timeval tv; + socklen_t tv_len = sizeof(tv); + EXPECT_EQ_INT(getsockopt(c->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &tv, &tv_len), 0); + EXPECT_EQ_INT((int)tv.tv_sec, 5); + tv_len = sizeof(tv); + EXPECT_EQ_INT(getsockopt(c->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &tv, &tv_len), 0); + EXPECT_EQ_INT((int)tv.tv_sec, 5); + client_disconnect(c); + client_delete(c); + tcp_set_timeouts(30, 10); + server_delete(&s); +} + void test_transport_tcp() { test_server_create_ephemeral(); test_server_delete_null(); @@ -263,4 +439,8 @@ void test_transport_tcp() { test_server_create_bind_address(); test_server_create_bind_ipv6(); test_tcp_nodelay_default_and_override(); + test_tcp_connect_falls_back_to_next_address(); + test_tcp_connect_failed_attempts_do_not_leak_fds(); + test_tcp_connect_socket_ex_applies_contimeout(); + test_tcp_connect_post_timeout_applied(); } diff --git a/tests/test_transport_tls.c b/tests/test_transport_tls.c index cc13aea..1385ff8 100644 --- a/tests/test_transport_tls.c +++ b/tests/test_transport_tls.c @@ -3,8 +3,11 @@ #include "test_utils.h" #include "transport_tcp.h" #include "transport_tls.h" +#include +#include #include #include +#include #include static void test_tls_global_init() { @@ -71,9 +74,53 @@ static void test_server_create_tls_empty_certs() { EXPECT_NULL(s); } +/* Count the process's open descriptors via /proc/self/fd (see the TCP tests). */ +static int tls_count_open_fds(void) { + DIR* dir = opendir("/proc/self/fd"); + if (!dir) + return -1; + int count = 0; + const struct dirent* ent; + while ((ent = readdir(dir)) != NULL) { + if (strcmp(ent->d_name, ".") == 0 || strcmp(ent->d_name, "..") == 0) + continue; + count++; + } + closedir(dir); + return count; +} + +/* #219 AC3: client_connect_tls_ex() reuses the shared tcp_connect_socket_ex() + * for the TCP connect, and a later TLS-setup failure must release that + * descriptor. Passing no CA path makes create_ssl_ctx() fail deterministically + * AFTER a successful TCP connect, so the cleanup path is exercised without a + * TLS handshake or a certificate. (The multi-address fallback itself is covered + * by the shared tcp_connect_socket_ex() tests in test_transport_tcp.c, which the + * TLS entry point calls.) */ +static void test_client_connect_tls_releases_fd_on_setup_failure() { + Server* s = server_create(0); + EXPECT_NOT_NULL(s); + EXPECT_EQ_INT(listen(s->file_descriptor, 1), 0); + struct sockaddr_in bound; + socklen_t bound_len = sizeof(bound); + EXPECT_EQ_INT(getsockname(s->file_descriptor, (struct sockaddr*)&bound, &bound_len), 0); + int port = ntohs(bound.sin_port); + + int before = tls_count_open_fds(); + Client* c = client_create(); + EXPECT_NOT_NULL(c); + EXPECT_FALSE(client_connect_tls(c, "127.0.0.1", port, NULL, NULL, NULL)); + EXPECT_TRUE(c->file_descriptor == -1); + if (before >= 0) + EXPECT_EQ_INT(tls_count_open_fds(), before); + client_delete(c); + server_delete(&s); +} + void test_transport_tls() { test_tls_global_init(); test_server_create_tls_without_certs(); test_client_connect_tls_fail(); + test_client_connect_tls_releases_fd_on_setup_failure(); test_server_create_tls_empty_certs(); } -- 2.54.0 From 3787695ba66d1c9501ece213427df01916d2d1cd Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 16:05:18 +0200 Subject: [PATCH 45/68] fix(xattr): apply --dirs directory xattrs fd-relative (#286) --- src/shared/file_save.c | 92 ++++++++++++++++++++++++++++-- tests/integration/test_features.py | 25 ++++++++ tests/test_xattr.c | 50 ++++++++++++++++ 3 files changed, 161 insertions(+), 6 deletions(-) diff --git a/src/shared/file_save.c b/src/shared/file_save.c index 98000ae..d207968 100644 --- a/src/shared/file_save.c +++ b/src/shared/file_save.c @@ -714,27 +714,107 @@ static FileSaveResult file_save_directory_to_disk(const FileSavePlan* plan, bool return FILE_SAVE_ERROR; bool dir_existed = file_path_exists_secure(dir_path); bool ok = file_ensure_directory_secure(dir_path); + /* One confined, no-follow descriptor drives ownership/mode/xattr/timestamp + application so none of them can follow a same-named symlink planted after + the mkdir. This mirrors the O_DIRECTORY|O_NOFOLLOW fd that + dir_metadata_list_apply() opens for the recursive path; the fd is reached + through the already-confined parent. */ + char* leaf = NULL; + int parent_fd = -1; + int dir_fd = -1; + if (ok) { + parent_fd = file_open_secure_parent(dir_path, &leaf, false); + if (parent_fd >= 0) + dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC); + } /* P7 Wave E: apply the negotiated ownership to the directory ITSELF (not just the files inside it). --copy-as and every explicit identity policy own every entry, so a directory must not keep the receiver's owner while its children get the policy owner. Applied no-follow on the confined - parent fd after the mkdir; identity_apply_ownership_link() is itself a - no-op unless an identity policy is active. */ + parent fd; identity_apply_ownership_link() is itself a no-op unless an + identity policy is active. Ownership runs before the mode because a chown + clears setuid/setgid. A failed REQUIRED --copy-as ownership fails the + entry; every other policy stays best-effort. */ if (ok && file->metadata && identity_active_enabled()) { - char* leaf = NULL; - int parent_fd = file_open_secure_parent(dir_path, &leaf, false); if (parent_fd >= 0) { if (!identity_apply_ownership_link(parent_fd, leaf, (int32_t)file->metadata->uid, (int32_t)file->metadata->gid)) ok = false; - close(parent_fd); } else if (identity_copy_as_active()) { /* The directory exists (ok) but its required --copy-as ownership could not be applied because the confined parent could not be opened. */ ok = false; } - free(leaf); + } else if (ok && identity_copy_as_active()) { + ok = false; } + /* Mode next: fchmod also rewrites the ACL mask, so the xattrs/ACLs below must + follow it. The --chmod/permission-bits handling matches the recursive + dir_metadata_list_apply() path exactly. */ + if (ok && file->metadata && plan->config && plan->config->preserve_perms) { + const Config* config = plan->config; + if (dir_fd < 0) { + char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to open directory %s to set its mode: %s", + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } else { + mode_t dir_mode = file->metadata->mode; + bool mode_ready = true; + if (config->chmod_spec && *config->chmod_spec && + !chmod_apply(dir_mode, config->chmod_spec, &dir_mode)) { + char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to apply --chmod to directory %s", + escaped_path ? escaped_path : ""); + free(escaped_path); + mode_ready = false; + } + if (mode_ready) { + /* rsync -p copies the source directory mode exactly, including + group/other write and the setgid/sticky bits. Setuid/setgid/sticky + are super-user activities: when the connection forbade them + (SUPER_MODE_OFF / --no-super), strip them even under -p. */ + mode_t safe_mode = dir_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777); + if (!privilege_super_mode_permitted(config->super_mode)) + safe_mode &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); + if (fchmod(dir_fd, safe_mode) != 0) { + char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to set directory mode on %s: %s", + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } + } + } + } + /* xattrs/ACLs after fchmod (the mode change can rewrite the ACL mask; the + ACL xattrs must be (re)applied last). Best-effort: a per-attribute failure + is logged and skipped by xattr_apply_fd(), never fatal. */ + if (ok && plan->config && plan->config->use_xattrs && dir_fd >= 0 && file->xattrs) + xattr_apply_fd(dir_fd, file->xattrs); + /* Timestamps last so no later chmod/xattr is mistaken for a content update. + -J/--omit-dir-times suppresses the directory mtime; --atimes/-U applies + only when the source atime is valid, exactly as the recursive path. */ + if (ok && file->metadata && plan->config && plan->config->preserve_times && + !plan->config->omit_dir_times) { + struct timespec times[2] = { + {.tv_sec = 0, .tv_nsec = UTIME_OMIT}, + {.tv_sec = file->metadata->mtime_sec, .tv_nsec = file->metadata->mtime_nsec}}; + if (plan->config->preserve_atimes && file->metadata->atime_valid) { + times[0].tv_sec = file->metadata->atime_sec; + times[0].tv_nsec = file->metadata->atime_nsec; + } + if (parent_fd >= 0 && utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW) != 0) { + char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); + log_message(LOG_LEVEL_WARNING, "Failed to set directory timestamps on %s: %s", + escaped_path ? escaped_path : "", strerror(errno)); + free(escaped_path); + } + } + if (dir_fd >= 0) + close(dir_fd); + if (parent_fd >= 0) + close(parent_fd); + free(leaf); free(dir_path); if (ok && created && !dir_existed) *created = true; diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index e60d8c7..be64da9 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -6750,6 +6750,31 @@ class TestExtendedAttributes: assert os.getxattr(received, "user.rootdir") == b"r" assert os.getxattr(os.path.join(received, "sub"), "user.subdir") == b"s" + @pytest.mark.ci + @pytest.mark.parametrize("mt", [False, True]) + def test_dirs_directory_xattr_applied(self, shared_server, mt): + """#286.3: -d/-X must apply a transferred directory's user.* xattr at the + destination through the --dirs STATUS_MKDIR path (both the + single-threaded and -m/--threads receiver paths).""" + source, dest = self._source_and_dest("dirsxattr") + sub = os.path.join(source, "sub") + os.makedirs(sub) + if not _xattr_supported(sub): + pytest.skip("filesystem does not support user xattrs") + os.setxattr(sub, "user.dirsdir", b"dirs-value") + lst = os.path.join(TEST_DATA_DIR, "dirs_xattr_list.txt") + with open(lst, "wb") as fh: + fh.write(b"sub\n") + + flags = ["--files-from", lst, "--dirs", "-R", "-X"] + (["--threads"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, \ + f"--dirs -X sync failed: {(result.stderr or result.stdout)[:300]}" + received = os.path.join(dest, "sub") + assert os.path.isdir(received), "--dirs directory entry was not created" + assert os.getxattr(received, "user.dirsdir") == b"dirs-value", \ + "the --dirs directory's user.* xattr was not applied at the destination" + @pytest.mark.ci def test_directory_default_acl_preserved(self, shared_server): """#286.3: -aA must preserve a directory's default POSIX ACL (the diff --git a/tests/test_xattr.c b/tests/test_xattr.c index 10447a3..96802f7 100644 --- a/tests/test_xattr.c +++ b/tests/test_xattr.c @@ -2,6 +2,7 @@ #include "xattr.h" #include "config.h" #include "file.h" +#include "file_save.h" #include "identity.h" #include "protocol.h" #include "test_utils.h" @@ -540,6 +541,54 @@ static void test_xattr_list_clone() { xattr_list_free(clone); } +/* #286.3: an explicit directory entry (--dirs, STATUS_MKDIR) that carries a + * captured user.* xattr must have it applied fd-relative by the directory + * install path itself -- not only by the receiver's deferred DirTimeList, which + * a direct file_save_to_disk_full() caller does not use. */ +static void test_file_save_directory_applies_xattrs() { + const char* root = "test_save_dir_xattr_tmp"; + const char* leaf = "subdir"; + const char* path = "test_save_dir_xattr_tmp/subdir"; + rmdir(path); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0700), 0); + /* The working directory may be a filesystem without user xattrs (e.g. some + tmpfs mounts): skip cleanly rather than fail the suite. */ + if (setxattr(root, "user.fastsync-dirprobe", "p", 1, 0) != 0) { + rmdir(root); + return; + } + removexattr(root, "user.fastsync-dirprobe"); + + File* dir = file_create(leaf); + EXPECT_NOT_NULL(dir); + dir->is_dir = true; + FileXattrList* xattrs = xattr_list_new(); + EXPECT_NOT_NULL(xattrs); + EXPECT_TRUE(xattr_list_append(xattrs, "user.dirxattr", "dirvalue", 8)); + dir->xattrs = xattrs; + + Config* config = config_create(); + EXPECT_NOT_NULL(config); + config->use_metadata = true; + config->use_xattrs = true; + config->preserve_xattrs = true; + + EXPECT_EQ_INT(file_save_to_disk_full(root, dir, config), FILE_SAVE_WRITTEN); + EXPECT_EQ_INT(access(path, F_OK), 0); + + char value[32]; + ssize_t got = getxattr(path, "user.dirxattr", value, sizeof(value)); + EXPECT_EQ_INT((int)got, 8); + EXPECT_TRUE(got == 8 && memcmp(value, "dirvalue", 8) == 0); + + file_destroy(dir); + config_delete(config); + removexattr(path, "user.dirxattr"); + rmdir(path); + rmdir(root); +} + void test_xattr() { test_xattr_list_clone(); test_xattr_wire_roundtrip(); @@ -553,4 +602,5 @@ void test_xattr() { test_fake_super_restore(); test_fake_super_no_real_chown(); test_fake_super_storage_resolution(); + test_file_save_directory_applies_xattrs(); } \ No newline at end of file -- 2.54.0 From 1f5f8dc5a7d5fbcca8bcfd31990035c94943e08a Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 16:20:03 +0200 Subject: [PATCH 46/68] fix(output): emit directory/root lines for -i and --out-format (#292) --- src/client/change_list.c | 15 +- src/client/client_report.c | 228 +++++++++++++++++++++--- src/client/client_send.c | 19 +- src/client/client_send_internal.h | 3 + tests/integration/test_features.py | 7 +- tests/integration/test_output_parity.py | 68 ++++++- 6 files changed, 306 insertions(+), 34 deletions(-) diff --git a/src/client/change_list.c b/src/client/change_list.c index a2c5096..892cef7 100644 --- a/src/client/change_list.c +++ b/src/client/change_list.c @@ -141,6 +141,10 @@ static void itemize_code(const Config* config, const ChangeEvent* event, char co update = 'h'; else if (created) update = (event->is_directory || event->is_symlink || event->is_special) ? 'c' : '>'; + else if (event->is_directory) + /* rsync: an existing directory that only has attribute changes carries no + transfer, so the update column is `.` rather than `>`. */ + update = '.'; else update = '>'; code[0] = update; @@ -168,12 +172,15 @@ static void itemize_code(const Config* config, const ChangeEvent* event, char co code[11] = '\0'; } -/* rsync %n: the transfer-relative name, with a trailing slash for directories. */ +/* rsync %n: the transfer-relative name, with a trailing slash for directories. + * The transfer root is `.` (so `%n` renders `./`), matching rsync's root entry. */ static bool append_name(StrBuf* buf, const ChangeEvent* event) { - if (!strbuf_append(buf, event->name != NULL ? event->name : "")) + const char* name = event->name != NULL ? event->name : ""; + if (event->is_directory && name[0] == '\0') + return strbuf_append(buf, "./"); + if (!strbuf_append(buf, name)) return false; - if (event->is_directory && (event->name == NULL || event->name[0] == '\0' || - event->name[strlen(event->name) - 1] != '/')) + if (event->is_directory && name[strlen(name) - 1] != '/') return strbuf_append_char(buf, '/'); return true; } diff --git a/src/client/client_report.c b/src/client/client_report.c index 08050df..2cbe5df 100644 --- a/src/client/client_report.c +++ b/src/client/client_report.c @@ -243,17 +243,31 @@ void transfer_stats_note_transferred(TransferStats* stats, const File* file) { #define RSYNC_PROGRESS_IO_WINDOW (32ULL * 1024ULL) /* Paths-only pre-count of the source file list, built once at transfer start - * when progress output is requested. rsync's `to-chk` denominator is the whole - * file list -- every regular file, directory, symlink and special plus the - * transfer root -- while the streaming scan never emits directories. A - * metadata-only walk (no file reads, no hashing) supplies that total and the - * directory names, so the opt-in pass leaves non-progress runs untouched. */ + * when progress output or -i/--out-format needs it. rsync's `to-chk` + * denominator is the whole file list -- every regular file, directory, symlink + * and special plus the transfer root -- while the streaming scan only emits + * empty directories. A metadata-only walk (no file reads, no hashing) supplies + * that total and a metadata-bearing File for every directory, so --progress can + * name them and -i/--out-format can itemize them without a second full scan. */ +/* One directory in the pre-count, keyed by its transfer-relative display name + * ("" is the transfer root). `file` is owned by ProgressPrecount.dir_files and + * carries the source metadata needed by -i/--out-format (%M/%B/%U/%G). */ +typedef struct { + char* name; /* owned */ + File* file; +} DirRef; + typedef struct { unsigned long long total; ArrayList* dir_paths; /* owned char* in transfer-relative display form */ + ArrayList* dir_files; /* owned File* captured during the metadata walk */ + ArrayList* dir_refs; /* owned DirRef*, sorted by name for prefix lookup */ } ProgressPrecount; static bool g_progress_active; +/* True when -i/--out-format need the pre-counted directory entries fed into the + * change-event stream (independent of --progress). */ +static bool g_change_dirs_active; static unsigned long long g_progress_xferred; static unsigned long long g_progress_index; static unsigned long long g_progress_total; @@ -270,14 +284,101 @@ bool progress_requested(const Config* config) { (config->show_progress || (config->info_level & LOG_INFO_PROGRESS) != 0); } +static void dir_ref_destroy(void* item) { + DirRef* ref = (DirRef*)item; + if (ref == NULL) + return; + free(ref->name); + free(ref); +} + +/* Sort DirRef pointers by their transfer-relative name for binary search. */ +static int dir_ref_compare(const void* left, const void* right) { + const DirRef* a = *(const DirRef* const*)left; + const DirRef* b = *(const DirRef* const*)right; + return strcmp(a->name, b->name); +} + +/* Look up the pre-counted directory File for a transfer-relative name ("" is + * the transfer root). Returns NULL when no pre-count was built or the name is + * not a known directory. */ +static File* progress_dir_lookup(const char* name) { + if (name == NULL || g_progress_precount.dir_refs == NULL) + return NULL; + ArrayList* refs = g_progress_precount.dir_refs; + size_t lo = 0; + size_t hi = (size_t)refs->size; + while (lo < hi) { + size_t mid = lo + (hi - lo) / 2; + DirRef* ref = (DirRef*)refs->items[mid]; + int cmp = strcmp(ref->name, name); + if (cmp < 0) + lo = mid + 1; + else if (cmp > 0) + hi = mid; + else + return ref->file; + } + return NULL; +} + static void progress_precount_dispose(ProgressPrecount* p) { if (p->dir_paths != NULL) { array_list_delete(p->dir_paths); p->dir_paths = NULL; } + if (p->dir_files != NULL) { + array_list_delete(p->dir_files); + p->dir_files = NULL; + } + if (p->dir_refs != NULL) { + array_list_delete(p->dir_refs); + p->dir_refs = NULL; + } p->total = 0; } +/* Record the transfer root's pre-transfer state for -i/--out-format. The + * receive root always exists, so rsync never marks it `cd`; its only observable + * change is its timestamp, which FastSync cannot observe remotely. Force a time + * mismatch so the root renders rsync's `.d..t...... ./` rather than the `cd` + * a zeroed destination state would produce. */ +static void progress_precount_mark_root(File* root) { + if (root == NULL) + return; + root->dest_state.known = true; + root->dest_state.existed = true; + root->dest_state.mode = root->metadata != NULL ? root->metadata->mode : 0; + root->dest_state.uid = root->metadata != NULL ? root->metadata->uid : 0; + root->dest_state.gid = root->metadata != NULL ? root->metadata->gid : 0; + root->dest_state.size = 0; + root->dest_state.mtime_sec = (root->metadata != NULL ? root->metadata->mtime_sec : 0) - 3600; + root->dest_state.mtime_nsec = root->metadata != NULL ? root->metadata->mtime_nsec : 0; +} + +/* Append one DirRef (name -> file) to the pre-count, marking the transfer + * root's destination state. Returns false on allocation failure. */ +static bool progress_precount_add_ref(ProgressPrecount* p, const Config* config, File* file) { + const char* rel = delete_display_path(config, file_wire_path(file)); + char* name = rel != NULL ? str_dup(rel) : NULL; + if (name == NULL) + return false; + DirRef* ref = malloc(sizeof(*ref)); + if (ref == NULL) { + free(name); + return false; + } + ref->name = name; + ref->file = file; + if (name[0] == '\0') + progress_precount_mark_root(file); + if (!array_list_add(p->dir_refs, ref)) { + dir_ref_destroy(ref); + return false; + } + return true; +} + void client_progress_cleanup(void) { if (g_progress_dir_index_valid) { path_index_free(&g_progress_dir_index); @@ -293,6 +394,7 @@ void client_progress_cleanup(void) { } progress_precount_dispose(&g_progress_precount); g_progress_active = false; + g_change_dirs_active = false; g_progress_total = 0; g_progress_index = 0; g_progress_xferred = 0; @@ -395,6 +497,11 @@ void print_delete_reports(const Config* config, const ArrayList* paths) { fflush(stdout); } +/* Emit every not-yet-seen ancestor directory of `rel`, outermost first, in the + * order rsync's depth-first flist walk visits them. With -i/--out-format each + * ancestor becomes a real change line (`cd+++++++++ sub/`, `.d..t...... ./`) + * rendered by the shared itemize code; otherwise it is the `--info=name` / + * --progress directory name line. */ static void client_progress_emit_ancestors(const Config* config, const char* rel) { if (!g_progress_dir_index_valid || !g_progress_emitted_valid || g_progress_emitted_keys == NULL || rel == NULL) @@ -413,9 +520,15 @@ static void client_progress_emit_ancestors(const Config* config, const char* rel char* key = str_dup(prefix); if (key != NULL && array_list_add(g_progress_emitted_keys, key)) { str_hash_set_insert_ref(&g_progress_emitted, key); - char* escaped = output_escape(prefix, config->eight_bit_output); - printf("%s/\n", escaped ? escaped : prefix); - free(escaped); + if (g_change_dirs_active) { + File* dir = progress_dir_lookup(prefix); + if (dir != NULL) + change_emit_dir_sent(config, dir); + } else { + char* escaped = output_escape(prefix, config->eight_bit_output); + printf("%s/\n", escaped ? escaped : prefix); + free(escaped); + } g_progress_index++; } else { free(key); @@ -425,6 +538,20 @@ static void client_progress_emit_ancestors(const Config* config, const char* rel } } +/* Feed a transferred entry's ancestor directories into the change-event stream + * before the entry's own line, so -i/--out-format and --progress report + * directories in rsync's depth-first order. Every directory is an ancestor of + * some emitted entry (a file, symlink, special, hard link or the empty-directory + * entry the scanner emits for a leaf), so this covers the whole tree. */ +void client_change_emit_ancestors(const Config* config, const File* file) { + if (config == NULL || file == NULL) + return; + if (!g_progress_active && !g_change_dirs_active) + return; + const char* rel = delete_display_path(config, file_wire_path(file)); + client_progress_emit_ancestors(config, rel); +} + /* rsync's --info=name/progress line for one entry: transfer-relative name (a * trailing slash for directories) plus the ` -> target` symlink suffix. */ static char* progress_entry_line(const File* file, const char* rel) { @@ -462,7 +589,7 @@ void client_progress_begin(const Config* config) { g_progress_active = progress_requested(config); g_progress_xferred = 0; g_progress_index = 1; /* the transfer root is file-list entry #0 */ - if (!g_progress_active) { + if (!g_progress_active && !g_change_dirs_active) { /* `--info=flist` prints rsync's file-list header even without progress. */ if (!config->quiet && info_flag_enabled(config, LOG_INFO_FLIST)) { printf("sending incremental file list\n"); @@ -470,11 +597,18 @@ void client_progress_begin(const Config* config) { } return; } - printf("sending incremental file list\n"); - /* rsync prints the transfer-root directory's name before the first file when - that directory is created; FastSync mirrors the source root below the - receive root and creates it on a fresh destination, so emit it here. */ - printf("./\n"); + if (g_progress_active) + printf("sending incremental file list\n"); + /* rsync prints the transfer-root directory before the first entry. Under + -i/--out-format it is the root change line (`.d..t...... ./`); otherwise it + is the plain --info=name / --progress name line. */ + if (g_change_dirs_active) { + File* root = progress_dir_lookup(""); + if (root != NULL) + change_emit_dir_sent(config, root); + } else { + printf("./\n"); + } fflush(stdout); } @@ -487,7 +621,6 @@ void client_progress_file(const Config* config, const File* file) { unsigned long long size = file->data->size; if (!config->itemize_changes && config->out_format == NULL) { const char* rel = delete_display_path(config, file_wire_path(file)); - client_progress_emit_ancestors(config, rel); char* escaped = output_escape(rel, config->eight_bit_output); printf("%s\n", escaped ? escaped : (rel ? rel : "")); free(escaped); @@ -510,7 +643,6 @@ void client_progress_name(const Config* config, const File* file) { return; const char* rel = delete_display_path(config, file_wire_path(file)); if (!config->itemize_changes && config->out_format == NULL) { - client_progress_emit_ancestors(config, rel); char* line = progress_entry_line(file, rel ? rel : ""); if (line != NULL) { char* escaped = output_escape(line, config->eight_bit_output); @@ -550,8 +682,12 @@ static bool progress_precount_add_dir(ProgressPrecount* p, const char* path) { * so the data pass's link-group state is never perturbed. */ static bool progress_precount_scan(const Config* config, ProgressPrecount* out) { out->dir_paths = array_list_create(free); - if (out->dir_paths == NULL) + out->dir_files = array_list_create(file_destroy); + out->dir_refs = array_list_create(dir_ref_destroy); + if (out->dir_paths == NULL || out->dir_files == NULL || out->dir_refs == NULL) { + progress_precount_dispose(out); return false; + } out->total = 0; PreparedScanner prepared; memset(&prepared, 0, sizeof(prepared)); @@ -568,12 +704,14 @@ static bool progress_precount_scan(const Config* config, ProgressPrecount* out) local.preserve_xattrs = false; local.preserve_acls = false; local.checksum = false; - local.capture_dir_times = false; + /* Capture one metadata-bearing File per traversed directory (including the + transfer root) so -i/--out-format can render %M/%B/%U/%G for directories. */ + local.capture_dir_times = true; local.excluded_paths = NULL; local.size_skipped_paths = NULL; local.synced_dirs = NULL; local.plan_dirs = NULL; - local.dir_entries = NULL; + local.dir_entries = out->dir_files; local.dir_entries_mutex = NULL; local.hardlinks = NULL; DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); @@ -598,6 +736,18 @@ static bool progress_precount_scan(const Config* config, ProgressPrecount* out) progress_precount_dispose(out); return false; } + /* Build the name -> File lookup from the captured directory Files. */ + for (int i = 0; i < out->dir_files->size; i++) { + File* f = (File*)out->dir_files->items[i]; + if (f == NULL) + continue; + if (!progress_precount_add_ref(out, config, f)) { + progress_precount_dispose(out); + return false; + } + } + if (out->dir_refs->size > 1) + qsort(out->dir_refs->items, (size_t)out->dir_refs->size, sizeof(DirRef*), dir_ref_compare); out->total += 1; /* the transfer root "." */ return true; } @@ -609,9 +759,26 @@ static bool progress_precount_from_plan_dirs(const Config* config, const ArrayLi unsigned long long non_dir_count, ProgressPrecount* out) { out->dir_paths = array_list_create(free); - if (out->dir_paths == NULL) + out->dir_files = array_list_create(file_destroy); + out->dir_refs = array_list_create(dir_ref_destroy); + if (out->dir_paths == NULL || out->dir_files == NULL || out->dir_refs == NULL) { + progress_precount_dispose(out); return false; + } out->total = non_dir_count + 1; + /* The delete pre-scan's plan list omits the transfer root, so synthesize its + entry here; it is only used for the root change line. */ + File* root = file_create(""); + if (root == NULL || !array_list_add(out->dir_files, root)) { + file_destroy(root); + progress_precount_dispose(out); + return false; + } + root->is_dir = true; + if (!progress_precount_add_ref(out, config, root)) { + progress_precount_dispose(out); + return false; + } for (int i = 0; i < plan_dirs->size; i++) { const char* path = (const char*)plan_dirs->items[i]; const char* rel = config->send_directory != NULL @@ -621,7 +788,25 @@ static bool progress_precount_from_plan_dirs(const Config* config, const ArrayLi progress_precount_dispose(out); return false; } + File* dir = file_create(""); + if (dir == NULL) { + progress_precount_dispose(out); + return false; + } + dir->is_dir = true; + dir->send_path = str_dup(rel != NULL ? rel : ""); + if (dir->send_path == NULL || !array_list_add(out->dir_files, dir)) { + file_destroy(dir); + progress_precount_dispose(out); + return false; + } + if (!progress_precount_add_ref(out, config, dir)) { + progress_precount_dispose(out); + return false; + } } + if (out->dir_refs->size > 1) + qsort(out->dir_refs->items, (size_t)out->dir_refs->size, sizeof(DirRef*), dir_ref_compare); out->total += (unsigned long long)out->dir_paths->size; return true; } @@ -633,7 +818,8 @@ void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, unsigned long long plan_non_dir_count) { client_progress_cleanup(); g_progress_active = progress_requested(config); - if (!g_progress_active) + g_change_dirs_active = config->itemize_changes || config->out_format != NULL; + if (!g_progress_active && !g_change_dirs_active) return; bool ok = plan_dirs != NULL ? progress_precount_from_plan_dirs( config, plan_dirs, plan_non_dir_count, &g_progress_precount) diff --git a/src/client/client_send.c b/src/client/client_send.c index 3b9d623..9bb783f 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -838,6 +838,10 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, if (chunk->items[i] == NULL) continue; transfer_stats_note_entry(stats, chunk->items[i]); + /* The chunk-serialization path emits no --progress name lines, so only + feed -i/--out-format its ancestor directory lines here. */ + if (config->itemize_changes || config->out_format != NULL) + client_change_emit_ancestors(config, chunk->items[i]); if (chunk->items[i]->is_dir) change_emit_dir_sent(config, chunk->items[i]); else @@ -859,6 +863,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, source to remove and no incremental check. */ if (!send_directory_entry(client, f, config)) return -1; + client_change_emit_ancestors(config, f); change_emit_dir_sent(config, f); client_progress_name(config, f); continue; @@ -873,6 +878,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, !send_int(client->file_descriptor, f->link_group) || !send_wire_str(client->file_descriptor, f->hardlink_target)) return -1; + client_change_emit_ancestors(config, f); change_emit_file_sent(config, f); client_progress_name(config, f); continue; @@ -881,6 +887,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, if (f->is_symlink) { if (!send_symlink_entry(client, f, config)) return -1; + client_change_emit_ancestors(config, f); change_emit_file_sent(config, f); client_progress_name(config, f); continue; @@ -890,6 +897,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, if (f->is_special) { if (!file_send_special(f, client->file_descriptor, config->use_metadata)) return -1; + client_change_emit_ancestors(config, f); change_emit_file_sent(config, f); client_progress_name(config, f); continue; @@ -913,6 +921,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config, return -1; } transfer_stats_note_transferred(stats, f); + client_change_emit_ancestors(config, f); change_emit_file_sent_bytes(config, f, protocol_bytes_written() - bytes_before, protocol_bytes_read() - read_before); client_progress_file(config, f); @@ -1567,7 +1576,10 @@ static bool send_files_prepare_delete(Config* config, SendFilesState* state) { * runs the shared cleanup). */ static bool send_files_run(Config* config, SendFilesState* state) { Client* client = state->client; - if (progress_requested(config)) + /* --progress needs the file-list total; -i/--out-format needs the directory + entries. Either way one paths-only pre-count supplies both, and a + --delete-during/--delete-delay pre-scan is reused when present. */ + if (progress_requested(config) || config->itemize_changes || config->out_format != NULL) client_progress_prepare(config, state->plan_dirs, state->per_dir_non_dir_count); /* Phase 6: compute the client-only stop deadline once at transfer start. The early-delete pre-scan above deliberately ignores it so the keep-set (and @@ -2045,9 +2057,10 @@ int send_files_multithreaded(Config* config) { return 1; } /* --progress/--info=progress: pre-count the file list for rsync's to-chk - denominator, reusing a --delete-during/--delete-delay pre-scan when one + denominator; -i/--out-format: pre-count the directory entries. One pass + supplies both, reusing a --delete-during/--delete-delay pre-scan when one already ran. */ - if (progress_requested(config)) + if (progress_requested(config) || config->itemize_changes || config->out_format != NULL) client_progress_prepare(config, context->plan_dirs, pre_scan_non_dir); thrd_t scanner, loader, sender; diff --git a/src/client/client_send_internal.h b/src/client/client_send_internal.h index 6d042b5..d82d45d 100644 --- a/src/client/client_send_internal.h +++ b/src/client/client_send_internal.h @@ -64,6 +64,9 @@ void client_progress_cleanup(void); void client_progress_begin(const Config* config); void client_progress_file(const Config* config, const File* file); void client_progress_name(const Config* config, const File* file); +/* Emit a transferred entry's ancestor directories (as -i/--out-format change + * lines or --progress name lines) before the entry's own line. */ +void client_change_emit_ancestors(const Config* config, const File* file); void client_progress_uptodate(const Config* config, const File* file); void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, unsigned long long plan_non_dir_count); diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index e60d8c7..03fcb9e 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -2786,7 +2786,12 @@ class TestItemizeChanges: flags=["--preserve", "-i", "--incremental"], port=shared_server.port) assert result.returncode == 0, f"incremental itemize failed: {result.stderr[:200]}" - itemized = [line for line in result.stdout.splitlines() if line and line[0] in ">.. fl, f"fastsync %b must include framing: {result.stdout!r}" @@ -421,7 +477,9 @@ class TestWireStatsParity: assert file_lines(result.stdout) == file_lines(rsync_result.stdout), ( f"rsync={rsync_result.stdout!r} fastsync={result.stdout!r}" ) - assert result.stdout.split()[0] == rsync_result.stdout.split()[0] == "16", ( + fs_c = _file_entry_line(result.stdout).split()[0] + rs_c = _file_entry_line(rsync_result.stdout).split()[0] + assert fs_c == rs_c == "16", ( f"%c must be rsync's 16-byte sum header: {result.stdout!r}" ) @@ -450,8 +508,8 @@ class TestWireStatsParity: "--out-format=" + fmt], port=shared_server.port) assert result.returncode == 0, result.stderr[:300] - rs_c = int(rsync_result.stdout.split()[0]) - fs_c = int(result.stdout.split()[0]) + rs_c = int(_file_entry_line(rsync_result.stdout).split()[0]) + fs_c = int(_file_entry_line(result.stdout).split()[0]) # No basis exists, so rsync still reports only its sum header. assert rs_c == 16, rsync_result.stdout # FastSync reports its own handshake bytes and is not aligned. -- 2.54.0 From 7bf25048f6ef4036a4b72ee643eaaa7f323be629 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 16:37:21 +0200 Subject: [PATCH 47/68] docs: correct parity claims for issues #286-#297 --- CHANGELOG.md | 30 +++++++++++++++- README.md | 38 ++++++++++---------- RSYNC_COMPAT.md | 93 ++++++++++++++++++++++++++++--------------------- 3 files changed, 103 insertions(+), 58 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 763c1c1..cd7502d 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -21,7 +21,9 @@ reclassification is `--filter=RULE` moving ✅ → ⚠️, because its merge-onl `e`/`n`/`w`/`-` modifiers are now accepted and consumed but their semantics remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ / 11 ⚠️ / 27 ❌** of 157 rows. The affected rows' notes and the summary tally in -`RSYNC_COMPAT.md` were updated. +`RSYNC_COMPAT.md` were updated. A following triage-fix cycle (see **Triage +fixes** below) moves `-F` and `-i` to ⚠️, for a final **117 ✅ / 13 ⚠️ / 27 ❌** +of 157 rows. ### Changed @@ -135,6 +137,32 @@ remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ / `CHANGELOG.md` and `HANDOFF.md` were updated for the audit cycle; the `RSYNC_COMPAT.md` summary tally was corrected to match the rows. +### Triage fixes + +- **`--dirs` directory xattrs applied inline.** A `-d/--dirs` transfer now + applies captured directory `-X`/`-A` xattrs fd-relative on the directory entry + instead of dropping them, so directory xattrs survive the non-recursive path + (`src/shared/file_save.c`, `tests/test_xattr.c`). +- **Directory/root itemize and `--out-format` lines.** `-i`/`--itemize-changes` + and `--out-format` now emit the transfer-root `./` line and per-directory + `cd...`/`.d..t...` lines, rendered by the shared itemize code. This matches + rsync's fresh-transfer output; because the root line is unconditional and an + incremental re-run may itemize directories/symlinks that rsync's quick-check + leaves silent, `-i` is now a ⚠️ Caveat row. +- **FROM name globs for identity maps.** `--usermap`/`--groupmap` `FROM` tokens + now accept `*`/`?`/`[...]` globs, expanded sender-side against the passwd/group + database and collapsed into bounded numeric ranges (`MAX_IDENTITY_MAP`), + matching rsync. +- **Transport fallback unit tests.** Added unit coverage for the TCP/TLS + transport fallback paths (`tests/test_transport_tcp.c`, + `tests/test_transport_tls.c`). +- **Docs corrections.** `RSYNC_COMPAT.md`/`README.md` corrected stale parity + claims for issues #286–#297: the `-F` and `-i` reclassifications, the + `--munge-links` direction, the accepted checksum/compression name sets, + `--bwlimit` parsing, `--stop-at` grammar, `--trust-sender`, symlink xattrs, and + the native/non-interoperable batch and credential notes. The summary tally is + now **117 ✅ / 13 ⚠️ / 27 ❌** of 157 rows. + ## [2.28.0] - 2026-09-20 The rsync-parity cycle. `PROTOCOL_VERSION` moves `2.26.0 → 2.27.0 → 2.28.0`; diff --git a/README.md b/README.md index 22b44dd..dee3e96 100644 --- a/README.md +++ b/README.md @@ -233,9 +233,9 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `-h, --human-readable` | Format transfer byte/rate counts with rsync's decimal (base-1000) units | | `--max-depth ` | Maximum directory depth to recurse (0 = unlimited, default: 0) | | `--log-file ` | Write log messages to file instead of stderr | -| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree | -| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server) | -| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server) | +| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree (FastSync-native format, not rsync-interoperable) | +| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server); FastSync-native format, not rsync-interoperable | +| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server); FastSync-native format, not rsync-interoperable | | `--source-dir ` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) | | `--dest-dir ` | Server destination directory (overrides `FASTSYNC_DEST_DIR`) | | `--save-to-disk` | Write received files to disk | @@ -248,12 +248,12 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `-4, --ipv4` | Force IPv4 for destination resolution | | `-6, --ipv6` | Force IPv6 for destination resolution | | `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect (`TCP_NODELAY`, `SO_KEEPALIVE`, `SO_RCVBUF`, `SO_SNDBUF`, `SO_REUSEADDR`) | -| `--bwlimit ` | Bandwidth limit in kilobytes per second; also paces `--sendfile` transfers | +| `--bwlimit ` | Bandwidth limit, using rsync's exact `parse_size_arg` grammar: a bare value is KiB/s; `K`/`M`/`G`/`T`/`P` are binary suffixes; `KB`/`MB` are decimal and `KiB`/`MiB` binary; decimals are accepted and quantized to whole KiB; `0` (or empty) means no limit. Also paces `--sendfile` transfers | | `--chunk-size ` | Chunk size in bytes (default: 10485760) | | `--timeout ` | I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline. Default `0` = disabled (matching rsync); `0` disables it. `--no-timeout` is the negation. The value is not sent on the wire; the server side keeps its own safe floor. | | `--contimeout ` | Connection timeout in seconds (default: 60, matching rsync); `0` disables it (`--no-contimeout` is the negation) | | `--stop-after=MINS` | Stop the transfer after MINS minutes (a positive integer); whatever was already transferred is kept | -| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`); an early stop skips the late `--delete` keep-set | +| `--stop-at=TIME` | Stop at an absolute time. Accepts rsync's `parse_time` forms (`Y-M-DTh:m`, `Y/M/DTh:m`, `Y-M-D`, `M-D`, `D`, `h:m`, `:m`, `T h:m`; omitted fields resolve to the next matching point in the local timezone), plus `now+N[smhd]` and FastSync's `HH:MM`/`HH:MM:SS` clock-time spelling. An early stop skips the late `--delete` keep-set | | `-b, --backup` | Backup existing destination files before overwriting | | `--backup-dir ` | Target directory for backups (requires `--backup`) | | `--tls` | Enable TLS encryption | @@ -535,7 +535,7 @@ features without changing the meaning of ordinary compatibility options. | `--server-host ` | Select the TCP server host. | | `--server-port ` | Select the TCP server port (`--port ` and `--port=` are rsync-friendly aliases). | | `--tls` | Enable TLS for TCP transport. | -| `--bwlimit ` | Apply token-bucket bandwidth limiting (also paces `--sendfile` transfers). | +| `--bwlimit ` | Apply token-bucket bandwidth limiting with rsync's exact `parse_size_arg` grammar (bare = KiB/s, `K`/`M`/`G`/`T`/`P` binary, `KB`/`MB` decimal, `KiB`/`MiB` binary, decimals quantized to whole KiB, `0`/empty = no limit; also paces `--sendfile` transfers). | | `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created). | | `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; `Number of files`/`Number of created files` carry rsync's per-type breakdown (deleted files are a single total). | | `--timeout ` | Set the socket **and** per-message protocol I/O timeout. Default `0` = disabled (matching rsync); `0` disables it. | @@ -613,11 +613,11 @@ remote SSH argv is already built injection-safe. | `--partial-dir ` | Set a relative partial-transfer directory below the server destination root. Implies `--partial`. Rejected together with `--inplace` (`--inplace cannot be used with --partial-dir`, matching rsync), because the inplace path bypasses partial/temp staging. | | `--inplace` | Write directly to the destination instead of using a temporary file. Cannot be combined with `--partial-dir`. | | `--fsync` | Fsync every written file before publication. | -| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree. | -| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server). | -| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server). | +| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree (FastSync-native format, not rsync-interoperable). | +| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server); FastSync-native format, not rsync-interoperable. | +| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server); FastSync-native format, not rsync-interoperable. | | `--stop-after=MINS` | Stop the transfer after MINS minutes; whatever was already transferred is kept. | -| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`). An early stop skips the late `--delete` keep-set. | +| `--stop-at=TIME` | Stop at an absolute time. Accepts rsync's `parse_time` forms (`Y-M-DTh:m`, `Y/M/DTh:m`, `Y-M-D`, `M-D`, `D`, `h:m`, `:m`, `T h:m`; omitted fields resolve to the next matching point in the local timezone), plus `now+N[smhd]` and FastSync's `HH:MM`/`HH:MM:SS` clock-time spelling. An early stop skips the late `--delete` keep-set. | ### Metadata and links @@ -689,7 +689,7 @@ remote SSH argv is already built injection-safe. | `--fastsync-server-path ` | Remote FastSync server path for SSH mode (client-only; never crosses the wire). | | `--rsync-path ` | Alias for `--fastsync-server-path`. | | `-M`, `--remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable; rejected for daemon/TCP destinations). | -| `--trust-sender` | Receiver-local: trust the remote sender's file list and skip path re-validation (does not affect symlink targets). | +| `--trust-sender` | Receiver-local: trust the remote sender's file list and skip path re-validation (does not affect symlink targets). **On the client this flag alone is inert** — it is never sent on the wire; the server must be started with its own `--trust-sender`, or the client must forward it with `-M--trust-sender` (SSH only). | | `--timeout ` | Socket + per-message I/O timeout; default `0` = disabled. | | `--contimeout ` | Connection timeout; default 60; `0` disables. | | `--source-dir ` | Set the source directory explicitly. | @@ -732,14 +732,14 @@ remote SSH argv is already built injection-safe. | `-6`, `--ipv6` | Bind an IPv6 socket. | | `--allow-delete` | Permit client delete manifests. Deletion is refused by default. This also gates `--force` (which can recursively replace/remove a destination directory tree). | | `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. Rejected with `--stdio` (the SSH remote argv is client-composed; use a forced command if the default must hold). No effect when not root. Daemon modules opt in per module with `client owner = yes`. | -| `--trust-sender` | Trust the remote sender's file list: skip the receiver's up-front path-traversal re-validation (fewer checks, faster, potentially unsafe; off by default). It does not affect symlink targets, which are stored verbatim either way. | +| `--trust-sender` | Trust the remote sender's file list: skip the receiver's up-front path-traversal re-validation (fewer checks, faster, potentially unsafe; off by default). It does not affect symlink targets, which are stored verbatim either way. A client `--trust-sender` is never sent over the wire — the server must set this flag itself, or the client must forward it via `-M--trust-sender`. | | `--no-super` | Operator veto: never attempt super-user activities (ownership, device nodes) even as root, and refuse any client `--copy-as`/`--super` request. | | `--allow-unauthenticated` | Permit plaintext/anonymous network clients; an auth-required module still accepts only opted-in loopback plaintext. | | `--iconv=LOCAL[,REMOTE]` | Declare this server's LOCAL charset for file-name conversion. | -| `--password-file=FILE` | Credential store for modules that declare `auth users`. Requires `--daemon`. | -| `--early-input=FILE` | Second credential store layered over `--password-file`. Requires `--daemon`. | -| `--hash-credentials ` | Read ``'s `user:password` lines and print PBKDF2 credential-store lines to stdout, then exit. Cannot be combined with `--daemon` or `--stdio`. | -| `--iterations N` | PBKDF2 iteration count for `--hash-credentials` (default 600000, range 100000–10000000). Requires `--hash-credentials`. | +| `--password-file=FILE` | Credential store for modules that declare `auth users`. Requires `--daemon`. FastSync-native SCRAM/PBKDF2 format, not rsync-interoperable. | +| `--early-input=FILE` | Second credential store layered over `--password-file`. Requires `--daemon`. FastSync-native format, not rsync-interoperable. | +| `--hash-credentials ` | Read ``'s `user:password` lines and print PBKDF2 credential-store lines to stdout, then exit. Cannot be combined with `--daemon` or `--stdio`. FastSync-native, not rsync-interoperable. | +| `--iterations N` | PBKDF2 iteration count for `--hash-credentials` (default 600000, range 100000–10000000). Requires `--hash-credentials`. FastSync-native, not rsync-interoperable. | | `-v`, `--verbose` | Enable debug logging. | | `--help` | Print server usage. | @@ -897,8 +897,10 @@ The project will reach the drop-in replacement goal in stages: completion wave's scope; the tests live in `tests/integration/` and skip cleanly when rsync is unavailable. 3. `-a` implements full rsync `-rlptgoD`; under `-p` the source mode is copied - exactly (no masking). Ownership application stays privilege-gated, as in - rsync. + exactly, including group/other-write bits, with setuid/setgid/sticky copied + only when super-user activities are permitted (masked under + `SUPER_MODE_OFF`/`--no-super`). Ownership application stays privilege-gated, + as in rsync. 4. Symlink (verbatim storage), sparse-file, metadata, delete-policy (including `--max-delete` partial + exit 25, per-directory `--delete-during`/ `--delete-delay`), codecs, and resumable-write semantics are implemented; diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index decd2d5..ed3f622 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -6,8 +6,8 @@ This document maps rsync's full feature set to FastSync's current implementation | Status | Count | Description | |--------|-------|-------------| -| ✅ Parity | 119 | Reproduces rsync's semantics for this option's scope | -| ⚠️ Caveat | 11 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) | +| ✅ Parity | 117 | Reproduces rsync's semantics for this option's scope | +| ⚠️ Caveat | 13 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) | | ❌ Divergent | 27 | Rejected, an accepted no-op, deliberately non-rsync (native config/auth/batch, privileged namespaces, safe-subset privilege), or impossible on any portable filesystem call | | **Total** | **157** | One row per rsync option/feature group; a row may name several spellings | @@ -62,7 +62,7 @@ matrix is **111 ✅ / 13 ⚠️ / 33 ❌ = 157**. - **Fuzzy eligibility.** The `-y/--fuzzy` candidate search no longer inherits the ordinary delta engine's 16 KiB minimum or 10× ratio bound, so an oversized or sub-16-KiB sibling is reused as rsync reuses it (`test_parity_basis_fuzzy.py`). - **Output partials.** `--info=mount`/`--info=stats`, the `--stats` `dir:` breakdown under `-r`, and real `--debug` output for `flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send` were added (`test_parity_info_mount_stats.py`, `test_output_parity.py`, `test_parity_debug.py`); those rows stay ⚠️ for their remaining documented residuals. `--delete-before`'s phase-0 late-file divergence and the `--progress` root/ancestor/symlink feedback remain open (they need a receiver→sender event channel), and the >256 MiB single-file streaming limit (B4) was not addressed. The matrix is now **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. -**Audit cycle (no wire change; `PROTOCOL_VERSION` stays 2.28.0).** A security-and-correctness audit pass ran against the parity-2.29 baseline, followed by a set of audit follow-ups (filter merge modifiers, the `--inplace`/`--partial-dir` conflict, credential-file hardening, and small leak/log/test fixes). The only classification change is `--filter=RULE` moving ✅ → ⚠️, because its merge-only `e`/`n`/`w`/`-` modifiers are now accepted and consumed but their semantics remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ / 11 ⚠️ / 27 ❌ = 157**. The affected rows (`-z`/`--compress`, `--bwlimit`, `-T`/`--temp-dir`, `-p`/`--chmod`, `--partial-dir`, `--filter`) had their notes updated in place: +**Audit cycle (no wire change; `PROTOCOL_VERSION` stays 2.28.0).** A security-and-correctness audit pass ran against the parity-2.29 baseline, followed by a set of audit follow-ups (filter merge modifiers, the `--inplace`/`--partial-dir` conflict, credential-file hardening, and small leak/log/test fixes). The only classification change is `--filter=RULE` moving ✅ → ⚠️, because its merge-only `e`/`n`/`w`/`-` modifiers are now accepted and consumed but their semantics remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ / 11 ⚠️ / 27 ❌ = 157**. The affected rows (`-z`/`--compress`, `--bwlimit`, `-T`/`--temp-dir`, `-p`/`--chmod`, `--partial-dir`, `--filter`) had their notes updated in place. A later triage cycle moved `-F` and `-i` ✅ → ⚠️ (see the triage-cycle note below), giving **117 ✅ / 13 ⚠️ / 27 ❌ = 157**: - **Decompression ceiling.** `MAX_DECOMPRESSED_SIZE` was 100 MiB while the receiver advertises and the sender compresses whole files up to `MAX_RECEIVE_WHOLE_FILE_SIZE` (256 MiB), so `-z` on a 100–256 MiB regular file failed with `Declared decompressed size exceeds 104857600 bytes`. The ceiling is now defined in terms of the protocol whole-file bound (still a real allocation-clamped bomb guard), so the two cannot drift; `-z` on 100–256 MiB files now works. - **`--bwlimit` with `--sendfile`.** The plaintext-TCP `--sendfile` fast path wrote through `sendfile(2)` without passing through the protocol's token bucket, so `--bwlimit` was ignored on that path. It is now paced through the same per-session leaky bucket, so TLS and plaintext transports share identical `--bwlimit` semantics. @@ -73,6 +73,8 @@ matrix is **111 ✅ / 13 ⚠️ / 33 ❌ = 157**. - **Bounds and wire validation.** `--filter` rule count is now checked client-side against `MAX_FILTER_RULES` (with an actionable message before any network I/O) rather than surfacing as an opaque receiver protocol error; `send_protect_entries()` still re-checks the expanded count. Unknown wire `Status` values are rejected as protocol errors (`status_is_valid()`), and the audit also fixed a mutex leak on an init-failure path, an `errno`-after-`free()` in deferred delete application, `log_perror` misuse for non-`errno` conditions, `SSL_read` length clamping, `sendfile` `poll` `EINTR` retry, and printf-format/attribute issues. - **Credential-file hardening follow-up.** `secret_file_open()` now opens `--password-file`/`--early-input`/`--hash-credentials` inputs with `O_NOFOLLOW`, so a symlinked credential path fails closed (`ELOOP`) instead of being followed before the owner/mode gate; literal fd-backed paths (`/dev/fd/`, `/proc/self/fd/`, which is what a bash process substitution passes) are exempt, so process substitution still works. A FIFO/process-substitution read now waits under a bounded ~3 s deadline for its writer, so a slow producer works while a connected-but-silent FIFO fails instead of hanging. The follow-up also fixed a `config_create` allocation leak on its `server_host` failure path (`config_delete` now releases it), corrected the decompression-limit log message to print the effective bound rather than the compile-time ceiling, and hardened the daemon umask/root test fixtures. +**Triage cycle (no wire change; `PROTOCOL_VERSION` stays 2.28.0).** A documentation-and-correctness triage pass over the parity baseline corrected stale prose and reclassified two rows that carried a real behavioral residual: `-F` moves ✅ → ⚠️ (its own note already documented that per-directory merge rules are not carried to the receiver filter engine, so a destination-only entry matching ONLY a `.rsync-filter` rule is not shielded from `--delete`), and `-i`/`--itemize-changes` moves ✅ → ⚠️ (directory and transfer-root lines are now emitted, but the root `./` line is emitted unconditionally and an incremental re-run itemizes directories/symlinks that rsync's quick-check leaves silent). The matrix is **117 ✅ / 13 ⚠️ / 27 ❌ = 157**. + **Parity completion wave (protocol 2.23.0 → 2.26.0).** This wave closed the remaining gaps the rsync-parity wave left open (delete timing, wire counters and output, codec breadth, general `-R`/`-d`, the filter grammar (the unsupported @@ -121,10 +123,10 @@ Every one of those has an entry below with its remaining caveats. |------|-------------------|-----------------|-------| | `--stats` | Give transfer stats | ⚠️ Caveat | Prints transfer statistics. Protocol 2.25.0 populates the receiver-only counters the sender cannot observe (`Matched data`, `Number of deleted files`) from the receiver's `STATUS_STATS` report; the sender tracks the scanned file list per type so `Number of files` carries rsync's `(reg: X, dir: Y, link: Z, special: W)` breakdown (directories come from the scanner's captured directory list for `-a`/`-t`/`-p`, or from a lightweight traversed-directory counter on a plain `-r` run so the `dir:` category is present there too), `Number of regular files transferred` excludes symlinks/specials and up-to-date files, `Total file size` includes symlink target lengths, and `Total transferred file size` counts only transferred files. **Protocol 2.28.0 extends `STATUS_STATS`** with receiver-observed `literal_bytes` and the four `created_*` counters: `Number of created files` now carries rsync's `(reg/dir/link/special)` breakdown (the receiver reports which destination entries it newly created, including implicitly-created parent directories below the transfer root) and `Literal data` is exact for a delta transfer (the receiver counts the literal fragments it stored, not the whole source size) — all differential-tested in the sequential and `--threads` paths against rsync 3.4.1 for fresh-create, update and delta shapes. **Remaining divergences:** rsync's per-type breakdown on `Number of deleted files` is not reproduced; and `Total bytes sent`/`received` are FastSync wire bytes framed differently from rsync's, so they are not numerically comparable | | `-h`, `--human-readable` | Human-readable numbers | ✅ Parity | Formats transfer byte and rate counts using rsync's **decimal** (base-1000) units, matching rsync `-h` (e.g. `1.23M`), not binary units. **A lone `-h` with no transfer arguments prints help instead** (protocol 2.26.0), matching the rsync idiom; `-h` alongside a transfer remains human-readable | -| `-i`, `--itemize-changes` | Per-file change summary | ✅ Parity | Prints rsync-style `>f+++++++++` lines to stdout only for files actually sent (also under `-j`/`--threads`); unchanged files print nothing, matching single-`-i` behavior | +| `-i`, `--itemize-changes` | Per-file change summary | ⚠️ Caveat | Prints rsync-style itemize lines to stdout for files actually sent (also under `-j`/`--threads`). Directory and transfer-root lines are now emitted too: a run produces rsync's `./` root line and per-directory `cd+++++++++`/`.d..t......` lines, rendered by the shared itemize code. **Residual:** the root `./` line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an incremental re-run itemizes directories/symlinks that lack a quick-check where rsync stays silent (unchanged regular files still print nothing, matching single-`-i`) | | `--progress` | Show progress | ⚠️ Caveat | Protocol 2.25.0 prints rsync-style per-file progress blocks (percent, transferred/total bytes, rate, elapsed, `(xfr#N, to-chk=M/T)`) fed by the receiver's `STATUS_STATS`, in both the sequential and `--threads` send paths. FastSync also prints rsync's leading `./` transfer-root line and, when progress is requested (`--progress`/`-P`/`--info=progress`) and not `--quiet`, runs a **paths-only metadata pre-scan** (no file reads, no hashing) that supplies rsync's file-list total `T` for the `to-chk` denominator and the directory names; `--delete-during`/`--delete-delay` reuse their existing keep-set pre-scan instead of walking twice, and non-progress runs are untouched. Per-directory name lines are emitted (trailing `/`), and symlink (` -> target`) and special entries are named too, so a **fresh multi-directory tree's name set and `to-chk` denominator match rsync 3.4.1** (differential test, sequential and `--threads`) and a **single-file transfer's name lines and deterministic frames remain byte-identical** to rsync. **Order parity (parity-2.29):** the sequential scanner now emits entries in rsync's sorted depth-first flist order (non-directories ascending, then directories ascending), so the interleaving and the `to-chk` numerator match rsync for the default single-threaded transfer (differential `test_parity_order.py`; `--threads` has no rsync analogue and stays unordered). **Remaining divergences:** the leading `./` root line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an ancestor directory line is emitted whenever a child transfers (rsync suppresses it when the directory itself is unchanged); on a re-run, entries without a quick-check (symlinks, empty directories) are still named where rsync stays silent; and the rate/ETA are wall-clock dependent | | `-P` | Same as --partial --progress | ✅ Parity | Parses to `--partial` + `--progress`. The independent `--partial` retention semantics are rsync parity: an interrupted write retains the already-written temp at the destination (best-effort) so a later `--append`/`--append-verify` can resume. Progress presentation is owned by the `--progress` row; there is no separate `-P` divergence | -| `--out-format=FORMAT` | Custom output format | ❌ Divergent | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%c` `%C` `%i` `%M` `%o` `%U` `%G` `%t` `%%`. `%C` now uses the negotiated transfer algorithm (`--checksum-choice`, default `xxh128`, seed 0) and renders every algorithm exactly like rsync — xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, `none` a blank 2-char column — differential-tested across all algorithms. `%f`/`%n`/`%l`/`%i`/`%M`/`%U`/`%G`/`%B` also match. **Reclassified because `%b`/`%c` are protocol-specific and cannot match:** a differential against rsync 3.4.1 shows whole-file `%c = 16` for both, but rsync whole-file `%b = filesize + 27 + transfer-digest-bytes` (39 for a 0-byte file; 43/35/47 for xxh128/xxh64/sha1 on a 12-byte file) while FastSync `%b` counts its own framing; in delta mode rsync `%c = 16 + 6·ceil(filesize/block_size)` (verified at block sizes 512/700/1024/2048) while FastSync counts its own signature handshake, and rsync `%b` is its token stream. FastSync's wire bytes are a different quantity, so exact `%b`/delta-`%c` equality is impossible | +| `--out-format=FORMAT` | Custom output format | ❌ Divergent | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%c` `%C` `%i` `%M` `%o` `%U` `%G` `%t` `%%`. `%C` now uses the negotiated transfer algorithm (`--checksum-choice`, default `xxh128`, seed 0) and renders every algorithm exactly like rsync — xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, `none` a blank 2-char column — differential-tested across all algorithms. `%f`/`%n`/`%l`/`%i`/`%M`/`%U`/`%G`/`%B` also match. **Reclassified because `%b`/`%c` are protocol-specific and cannot match:** a differential against rsync 3.4.1 shows whole-file `%c = 16` for both, but rsync whole-file `%b = filesize + 27 + transfer-digest-bytes` (39 for a 0-byte file; 43/35/47 for xxh128/xxh64/sha1 on a 12-byte file) while FastSync `%b` counts its own framing; in delta mode rsync `%c = 16 + 6·ceil(filesize/block_size)` (verified at block sizes 512/700/1024/2048) while FastSync counts its own signature handshake, and rsync `%b` is its token stream. FastSync's wire bytes are a different quantity, so exact `%b`/delta-`%c` equality is impossible. Directory and transfer-root lines are now emitted (rsync's `./` root line and per-directory `cd...`/`.d..t...` lines), with the same residual as `-i`: the root line is emitted unconditionally and an incremental re-run may itemize directories/symlinks without a quick-check | | `--log-file=FILE` | Log to file | ✅ Parity | `log_file` config field | | `--log-file-format=FMT` | Log format | ✅ Parity | Requires `--log-file`; writes one template line per transferred file using the same token set as `--out-format` (including `%b` as the wire byte count) | | `--8-bit-output`, `-8` | Leave high-bit chars unescaped | ✅ Parity | Applies to displayed paths and protocol debug output | @@ -148,7 +150,7 @@ Every one of those has an entry below with its remaining caveats. | `--ignore-existing` | Skip updating existing files | ✅ Parity | `ignore_existing` config field (crosses the wire; receiver-side policy). Protocol 2.26.0 short-circuits in the per-file check **before any payload**: when the destination entry already exists, the receiver answers the skip during the incremental handshake instead of letting the sender stream data that would be discarded, so an existing 4 MiB destination costs only the config/check frames (verified with a counting proxy, matching rsync). The write-time paths (regular, delay-updates-staged, hardlink-sibling, special/device) still return `FILE_SAVE_SKIPPED` without overwriting, and `--backup` is disabled for skipped files. Like rsync, it does not apply to directories/symlinks. Combines with `-j`/`--threads` and `--delay-updates` | | `--remove-source-files` | Sender removes regular files after confirmed transfer | ✅ Parity | | | `-x`, `--one-file-system` | Do not cross filesystem boundaries | ✅ Parity | Sender scanner captures the root device and does not descend into mount-point crossings (`st_dev` differs). **Protocol 2.23.0 matches rsync's entry emission:** the mount-point directory itself is emitted as a payload-less directory entry (so the destination gets an empty directory) while its contents are skipped; previously the crossing subdirectory was dropped entirely | -| `-F` | Add the default `.rsync-filter` rules | ✅ Parity | Reads one filter rule per line from each directory's `.rsync-filter` file during traversal and applies it to that directory's subtree; the current directory's rules are evaluated before its ancestors', so deeper files override shallower ones and per-directory files override the command-line `--filter`/`-C` base by default (first match wins). **A single `-F` transfers the `.rsync-filter` files themselves, matching rsync; a repeated `-FF` additionally excludes them** (rsync 3.4.1's `-F`/`-FF` are exactly these two rules, with no `.cvsignore` branch). Unsupported/unparseable rules inside a per-directory file fail the scan with a clear error. **Residual (track 4a):** per-directory rules are still enforced receiver-side only through the sender-derived source-mirror protected prefixes; the base-rule receiver filter engine does not carry per-directory rules, so a destination-only entry matching ONLY a `.rsync-filter` rule is not yet shielded from `--delete` | +| `-F` | Add the default `.rsync-filter` rules | ⚠️ Caveat | Reads one filter rule per line from each directory's `.rsync-filter` file during traversal and applies it to that directory's subtree; the current directory's rules are evaluated before its ancestors', so deeper files override shallower ones and per-directory files override the command-line `--filter`/`-C` base by default (first match wins). **A single `-F` transfers the `.rsync-filter` files themselves, matching rsync; a repeated `-FF` additionally excludes them** (rsync 3.4.1's `-F`/`-FF` are exactly these two rules, with no `.cvsignore` branch). Unsupported/unparseable rules inside a per-directory file fail the scan with a clear error. **Residual (track 4a):** per-directory rules are still enforced receiver-side only through the sender-derived source-mirror protected prefixes; the base-rule receiver filter engine does not carry per-directory rules, so a destination-only entry matching ONLY a `.rsync-filter` rule is not yet shielded from `--delete` | ## 4. Directory Options @@ -337,7 +339,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | `-E`, `--executability` | Preserve executability | ✅ Parity | Preserves executable permission bits (implies metadata preservation) | | `--chmod=CHMOD` | Affect file permissions | ✅ Parity | Faithful port of rsync 3.4.1's `parse_chmod`/`tweak_mode`: numeric octal and symbolic `ugo`/`rwx` changes, `D`/`F` directory/file selectors, `X` (execute only on directories or already-executable files), `s`/`t` setuid/setgid/sticky, and append semantics — repeated clauses and repeated `--chmod` options accumulate in order (joined with commas). The changes are applied to the new mode **without sanitization** (matching rsync), except that setuid/setgid/sticky are masked when the connection forbids super-user activities (audit-cycle fix, see `-p`), and `--chmod` does **not** imply `-p` (rsync parity). Applied to files and directories on the receiver | | `-A`, `--acls` | Preserve ACLs | ✅ Parity | Implemented on Linux via the POSIX-ACL xattr representation: the sender captures the `system.posix_acl_access` / `system.posix_acl_default` xattrs and the receiver re-applies them fd-relative. A differential test with `setfacl` confirms the complete access and default ACL sets (including `mask`) are identical to rsync's on a directory. libacl is not required; a `fsetxattr` an unprivileged receiver may not perform is logged and skipped, never fatal. Only the `system.posix_acl_*` namespaces plus `user.*` are ever applied; privileged namespaces are never applied. Implies metadata transmission | -| `-X`, `--xattrs` | Preserve extended attributes | ❌ Divergent | Deliberately restricted to unprivileged `user.*` extended attributes plus the two POSIX ACL xattrs; `security.*` (SELinux, capabilities, ...) and `trusted.*` are **never** captured or applied — a client can never force a privileged attribute onto the destination, and the receiver independently re-validates every incoming name against the whitelist. This is a security-policy divergence from rsync, which can preserve the privileged namespaces with the needed privilege; implementing them would defeat FastSync's privilege-escalation guard. `user.*` capture/apply matches rsync in a differential test. Payloads are bounded on both ends. Incompatible with `-s` | +| `-X`, `--xattrs` | Preserve extended attributes | ❌ Divergent | Deliberately restricted to unprivileged `user.*` extended attributes plus the two POSIX ACL xattrs; `security.*` (SELinux, capabilities, ...) and `trusted.*` are **never** captured or applied — a client can never force a privileged attribute onto the destination, and the receiver independently re-validates every incoming name against the whitelist. This is a security-policy divergence from rsync, which can preserve the privileged namespaces with the needed privilege; implementing them would defeat FastSync's privilege-escalation guard. `user.*` capture/apply matches rsync in a differential test. Payloads are bounded on both ends. Incompatible with `-s`. **Also divergent: symlink xattrs/ACLs are not captured or applied** — `-X`/`-A` with `-l` carries only the link's owner/times/mode, not its xattrs (the capture uses path-following `listxattr`/`getxattr`, so the link's own xattrs are never read, and the receiver's symlink write path applies no xattr block). Closing this needs a dedicated symlink-xattr wire block and a `PROTOCOL_VERSION` bump | | `-H`, `--hard-links` | Preserve hard links | ✅ Parity | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below | | `-D` | Same as --devices --specials | ✅ Parity | Implies `--devices --specials`. `-D` was unassigned in FastSync (verified: no collision), so it is free to imply both device-node and special-file preservation. As of protocol 2.23.0 `--specials` genuinely covers **both FIFOs and unix sockets**, so `-D` covers the full rsync set. See the `--devices`/`--specials` rows and the Phase-4 devices notes below | | `--devices` | Preserve device files | ❌ Divergent | Recreates char/block device nodes with `mknodat` (type + rdev strictly validated, confined fd-relative below the receive root), but only when the receiver has `CAP_MKNOD`: a non-root receiver logs a warning and skips the entry instead of erroring, so a transfer with devices never aborts. Deliberate privilege-model divergence from rsync, which errors when it cannot create the node. `--specials` (FIFOs and unix sockets) is unprivileged and remains parity | @@ -352,7 +354,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | `--fake-super` | Store/recover privileged attrs via xattrs | ❌ Divergent | Records the resolved `uid:gid:mode:mtime_sec:mtime_nsec` in a reserved `user.fastsync.stat` xattr and immediately replays mode/times fd-relative, but **never performs a real `chown`** (the owner is recorded for a later privileged restore). The on-disk key and format are FastSync-native, not rsync's `user.rsync.%stat%`, so recordings are not interoperable with rsync — the same class as the native auth and batch formats. Implies metadata transmission; incompatible with `-s` | | `--open-noatime` | Avoid changing access time when opening files | ✅ Parity | Sender-side policy: the sender opens source files with `O_NOATIME` (Linux) when reading them for transfer, so the open/read does NOT bump the source's on-disk access time. Degrades safely when `O_NOATIME` is unavailable (not defined) or refused (`EPERM`, since it needs `CAP_FOWNER` or file ownership): the code falls back to a normal open, so the data always transfers — only the atime-bump is skipped. It does not itself capture/preserve atime; it only avoids modifying it. **Client-only, never crosses the wire.** Exposed as `file_open_for_read()` and applied to both the buffered data path and the sendfile path | | `--numeric-ids` | Do not map uid/gid by name | ✅ Parity | **A mapping modifier only:** when ownership is being applied it uses the transmitted numeric uid/gid directly, skipping the name lookup. It does **not** request ownership application on its own — combine it with `-o`/`-g`, `-a`, or an explicit map (`--chown`/`--usermap`/`--groupmap`) — and it does not need any metadata flag merely to parse. Ownership is only applied when metadata (hence the source uid/gid) is actually transmitted (see the Phase-4 identity notes) | -| `--usermap=STRING` | Map usernames | ✅ Parity | Opt-in ownership application. Comma-separated `FROM:TO` rules evaluated in order, first match wins. `FROM` accepts a source-resolved user name, an `@N`/bare `N` numeric id, an inclusive `LOW-HIGH` id range, `*`, or an empty field (ids with no source name). `TO` accepts a receiver-resolved **name** (protocol 2.26.0 resolves it on the receiving side against the receiver's account database, matching rsync), an `@N`/bare `N` id, or `*` (the receiving process's euid). Rules travel as resolved numeric pairs plus an optional TO name; the receiver applies a matching rule, else falls back to `--chown`, `--numeric-ids`, then a best-effort name lookup, via fd-relative `fchown`. Malformed specs are clear errors. Implies metadata; only effective where the receiver can chown (otherwise a warning) | +| `--usermap=STRING` | Map usernames | ✅ Parity | Opt-in ownership application. Comma-separated `FROM:TO` rules evaluated in order, first match wins. `FROM` accepts a source-resolved user name, a name **glob** (`*`/`?`/`[...]`, expanded sender-side at CLI-parse time against the sender's passwd/group database and collapsed into numeric `LOW-HIGH` ranges, bounded by `MAX_IDENTITY_MAP`), an `@N`/bare `N` numeric id, an inclusive `LOW-HIGH` id range, `*`, or an empty field (ids with no source name). `TO` accepts a receiver-resolved **name** (protocol 2.26.0 resolves it on the receiving side against the receiver's account database, matching rsync), an `@N`/bare `N` id, or `*` (the receiving process's euid). Rules travel as resolved numeric pairs plus an optional TO name; the receiver applies a matching rule, else falls back to `--chown`, `--numeric-ids`, then a best-effort name lookup, via fd-relative `fchown`. Malformed specs are clear errors. Implies metadata; only effective where the receiver can chown (otherwise a warning) | | `--groupmap=STRING` | Map group names | ✅ Parity | Same rules and receiver-side `TO`-name resolution as `--usermap`, applied to the group (gid) side | | `--chown=USER:GROUP` | Map owner and group | ✅ Parity | Opt-in ownership override. Forms `USER:GROUP`, `USER`, `:GROUP`; `*` means the current user/group as appropriate; `@N`/bare `N` ids; a name may escape `:` as `\:`. A name that resolves on the sender is sent as an id; an unresolvable name is carried as a receiver-resolved `TO` name (protocol 2.26.0), matching rsync's receiver-side resolution. Equivalent to a trailing `*:*` usermap+groupmap rule (an explicit map match wins). Conflicts with `--usermap`/`--groupmap` on the same side are a clear configuration error. Implies metadata; a non-root receiver warns and continues (rsync parity) | | `--copy-as=USER[:GROUP]` | Perform the copy as another user/group | ❌ Divergent | Close-refusal safe subset. FastSync never switches process credentials (its receiver is multithreaded, so a real `setuid`/`setgid` would be unsafe); instead the receiver forces the ownership of every entry it writes to the client-resolved ids through the confined fd-relative identity path. A privileged (root) receiver is required: an unprivileged receiver refuses the whole transfer at the config handshake, before any data, rather than produce wrong ownership. Deliberate divergence from rsync's real identity switching; a daemon refuses it unless the module sets `client owner = yes` | @@ -497,8 +499,10 @@ names, `--chown` names) are resolved to numbers at CLI parse time against the **client (sender) machine's** account databases; this reproduces rsync's semantics on a shared-account source/destination and is documented for a genuinely different destination. The interesting named-value subset is -supported (`*` FROM wildcard, `*` TO = current user, `@N`/bare-`N` numerics); a -lone-`@` "use the FROM value unchanged" rsync form is not implemented. Also +supported (FROM name globs `*`/`?`/`[...]` expanded sender-side against the +passwd/group database and bounded by `MAX_IDENTITY_MAP`, `*` FROM wildcard, +`*` TO = current user, `@N`/bare-`N` numerics); a lone-`@` "use the FROM value +unchanged" rsync form is not implemented. Also unlike rsync, plain `-M` never applies ownership and `--usermap`/`--groupmap`/ `--chown` each imply metadata preservation so the source uid/gid actually travel (the flags only take effect where ownership is being preserved/applied). @@ -599,7 +603,7 @@ warning + skip, never a system-clobbering write or an abort. | `-L`, `--copy-links` | Transform symlink to referent | ✅ Parity | Sender-side: every symlink is replaced by its referent's content. A referent that cannot be read, including a broken symlink, makes the run exit 23 (`RERR_PARTIAL`) like rsync while the rest of the tree still transfers, in both the sequential and `--threads` paths (differential test). The transferred tree matches rsync | | `--copy-unsafe-links` | Transform unsafe symlinks | ✅ Parity | Sender-side: only symlinks whose target is unsafe (absolute or escaping via `..`, matching rsync's `unsafe_symlink()` semantics) are dereferenced into their referent; safe links stay symlinks. A broken unsafe referent makes the run exit 23 like rsync (differential test), while a safe broken symlink is not dereferenced and exits 0 | | `--safe-links` | Ignore symlinks outside tree | ✅ Parity | Sender-side: a symlink whose target is unsafe is not transmitted at all (skipped), matching rsync's `--safe-links`. Because FastSync applies this while scanning the source, the receiver does not need to repeat it (`safe_links` config field) | -| `--munge-links` | Munge symlinks for safety | ✅ Parity | Sender rewrites each transmitted symlink target with rsync's `/rsyncd-munged/` prefix; the receiver strips the marker (only when the negotiated `munge_links` policy is on, so a source link that genuinely begins with the marker round-trips verbatim) and restores the exact real target. Unlike rsync, FastSync prefixes on the *sender* and un-munges on the receiver, but the wire result and the stored marker match rsync. See the Phase-4 symlink-trust notes | +| `--munge-links` | Munge symlinks for safety | ✅ Parity | The **receiver** munges: it prefixes each stored symlink target with rsync's `/rsyncd-munged/` marker (only when the negotiated `munge_links` policy is on, so a source link that genuinely begins with the marker round-trips verbatim). The **sender** un-munges a source target that already begins with the marker before transmitting, so a munged tree round-trips through the receiver's re-munging exactly like rsync. Matching rsync, the prefix is applied on the receiver and stripped on the sender; the wire result and the stored marker match rsync. See the Phase-4 symlink-trust notes | | `-k`, `--copy-dirlinks` | Transform symlink to dir | ✅ Parity | A symlink whose referent is a directory is dereferenced and recursed as a real directory; a symlink to a regular file stays a symlink. Sender-side only. See the Phase-4 symlink-trust notes | | `-K`, `--keep-dirlinks` | Treat symlinked dir as dir | ✅ Parity | On the receiver, an existing destination symlink-to-a-directory is used as that directory (followed) instead of being replaced; it is followed only when it resolves to a directory that stays beneath the receive root. See the Phase-4 symlink-trust notes | @@ -649,14 +653,15 @@ was bumped **2.12.0 → 2.13.0** (peers must match, exactly as prior phases did) divergence for `--delete` over an existing symlinked dir). Without `-K` the destination symlink is not followed (the O_NOFOLLOW walk fails the write), which is the safe default. -- **`--munge-links`** (sender rewrite; crosses the wire so the receiver - unmunges): every transmitted symlink target is prefixed with rsync's marker - `SYMLINK_MUNGE_PREFIX` = `/rsyncd-munged/`; the receiver strips the marker - (only when the negotiated `munge_links` policy is on — a plain `-l` run never - strips the prefix, so a source symlink that genuinely begins with - `/rsyncd-munged/` round-trips verbatim) and restores the exact real target. - This matches rsync's stored marker and its both-ends-negotiated model, with the - prefix applied on the sender rather than the receiver. The link *value* is +- **`--munge-links`** (receiver rewrite; crosses the wire so the receiver + munges): every stored symlink target is prefixed with rsync's marker + `SYMLINK_MUNGE_PREFIX` = `/rsyncd-munged/` by the **receiver**; the sender + un-munges a source target that already begins with the marker before + transmitting, so a munged tree round-trips verbatim. The marker is applied only + when the negotiated `munge_links` policy is on — a plain `-l` run never + prefixes, so a source symlink that genuinely begins with + `/rsyncd-munged/` round-trips verbatim. This matches rsync's stored marker and + its both-ends-negotiated model, with the prefix applied on the receiver. The link *value* is otherwise stored verbatim; the *placement* path still goes through `file_symlink_at_secure`'s confined fd walk (`has_path_traversal` on the destination path, no symlink follow). When no symlink is being transmitted @@ -843,7 +848,7 @@ These features are moderate because they affect traversal, temporary files, mani | `--relative`, `-R`; `--no-implied-dirs`; `--dirs`, `-d`; `--mkpath` | M | Extend path-list construction and destination directory creation while preserving traversal safety. | | `--temp-dir`, `-T` | M | Separate temporary-file placement from FastSync's timeout alias and define collision, permissions, and cleanup rules. | | `--delay-updates` | L | Stage all successful updates and publish them at completion, including crash and cancellation cleanup. | -| `--files-from=FILE`; `--from0`, `-0`; `--filter=RULE`, `-f`; `-F`; `--cvs-exclude`, `-C` | L | Build a complete filter/parser layer and integrate it with scanner pruning, manifests, and delete behavior. `-f` conflicts with FastSync sendfile mode. | +| `--files-from=FILE`; `--from0`, `-0`; `--filter=RULE`, `-f`; `-F`; `--cvs-exclude`, `-C` | L | Build a complete filter/parser layer and integrate it with scanner pruning, manifests, and delete behavior. (Done: `-f` is bound to `--filter`; the old sendfile conflict is gone, since sendfile is long-only `--sendfile`.) | | `--list-only`; `--itemize-changes`, `-i`; `--out-format=FORMAT`; `--log-file-format=FMT` | M | Add a structured change-event model so output modes share one source of truth. | ### Phase 3: Deletion, Comparison, and Delta Compatibility @@ -951,7 +956,7 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass and the audit-cycle follow-ups.** ✅ Parity 119 / ⚠️ Caveat 11 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `--delete-before`, `--filter`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, and the triage cycle.** ✅ Parity 117 / ⚠️ Caveat 13 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--delete-before`, `--filter`, `-F`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that receiver filter engine, flipping `--filter=RULE` back to ✅ (see above; the @@ -990,13 +995,17 @@ integration tests unless it is explicitly listed as a limitation. ### Checksums and compression -- **`--checksum-choice`/`--cc`** accepts `xxh64` (default), `xxhash`, `xxh3`, - `xxh128`, `md5`, and `auto`; `md4`, `sha1`, `none`, and the two-name - `transfer,pre-transfer` form are **rejected by name**. +- **`--checksum-choice`/`--cc`** accepts the full rsync 3.4.1 set: `xxh64` + (default), `xxhash`, `xxh3`, `xxh128`, `md5`, `md4`, `sha1`, `none`, the + two-name `transfer,pre-transfer` form, and `auto` (which honors + `RSYNC_CHECKSUM_LIST` before the compiled-in order). A genuinely unknown name + is still rejected by name, matching rsync. - **`--checksum-seed=0` is randomized per transfer** (the chosen seed is sent to the receiver), matching rsync; an explicit non-zero seed is used verbatim. -- **`--compress-choice`/`--zc`** accepts `zstd` (default), `none`, and `auto`; - rsync's `lz4`/`zlib`/`zlibx` are **rejected by name**. +- **`--compress-choice`/`--zc`** accepts the full rsync 3.4.1 set: `zstd` + (default), `lz4`, `zlib`, `zlibx`, `none`, and `auto` (which honors + `RSYNC_COMPRESS_LIST` before the compiled-in order). A genuinely unknown name + is still rejected by name, matching rsync. - **`--skip-compress`** uses rsync 3.4.1's built-in default suffix list when no list is supplied; an explicit list replaces it. - **`--no-whole-file`** is accepted as the rsync spelling that clears @@ -1040,9 +1049,10 @@ integration tests unless it is explicitly listed as a limitation. - **`--numeric-ids` is a mapping modifier only** — it changes *how* ids map, not *whether* ownership is applied; combine it with `-o`/`-g`, `-a`, or an explicit map. -- **`--usermap`/`--groupmap`** support names, `@N`/bare `N` ids, inclusive - `LOW-HIGH` ranges, `*`, empty-`FROM` (unnamed ids), and receiver-resolved `TO` - names. +- **`--usermap`/`--groupmap`** support names, FROM name **globs** + (`*`/`?`/`[...]`, expanded sender-side against the passwd/group database and + bounded by `MAX_IDENTITY_MAP`), `@N`/bare `N` ids, inclusive `LOW-HIGH` ranges, + `*`, empty-`FROM` (unnamed ids), and receiver-resolved `TO` names. - **`--chown` conflicts with `--usermap`/`--groupmap` on the same side** and is a clear configuration error (matching rsync) instead of an order-dependent winner. @@ -1056,19 +1066,23 @@ integration tests unless it is explicitly listed as a limitation. ### Symlinks and special files - **`-l`/`--links` stores symlink targets verbatim** (absolute and `..`-bearing - targets included), matching rsync. `--safe-links`, `--copy-unsafe-links`, and - `--munge-links` (which now uses rsync's `/rsyncd-munged/` marker) match rsync - and are applied sender-side. + targets included), matching rsync. `--safe-links` and `--copy-unsafe-links` + match rsync and are applied sender-side; `--munge-links` (which now uses + rsync's `/rsyncd-munged/` marker) matches rsync too but is applied + **receiver-side** (the sender un-munges an already-marked source target). - **`--specials` recreates unix sockets** with `mknodat(..., S_IFSOCK)`, so `-D`/`--devices --specials` now covers the full rsync node set. - **`--copy-devices`** is implemented (see its caveat below). ### Output -- **`-i`/`--out-format`** print rsync-style change lines; **`--list-only`** +- **`-i`/`--out-format`** print rsync-style change lines, including the + transfer-root `./` and per-directory `cd...`/`.d..t...` lines; **`--list-only`** scans the source only and contacts no server; **`-h`** uses rsync's decimal - units; **`--progress`** is an aggregate line; **`--stats`** prints the counters - FastSync can observe locally (receiver-only counters are 0). + units; **`--progress`** prints rsync-style per-file progress blocks; + **`--stats`** prints the transfer-statistics block, whose receiver-only + counters (`Matched data`, `Number of deleted files`) are populated from the + receiver's `STATUS_STATS` report. - **Server `--port`** is an alias of the `-p ` TCP listen port (`--dparam port=` overrides the daemon config). @@ -1090,8 +1104,10 @@ These remain after the wave; they are the reasons a row above is ⚠️. - **New directories without `-p` still use FastSync's `0755` creation default** rather than `source & ~umask`; directory metadata is only applied when a directory attribute is requested. -- **`--stats` receiver-only counters** (matched data, file-list bytes, deleted - count) are reported as 0; `--progress` is an aggregate line, not per-file. +- **`--stats` byte totals** (`Total bytes sent`/`received`) are FastSync wire + bytes framed differently from rsync's, so they are not numerically comparable; + the remaining `--stats`/`--progress` divergences are the ones named in their + rows (per-type deleted-file breakdown, root-line/ancestor suppression). - **`--password-file`/`--early-input`/`--hash-credentials`/`--iterations` are FastSync-native** (SCRAM/PBKDF2), not rsync semantics; the batch format is not rsync-interoperable. Credential files are opened with `O_NOFOLLOW` (a symlinked @@ -1230,8 +1246,7 @@ These remain after the wave; the individual rows carry the precise wording. name heuristic, but its candidate eligibility is bounded by the delta engine (both files ≥ 16 KiB, size ratio ≤ 10×), a narrower window than rsync's, so the selected basis — and the `--stats` bandwidth counters — - can differ while the tree stays byte-exact; and **`--bwlimit`** rejects rsync's - `0`/decimal/suffixed rates. + can differ while the tree stays byte-exact. - **`--inc-recursive`/`--no-inc-recursive`** are not implemented (rejected). ### Intentional divergences (explicit ❌ rows) -- 2.54.0 From 309c9aed985d4c7f92454c6c4a1462024ecbdc1d Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 16:46:23 +0200 Subject: [PATCH 48/68] refactor: const-correct dir/root locals in progress itemize --- src/client/client_report.c | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/src/client/client_report.c b/src/client/client_report.c index 2cbe5df..cb787c4 100644 --- a/src/client/client_report.c +++ b/src/client/client_report.c @@ -521,7 +521,7 @@ static void client_progress_emit_ancestors(const Config* config, const char* rel if (key != NULL && array_list_add(g_progress_emitted_keys, key)) { str_hash_set_insert_ref(&g_progress_emitted, key); if (g_change_dirs_active) { - File* dir = progress_dir_lookup(prefix); + const File* dir = progress_dir_lookup(prefix); if (dir != NULL) change_emit_dir_sent(config, dir); } else { @@ -603,7 +603,7 @@ void client_progress_begin(const Config* config) { -i/--out-format it is the root change line (`.d..t...... ./`); otherwise it is the plain --info=name / --progress name line. */ if (g_change_dirs_active) { - File* root = progress_dir_lookup(""); + const File* root = progress_dir_lookup(""); if (root != NULL) change_emit_dir_sent(config, root); } else { -- 2.54.0 From 6537227467aebe05a5e044e855e9f24826e08bf8 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 17:05:54 +0200 Subject: [PATCH 49/68] test(transport): de-race fallback/fd tests and gate for valgrind --- tests/test_transport_tcp.c | 69 ++++++++++++++++++++++++++++++++------ tests/test_transport_tls.c | 8 +++-- 2 files changed, 64 insertions(+), 13 deletions(-) diff --git a/tests/test_transport_tcp.c b/tests/test_transport_tcp.c index 6dd09f5..2010f14 100644 --- a/tests/test_transport_tcp.c +++ b/tests/test_transport_tcp.c @@ -266,12 +266,36 @@ static int count_open_fds(void) { return count; } +/* True when two sockaddrs name the same endpoint (family, address, and port). + * Comparing only the IP would let a connection to a different port on the same + * host pass, so the port is part of the identity. */ +static bool sockaddr_same_endpoint(const struct sockaddr_storage* a, + const struct sockaddr_storage* b) { + if (a->ss_family != b->ss_family) + return false; + if (a->ss_family == AF_INET) { + const struct sockaddr_in* ia = (const struct sockaddr_in*)a; + const struct sockaddr_in* ib = (const struct sockaddr_in*)b; + return ia->sin_port == ib->sin_port && ia->sin_addr.s_addr == ib->sin_addr.s_addr; + } + if (a->ss_family == AF_INET6) { + const struct sockaddr_in6* ia = (const struct sockaddr_in6*)a; + const struct sockaddr_in6* ib = (const struct sockaddr_in6*)b; + return ia->sin6_port == ib->sin6_port && + memcmp(&ia->sin6_addr, &ib->sin6_addr, sizeof(ia->sin6_addr)) == 0; + } + return false; +} + /* Bind + listen on the SECOND address getaddrinfo returns for "localhost", so * the first candidate is connection-refused and the shared connect loop must - * fall back to a later one. Returns the listener fd and its port, or -1 when - * this host does not resolve localhost to at least two addresses (the test then - * skips rather than claiming coverage it does not have). */ -static int bind_second_localhost_address(int* out_port) { + * fall back to a later one. On success the actual bound endpoint is written to + * out_bound/out_bound_len (the caller asserts the winning connect landed on it). + * Returns the listener fd and its port, or -1 when this host does not resolve + * localhost to at least two addresses (the test then skips rather than claiming + * coverage it does not have). */ +static int bind_second_localhost_address(int* out_port, struct sockaddr_storage* out_bound, + socklen_t* out_bound_len) { struct addrinfo hints; memset(&hints, 0, sizeof(hints)); hints.ai_family = AF_UNSPEC; @@ -303,6 +327,10 @@ static int bind_second_localhost_address(int* out_port) { freeaddrinfo(res); return -1; } + if (out_bound) + *out_bound = bound; + if (out_bound_len) + *out_bound_len = bound_len; if (bound.ss_family == AF_INET6) *out_port = ntohs(((struct sockaddr_in6*)&bound)->sin6_port); else @@ -316,18 +344,29 @@ static int bind_second_localhost_address(int* out_port) { * (proving the failed attempt's fd was closed before the retry). */ static void test_tcp_connect_falls_back_to_next_address() { int port = 0; - int listener = bind_second_localhost_address(&port); + struct sockaddr_storage bound; + int listener = bind_second_localhost_address(&port, &bound, NULL); if (listener < 0) return; /* localhost is single-address on this host: cannot exercise fallback */ - int before = count_open_fds(); + + /* The /proc/self/fd delta is unreliable under valgrind (its own lazy fd + * activity perturbs the baseline), so only the functional assertions run + * there; the fd-count checks are skipped. */ + bool check_fds = !is_running_under_valgrind(); + int before = check_fds ? count_open_fds() : -1; + Client* c = client_create(); EXPECT_NOT_NULL(c); EXPECT_TRUE(client_connect(c, "localhost", port)); EXPECT_TRUE(c->file_descriptor >= 0); - if (before >= 0) + /* The winning candidate must be the endpoint we bound (the second + * getaddrinfo entry). Without this, a re-resolution that dropped the second + * address would make the test pass without ever exercising fallback. */ + EXPECT_TRUE(sockaddr_same_endpoint(&c->address, &bound)); + if (check_fds && before >= 0) EXPECT_EQ_INT(count_open_fds(), before + 1); client_disconnect(c); - if (before >= 0) + if (check_fds && before >= 0) EXPECT_EQ_INT(count_open_fds(), before); client_delete(c); close(listener); @@ -336,7 +375,13 @@ static void test_tcp_connect_falls_back_to_next_address() { /* #219 AC3: a connect that fails on every candidate leaves at most one * descriptor (the last failed attempt) and none after client_disconnect. */ static void test_tcp_connect_failed_attempts_do_not_leak_fds() { - /* Reserve an ephemeral port, then close it: connecting to it must fail. */ + if (is_running_under_valgrind()) + return; /* /proc/self/fd delta is perturbed by valgrind's own lazy fds */ + + /* Keep an ephemeral loopback port bound (but NOT listening) for the whole + * assertion: the port stays occupied by our own socket, so the kernel + * deterministically refuses a connect() to it. This closes the bind/close/ + * connect TOCTOU window in which a parallel test could claim the port. */ int probe = socket(AF_INET, SOCK_STREAM, 0); EXPECT_TRUE(probe >= 0); struct sockaddr_in addr; @@ -348,18 +393,20 @@ static void test_tcp_connect_failed_attempts_do_not_leak_fds() { socklen_t addr_len = sizeof(addr); EXPECT_EQ_INT(getsockname(probe, (struct sockaddr*)&addr, &addr_len), 0); int port = ntohs(addr.sin_port); - close(probe); int before = count_open_fds(); Client* c = client_create(); EXPECT_NOT_NULL(c); - EXPECT_FALSE(client_connect(c, "localhost", port)); + /* The literal loopback address has a single getaddrinfo candidate -- the one + * our bound socket owns -- so the connect is deterministically refused. */ + EXPECT_FALSE(client_connect(c, "127.0.0.1", port)); if (before >= 0) EXPECT_TRUE(count_open_fds() <= before + 1); client_disconnect(c); if (before >= 0) EXPECT_EQ_INT(count_open_fds(), before); client_delete(c); + close(probe); } /* #219 AC3: the shared tcp_connect_socket_ex() (used by both the plain and TLS diff --git a/tests/test_transport_tls.c b/tests/test_transport_tls.c index 1385ff8..aeab847 100644 --- a/tests/test_transport_tls.c +++ b/tests/test_transport_tls.c @@ -106,12 +106,16 @@ static void test_client_connect_tls_releases_fd_on_setup_failure() { EXPECT_EQ_INT(getsockname(s->file_descriptor, (struct sockaddr*)&bound, &bound_len), 0); int port = ntohs(bound.sin_port); - int before = tls_count_open_fds(); + /* The /proc/self/fd delta is unreliable under valgrind (its own lazy fd + * activity perturbs the baseline); keep the functional assertions and skip + * only the count checks there. */ + bool check_fds = !is_running_under_valgrind(); + int before = check_fds ? tls_count_open_fds() : -1; Client* c = client_create(); EXPECT_NOT_NULL(c); EXPECT_FALSE(client_connect_tls(c, "127.0.0.1", port, NULL, NULL, NULL)); EXPECT_TRUE(c->file_descriptor == -1); - if (before >= 0) + if (check_fds && before >= 0) EXPECT_EQ_INT(tls_count_open_fds(), before); client_delete(c); server_delete(&s); -- 2.54.0 From 79e45441c0dd3e9cd7d546444ec87f17f86cb7d9 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 17:06:54 +0200 Subject: [PATCH 50/68] fix(receiver): defer --dirs directory mode to avoid EACCES on children file_save_directory_to_disk() applied the exact source mode (fchmod) inline for explicit --dirs/STATUS_MKDIR entries. A restrictive source mode (e.g. 0555) then made the directory read-only before its children were written, so a non-root receiver failed each child with EACCES. The recursive -a path never hit this because it defers directory metadata. Remove the inline fchmod and let the existing deferred dir_metadata_list_apply() stamp the exact mode at end of transfer, as the recursive path does. Keep the inline ownership and xattrs (a direct file_save_to_disk_full() caller has no deferred pass) and document the resulting intentional ordering. Capture errno before output_escape() in the inline timestamp diagnostic so strerror() reports the real error, and add the missing trailing newline to tests/test_xattr.c. --- src/shared/file_save.c | 64 ++++++++++++++---------------------------- tests/test_xattr.c | 2 +- 2 files changed, 22 insertions(+), 44 deletions(-) diff --git a/src/shared/file_save.c b/src/shared/file_save.c index d207968..4cab382 100644 --- a/src/shared/file_save.c +++ b/src/shared/file_save.c @@ -748,50 +748,27 @@ static FileSaveResult file_save_directory_to_disk(const FileSavePlan* plan, bool } else if (ok && identity_copy_as_active()) { ok = false; } - /* Mode next: fchmod also rewrites the ACL mask, so the xattrs/ACLs below must - follow it. The --chmod/permission-bits handling matches the recursive - dir_metadata_list_apply() path exactly. */ - if (ok && file->metadata && plan->config && plan->config->preserve_perms) { - const Config* config = plan->config; - if (dir_fd < 0) { - char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "Failed to open directory %s to set its mode: %s", - escaped_path ? escaped_path : "", strerror(errno)); - free(escaped_path); - } else { - mode_t dir_mode = file->metadata->mode; - bool mode_ready = true; - if (config->chmod_spec && *config->chmod_spec && - !chmod_apply(dir_mode, config->chmod_spec, &dir_mode)) { - char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "Failed to apply --chmod to directory %s", - escaped_path ? escaped_path : ""); - free(escaped_path); - mode_ready = false; - } - if (mode_ready) { - /* rsync -p copies the source directory mode exactly, including - group/other write and the setgid/sticky bits. Setuid/setgid/sticky - are super-user activities: when the connection forbade them - (SUPER_MODE_OFF / --no-super), strip them even under -p. */ - mode_t safe_mode = dir_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777); - if (!privilege_super_mode_permitted(config->super_mode)) - safe_mode &= ~(mode_t)(S_ISUID | S_ISGID | S_ISVTX); - if (fchmod(dir_fd, safe_mode) != 0) { - char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); - log_message(LOG_LEVEL_WARNING, "Failed to set directory mode on %s: %s", - escaped_path ? escaped_path : "", strerror(errno)); - free(escaped_path); - } - } - } - } - /* xattrs/ACLs after fchmod (the mode change can rewrite the ACL mask; the - ACL xattrs must be (re)applied last). Best-effort: a per-attribute failure - is logged and skipped by xattr_apply_fd(), never fatal. */ + /* The final source MODE is deliberately NOT applied inline. A restrictive + source mode (for example 0555) would make the directory unwritable before + its children are created, so a non-root receiver fails each child with + EACCES. The receiver feeds every is_dir entry -- including this explicit + --dirs/STATUS_MKDIR one -- into the deferred DirTimeList, and + dir_metadata_list_apply() stamps the exact mode once the whole transfer has + finished, exactly as it does for the recursive path. Leaving the directory + at its creation mode keeps it writable for the children until then. + + The xattrs below are still applied inline so a direct + file_save_to_disk_full() caller (which has no deferred pass) also gets + --dirs directory xattrs. Because the inline mode is absent, the inline + order here is ownership, then xattrs, then timestamps; the recursive path + (which DOES apply a mode) orders them times, mode, xattrs -- the difference + is intentional, and the deferred pass re-stamps mode and xattrs last. + Best-effort: a per-attribute failure is logged and skipped by + xattr_apply_fd(), never fatal. */ if (ok && plan->config && plan->config->use_xattrs && dir_fd >= 0 && file->xattrs) xattr_apply_fd(dir_fd, file->xattrs); - /* Timestamps last so no later chmod/xattr is mistaken for a content update. + /* Timestamps last so no later inline ownership/xattr change is mistaken for a + content update; the deferred pass re-stamps them after every child write. -J/--omit-dir-times suppresses the directory mtime; --atimes/-U applies only when the source atime is valid, exactly as the recursive path. */ if (ok && file->metadata && plan->config && plan->config->preserve_times && @@ -804,9 +781,10 @@ static FileSaveResult file_save_directory_to_disk(const FileSavePlan* plan, bool times[0].tv_nsec = file->metadata->atime_nsec; } if (parent_fd >= 0 && utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW) != 0) { + int saved_errno = errno; char* escaped_path = output_escape(dir_path, log_get_8_bit_output()); log_message(LOG_LEVEL_WARNING, "Failed to set directory timestamps on %s: %s", - escaped_path ? escaped_path : "", strerror(errno)); + escaped_path ? escaped_path : "", strerror(saved_errno)); free(escaped_path); } } diff --git a/tests/test_xattr.c b/tests/test_xattr.c index 96802f7..5762503 100644 --- a/tests/test_xattr.c +++ b/tests/test_xattr.c @@ -603,4 +603,4 @@ void test_xattr() { test_fake_super_no_real_chown(); test_fake_super_storage_resolution(); test_file_save_directory_applies_xattrs(); -} \ No newline at end of file +} -- 2.54.0 From ff4db2383166f837eadc164cc9dbb9dc78c8b119 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 17:08:03 +0200 Subject: [PATCH 51/68] fix(identity): free TO name on glob success path; harden test oracle --- src/shared/identity.c | 5 +++ tests/test_client_cli.c | 87 +++++++++++++++++++++++++++++++++++++---- 2 files changed, 85 insertions(+), 7 deletions(-) diff --git a/src/shared/identity.c b/src/shared/identity.c index 31b205b..f2b32ad 100644 --- a/src/shared/identity.c +++ b/src/shared/identity.c @@ -654,6 +654,11 @@ int identity_parse_map(Config* config, const char* value, bool is_group) { free(list); return -1; } + /* Every rule emitted by the expansion took its own str_dup of the name, + * so the parse-time copy is unreachable on success: release it here (the + * failure path above already does). `parsed.to_name` is NULL for a + * numeric TO. */ + free(parsed.to_name); continue; } if (identity_parse_from(from_token, is_group, &parsed.from, &parsed.from_hi) != 0) { diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index e6c91b8..ba85fa1 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -3541,9 +3541,14 @@ static void test_parse_args_usermap_rsync_forms() { /* Independent oracle for the FROM name-glob tests: enumerate the sender's * account database and fill `ids` with the DISTINCT ids whose name matches - * `glob`, sorted ascending. Returns the count (bounded by `max`). */ + * `glob`, sorted ascending. Returns the count, or -1 if the matching set + * exceeds `max` distinct ids -- production has no such bound on the number of + * candidates it scans, so a truncated set would under-count runs and flake on + * hosts with very large account databases. Callers must skip (not fail) on + * -1. */ static int cli_collect_glob_ids(const char* glob, bool is_group, int32_t* ids, int max) { int n = 0; + bool overflow = false; if (is_group) { setgrent(); struct group* gr; @@ -3557,8 +3562,13 @@ static int cli_collect_glob_ids(const char* glob, bool is_group, int32_t* ids, i for (int i = 0; i < n; i++) if (ids[i] == id) dup = true; - if (!dup && n < max) - ids[n++] = id; + if (dup) + continue; + if (n >= max) { + overflow = true; + break; + } + ids[n++] = id; } endgrent(); } else { @@ -3574,8 +3584,13 @@ static int cli_collect_glob_ids(const char* glob, bool is_group, int32_t* ids, i for (int i = 0; i < n; i++) if (ids[i] == id) dup = true; - if (!dup && n < max) - ids[n++] = id; + if (dup) + continue; + if (n >= max) { + overflow = true; + break; + } + ids[n++] = id; } endpwent(); } @@ -3588,7 +3603,7 @@ static int cli_collect_glob_ids(const char* glob, bool is_group, int32_t* ids, i } ids[j + 1] = key; } - return n; + return overflow ? -1 : n; } static int cli_count_runs(const int32_t* ids, int n) { @@ -3615,8 +3630,10 @@ static void test_parse_args_identity_map_from_name_glob(bool is_group) { const char glob[3] = {c, '*', '\0'}; int n = cli_collect_glob_ids(glob, is_group, ids, (int)(sizeof(ids) / sizeof(ids[0]))); if (n < 2) - continue; + continue; /* no matches, or the set overflowed the oracle's buffer */ int runs = cli_count_runs(ids, n); + if (runs > MAX_IDENTITY_MAP) + continue; /* production would reject this expansion; try another prefix */ if (runs >= 2 || chosen_c == 0) { chosen_c = c; chosen_n = n; @@ -3630,6 +3647,8 @@ static void test_parse_args_identity_map_from_name_glob(bool is_group) { const char glob[3] = {chosen_c, '*', '\0'}; chosen_n = cli_collect_glob_ids(glob, is_group, ids, (int)(sizeof(ids) / sizeof(ids[0]))); + if (chosen_n < 2) + return; /* account DB changed under us: skip, don't flake */ chosen_runs = cli_count_runs(ids, chosen_n); EXPECT_TRUE(chosen_n >= 2); @@ -3677,6 +3696,58 @@ static void test_parse_args_groupmap_from_name_glob() { test_parse_args_identity_map_from_name_glob(true); } +/* Regression for a leak in the FROM name-glob success path: the TO side is + * parsed into `parsed.to_name` before the glob is expanded, and every emitted + * rule takes its own str_dup of that name -- so the parse-time copy must be + * released before the branch continues. A numeric TO has to_name == NULL and + * cannot expose the leak, hence this uses a NAME TO. The name is resolved on + * the receiver (not here), so any well-formed non-glob name works. Run this + * under ASan/valgrind to catch the leak. */ +static void test_parse_args_identity_map_from_name_glob_name_to(bool is_group) { + int32_t ids[512]; + char chosen_c = 0; + for (char c = 'a'; c <= 'z'; c++) { + const char glob[3] = {c, '*', '\0'}; + int n = cli_collect_glob_ids(glob, is_group, ids, (int)(sizeof(ids) / sizeof(ids[0]))); + if (n < 2) + continue; + if (cli_count_runs(ids, n) > MAX_IDENTITY_MAP) + continue; + chosen_c = c; + break; + } + if (chosen_c == 0) + return; /* no multi-match prefix on this host (skipped, not failed) */ + + const char glob[3] = {chosen_c, '*', '\0'}; + char map_value[32]; + snprintf(map_value, sizeof(map_value), "%s:nobody", glob); + Config* cfg = config_create(); + char* argv[] = {"fastsync", is_group ? "--groupmap" : "--usermap", map_value, "/src", "/dst"}; + int positional_args[2]; + int positional_count = 0; + EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0); + + int got = is_group ? cfg->groupmap_count : cfg->usermap_count; + EXPECT_TRUE(got >= 1); + const IdentityMap* map = is_group ? cfg->groupmap : cfg->usermap; + for (int r = 0; r < got; r++) { + EXPECT_EQ_INT(map[r].to, 0); + EXPECT_NOT_NULL(map[r].to_name); + if (map[r].to_name) + EXPECT_EQ_STR(map[r].to_name, "nobody"); + } + config_delete(cfg); +} + +static void test_parse_args_usermap_from_name_glob_name_to() { + test_parse_args_identity_map_from_name_glob_name_to(false); +} + +static void test_parse_args_groupmap_from_name_glob_name_to() { + test_parse_args_identity_map_from_name_glob_name_to(true); +} + /* #294: an expansion that would push the map past MAX_IDENTITY_MAP must fail * with a clear error rather than silently truncating. Prefill the map to the * cap and then add a wildcard guaranteed to match at least the current user. */ @@ -5053,6 +5124,8 @@ void test_client_cli() { test_parse_args_usermap_rsync_forms(); test_parse_args_usermap_from_name_glob(); test_parse_args_groupmap_from_name_glob(); + test_parse_args_usermap_from_name_glob_name_to(); + test_parse_args_groupmap_from_name_glob_name_to(); test_parse_args_identity_map_from_name_glob_over_cap(); test_parse_args_identity_map_chown_conflict(); test_parse_args_chown(); -- 2.54.0 From aa15099d22fc707c82ca463655b218bd5754e255 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 17:17:04 +0200 Subject: [PATCH 52/68] fix(output): restore --info=flist header; dir metadata on plan path; root line for -d --- RSYNC_COMPAT.md | 4 +- src/client/client_report.c | 175 +++++++++++++++++++++++- tests/integration/test_output_parity.py | 110 +++++++++++++++ 3 files changed, 281 insertions(+), 8 deletions(-) diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index ed3f622..5a6b5b8 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -123,10 +123,10 @@ Every one of those has an entry below with its remaining caveats. |------|-------------------|-----------------|-------| | `--stats` | Give transfer stats | ⚠️ Caveat | Prints transfer statistics. Protocol 2.25.0 populates the receiver-only counters the sender cannot observe (`Matched data`, `Number of deleted files`) from the receiver's `STATUS_STATS` report; the sender tracks the scanned file list per type so `Number of files` carries rsync's `(reg: X, dir: Y, link: Z, special: W)` breakdown (directories come from the scanner's captured directory list for `-a`/`-t`/`-p`, or from a lightweight traversed-directory counter on a plain `-r` run so the `dir:` category is present there too), `Number of regular files transferred` excludes symlinks/specials and up-to-date files, `Total file size` includes symlink target lengths, and `Total transferred file size` counts only transferred files. **Protocol 2.28.0 extends `STATUS_STATS`** with receiver-observed `literal_bytes` and the four `created_*` counters: `Number of created files` now carries rsync's `(reg/dir/link/special)` breakdown (the receiver reports which destination entries it newly created, including implicitly-created parent directories below the transfer root) and `Literal data` is exact for a delta transfer (the receiver counts the literal fragments it stored, not the whole source size) — all differential-tested in the sequential and `--threads` paths against rsync 3.4.1 for fresh-create, update and delta shapes. **Remaining divergences:** rsync's per-type breakdown on `Number of deleted files` is not reproduced; and `Total bytes sent`/`received` are FastSync wire bytes framed differently from rsync's, so they are not numerically comparable | | `-h`, `--human-readable` | Human-readable numbers | ✅ Parity | Formats transfer byte and rate counts using rsync's **decimal** (base-1000) units, matching rsync `-h` (e.g. `1.23M`), not binary units. **A lone `-h` with no transfer arguments prints help instead** (protocol 2.26.0), matching the rsync idiom; `-h` alongside a transfer remains human-readable | -| `-i`, `--itemize-changes` | Per-file change summary | ⚠️ Caveat | Prints rsync-style itemize lines to stdout for files actually sent (also under `-j`/`--threads`). Directory and transfer-root lines are now emitted too: a run produces rsync's `./` root line and per-directory `cd+++++++++`/`.d..t......` lines, rendered by the shared itemize code. **Residual:** the root `./` line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an incremental re-run itemizes directories/symlinks that lack a quick-check where rsync stays silent (unchanged regular files still print nothing, matching single-`-i`) | +| `-i`, `--itemize-changes` | Per-file change summary | ⚠️ Caveat | Prints rsync-style itemize lines to stdout for files actually sent (also under `-j`/`--threads`). Directory and transfer-root lines are now emitted too: a run produces rsync's `./` root line and per-directory `cd+++++++++`/`.d..t......` lines, rendered by the shared itemize code. **Residual:** the root `./` line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision; **every** non-root directory is rendered as created (`cd+++++++++`) because the sender never probes a directory's destination state, so a pre-existing destination directory that rsync reports as unchanged (`.d..t......`) is still itemized as created — this is not limited to re-runs; an incremental re-run additionally itemizes directories/symlinks that lack a quick-check where rsync stays silent (unchanged regular files still print nothing, matching single-`-i`); and directory attribute columns (`%M`/`%U`/`%G`) come from the source | | `--progress` | Show progress | ⚠️ Caveat | Protocol 2.25.0 prints rsync-style per-file progress blocks (percent, transferred/total bytes, rate, elapsed, `(xfr#N, to-chk=M/T)`) fed by the receiver's `STATUS_STATS`, in both the sequential and `--threads` send paths. FastSync also prints rsync's leading `./` transfer-root line and, when progress is requested (`--progress`/`-P`/`--info=progress`) and not `--quiet`, runs a **paths-only metadata pre-scan** (no file reads, no hashing) that supplies rsync's file-list total `T` for the `to-chk` denominator and the directory names; `--delete-during`/`--delete-delay` reuse their existing keep-set pre-scan instead of walking twice, and non-progress runs are untouched. Per-directory name lines are emitted (trailing `/`), and symlink (` -> target`) and special entries are named too, so a **fresh multi-directory tree's name set and `to-chk` denominator match rsync 3.4.1** (differential test, sequential and `--threads`) and a **single-file transfer's name lines and deterministic frames remain byte-identical** to rsync. **Order parity (parity-2.29):** the sequential scanner now emits entries in rsync's sorted depth-first flist order (non-directories ascending, then directories ascending), so the interleaving and the `to-chk` numerator match rsync for the default single-threaded transfer (differential `test_parity_order.py`; `--threads` has no rsync analogue and stays unordered). **Remaining divergences:** the leading `./` root line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an ancestor directory line is emitted whenever a child transfers (rsync suppresses it when the directory itself is unchanged); on a re-run, entries without a quick-check (symlinks, empty directories) are still named where rsync stays silent; and the rate/ETA are wall-clock dependent | | `-P` | Same as --partial --progress | ✅ Parity | Parses to `--partial` + `--progress`. The independent `--partial` retention semantics are rsync parity: an interrupted write retains the already-written temp at the destination (best-effort) so a later `--append`/`--append-verify` can resume. Progress presentation is owned by the `--progress` row; there is no separate `-P` divergence | -| `--out-format=FORMAT` | Custom output format | ❌ Divergent | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%c` `%C` `%i` `%M` `%o` `%U` `%G` `%t` `%%`. `%C` now uses the negotiated transfer algorithm (`--checksum-choice`, default `xxh128`, seed 0) and renders every algorithm exactly like rsync — xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, `none` a blank 2-char column — differential-tested across all algorithms. `%f`/`%n`/`%l`/`%i`/`%M`/`%U`/`%G`/`%B` also match. **Reclassified because `%b`/`%c` are protocol-specific and cannot match:** a differential against rsync 3.4.1 shows whole-file `%c = 16` for both, but rsync whole-file `%b = filesize + 27 + transfer-digest-bytes` (39 for a 0-byte file; 43/35/47 for xxh128/xxh64/sha1 on a 12-byte file) while FastSync `%b` counts its own framing; in delta mode rsync `%c = 16 + 6·ceil(filesize/block_size)` (verified at block sizes 512/700/1024/2048) while FastSync counts its own signature handshake, and rsync `%b` is its token stream. FastSync's wire bytes are a different quantity, so exact `%b`/delta-`%c` equality is impossible. Directory and transfer-root lines are now emitted (rsync's `./` root line and per-directory `cd...`/`.d..t...` lines), with the same residual as `-i`: the root line is emitted unconditionally and an incremental re-run may itemize directories/symlinks without a quick-check | +| `--out-format=FORMAT` | Custom output format | ❌ Divergent | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%c` `%C` `%i` `%M` `%o` `%U` `%G` `%t` `%%`. `%C` now uses the negotiated transfer algorithm (`--checksum-choice`, default `xxh128`, seed 0) and renders every algorithm exactly like rsync — xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, `none` a blank 2-char column — differential-tested across all algorithms. `%f`/`%n`/`%l`/`%i`/`%M`/`%U`/`%G`/`%B` also match. **Reclassified because `%b`/`%c` are protocol-specific and cannot match:** a differential against rsync 3.4.1 shows whole-file `%c = 16` for both, but rsync whole-file `%b = filesize + 27 + transfer-digest-bytes` (39 for a 0-byte file; 43/35/47 for xxh128/xxh64/sha1 on a 12-byte file) while FastSync `%b` counts its own framing; in delta mode rsync `%c = 16 + 6·ceil(filesize/block_size)` (verified at block sizes 512/700/1024/2048) while FastSync counts its own signature handshake, and rsync `%b` is its token stream. FastSync's wire bytes are a different quantity, so exact `%b`/delta-`%c` equality is impossible. Directory and transfer-root lines are now emitted (rsync's `./` root line and per-directory `cd...`/`.d..t...` lines), with the same residual as `-i`: the root line is emitted unconditionally, every non-root directory renders as created because the sender does not probe directory destination state (so a pre-existing unchanged directory still shows `cd+++++++++`), and directory attribute columns (`%M`/`%U`/`%G`) come from the source | | `--log-file=FILE` | Log to file | ✅ Parity | `log_file` config field | | `--log-file-format=FMT` | Log format | ✅ Parity | Requires `--log-file`; writes one template line per transferred file using the same token set as `--out-format` (including `%b` as the wire byte count) | | `--8-bit-output`, `-8` | Leave high-bit chars unescaped | ✅ Parity | Applies to displayed paths and protocol debug output | diff --git a/src/client/client_report.c b/src/client/client_report.c index cb787c4..ff45260 100644 --- a/src/client/client_report.c +++ b/src/client/client_report.c @@ -597,7 +597,10 @@ void client_progress_begin(const Config* config) { } return; } - if (g_progress_active) + /* -i/--out-format alone do not print the header, but --progress always does + and --info=flist does under any output mode (rsync prints it for + `-i --info=flist` and `--out-format=... --info=flist` too). */ + if (g_progress_active || (!config->quiet && info_flag_enabled(config, LOG_INFO_FLIST))) printf("sending incremental file list\n"); /* rsync prints the transfer-root directory before the first entry. Under -i/--out-format it is the root change line (`.d..t...... ./`); otherwise it @@ -677,6 +680,125 @@ static bool progress_precount_add_dir(ProgressPrecount* p, const char* path) { return false; } +static int progress_path_compare(const void* left, const void* right) { + const char* const* a = (const char* const*)left; + const char* const* b = (const char* const*)right; + return strcmp(*a, *b); +} + +/* Sort the collected directory paths and drop duplicates so a large + * --files-from list (many entries sharing an implied ancestor) cannot grow the + * list without bound. */ +static void progress_precount_dedup_dirs(ArrayList* dir_paths) { + if (dir_paths == NULL || dir_paths->size < 2) + return; + qsort(dir_paths->items, (size_t)dir_paths->size, sizeof(char*), progress_path_compare); + int write = 0; + for (int read = 0; read < dir_paths->size; read++) { + char* current = (char*)dir_paths->items[read]; + if (write > 0 && strcmp((char*)dir_paths->items[write - 1], current) == 0) { + free(current); + continue; + } + dir_paths->items[write++] = current; + } + dir_paths->size = write; +} + +/* Create a metadata-bearing directory File for the transfer-relative directory + * `rel` ("" is the transfer root), stat'ing it below config->send_directory. + * The -d/--files-from dirs generator never traverses directories, so this + * synthesizes the metadata the recursive scanner captures through + * scanner_capture_dir_time, letting -i/--out-format render %M/%B/%U/%G and the + * transfer-root/ancestor lines identically on both paths. Returns NULL when + * the path cannot be stat'd as a directory or on allocation failure (the line + * is then simply omitted, exactly as before). */ +static File* progress_precount_make_dir(const Config* config, const char* rel) { + if (config == NULL || config->send_directory == NULL) + return NULL; + char* fs_path = (rel == NULL || rel[0] == '\0') ? str_dup(config->send_directory) + : path_cat(config->send_directory, rel); + if (fs_path == NULL) + return NULL; + struct stat st; + if (stat(fs_path, &st) != 0 || !S_ISDIR(st.st_mode)) { + free(fs_path); + return NULL; + } + File* file = file_create(fs_path); + if (file == NULL) { + free(fs_path); + return NULL; + } + file->is_dir = true; + file->metadata = + file_metadata_create(fs_path, &st, config->preserve_atimes, config->preserve_crtimes); + file->send_path = str_dup(rel != NULL ? rel : ""); + free(fs_path); + if (file->metadata == NULL || file->send_path == NULL) { + file_destroy(file); + return NULL; + } + return file; +} + +/* The -d/--files-from dirs generator neither traverses nor records directories, + * so its metadata walk captures no Files. Synthesize the transfer root and + * every listed/implied directory from `entry_rels` so -i/--out-format emits the + * same root and ancestor lines the recursive scan does. `root_emitted` is true + * when the generator itself emits the root entry (bare `-d `), whose + * data-pass line must not be duplicated. Returns false only on allocation + * failure. */ +static bool progress_precount_synthesize_dirs(const Config* config, ProgressPrecount* out, + const ArrayList* entry_rels, bool root_emitted) { + for (int i = 0; i < entry_rels->size; i++) { + const char* rel = (const char*)entry_rels->items[i]; + if (rel == NULL) + continue; + size_t len = strlen(rel); + for (size_t j = 1; j < len; j++) { + if (rel[j] != '/') + continue; + /* --no-implied-dirs: rsync neither creates nor itemizes an implied parent, + so only explicitly listed directories get a line. */ + if (config->no_implied_dirs) + break; + char* prefix = malloc(j + 1); + if (prefix == NULL) + return false; + memcpy(prefix, rel, j); + prefix[j] = '\0'; + if (!progress_precount_add_dir(out, prefix)) { + free(prefix); + return false; + } + free(prefix); + } + } + progress_precount_dedup_dirs(out->dir_paths); + for (int i = 0; i < out->dir_paths->size; i++) { + const char* rel = (const char*)out->dir_paths->items[i]; + File* dir = progress_precount_make_dir(config, rel); + if (dir == NULL) + continue; + if (!array_list_add(out->dir_files, dir)) { + file_destroy(dir); + return false; + } + } + /* Emit the transfer root only when the generator actually emitted an entry: + --prune-empty-dirs (or an empty --files-from list) transfers nothing, and + rsync prints no root line then either. */ + if (!root_emitted && entry_rels->size > 0) { + File* root = progress_precount_make_dir(config, ""); + if (root != NULL && !array_list_add(out->dir_files, root)) { + file_destroy(root); + return false; + } + } + return true; +} + /* Metadata-only walk collecting the full file-list total and every directory * name. It uses its own scanner (fresh filter compilation and hard-link table) * so the data pass's link-group state is never perturbed. */ @@ -689,9 +811,21 @@ static bool progress_precount_scan(const Config* config, ProgressPrecount* out) return false; } out->total = 0; + /* The -d/--files-from dirs generator never calls scanner_capture_dir_time, so + the walk below captures no directory Files. Record every emitted entry's + transfer-relative name so the implied ancestors can be synthesized once the + walk is done. */ + bool synthesize = g_change_dirs_active && config->dirs; + ArrayList* entry_rels = synthesize ? array_list_create(free) : NULL; + if (synthesize && entry_rels == NULL) { + progress_precount_dispose(out); + return false; + } + bool root_emitted = false; PreparedScanner prepared; memset(&prepared, 0, sizeof(prepared)); if (!prepare_scanner(config, 0, &prepared)) { + array_list_delete(entry_rels); progress_precount_dispose(out); return false; } @@ -722,8 +856,23 @@ static bool progress_precount_scan(const Config* config, ProgressPrecount* out) out->total += (unsigned long long)chunk->element_count; for (int i = 0; i < chunk->element_count && ok; i++) { const File* f = chunk->items[i]; - if (f != NULL && f->is_dir) - ok = progress_precount_add_dir(out, delete_display_path(config, file_wire_path(f))); + if (f == NULL) + continue; + const char* rel = delete_display_path(config, file_wire_path(f)); + if (f->is_dir) { + if (rel != NULL && rel[0] == '\0') + root_emitted = true; + ok = progress_precount_add_dir(out, rel); + if (!ok) + break; + } + if (synthesize && rel != NULL) { + char* dup = str_dup(rel); + if (dup == NULL || !array_list_add(entry_rels, dup)) { + free(dup); + ok = false; + } + } } chunk_destroy(chunk); } @@ -733,9 +882,18 @@ static bool progress_precount_scan(const Config* config, ProgressPrecount* out) } prepared_scanner_destroy(&prepared); if (!ok) { + array_list_delete(entry_rels); progress_precount_dispose(out); return false; } + if (synthesize) { + bool synth_ok = progress_precount_synthesize_dirs(config, out, entry_rels, root_emitted); + array_list_delete(entry_rels); + if (!synth_ok) { + progress_precount_dispose(out); + return false; + } + } /* Build the name -> File lookup from the captured directory Files. */ for (int i = 0; i < out->dir_files->size; i++) { File* f = (File*)out->dir_files->items[i]; @@ -821,9 +979,14 @@ void client_progress_prepare(const Config* config, const ArrayList* plan_dirs, g_change_dirs_active = config->itemize_changes || config->out_format != NULL; if (!g_progress_active && !g_change_dirs_active) return; - bool ok = plan_dirs != NULL ? progress_precount_from_plan_dirs( - config, plan_dirs, plan_non_dir_count, &g_progress_precount) - : progress_precount_scan(config, &g_progress_precount); + /* -i/--out-format render directory metadata (%M/%B/%U/%G) that only the + metadata walk captures; the --delete-during/--delete-delay plan list has no + metadata, so prefer the walk whenever a change line is rendered. Pure + --progress keeps reusing the plan list and its cheaper path-only pass. */ + bool ok = (plan_dirs != NULL && !g_change_dirs_active) + ? progress_precount_from_plan_dirs(config, plan_dirs, plan_non_dir_count, + &g_progress_precount) + : progress_precount_scan(config, &g_progress_precount); if (!ok) { g_progress_total = 0; return; diff --git a/tests/integration/test_output_parity.py b/tests/integration/test_output_parity.py index b3b2f89..577a1ee 100644 --- a/tests/integration/test_output_parity.py +++ b/tests/integration/test_output_parity.py @@ -240,6 +240,75 @@ class TestItemizeParity: ) assert ".d..t...... ./" in fast, f"missing root line: {result.stdout!r}" + @requires_rsync + @pytest.mark.ci + def test_itemize_info_flist_header_matches_rsync(self, shared_server): + """`-i --info=flist` prints rsync's file-list header: the -i change + lines alone do not enable the flist category, but an explicit --info=flist + must not be suppressed when itemizing.""" + source = os.path.join(TEST_DATA_DIR, "out_itemfl_src") + dest = os.path.join(TEST_DATA_DIR, "out_itemfl_dst") + rdst = os.path.join(TEST_DATA_DIR, "out_itemfl_rdst") + _make_output_tree(source) + clean_dir(dest) + clean_dir(rdst) + flags = ["-a", "-i", "--info=flist"] + rsync_result = _rsync(flags + [source + "/", rdst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, result.stderr[:300] + assert "sending incremental file list" in rsync_result.stdout + assert "sending incremental file list" in result.stdout, result.stdout + # -i alone (no explicit --info=flist) must stay silent like rsync. + clean_dir(dest) + clean_dir(rdst) + rsync_plain = _rsync(["-a", "-i", source + "/", rdst + "/"]) + plain, _ = run_client(source, dest, flags=["-a", "-i"], + port=shared_server.port) + assert "sending incremental file list" not in rsync_plain.stdout + assert "sending incremental file list" not in plain.stdout, plain.stdout + + @requires_rsync + @pytest.mark.ci + def test_itemize_files_from_dirs_root_and_ancestors(self, shared_server): + """The -d/--files-from dirs generator emits the transfer-root line and + rsync's implied ancestor directory lines. The generator traverses no + directories, so those must be synthesized from the listed entries.""" + source = os.path.join(TEST_DATA_DIR, "out_itemff_src") + dest = os.path.join(TEST_DATA_DIR, "out_itemff_dst") + rdst = os.path.join(TEST_DATA_DIR, "out_itemff_rdst") + clean_dir(source) + os.makedirs(os.path.join(source, "sub", "deep")) + with open(os.path.join(source, "sub", "deep", "d.txt"), "wb") as fh: + fh.write(b"deep\n") + clean_dir(dest) + clean_dir(rdst) + listing = os.path.join(TEST_DATA_DIR, "out_itemff.list") + with open(listing, "w") as fh: + fh.write("sub/deep/d.txt\n") + + flags = ["-d", "-i", "--files-from=" + listing] + rsync_result = _rsync(flags + [source + "/", rdst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + result, _ = run_client(source, dest, flags=flags, port=shared_server.port) + assert result.returncode == 0, result.stderr[:300] + + def dir_lines(text): + return sorted(line for line in text.splitlines() + if line.rsplit(" ", 1)[-1].endswith("/")) + + # rsync emits the implied parents (sub/, sub/deep/) but never the root + # here; FastSync emits the same set plus its unconditional root line. + expected = [line for line in dir_lines(rsync_result.stdout) + if not line.rsplit(" ", 1)[-1] == "./"] + fast = dir_lines(result.stdout) + assert [line for line in fast if not line.rsplit(" ", 1)[-1] == "./"] == expected, ( + f"rsync={rsync_result.stdout!r} fastsync={result.stdout!r}" + ) + assert "cd+++++++++ sub/" in fast, result.stdout + assert "cd+++++++++ sub/deep/" in fast, result.stdout + assert any(line.rsplit(" ", 1)[-1] == "./" for line in fast), result.stdout + @requires_rsync @pytest.mark.ci def test_itemize_modified_file_matches_rsync(self, shared_server): @@ -318,6 +387,47 @@ class TestOutFormatParity: if line: assert pattern.match(line), f"bad %M format: {line!r}" + @requires_rsync + @pytest.mark.ci + def test_out_format_directory_metadata_with_delete_during(self): + """--delete-during/--delete-delay reuse the per-directory plan pre-scan, + whose list carries no metadata. Directory %M/%B/%U/%G must still come + from the source, exactly as the plain recursive scan renders them.""" + source = os.path.join(TEST_DATA_DIR, "out_fmtmeta_src") + dest = os.path.join(TEST_DATA_DIR, "out_fmtmeta_dst") + rdst = os.path.join(TEST_DATA_DIR, "out_fmtmeta_rdst") + clean_dir(source) + os.makedirs(os.path.join(source, "sub", "deep")) + with open(os.path.join(source, "a.txt"), "wb") as fh: + fh.write(b"hello\n") + with open(os.path.join(source, "sub", "b.txt"), "wb") as fh: + fh.write(b"world\n") + clean_dir(dest) + clean_dir(rdst) + + def dir_lines(text): + # Directory names are the last whitespace-separated token. + return sorted(line for line in text.splitlines() + if line.rsplit(" ", 1)[-1].endswith("/") + and line.rsplit(" ", 1)[-1] != "./") + + for timing in ("--delete-during", "--delete-delay"): + for fmt in ("%M %n", "%B %n", "%U %G %n"): + clean_dir(dest) + clean_dir(rdst) + flags = ["-a", "--out-format=" + fmt, timing] + rsync_result = _rsync(flags + [source + "/", rdst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + result, _ = run_client(source, dest, flags=flags, port=server.port) + assert result.returncode == 0, result.stderr[:300] + assert dir_lines(result.stdout) == dir_lines(rsync_result.stdout), ( + f"{timing} {fmt}: rsync={rsync_result.stdout!r} " + f"fastsync={result.stdout!r}" + ) + assert "1970/" not in result.stdout, result.stdout + class TestListOnlyParity: @requires_rsync -- 2.54.0 From f96f1764af74442494c3e3cf1804002fb104adfb Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 21:58:54 +0200 Subject: [PATCH 53/68] feat(cli): accept --inc-recursive no-op and in-root absolute --temp-dir --- README.md | 4 +- RSYNC_COMPAT.md | 30 +++-- src/client/client_cli.c | 13 +++ src/client/usage.c | 3 + src/shared/file_save.c | 113 +++++++++++++----- tests/integration/test_temp_dir_absolute.py | 120 ++++++++++++++++++++ tests/test_client_cli.c | 8 +- tests/test_file.c | 12 ++ 8 files changed, 257 insertions(+), 46 deletions(-) create mode 100644 tests/integration/test_temp_dir_absolute.py diff --git a/README.md b/README.md index dee3e96..1eb97a4 100644 --- a/README.md +++ b/README.md @@ -219,7 +219,7 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) | | `--max-delete ` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync | | `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication; the fixed `.fastsync-stage` staging name diverges from rsync — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) | -| `-T, --temp-dir ` | Scratch directory for temp files before the atomic install; confined to the receive root (relative only), with an `EXDEV` non-atomic copy fallback | +| `-T, --temp-dir ` | Scratch directory for temp files before the atomic install; confined to the receive root (a relative path resolves below it; an absolute path is accepted only when it canonicalizes inside it), with an `EXDEV` non-atomic copy fallback | | `-n, --dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. | | `-v, --verbose` | Enable debug logging | | `-q, --quiet` | Suppress non-error output | @@ -606,7 +606,7 @@ remote SSH argv is already built injection-safe. | `--max-alloc ` | Maximum single allocation (binary units; default 1G; `0` = no local limit). | | `--max-depth ` | Limit recursive scanning depth; zero means unlimited. | | `-b, --backup` | Back up overwritten files. | -| `-T, --temp-dir ` | Scratch directory for temp files before the atomic install (confined to the receive root; `EXDEV` falls back to a non-atomic copy). | +| `-T, --temp-dir ` | Scratch directory for temp files before the atomic install (confined to the receive root: relative resolves below it, absolute must canonicalize inside it; `EXDEV` falls back to a non-atomic copy). | | `--backup-dir ` | Store backups under a separate directory (requires `--backup`). | | `--suffix ` | Set the backup filename suffix (default: `~`). | | `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir `, completed files are written under the partial directory and installed atomically. | diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 5a6b5b8..f70eaf2 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -6,15 +6,15 @@ This document maps rsync's full feature set to FastSync's current implementation | Status | Count | Description | |--------|-------|-------------| -| ✅ Parity | 117 | Reproduces rsync's semantics for this option's scope | +| ✅ Parity | 118 | Reproduces rsync's semantics for this option's scope | | ⚠️ Caveat | 13 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) | -| ❌ Divergent | 27 | Rejected, an accepted no-op, deliberately non-rsync (native config/auth/batch, privileged namespaces, safe-subset privilege), or impossible on any portable filesystem call | +| ❌ Divergent | 26 | Rejected, an accepted no-op, deliberately non-rsync (native config/auth/batch, privileged namespaces, safe-subset privilege), or impossible on any portable filesystem call | | **Total** | **157** | One row per rsync option/feature group; a row may name several spellings | This matrix reports honest rsync parity, not "implemented" as a synonym for "parsed". A ✅ row matches rsync for the option's scope. A ⚠️ row is real and tested but diverges in at least one documented way. An ❌ row is either -rejected (`--protocol` with any value but the current one, `--inc-recursive`), +rejected (`--protocol` with any value but the current one), an accepted no-op (`-s`/`--secluded-args`, `--protect-args`, `--old-args`), deliberately non-rsync and non-interoperable (the FastSync daemon config/auth, the batch container, `--fake-super`'s xattr format, `--copy-as` credential @@ -161,7 +161,7 @@ Every one of those has an entry below with its remaining caveats. | `--no-implied-dirs` | Don't send implied dirs with -R | ✅ Parity | With `-R`, rsync creates the ancestor directories implied by a listed path and, with `--no-implied-dirs`, omits their attributes from the transfer so they keep the destination's own state (or are created with default attributes when absent). Protocol 2.26.0 matches this: without `--files-from` the implied-dir walk applies per-attribute metadata only to explicitly transferred directories, and with `-R --files-from` a listed file whose parent is not itself listed is placed normally — the missing implied parent is created with default attributes (not the source's) and the file transfers with `rc 0`, exactly like rsync 3.4.1 (a differential test verifies the modes and mtimes with and without the flag). Works single-threaded and under `-j`/`--threads` | | `-d`, `--dirs`, `--old-dirs`, `--old-d` | Transfer dirs without recursing | ✅ Parity | Protocol 2.26.0 implements rsync's one-level `-d` listing for `dir`, `dir/` and `.`: the source's immediate contents are transferred (files with content, directories as explicit entries), matching rsync's destination tree in a differential test. `--dirs --files-from` transfers exactly the listed items — a listed directory is created empty and a listed file with content — under the same `-R` layout rules. A plain recursive scan also recreates empty source directories now: the scanner emits a payload-less directory entry (with metadata) for every traversed directory that produced no transferred or descended child, unless `-m/--prune-empty-dirs` suppresses it or the run is `--files-from`/`--list-only` (a directory emptied by filtering is recreated too, matching rsync). Directory entries cross as `STATUS_MKDIR` and appear in the delete manifest, so `--delete` prunes correctly and an empty listed directory survives; an incoming directory replaces a destination regular file (rsync removes the non-directory and creates the directory), verified differentially. Directory times are applied at the end of the transfer; modes/ownership follow the per-attribute policy. Under `--delay-updates` directories are created immediately while only regular files are staged, exactly as rsync does | | `--mkpath` | Create missing path components | ✅ Parity | Wire option (client → server). At connection start the server creates the client's destination root directory (and any missing leading components below its own authorized root) when `--mkpath` is set, failing the connection cleanly if it cannot. Without `--mkpath` a destination root that does not exist yet is rejected up front (rsync semantics), so the flag is the only way to transfer into a not-yet-created destination directory. Creation is confined by the same secure mkdir walk as file writes (`O_NOFOLLOW`, no `..`) | -| `--inc-recursive`, `--no-inc-recursive` | Incremental recursion mode | ❌ Divergent | rsync's man-page-only scanning-mode switch (and its short aliases). FastSync always performs a single full recursive scan, so both spellings are rejected as unknown options rather than accepted as a no-op; there is no incremental-recursion engine to toggle. A genuine implementation would be a scan-architecture change with no benefit for FastSync's push model | +| `--inc-recursive`, `--no-inc-recursive` | Incremental recursion mode | ✅ Parity | rsync's man-page-only scanning-mode switch. FastSync always performs a single full recursive scan, so both spellings are accepted as inert no-ops and the destination is identical whichever mode the caller requests — the same treatment as `-r`/`--recursive`, which is likewise a no-op. The switch is a scan-implementation detail with no observable effect on the final tree (rsync's own `--no-inc-recursive` selects a full scan, which is exactly FastSync's behavior) | ## 5. Transfer Modifications @@ -184,7 +184,7 @@ Every one of those has an entry below with its remaining caveats. | `--backup-dir=DIR` | Backup directory hierarchy | ✅ Parity | `backup_dir` config field | | `--suffix=SUFFIX` | Backup suffix (default ~) | ✅ Parity | `suffix` config field | | `--delay-updates` | Put updated files in place at end | ❌ Divergent | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication, and **`--force` is honored at publication** (protocol 2.23.0): a staged regular file or symlink may replace a destination directory that blocks it. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). **Reclassified Divergent (differential evidence):** the staging name is fixed and a delayed run wipes a pre-existing destination tree of that name at start even without `--delete`, whereas rsync uses its own internal temp name and leaves a genuine destination entry named `.fastsync-stage` untouched (`test_delay_updates_staging_name_collision_residual`); deletion also runs before publication while rsync's `--delay-updates` implies `--delete-after`. Works in single-threaded and `-j`/`--threads` modes | -| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ❌ Divergent | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). A **relative** dir matches rsync exactly: it is resolved below the receive/destination root and must already exist (differentially verified: `rsync -a --temp-dir=scratch src/ dst/` and FastSync produce identical trees and an empty scratch dir). **Reclassified as a deliberate divergence because an absolute `--temp-dir` is rejected by the receiver** — it is resolved verbatim by rsync standalone (which will use `/tmp` or any other absolute directory, including one outside the destination), but FastSync's security-reviewed receiver confines the scratch dir to the authorized receive root and rejects any absolute path or one containing `..`. **Audit-cycle hardening:** the opened dir is additionally judged by the real path of its fd (`/proc/self/fd`), so a client-planted symlink under the receive root cannot redirect receiver scratch files outside the authorized root (an escaping target is refused with `EACCES`), while an in-root symlink to another filesystem — the `EXDEV` fallback case — still works. A differential test confirms rsync exits 0 using an absolute scratch dir while FastSync refuses before writing anything into it (the scratch dir stays empty). Its daemon mode also confines relative to the module, but standalone rsync's absolute-temp-dir behavior is not reproduced because it would let a client place receiver scratch files outside the sandbox. Temp copies use a unique name in the scratch dir and are atomically renamed into place; **on `EXDEV` (scratch dir and destination on different filesystems, reachable via a confined relative symlink) the receiver falls back to a non-atomic copy instead of aborting**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir | +| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ❌ Divergent | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). A **relative** dir matches rsync exactly: it is resolved below the receive/destination root and must already exist (differentially verified: `rsync -a --temp-dir=scratch src/ dst/` and FastSync produce identical trees and an empty scratch dir). An **absolute** dir is accepted when it canonicalizes (`realpath(3)`) inside the receive root, so an in-root absolute scratch path is usable and used (unit- and integration-tested for both the local batch apply and a live TCP transfer). **Remaining divergence:** an absolute `--temp-dir` that escapes the receive root is rejected, and so is a relative one containing `..` — rsync standalone resolves an absolute `--temp-dir` verbatim (it will use `/tmp` or any other directory, including one outside the destination), but FastSync's security-reviewed receiver confines the scratch dir to the authorized receive root and refuses an out-of-root path before writing anything. **Audit-cycle hardening:** the opened dir is additionally judged by the real path of its fd (`/proc/self/fd`), so a client-planted symlink under the receive root cannot redirect receiver scratch files outside the authorized root (an escaping target is refused with `EACCES`), while an in-root symlink to another filesystem — the `EXDEV` fallback case — still works. A differential test confirms rsync exits 0 using an out-of-root absolute scratch dir while FastSync refuses before writing anything into it (the scratch dir stays empty). Its daemon mode also confines relative to the module. Temp copies use a unique name in the scratch dir and are atomically renamed into place; **on `EXDEV` (scratch dir and destination on different filesystems, reachable via a confined relative symlink) the receiver falls back to a non-atomic copy instead of aborting**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir | | `--partial` | Keep partially transferred files | ✅ Parity | On a failed/interrupted write the already-written temp file is retained at the destination path (best-effort rename instead of unlink) so a later `--append`/`--append-verify` run can resume it. Retention never runs when no data was actually written or under `--ignore-existing`/`--existing` (the destination is not ours to overwrite), and it only ever renames the already-written temp. A failed rename falls back to the normal unlink | | `--partial-dir=DIR` | Keep partial files in DIR | ✅ Parity | The working file is written under the confined partial directory (a relative dir below the receive root) and atomically renamed into place once complete, so an interrupted transfer leaves a resumable copy there and completed transfers do not linger under it. `--inplace` bypasses the partial dir (rsync parity), and combining `--inplace` with `--partial-dir` is now **rejected up front** with rsync's message (`--inplace cannot be used with --partial-dir`) instead of silently ignoring the partial dir. **Implies `--partial`** (audit-cycle fix, matching rsync 3.4.1, which sets `keep_partial` after option parsing): `--partial-dir=DIR` alone retains an interrupted transfer's partial, and the implication wins over an explicit `--no-partial` regardless of order | @@ -956,7 +956,7 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, and the triage cycle.** ✅ Parity 117 / ⚠️ Caveat 13 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--delete-before`, `--filter`, `-F`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, the triage cycle, and a later no-wire CLI parity fix.** ✅ Parity 118 / ⚠️ Caveat 13 / ❌ Divergent 26 = 157 rows. A no-wire CLI-parity pass accepted `--inc-recursive`/`--no-inc-recursive` as inert no-ops (❌ → ✅, since FastSync's full scan is rsync's `--no-inc-recursive` and the destination is identical) and narrowed the `--temp-dir` divergence by accepting an absolute path that canonicalizes inside the receive root (the row stays ❌ for out-of-root absolute paths). The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--delete-before`, `--filter`, `-F`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that receiver filter engine, flipping `--filter=RULE` back to ✅ (see above; the @@ -1021,7 +1021,9 @@ integration tests unless it is explicitly listed as a limitation. ### Filesystem and deletion semantics - **`--temp-dir` is confined to the receive root on the receiver:** a relative - dir resolves below it; an absolute path or one containing `..` is rejected. + dir resolves below it, and an absolute path is accepted only when + `realpath(3)` confirms it is inside the canonical receive root; an + out-of-root absolute path or one containing `..` is rejected. An `EXDEV` install falls back to a non-atomic copy instead of aborting. (The confined receiver path cannot be mount-tested in the CI container — no `CAP_SYS_ADMIN` and unprivileged user namespaces are disabled — so the @@ -1095,8 +1097,9 @@ These remain after the wave; they are the reasons a row above is ⚠️. link-following tool can follow a link outside the receive root. Use `--safe-links` when the source is untrusted. `--trust-sender` does **not** affect symlink targets. -- **`--temp-dir` absolute/foreign-filesystem paths are rejected by the - receiver** (rsync's daemon also confines; standalone rsync differs). +- **`--temp-dir` out-of-root absolute and foreign-filesystem paths are rejected + by the receiver** (an absolute path that canonicalizes inside the receive root + is accepted; rsync's daemon also confines; standalone rsync differs). - **`--copy-devices` reads a bounded `st_size`** rather than rsync's unbounded device read. - **A broken symlink referent under `--copy-links`/`--copy-unsafe-links` exits 0** @@ -1235,8 +1238,9 @@ These remain after the wave; the individual rows carry the precise wording. `--ignore-errors` exits 23 but its EACCES differential is not exercised in CI. - **`--delay-updates`** uses a fixed staging name with an advisory lock and - deletes before publication; **`--temp-dir`** rejects absolute/foreign paths - (deliberately confined, see the row); **`--remote-option`** is SSH-only. + deletes before publication; **`--temp-dir`** rejects out-of-root absolute and + foreign paths (in-root absolute paths are accepted; deliberately confined, see + the row); **`--remote-option`** is SSH-only. **`--iconv`** now matches rsync's push direction (destination charset = the spec's REMOTE half; a server `--iconv` overrides it). - **Basis dirs** now use rsync's metadata quick-check by default (track 5a) and @@ -1247,7 +1251,9 @@ These remain after the wave; the individual rows carry the precise wording. engine (both files ≥ 16 KiB, size ratio ≤ 10×), a narrower window than rsync's, so the selected basis — and the `--stats` bandwidth counters — can differ while the tree stays byte-exact. -- **`--inc-recursive`/`--no-inc-recursive`** are not implemented (rejected). +- **`--inc-recursive`/`--no-inc-recursive`** are accepted as inert no-ops: + FastSync always performs a single full recursive scan (equivalent to + rsync's `--no-inc-recursive`), so the destination is identical either way. ### Intentional divergences (explicit ❌ rows) diff --git a/src/client/client_cli.c b/src/client/client_cli.c index 12f0c6d..d89177a 100644 --- a/src/client/client_cli.c +++ b/src/client/client_cli.c @@ -962,6 +962,13 @@ static const OptionEntry OPTION_TABLE[] = { /* rsync -r/--recursive: FastSync is always recursive, so this is a * faithful no-op (accepted silently, never consumes an argument). */ {"--recursive", "-r", OPT_NOOP, 0}, + /* rsync's incremental-recursion scan-mode switch. FastSync always performs + * a single full recursive scan, so both spellings are accepted as no-ops: + * the destination is identical whichever mode the caller requests. + * --no-inc-recursive is handled before the generic --no-* negation branch + * (see cli_handle_pre_negation) but is registered here for discoverability. */ + {"--inc-recursive", NULL, OPT_NOOP, 0}, + {"--no-inc-recursive", NULL, OPT_NOOP, 0}, {"--update", "-u", OPT_FLAG, offsetof(Config, update)}, /* rsync's --old-args: accepted for CLI compatibility as a documented no-op * (the remote server path is always safely quoted; see usage.c). It is @@ -1387,6 +1394,12 @@ static bool cli_handle_pre_negation(CliParseCtx* ctx) { ctx->no_delta = true; else if (strcmp(arg, "--no-incremental") == 0) ctx->no_incremental = true; + /* Real rsync option names that merely start with "--no-" and are inert + * no-ops (e.g. --no-inc-recursive) are registered as OPT_NOOP entries; + * accept them before the generic negation table would reject the name. */ + const OptionEntry* noop = find_table_option(arg); + if (noop && noop->kind == OPT_NOOP) + return true; if (apply_negation(config, arg) != 0) { ctx->exit_code = -1; return true; diff --git a/src/client/usage.c b/src/client/usage.c index cd5e35c..d10387b 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -26,6 +26,9 @@ void print_usage(void) { printf(" owner, group, devices and specials; not\n"); printf(" compression/multithreading\n"); printf(" -r, --recursive Recurse into directories (FastSync is always recursive)\n"); + printf(" --inc-recursive Accepted for rsync CLI compatibility; no effect (FastSync\n"); + printf(" always performs a full scan, so the destination is identical)\n"); + printf(" --no-inc-recursive Accepted for rsync CLI compatibility; no effect\n"); printf(" -n, --dry-run Show what would be transferred\n"); printf(" --remove-source-files Remove regular source files after successful transfer\n"); printf(" -p, --perms Preserve permission bits\n"); diff --git a/src/shared/file_save.c b/src/shared/file_save.c index 4cab382..73d7640 100644 --- a/src/shared/file_save.c +++ b/src/shared/file_save.c @@ -2,6 +2,7 @@ #include #include #include +#include #include #include #include @@ -198,6 +199,56 @@ static FileSaveResult hardlink_sibling_absent_first(const char* destination_path return FILE_SAVE_ERROR; } +/* Resolve a user-supplied --temp-dir against the receive `root`. + * + * A relative, traversal-free name is joined below the root (the historical + * behavior). An absolute path is canonicalized with realpath(3) and accepted + * only when it lies inside the canonicalized receive root; this is the parity + * win over rejecting every absolute path, without weakening the confinement + * invariant: an absolute path that escapes the root (including one reached + * through a symlinked component) is still refused. A `..` component in a + * relative name is likewise refused. The root itself is treated as an + * absolute path free of `..`; its realpath() resolves any symlinks so the + * prefix comparison is against one canonical form. + * + * Logs a clear error on rejection (the scratch dir must stay confined) and + * returns a newly allocated scratch path, or NULL on rejection/allocation + * failure. */ +static char* file_save_resolve_temp_dir(const char* root, const char* temp_dir) { + if (temp_dir[0] != '/') { + if (has_path_traversal(temp_dir)) { + log_message( + LOG_LEVEL_ERROR, + "receiver rejected --temp-dir '%s': a '..' component would escape the receive root", + temp_dir); + return NULL; + } + return path_cat(root, temp_dir); + } + char canonical_temp[PATH_MAX]; + char canonical_root[PATH_MAX]; + if (!realpath(temp_dir, canonical_temp)) { + log_message(LOG_LEVEL_ERROR, + "receiver rejected --temp-dir '%s': could not resolve the absolute path (%s)", + temp_dir, strerror(errno)); + return NULL; + } + if (!realpath(root, canonical_root)) { + log_message(LOG_LEVEL_ERROR, + "receiver rejected --temp-dir '%s': could not resolve the receive root (%s)", + temp_dir, strerror(errno)); + return NULL; + } + if (strcmp(canonical_root, "/") != 0 && !path_is_within_root(canonical_root, canonical_temp)) { + log_message(LOG_LEVEL_ERROR, + "receiver rejected --temp-dir '%s': an absolute temp dir must be inside the " + "receive root '%s'", + temp_dir, canonical_root); + return NULL; + } + return str_dup(canonical_temp); +} + /* Install a --hard-links/-H sibling: the destination entry is atomically replaced (temp + rename) with a hard link to the group's first member. The first member is guaranteed already installed at `hardlink_target` under the @@ -303,17 +354,13 @@ static FileSaveResult file_save_hardlink_sibling(const char* root_directory, con free(destination_path); return absent_result; } - /* Resolve a relative --temp-dir under the destination root, exactly as the - * primary save path does; an absolute or `..`-escaping value is rejected. */ + /* Resolve the --temp-dir under the destination root, exactly as the primary + * save path does: a relative dir joins below the root, an absolute dir is + * accepted only when it canonicalizes inside the root, and any escaping value + * is rejected. */ char* resolved_temp = NULL; if (cfg->temp_dir) { - if (cfg->temp_dir[0] == '/' || has_path_traversal(cfg->temp_dir)) { - free(content); - free(first_disk); - free(destination_path); - return FILE_SAVE_ERROR; - } - resolved_temp = path_cat(root_directory, cfg->temp_dir); + resolved_temp = file_save_resolve_temp_dir(root_directory, cfg->temp_dir); if (!resolved_temp) { free(content); free(first_disk); @@ -896,17 +943,17 @@ static bool file_save_try_special_dispatch(const FileSavePlan* plan, bool* creat and disk paths. Returns false on an invalid/escaping option or an allocation failure (the caller routes to the cleanup epilogue). */ static bool file_save_resolve_paths(FileSavePlan* plan) { - /* These options arrive from the client. --backup-dir, --partial-dir and - --temp-dir are names below the server root, never independent filesystem - roots: an absolute or `..`-escaping value is rejected outright (rsync's - daemon confines temp-dir to the module the same way). A relative temp dir - is resolved under the receive root below; if that resolution still lands on - a different filesystem than the destination the install falls back to a - non-atomic copy (see file_to_disk_secure_impl), never an abort. */ + /* These options arrive from the client. --backup-dir and --partial-dir are + names below the server root, never independent filesystem roots: an + absolute or `..`-escaping value is rejected outright. --temp-dir is + resolved by file_save_resolve_temp_dir below: a relative name joins below + the root, an absolute name is accepted only when it canonicalizes inside + the root, and any escaping value is rejected. If the resolved scratch dir + still lands on a different filesystem than the destination the install + falls back to a non-atomic copy (see file_to_disk_secure_impl), never an + abort. */ if ((plan->backup_dir && (plan->backup_dir[0] == '/' || has_path_traversal(plan->backup_dir))) || - (plan->partial_dir && - (plan->partial_dir[0] == '/' || has_path_traversal(plan->partial_dir))) || - (plan->temp_dir && (plan->temp_dir[0] == '/' || has_path_traversal(plan->temp_dir)))) + (plan->partial_dir && (plan->partial_dir[0] == '/' || has_path_traversal(plan->partial_dir)))) return false; if (plan->backup_dir && !(plan->confined_backup = path_cat(plan->root_directory, plan->backup_dir))) @@ -914,6 +961,9 @@ static bool file_save_resolve_paths(FileSavePlan* plan) { if (plan->partial_dir && !(plan->confined_partial = path_cat(plan->root_directory, plan->partial_dir))) return false; + if (plan->temp_dir && + !(plan->confined_temp = file_save_resolve_temp_dir(plan->root_directory, plan->temp_dir))) + return false; const char* actual_root = plan->use_partial_root ? plan->confined_partial : plan->root_directory; plan->destination_path = path_cat(plan->root_directory, plan->file->path); @@ -1146,17 +1196,22 @@ FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* /* A configured --temp-dir sends the temporary working copy to a scratch directory; the engine then atomically renames the completed file into the - final destination directory. A relative temp dir is resolved under the - receive root and must already exist (an absolute or `..`-escaping value was - rejected above); the engine falls back to a non-atomic copy on EXDEV. The - partial-dir flow already keeps its working copy in a separate directory and - --inplace writes directly, so neither diverts through the scratch dir - (matching rsync, where --inplace/--partial-dir supersede --temp-dir). */ - bool use_temp_dir = plan.temp_dir != NULL && !plan.inplace && !plan.use_partial_root; + final destination directory. The scratch path was confined to the receive + root (and canonicalized) in file_save_resolve_paths and must already exist; + the engine falls back to a non-atomic copy on EXDEV. The partial-dir flow + already keeps its working copy in a separate directory and --inplace writes + directly, so neither diverts through the scratch dir (matching rsync, where + --inplace/--partial-dir supersede --temp-dir). */ + /* --inplace and --partial-dir supersede --temp-dir in rsync, so the scratch + dir is not used on those paths. The value was still validated/confined by + file_save_resolve_paths; drop the resolved path so it is never handed to the + install engine. */ + if (plan.confined_temp && (plan.inplace || plan.use_partial_root)) { + free(plan.confined_temp); + plan.confined_temp = NULL; + } + bool use_temp_dir = plan.confined_temp != NULL; if (use_temp_dir) { - plan.confined_temp = path_cat(root_directory, plan.temp_dir); - if (!plan.confined_temp) - goto out; /* A user-supplied trailing slash would leave the scratch path ending in "/", which has no final component to create/open. Normalize it away. */ size_t temp_len = strlen(plan.confined_temp); diff --git a/tests/integration/test_temp_dir_absolute.py b/tests/integration/test_temp_dir_absolute.py new file mode 100644 index 0000000..ff7f761 --- /dev/null +++ b/tests/integration/test_temp_dir_absolute.py @@ -0,0 +1,120 @@ +"""Parity coverage for an absolute ``--temp-dir`` that lies inside the receive root. + +FastSync confines the ``--temp-dir`` scratch directory to the receive root. It +previously rejected *every* absolute path; it now canonicalizes an absolute path +with ``realpath(3)`` and accepts it when it resolves inside the canonical receive +root (the destination is identical, so this is a pure parity win), while an +absolute path that escapes the root stays rejected with a clear error. + +Two paths exercise the same receiver-side resolution: + +* the local ``--read-batch`` apply (no network; the batch destination is the + receive root), and +* a real TCP transfer against the shared server (the destination root is the + client-supplied absolute path). + +The out-of-root case asserts the run fails without writing a single scratch +file, so the confinement invariant is preserved. +""" +import os +import subprocess +import sys + +import pytest + +sys.path.insert(0, os.path.dirname(__file__)) +from common import CLIENT_CMD, TEST_DATA_DIR, clean_dir, get_dest_received_dir, run_client + +FILES = { + "top.txt": b"top level\n", + "sub/nested.txt": b"nested file\n" * 16, +} + + +def _run(args): + return subprocess.run(CLIENT_CMD + args, capture_output=True, text=True, timeout=180) + + +def _seed_source(root): + clean_dir(root) + for rel, data in FILES.items(): + full = os.path.join(root, rel) + os.makedirs(os.path.dirname(full), exist_ok=True) + with open(full, "wb") as fh: + fh.write(data) + + +def _make_batch(tmp, source): + batch = os.path.join(tmp, "tree.batch") + result = _run(["--only-write-batch", batch, source]) + assert result.returncode == 0, (result.stdout, result.stderr) + return batch + + +def _read(path): + with open(path, "rb") as fh: + return fh.read() + + +@pytest.mark.ci +def test_read_batch_absolute_temp_dir_inside_root_accepted(tmp_path): + source = os.path.join(tmp_path, "src") + dest = os.path.join(tmp_path, "dst") + _seed_source(source) + clean_dir(dest) + scratch = os.path.join(dest, "scratch") + os.makedirs(scratch) + + batch = _make_batch(str(tmp_path), source) + result = _run(["--read-batch", batch, dest, "--temp-dir", scratch]) + assert result.returncode == 0, (result.stdout, result.stderr) + + received = get_dest_received_dir(dest, source) + for rel, data in FILES.items(): + assert _read(os.path.join(received, rel)) == data, f"content mismatch for {rel}" + assert os.listdir(scratch) == [], "scratch dir was not left clean" + + +@pytest.mark.ci +def test_read_batch_absolute_temp_dir_outside_root_rejected(tmp_path): + source = os.path.join(tmp_path, "src") + dest = os.path.join(tmp_path, "dst") + _seed_source(source) + clean_dir(dest) + outside = os.path.join(tmp_path, "outside") + os.makedirs(outside) + + batch = _make_batch(str(tmp_path), source) + result = _run(["--read-batch", batch, dest, "--temp-dir", outside]) + assert result.returncode != 0, "an absolute temp dir outside the receive root must be rejected" + assert os.listdir(outside) == [], "receiver wrote into an unconfined temp dir" + assert "temp-dir" in (result.stdout + result.stderr), (result.stdout, result.stderr) + + +def test_tcp_absolute_temp_dir_inside_root_accepted(shared_server): + source = os.path.join(TEST_DATA_DIR, "tempdir_abs_in_src") + dest = os.path.join(TEST_DATA_DIR, "tempdir_abs_in_dst") + _seed_source(source) + clean_dir(dest) + scratch = os.path.join(dest, "scratch") + os.makedirs(scratch) + + result, _ = run_client(source, dest, flags=["--temp-dir", scratch], port=shared_server.port) + assert result.returncode == 0, (result.stdout, result.stderr)[:300] + received = get_dest_received_dir(dest, source) + for rel, data in FILES.items(): + assert _read(os.path.join(received, rel)) == data, f"content mismatch for {rel}" + assert os.listdir(scratch) == [], "scratch dir was not left clean" + + +def test_tcp_absolute_temp_dir_outside_root_rejected(shared_server): + source = os.path.join(TEST_DATA_DIR, "tempdir_abs_out_src") + dest = os.path.join(TEST_DATA_DIR, "tempdir_abs_out_dst") + _seed_source(source) + clean_dir(dest) + outside = os.path.join(TEST_DATA_DIR, "tempdir_abs_out_scratch") + clean_dir(outside) + + result, _ = run_client(source, dest, flags=["--temp-dir", outside], port=shared_server.port) + assert result.returncode != 0, "an absolute temp dir outside the receive root must be rejected" + assert os.listdir(outside) == [], "receiver wrote into an unconfined temp dir" diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index ba85fa1..64f9610 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -5040,10 +5040,12 @@ static void test_parse_args_include_exclude_order() { config_delete(cfg3); } -/* OPT_NOOP compatibility flags (-s/--secluded-args, -r/--recursive) must never - * swallow the next argv: `fastsync -s SRC DST` keeps both positionals. */ +/* OPT_NOOP compatibility flags (-s/--secluded-args, -r/--recursive, and the + * --inc-recursive/--no-inc-recursive scan-mode pair) must never swallow the + * next argv: `fastsync -s SRC DST` keeps both positionals. */ static void test_parse_args_noop_does_not_consume_argv() { - static const char* const noops[] = {"-s", "--secluded-args", "-r", "--recursive"}; + static const char* const noops[] = {"-s", "--secluded-args", "-r", + "--recursive", "--inc-recursive", "--no-inc-recursive"}; for (size_t i = 0; i < sizeof(noops) / sizeof(noops[0]); i++) { Config* cfg = config_create(); int positional_args[2]; diff --git a/tests/test_file.c b/tests/test_file.c index efb87a7..786ab4c 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -339,12 +339,16 @@ static void test_file_save_to_disk_temp_dir_confined() { const char* root = "test_temp_confine_tmp"; const char* dest_file = "test_temp_confine_tmp/file.txt"; char outside[PATH_MAX]; + char inside_abs[PATH_MAX]; snprintf(outside, sizeof(outside), "/tmp/fastsync_temp_outside_%d", (int)getpid()); unlink(dest_file); rmdir("test_temp_confine_tmp/scratch"); + rmdir("test_temp_confine_tmp/abs_scratch"); rmdir(root); mkdir(root, 0755); mkdir("test_temp_confine_tmp/scratch", 0755); + mkdir("test_temp_confine_tmp/abs_scratch", 0755); + EXPECT_NOT_NULL(realpath("test_temp_confine_tmp/abs_scratch", inside_abs)); mkdir(outside, 0755); File* f = file_create("file.txt"); @@ -364,6 +368,13 @@ static void test_file_save_to_disk_temp_dir_confined() { config->temp_dir = str_dup("../escape"); EXPECT_EQ_INT(file_save_to_disk_full(root, f, config), FILE_SAVE_ERROR); EXPECT_EQ_INT(access(dest_file, F_OK), -1); + /* An absolute temp dir that canonicalizes INSIDE the receive root is + accepted and used (the parity win); destination is still written. */ + free(config->temp_dir); + config->temp_dir = str_dup(inside_abs); + EXPECT_EQ_INT(file_save_to_disk_full(root, f, config), FILE_SAVE_WRITTEN); + EXPECT_EQ_INT(access(dest_file, F_OK), 0); + unlink(dest_file); free(config->temp_dir); config->temp_dir = str_dup("scratch"); EXPECT_EQ_INT(file_save_to_disk_full(root, f, config), FILE_SAVE_WRITTEN); @@ -373,6 +384,7 @@ static void test_file_save_to_disk_temp_dir_confined() { config_delete(config); unlink(dest_file); rmdir("test_temp_confine_tmp/scratch"); + rmdir("test_temp_confine_tmp/abs_scratch"); rmdir(root); rmdir(outside); } -- 2.54.0 From 86741725fd659069d347aaf5816bbc68a2463cc2 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 22:02:09 +0200 Subject: [PATCH 54/68] feat(daemon): accept rsync rsyncd.conf key subset and --dparam mapping --- RSYNC_COMPAT.md | 6 +- src/shared/daemon_conf.c | 132 +++++++++++++++++++++- src/shared/daemon_conf.h | 46 +++++--- tests/integration/test_daemon.py | 70 ++++++++++-- tests/test_daemon_conf.c | 181 +++++++++++++++++++++++++++++++ 5 files changed, 411 insertions(+), 24 deletions(-) diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 5a6b5b8..2dca0d8 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -730,8 +730,8 @@ targets verbatim, matching rsync. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| | `--daemon` | Run as rsync daemon | ❌ Divergent | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding | -| `--config=FILE` | Alternate rsyncd.conf file | ❌ Divergent | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` | -| `--dparam=OVERRIDE` | Override global daemon config | ❌ Divergent | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` | +| `--config=FILE` | Alternate rsyncd.conf file | ❌ Divergent | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and still strictly rejects a genuinely unknown key so a typo can never silently change what a module serves; requires `--daemon`. **rsync 3.4.1 key subset accepted:** the common rsyncd.conf GLOBAL keys (`port`, `address`, `motd file`, `max connections`, `hosts allow`/`hosts deny`, plus the inert `pid file`, `log file`, `socket options`/`sockopts`, `listen backlog`, `syslog facility`, `syslog tag`, `log format`, `use chroot`, `uid`, `gid`, `timeout`, `max verbosity`/`min verbosity`, `lock file`, `transfer logging`, `strict modes`, `reverse lookup`/`forward lookup`, `ignore errors`, `ignore nonreadable`, `dont compress`) and MODULE keys (`path`, `read only`, `max connections`, `auth users`, `hosts allow`/`hosts deny`, plus the inert `comment`, `use chroot`, `uid`/`gid`/`daemon uid`/`daemon gid`, `exclude`, `include`, `exclude from`/`include from`, `filter`, `secrets file`, `auth digest`, `max verbosity`/`min verbosity`, `lock file`, `transfer logging`, `log file`/`log format`/`syslog facility`/`syslog tag`, `timeout`, `strict modes`, `numeric ids`, `fake super`, `munge symlinks`, `write only`, `list`, `dont compress`, `charset`, `refuse options`, `incoming chmod`/`outgoing chmod`, `open noatime`, `max size`/`min size`, `temp dir`, `pre-xfer exec`/`post-xfer exec`, `name converter`, `proxy protocol`/`proxy protocol hosts`, `reverse lookup`/`forward lookup`, `ignore errors`, `ignore nonreadable`) are recognized. Keys with a FastSync equivalent map onto it (a global `read only` is honored as the default for later modules); keys with no FastSync equivalent load **inert** (no effect) rather than failing the whole config. Residual: the native grammar still differs from rsync's (no `\` line continuation, `%VAR%` expansion, `[global]` re-entry, or inline `#` comments), and the inert keys are genuinely not enforced — in particular a daemon-side `exclude`/`filter` is NOT applied and `secrets file` is NOT read (use `path`, `--password-file`, and client-side filters instead) | +| `--dparam=OVERRIDE` | Override global daemon config | ❌ Divergent | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Reuses the exact same global-key dispatch as `--config`, so it accepts the native global keys (`port`, `motd file`, `address`, `read only`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`), the recognized inert rsync global keys, and rsync's compact spellings (`motdfile`, `pidfile`, `logfile`); keys are case-insensitive. `read only` sets the global default and re-applies it to every module that did not set its own value. Genuinely unknown keys and invalid values are rejected. Requires `--daemon` | | `--no-detach` | Don't detach from parent | ✅ Parity | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` | | `--password-file=FILE` | Read daemon password from file | ❌ Divergent | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. **Hardening follow-up:** the file is opened with `O_NOFOLLOW`, so a symlinked credential path fails closed (`ELOOP`) instead of being followed before the owner/mode gate; literal fd-backed paths (`/dev/fd/`, `/proc/self/fd/`, which is what a bash process substitution passes) are exempt, so process substitution still works. A FIFO/process-substitution read now waits under a bounded ~3 s deadline for its writer, so a slow producer works while a connected-but-silent FIFO fails instead of hanging. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | | `--early-input=FILE` | Use FILE for daemon early exec | ❌ Divergent | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. Opened with the same `O_NOFOLLOW` hardening as `--password-file` (a symlinked path fails closed with `ELOOP`; fd-backed `/dev/fd/N`/`/proc/self/fd/N` process-substitution paths are exempt) and a FIFO read is bound-waited (~3 s) so a slow producer works while a writer-less FIFO cannot hang. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | @@ -739,7 +739,7 @@ targets verbatim, matching rsync. **Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. -- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. +- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. To reduce the rsync divergence, the parser additionally accepts the common rsync 3.4.1 GLOBAL and MODULE keys: the keys with a FastSync equivalent (`path`, `read only`, `max connections`, `auth users`, `hosts allow`/`hosts deny`, and the global `port`/`address`/`motd file`) map onto it, a global `read only` becomes the default for modules defined after it, and the keys with no FastSync equivalent (e.g. `pid file`, `log file`, `use chroot`, `uid`/`gid`, `comment`, `exclude`/`include`, `max verbosity`, `lock file`, `transfer logging`, `timeout`, `secrets file`) are recognized and loaded **inert** (accepted-but-ignored) instead of failing the whole file. `--dparam` reuses the same dispatch, so it also accepts the inert rsync global keys and the compact spellings `motdfile`/`pidfile`/`logfile`. A key outside both sets is still rejected. The inert keys are genuinely not enforced: a daemon-side `exclude`/`include`/`filter` is not applied and a `secrets file` is not read (use `--password-file`/`--early-input`), so an rsync config that relies on those must be edited rather than trusted. - **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. - **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`) and re-derives the per-module and per-source occupancy counts from the surviving REGISTERED slots, so a child killed mid-registration cannot leak a count. The per-source table has a bounded lifetime: an entry with no live connection is reclaimed after its lockout expires or it has been idle (300 s); if the table is genuinely full the per-source cap/lockout fails open for new sources (per-module cap and ACLs still apply) with a rate-limited warning. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4), and a trusted loopback peer (127.0.0.0/8 / `::1`, `utils_fd_peer_is_local`) is exempt from the per-source cap and the auth lockout because all local clients share one address (the per-module/global caps still apply). Clients behind a shared NAT/proxy address likewise share one per-source budget and lockout counter. A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index 2297a48..3b0a351 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -33,6 +33,108 @@ static bool key_equals(const char* key, const char* canonical) { return strcasecmp(key, canonical) == 0; } +/* True when `key` matches one of the NUL-terminated names in `list`. */ +static bool key_in_list(const char* key, const char* const* list, size_t count) { + for (size_t i = 0; i < count; i++) { + if (strcasecmp(key, list[i]) == 0) + return true; + } + return false; +} + +/* rsync 3.4.1 rsyncd.conf GLOBAL keys accepted in the pre-module section that + * have no FastSync equivalent. They are recognized and documented as inert: + * accepting a real rsync config must not fail on a logging/process key, but a + * silently-reinterpreted key is never invented. `pidfile`/`logfile` are the + * compact --dparam spellings rsync documents. The same list is used by the + * `--dparam` dispatch (apply_global_key), so there is a single impl. */ +static const char* const kRsyncInertGlobalKeys[] = { + "pid file", + "pidfile", + "log file", + "logfile", + "socket options", + "sockopts", + "listen backlog", + "syslog facility", + "syslog tag", + "log format", + "use chroot", + "uid", + "gid", + "timeout", + "max verbosity", + "min verbosity", + "lock file", + "transfer logging", + "strict modes", + "reverse lookup", + "forward lookup", + "ignore errors", + "ignore nonreadable", + "dont compress", +}; + +/* rsync 3.4.1 rsyncd.conf MODULE keys accepted in a [module] section that have + * no FastSync equivalent (accepted-and-documented inert). Keys with a FastSync + * meaning (`path`, `read only`, `auth users`, `max connections`, + * `hosts allow`/`hosts deny`, `client owner`) are handled by apply_module_key + * before this list is consulted. Security-relevant keys (`exclude`, `filter`, + * `secrets file`, `refuse options`, ...) are inert, so a daemon-side filter or + * rsync secrets file is NOT enforced: see RSYNC_COMPAT.md for the residual. */ +static const char* const kRsyncInertModuleKeys[] = { + "comment", + "use chroot", + "daemon chroot", + "uid", + "gid", + "daemon uid", + "daemon gid", + "exclude", + "include", + "exclude from", + "include from", + "filter", + "max verbosity", + "min verbosity", + "lock file", + "transfer logging", + "log file", + "log format", + "syslog facility", + "syslog tag", + "timeout", + "secrets file", + "auth digest", + "strict modes", + "numeric ids", + "fake super", + "munge symlinks", + "write only", + "list", + "dont compress", + "charset", + "refuse options", + "incoming chmod", + "outgoing chmod", + "open noatime", + "max size", + "min size", + "temp dir", + "pre-xfer exec", + "post-xfer exec", + "name converter", + "proxy protocol", + "proxy protocol hosts", + "reverse lookup", + "forward lookup", + "ignore errors", + "ignore nonreadable", +}; + +#define kRsyncInertGlobalCount (sizeof(kRsyncInertGlobalKeys) / sizeof(kRsyncInertGlobalKeys[0])) +#define kRsyncInertModuleCount (sizeof(kRsyncInertModuleKeys) / sizeof(kRsyncInertModuleKeys[0])) + static bool parse_bool_value(const char* value, bool* out) { if (strcasecmp(value, "yes") == 0 || strcasecmp(value, "true") == 0 || strcmp(value, "1") == 0) { *out = true; @@ -250,6 +352,7 @@ DaemonConf* daemon_conf_create(void) { if (!conf) return NULL; conf->global.port = DAEMON_CONF_DEFAULT_PORT; + conf->global.read_only_default = false; conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS; conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS; conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST; @@ -324,7 +427,7 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo char* err, size_t err_size) { if (key_equals(key, "port")) return store_port(&conf->global.port, value, err, err_size); - if (key_equals(key, "motd file")) { + if (key_equals(key, "motd file") || key_equals(key, "motdfile")) { if (!store_string(&conf->global.motd_file, value)) { set_error(err, err_size, "out of memory parsing 'motd file'"); return false; @@ -338,6 +441,25 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo } return true; } + /* rsync allows the `read only` module key in the global section as the + * default for modules defined after it. Map it to that default (a later + * --dparam re-applies it to modules that did not set their own value) so a + * global `read only = yes` cannot be silently dropped into a writable + * default. */ + if (key_equals(key, "read only")) { + bool parsed; + if (!parse_bool_value(value, &parsed)) { + set_error(err, err_size, "global 'read only' must be yes/no (or true/false/1/0), got '%s'", + value); + return false; + } + conf->global.read_only_default = parsed; + for (int i = 0; i < conf->module_count; i++) { + if (!conf->modules[i].read_only_explicit) + conf->modules[i].read_only = parsed; + } + return true; + } if (key_equals(key, "max connections")) return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size); if (key_equals(key, "max connections per host")) @@ -360,6 +482,9 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo if (key_equals(key, "hosts deny")) return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value, "hosts deny", NULL, replace_hosts, err, err_size); + /* A recognized rsync global key with no FastSync equivalent loads inert. */ + if (key_in_list(key, kRsyncInertGlobalKeys, kRsyncInertGlobalCount)) + return true; set_error(err, err_size, "unknown global key '%s'", key); return false; } @@ -388,6 +513,7 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* return false; } module->read_only = parsed; + module->read_only_explicit = true; return true; } if (key_equals(key, "client owner")) { @@ -457,6 +583,9 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* if (key_equals(key, "hosts deny")) return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny", module->name, false, err, err_size); + /* A recognized rsync module key with no FastSync equivalent loads inert. */ + if (key_in_list(key, kRsyncInertModuleKeys, kRsyncInertModuleCount)) + return true; set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name); return false; } @@ -507,6 +636,7 @@ static int open_module(DaemonConf* conf, int* current_module, const char* name, } conf->modules = grown; memset(&conf->modules[conf->module_count], 0, sizeof(DaemonModule)); + conf->modules[conf->module_count].read_only = conf->global.read_only_default; conf->modules[conf->module_count].name = str_dup(name); if (!conf->modules[conf->module_count].name) { set_error(err, err_size, "out of memory adding module '%s'", name); diff --git a/src/shared/daemon_conf.h b/src/shared/daemon_conf.h index d04399e..81af6db 100644 --- a/src/shared/daemon_conf.h +++ b/src/shared/daemon_conf.h @@ -17,7 +17,16 @@ * DAEMON_CONF_MAX_LINE all fail the whole load with a clear, line-numbered * error instead of being silently ignored. This keeps a typo from silently * changing what a module serves. - */ + * + * rsync compatibility: to reduce the divergence from rsync 3.4.1's rsyncd.conf + * grammar, the parser also ACCEPTS the common rsync GLOBAL and MODULE keys. + * Keys with a FastSync equivalent are mapped onto it (the native spellings are + * unchanged). Keys with no FastSync equivalent are accepted and documented as + * inert (they load successfully but have no effect) rather than failing the + * whole config; the accepted inert set is listed in kRsyncInertGlobalKeys / + * kRsyncInertModuleKeys in daemon_conf.c and in RSYNC_COMPAT.md. A key + * outside both the FastSync-native grammar and the recognized rsync subset is + * still rejected as unknown. */ /* A daemon module's configured root is used exactly like the standalone * server's --destination-root: the daemon confines every connection that @@ -42,15 +51,20 @@ * store refuses (fail closed) rather than falling open; see server.c. Auth is * never bypassed by ignoring the list. */ typedef struct DaemonModule { - char* name; /* module name, as the client requests it */ - char* path; /* module root (daemon-side authorized root) */ - bool read_only; /* `read only = yes/no`; default no */ - bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in - that lets this module's clients choose ownership - (--numeric-ids/--chown/--usermap/--groupmap/--fake-super/ - --copy-as) and request explicit --super super-user - activities. Without it the daemon refuses all of them. */ - char** auth_users; /* `auth users = a,b`; Wave B credential list */ + char* name; /* module name, as the client requests it */ + char* path; /* module root (daemon-side authorized root) */ + bool read_only; /* `read only = yes/no`; defaults to the global `read only` + default (rsync allows it in the global section), which is + itself default no */ + bool read_only_explicit; /* set when this module set its own `read only`, so a + later global default (from a `--dparam read only=`) + does not override it */ + bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in + that lets this module's clients choose ownership + (--numeric-ids/--chown/--usermap/--groupmap/--fake-super/ + --copy-as) and request explicit --super super-user + activities. Without it the daemon refuses all of them. */ + char** auth_users; /* `auth users = a,b`; Wave B credential list */ int auth_user_count; /* `max connections = N` (optional per-module cap). 0 means unlimited. The * per-connection child records the selected module in the shared registry @@ -69,6 +83,9 @@ typedef struct DaemonConfGlobals { int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */ char* motd_file; /* `motd file`, may be NULL */ char* address; /* `address` (optional bind address), may be NULL */ + bool read_only_default; /* global `read only` default for modules defined + after it (rsync allows the module key in the + global section); default no */ int max_connections; /* `max connections`, default DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */ int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default @@ -152,10 +169,13 @@ const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char* bool daemon_module_name_valid(const char* name); /* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and - * apply it to the global keys only. Keys are case-insensitive and limited to - * the global keys defined by the grammar (port, motd file, address, + * apply it to the global keys only. Keys are case-insensitive and cover the + * global keys defined by the grammar (port, motd file, address, read only, * max connections, max connections per host, auth failure delay, - * auth lockout threshold, auth lockout duration, hosts allow, hosts deny). + * auth lockout threshold, auth lockout duration, hosts allow, hosts deny) plus + * the recognized inert rsync global keys and the compact rsync spellings + * (`motdfile`, `pidfile`, `logfile`). Applying `read only` sets the global + * default and re-applies it to every module that did not set its own value. * Returns 0 on success, -1 on error (err filled). */ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size); diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index ae1d765..2b9917a 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -141,6 +141,7 @@ class DaemonManager: def __init__(self): self._proc = None self._port = None + self.log_path = None def start(self, config_path, port_override=None, extra_args=None, log_path=None): self.stop() @@ -154,7 +155,11 @@ class DaemonManager: if extra_args: cmd += extra_args if log_path is None: - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + # A unique log per manager: several managers run in one xdist + # worker, and a shared log lets one daemon's truncate/write offset + # corrupt the other's appended lines (a flaky log assertion). + log_path = os.path.join(TEST_DATA_DIR, f"fastsyncd_{id(self):x}.log") + self.log_path = log_path log = open(log_path, "w") self._proc = subprocess.Popen( cmd, stdout=log, stderr=log, stdin=subprocess.DEVNULL, start_new_session=True) @@ -375,6 +380,58 @@ class TestDaemonModuleSelection: proc.kill() +class TestRsyncConfigCompat: + """A real rsyncd.conf can be pointed at FastSync: the common rsync GLOBAL + and MODULE keys are accepted, the ones with a FastSync equivalent (port, + path, read only, max connections) take effect, and the inert ones (pid + file, log file, comment, use chroot, uid, gid, exclude, timeout, ...) are + documented no-ops. --dparam accepts the same expanded key set.""" + + @pytest.mark.ci + def test_rsync_style_config_round_trip(self): + module = os.path.join(MODULE_ROOT, "rsync_style") + shutil.rmtree(module, ignore_errors=True) + os.makedirs(module, exist_ok=True) + port = _find_free_port() + conf = os.path.join(TEST_DATA_DIR, "fastsyncd_rsync_style.conf") + with open(conf, "w") as f: + f.write( + "# an rsync 3.4.1-style rsyncd.conf\n" + "pid file = /tmp/fastsyncd_rsync_style.pid\n" + "log file = /tmp/fastsyncd_rsync_style.log\n" + "socket options = TCP_NODELAY\n" + "use chroot = no\n" + "uid = nobody\n" + "gid = nogroup\n" + "timeout = 600\n" + "max verbosity = 2\n" + "transfer logging = yes\n" + "port = %d\n" + "\n" + "[rsync_style]\n" + "path = %s\n" + "comment = rsync-style module\n" + "use chroot = no\n" + "exclude = *.tmp\n" + "read only = no\n" + "max connections = 4\n" + % (port, module)) + d = DaemonManager() + # --dparam borrows rsync's compact spelling; `pidfile` is inert but must + # not be rejected, proving dparam reuses the expanded global key set. + d.start(conf, extra_args=["--dparam", "pidfile=/tmp/rsync_style.pid"], + log_path=os.path.join(TEST_DATA_DIR, "fastsyncd_rsync_style.log")) + try: + result = _push("127.0.0.1::rsync_style", d.port) + assert result.returncode == 0, result.stderr or result.stdout + received = get_dest_received_dir(module, SOURCE_DIR) + mismatches, missing = verify_transfer(SOURCE_DIR, received) + assert not missing, f"missing: {missing[:5]}" + assert not mismatches, f"mismatch: {mismatches[:5]}" + finally: + d.stop() + + class TestDaemonRejection: def _tree_files(self): """Snapshot every file path (module-relative) currently under the module @@ -477,7 +534,7 @@ class TestDaemonRejection: before any data lands. `accept` lists the log phrases that count as the refusal (a non-root daemon refuses --copy-as earlier, at the privilege check, so the caller accepts that phrase too).""" - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + log_path = daemon.log_path before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 before_files = self._tree_files() result, _ = run_client(SOURCE_DIR, f"127.0.0.1::{module}", port=daemon.port, flags=flags) @@ -514,14 +571,13 @@ class TestDaemonRejection: the refusal into a silent accept.""" port = _find_free_port() d = DaemonManager() - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") try: d.start(CONF_FILE, port_override=port, extra_args=["--password-file", CRED_FILE, "--no-super"]) result, _ = run_client(SOURCE_DIR, "127.0.0.1::files", port=d.port, flags=["--super", "--preserve"]) assert result.returncode != 0, "the --no-super daemon must refuse --super" - with open(log_path, "rb") as f: + with open(d.log_path, "rb") as f: tail = f.read().decode("utf-8", "replace") assert "client-chosen ownership" in tail, ( f"daemon did not log the --super refusal: {tail[-400:]!r}" @@ -1010,7 +1066,7 @@ class TestDaemonAuthentication: def test_auth_log_does_not_leak_password(self, daemon): """The daemon log must never contain the password or the store verifier.""" - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + log_path = daemon.log_path before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 _push_with_creds("127.0.0.1::locked", daemon.port, "alice", WRONG_PASS) _push_with_creds("127.0.0.1::locked", daemon.port, "alice", ALICE_PASS) @@ -1037,7 +1093,7 @@ class TestDaemonAuthentication: _push_with_creds("127.0.0.1::locked", port, "alice", ALICE_PASS) _push_with_creds("127.0.0.1::locked", port, "alice", WRONG_PASS) time.sleep(0.3) - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") + log_path = d.log_path with open(log_path, "rb") as f: log = f.read().decode("utf-8", "replace") finally: @@ -1256,12 +1312,12 @@ class TestDaemonTLSAuth: _write_client_password_file(client_creds, "alice", ALICE_PASS) d = DaemonManager() port = _find_free_port() - log_path = os.path.join(TEST_DATA_DIR, "fastsyncd.log") try: d.start(CONF_FILE, port_override=port, extra_args=[ "--tls", "--cert", certs["server_cert"], "--key", certs["server_key"], "--ca", certs["ca"], "--client-cn", "fastsync-client", "--password-file", CRED_FILE]) + log_path = d.log_path before_files = _tree_file_count(AUTH_MODULE) log_before = os.path.getsize(log_path) if os.path.exists(log_path) else 0 tls_flags = ["--tls", diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index 5d0a46d..fde819c 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -626,6 +626,182 @@ static void test_daemon_conf_module_count_capped() { EXPECT_TRUE(strstr(err, "too many modules") != NULL); } +/* rsync rsyncd.conf compatibility: the common GLOBAL keys FastSync does not + * implement (pid file, log file, use chroot, uid/gid, timeout, ...) are + * accepted as documented inert keys, while the keys with a FastSync equivalent + * keep working and the compact rsync --dparam spellings (`pidfile`, `logfile`, + * `motdfile`) are recognized. A global `read only` is rsync's module default + * and must not be silently dropped. */ +static void test_daemon_conf_rsync_global_keys() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("pid file = /run/fastsyncd.pid\n" + "log file = /var/log/fastsyncd.log\n" + "socket options = TCP_NODELAY\n" + "listen backlog = 10\n" + "syslog facility = daemon\n" + "syslog tag = fastsyncd\n" + "use chroot = no\n" + "uid = nobody\n" + "gid = nogroup\n" + "timeout = 600\n" + "max verbosity = 3\n" + "lock file = /var/run/fastsyncd.lock\n" + "transfer logging = yes\n" + "strict modes = yes\n" + "reverse lookup = no\n" + "dont compress = *.gz\n" + "read only = yes\n" + "port = 8734\n" + "address = 127.0.0.1\n" + "pidfile = /run/other.pid\n" + "logfile = /var/log/other.log\n" + "motdfile = /etc/fastsync/motd.alt\n" + "\n" + "[pub]\n" + "path = /srv/pub\n" + "\n" + "[explicit]\n" + "path = /srv/explicit\n" + "read only = no\n", + &path), + 0); + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + /* Mapped globals took effect; the compact aliases too. */ + EXPECT_EQ_INT(conf->global.port, 8734); + EXPECT_EQ_STR(conf->global.address, "127.0.0.1"); + EXPECT_EQ_STR(conf->global.motd_file, "/etc/fastsync/motd.alt"); + /* The global `read only = yes` is the default for modules defined after it. */ + EXPECT_TRUE(conf->global.read_only_default); + EXPECT_EQ_INT(conf->module_count, 2); + EXPECT_TRUE(conf->modules[0].read_only); + /* An explicit per-module value wins over the global default. */ + EXPECT_FALSE(conf->modules[1].read_only); + daemon_conf_free(conf); +} + +/* rsync module keys with no FastSync equivalent load inert; the keys with a + * FastSync meaning still map onto their native fields. */ +static void test_daemon_conf_rsync_module_keys() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("[data]\n" + "path = /srv/data\n" + "comment = Public data\n" + "use chroot = yes\n" + "uid = nobody\n" + "gid = nogroup\n" + "exclude = *.tmp\n" + "include = keep.tmp\n" + "exclude from = /etc/rsync.exclude\n" + "max verbosity = 2\n" + "lock file = /var/run/rsyncd.lock\n" + "transfer logging = yes\n" + "timeout = 300\n" + "secrets file = /etc/rsyncd.secrets\n" + "auth digest = sha256\n" + "numeric ids = yes\n" + "write only = no\n" + "list = yes\n" + "dont compress = *.gz\n" + "refuse options = delete\n" + "read only = yes\n" + "max connections = 5\n" + "hosts allow = 10.0.0.0/8\n" + "auth users = alice\n", + &path), + 0); + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_EQ_STR(conf->modules[0].path, "/srv/data"); + EXPECT_TRUE(conf->modules[0].read_only); + EXPECT_EQ_INT(conf->modules[0].max_connections, 5); + EXPECT_EQ_INT(conf->modules[0].hosts_allow_count, 1); + EXPECT_EQ_STR(conf->modules[0].hosts_allow[0], "10.0.0.0/8"); + EXPECT_EQ_INT(conf->modules[0].auth_user_count, 1); + EXPECT_EQ_STR(conf->modules[0].auth_users[0], "alice"); + daemon_conf_free(conf); +} + +/* A genuinely unknown key is still rejected in both contexts, so accepting the + * rsync subset did not turn typos into silent no-ops. */ +static void test_daemon_conf_rsync_unknown_keys_rejected() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("bogus rsync key = 1\n", &path), 0); + const DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "unknown global key") != NULL); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nnot a real key = 1\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "unknown key 'not a real key'") != NULL); +} + +/* A recognized rsync key with an invalid value is still a clear parse error. */ +static void test_daemon_conf_rsync_read_only_invalid() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("read only = maybe\n[m]\npath = /x\n", &path), 0); + const DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "read only") != NULL); + + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nread only = maybe\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "read only") != NULL); +} + +/* --dparam reuses the same global dispatch: it accepts the compact rsync + * spellings and the inert rsync global keys, and `read only` sets the default + * for modules that did not set their own value. */ +static void test_daemon_conf_dparam_rsync_keys() { + DaemonConf* conf = daemon_conf_create(); + EXPECT_NOT_NULL(conf); + char err[256]; + + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "pidfile=/run/x.pid", err, sizeof(err)), 0); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "pid file=/run/y.pid", err, sizeof(err)), 0); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "logfile=/tmp/x.log", err, sizeof(err)), 0); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "log file=/tmp/y.log", err, sizeof(err)), 0); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "motdfile=/tmp/alt.motd", err, sizeof(err)), 0); + EXPECT_EQ_STR(conf->global.motd_file, "/tmp/alt.motd"); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "timeout=600", err, sizeof(err)), 0); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "use chroot=no", err, sizeof(err)), 0); + + /* A module already parsed without an explicit `read only` takes the + * --dparam default; an explicit module value is preserved. */ + { + char* path; + EXPECT_EQ_INT(write_conf("[plain]\npath = /p\n[explicit]\npath = /e\nread only = no\n", &path), + 0); + DaemonConf* loaded = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(loaded); + EXPECT_EQ_INT(daemon_conf_apply_dparam(loaded, "read only=yes", err, sizeof(err)), 0); + EXPECT_TRUE(loaded->global.read_only_default); + EXPECT_TRUE(loaded->modules[0].read_only); + EXPECT_FALSE(loaded->modules[1].read_only); + daemon_conf_free(loaded); + } + + /* Invalid values and genuinely unknown keys are still rejected. */ + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "read only=maybe", err, sizeof(err)), -1); + EXPECT_EQ_INT(daemon_conf_apply_dparam(conf, "definitely not rsync=1", err, sizeof(err)), -1); + EXPECT_TRUE(strstr(err, "unknown global key") != NULL); + + daemon_conf_free(conf); +} + void test_daemon_conf() { test_daemon_conf_create_defaults(); test_daemon_conf_full_parse(); @@ -645,4 +821,9 @@ void test_daemon_conf() { test_daemon_conf_module_count_capped(); test_daemon_hosts_allowed(); test_daemon_module_name_valid(); + test_daemon_conf_rsync_global_keys(); + test_daemon_conf_rsync_module_keys(); + test_daemon_conf_rsync_unknown_keys_rejected(); + test_daemon_conf_rsync_read_only_invalid(); + test_daemon_conf_dparam_rsync_keys(); } \ No newline at end of file -- 2.54.0 From 80e8d7f4506fe9cb33118f4b54e698c9e0355f03 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 22:03:41 +0200 Subject: [PATCH 55/68] feat(xattr): rsync-interoperable --fake-super stat; fix --devices error parity --- README.md | 4 +- RSYNC_COMPAT.md | 78 ++++++++------- src/client/usage.c | 9 +- src/shared/config.h | 10 +- src/shared/file.c | 9 +- src/shared/file_save.c | 39 ++++++-- src/shared/xattr.c | 68 ++++++------- src/shared/xattr.h | 49 +++++---- tests/integration/test_features.py | 120 ++++++++++++++++++----- tests/integration/test_preserve_attrs.py | 8 +- tests/test_xattr.c | 70 +++++++++++-- 11 files changed, 315 insertions(+), 149 deletions(-) diff --git a/README.md b/README.md index dee3e96..5689751 100644 --- a/README.md +++ b/README.md @@ -187,7 +187,7 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is | `--groupmap=MAP` | Map group names when applying ownership | | `--numeric-ids` | Apply source numeric uid/gid directly instead of mapping by name | | `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP] (requires a privileged receiver) | -| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown | +| `--fake-super` | Record the resolved owner plus full mode/rdev in rsync's reserved `user.rsync.%stat` xattr (rsync 3.4.1 grammar) and replay the permission bits; never performs a real chown | | `--super` | Permit the receiver to attempt confined super-user activities (device nodes) | | `-D` | Preserve device and special files (implies `--devices --specials`) | | `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`) | @@ -643,7 +643,7 @@ remote SSH argv is already built injection-safe. | `--groupmap=MAP` | Map group names when applying ownership (same syntax as `--usermap`). | | `--numeric-ids` | Mapping modifier: apply the source numeric uid/gid directly instead of mapping by name (combine with `-o`/`-g`, `-a`, or a map). | | `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP]; requires a privileged receiver. | -| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown. | +| `--fake-super` | Record the resolved owner plus full mode/rdev in rsync's reserved `user.rsync.%stat` xattr (rsync 3.4.1 grammar) and replay the permission bits; never performs a real chown. | | `--super` | Permit the receiver to attempt confined super-user activities (device nodes). | | `--no-super` | Forbid those super-user activities even when the receiver is root. | | `-l`, `--links` | Copy symlinks as symlinks; the target is stored verbatim (absolute and `..`-bearing targets included), matching rsync. | diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 5a6b5b8..cbea97f 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -17,7 +17,7 @@ tested but diverges in at least one documented way. An ❌ row is either rejected (`--protocol` with any value but the current one, `--inc-recursive`), an accepted no-op (`-s`/`--secluded-args`, `--protect-args`, `--old-args`), deliberately non-rsync and non-interoperable (the FastSync daemon config/auth, -the batch container, `--fake-super`'s xattr format, `--copy-as` credential +the batch container, `--copy-as` credential switching), or impossible (`-N`/`--crtimes`). The counts are derived from the rows below; update them together with the table. @@ -83,8 +83,8 @@ output, codec breadth, general `-R`/`-d`, the filter grammar (the unsupported rejected elsewhere — see the audit-cycle follow-up note above), receiver-side name resolution, absolute basis dirs, and the remaining client quick wins) and reclassified the inherently non-rsync rows as **divergent** (native daemon -config/auth, the non-interoperable batch container, `--fake-super`'s xattr -format, `-X`'s privileged namespaces, and the safe-subset device/privilege +config/auth, the non-interoperable batch container, +`-X`'s privileged namespaces, and the safe-subset device/privilege flags). It moved `PROTOCOL_VERSION` three times (`2.23.0 → 2.24.0` delete timing, `2.24.0 → 2.25.0` wire stats, `2.25.0 → 2.26.0` codecs). See the **Parity Completion Wave (protocol 2.26.0)** section near the end for the full @@ -342,7 +342,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | `-X`, `--xattrs` | Preserve extended attributes | ❌ Divergent | Deliberately restricted to unprivileged `user.*` extended attributes plus the two POSIX ACL xattrs; `security.*` (SELinux, capabilities, ...) and `trusted.*` are **never** captured or applied — a client can never force a privileged attribute onto the destination, and the receiver independently re-validates every incoming name against the whitelist. This is a security-policy divergence from rsync, which can preserve the privileged namespaces with the needed privilege; implementing them would defeat FastSync's privilege-escalation guard. `user.*` capture/apply matches rsync in a differential test. Payloads are bounded on both ends. Incompatible with `-s`. **Also divergent: symlink xattrs/ACLs are not captured or applied** — `-X`/`-A` with `-l` carries only the link's owner/times/mode, not its xattrs (the capture uses path-following `listxattr`/`getxattr`, so the link's own xattrs are never read, and the receiver's symlink write path applies no xattr block). Closing this needs a dedicated symlink-xattr wire block and a `PROTOCOL_VERSION` bump | | `-H`, `--hard-links` | Preserve hard links | ✅ Parity | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below | | `-D` | Same as --devices --specials | ✅ Parity | Implies `--devices --specials`. `-D` was unassigned in FastSync (verified: no collision), so it is free to imply both device-node and special-file preservation. As of protocol 2.23.0 `--specials` genuinely covers **both FIFOs and unix sockets**, so `-D` covers the full rsync set. See the `--devices`/`--specials` rows and the Phase-4 devices notes below | -| `--devices` | Preserve device files | ❌ Divergent | Recreates char/block device nodes with `mknodat` (type + rdev strictly validated, confined fd-relative below the receive root), but only when the receiver has `CAP_MKNOD`: a non-root receiver logs a warning and skips the entry instead of erroring, so a transfer with devices never aborts. Deliberate privilege-model divergence from rsync, which errors when it cannot create the node. `--specials` (FIFOs and unix sockets) is unprivileged and remains parity | +| `--devices` | Preserve device files | ⚠️ Caveat | Recreates char/block device nodes with `mknodat` (type + rdev strictly validated, confined fd-relative below the receive root). A device whose `mknodat` fails with `EPERM`/`EACCES` (no `CAP_MKNOD`, or super-user activity forbidden) is now a **transfer error** surfaced through the receiver's outcome aggregation — rsync parity: rsync reports `mknod ... failed` and exits partial (23) when it attempts the node (as root or with `--super`). Residual: FastSync's default AUTO still *attempts* the node on a non-root receiver and therefore errors, whereas rsync without `--super` silently ignores `--devices` and skips the non-regular entry with exit 0; use `--no-super` for rsync's silent-skip behavior. (FastSync's process exit code for a receiver-side transfer error is the general error code 1, not rsync's partial 23 — a client exit-code-mapping residual that applies to every receiver file error, not just this branch.) `--specials` (FIFOs and unix sockets) keeps the unprivileged skip path and remains parity | | `--specials` | Preserve special files | ✅ Parity | **FIFO and unix-socket recreation work** (protocol 2.23.0): FIFOs are recreated with `mkfifoat`, and sockets with `mknodat(..., S_IFSOCK)` — the latter is unprivileged on Linux because it materializes the socket *node*, not a live bound socket, so it is a real, assertable behavior under CI (it matches rsync, which also recreates a socket by `mknod`). Node creation is confined below the receive root (fd-relative parent; no `..`, no symlink follow) and type/rdev are validated strictly; a matching existing node is left in place and an unrelated entry is never replaced. Crosses the wire like `--devices` (the `STATUS_SPECIAL` frame). See the Phase-4 devices notes | | `--copy-devices` | Copy device contents as file | ❌ Divergent | Copies a device/FIFO's reported `st_size` into an ordinary regular file and never reads an unbounded pseudo-device, so `--sendfile` cannot hang and the run always succeeds. Deliberate safe divergence from rsync's dd-like unbounded device read, which can block; the dangerous behavior will not be implemented | | `--write-devices` | Write to devices as files | ❌ Divergent | Writes only into an existing char/block node under the confined receive root (`O_NOFOLLOW` + `O_NONBLOCK`); a missing, symlinked, FIFO-with-no-reader, non-device, or otherwise unusable destination is skipped with a warning rather than allowed or aborted. Deliberate confinement divergence from rsync's more permissive behavior | @@ -351,7 +351,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | `-O`, `--omit-dir-times` | Omit dirs from --times | ✅ Parity | Real modifier now that FastSync preserves directory times. With metadata on, the scanner captures every traversed source directory's mtime (and atime under `-U`) and the sender transmits them in trailing `STATUS_DIR_TIMES` frame(s) **after all file data and the optional delete manifest** (chunked at the receiver's `MAX_MANIFEST_ENTRIES` per-frame cap); a dir-time entry only RECORDS metadata and never creates the directory (an empty source directory is created by the separate `STATUS_MKDIR` entry the scanner now emits, and `-m/--prune-empty-dirs` suppresses that; the trailing dir-time simply re-applies the metadata). The receiver defers applying them until its delete / `--delay-updates` publication phases have committed, so writing or removing a child never clobbers a parent directory's mtime (rsync applies directory times at the end for exactly this reason). When `-O` is set (the boolean crosses the wire) the receiver does not apply any of them; without `-O` an `-a`/`--preserve` transfer now restores directory times (reversing the old "never preserves dir times" divergence). Wire change: the terminal `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** | | `-J`, `--omit-link-times` | Omit symlinks from --times | ✅ Parity | Real modifier now that FastSync preserves symlink times. Symlink entries already carried their metadata on `STATUS_SYMLINK`; the receiver now applies it with **no-follow primitives only** (`utimensat(..., AT_SYMLINK_NOFOLLOW)`, plus best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)`), so the link itself is stamped without ever dereferencing it, confined fd-relative below the authorized receive root. A symlink has no children, so the times are applied immediately at creation. When `-J` is set (the boolean crosses the wire) the receiver skips the timestamps (mode/ownership are unaffected); without `-J` an `-a`/`-l` transfer restores symlink mtimes. Wire change alongside `-O`: the shared `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** | | `--super` | Receiver attempts super-user activities | ❌ Divergent | Safe-subset privilege model. `--super` permits the receiver to attempt already-confined super-user activities (ownership application, char/block device-node creation, `--write-devices`); `--no-super` forbids them even for root; `auto` keeps the historical best-effort attempt. **FastSync never elevates** — no `setuid`/`seteuid`/`setgid` — and `--super` never bypasses the confinement floor, so it diverges from rsync's real elevation. A server `--no-super` veto forces it off for every connection; a privileged standalone listener defaults off without `--allow-super`; daemon modules opt in with `client owner = yes` | -| `--fake-super` | Store/recover privileged attrs via xattrs | ❌ Divergent | Records the resolved `uid:gid:mode:mtime_sec:mtime_nsec` in a reserved `user.fastsync.stat` xattr and immediately replays mode/times fd-relative, but **never performs a real `chown`** (the owner is recorded for a later privileged restore). The on-disk key and format are FastSync-native, not rsync's `user.rsync.%stat%`, so recordings are not interoperable with rsync — the same class as the native auth and batch formats. Implies metadata transmission; incompatible with `-s` | +| `--fake-super` | Store/recover privileged attrs via xattrs | ⚠️ Caveat | Writes rsync 3.4.1's reserved `user.rsync.%stat` xattr with rsync's exact value grammar ` , :` (e.g. `104711 0,0 1234:5678`), recording the RESOLVED owner (the `--chown`/`--usermap`/`--groupmap`/`--copy-as` mapping when active, else the source's own id) plus the full mode and rdev; it **never performs a real `chown`**. mtime is carried by the file's own timestamp, exactly as rsync does it (there is no mtime field). The receiver parses the same grammar and replays the permission bits fd-relative, stripping the recorded special bits on disk exactly like rsync's fake-super receiver. Regular files are interoperable with real rsync 3.4.1 in both directions (the differential test has rsync read a FastSync fake-super tree and re-emit the identical record). Residual: directories and device nodes are not yet faked — no `%stat` record is written for a directory, and a char/block node is still recreated/skipped rather than stored as a regular file carrying the stat. Implies metadata transmission; incompatible with `-s` | | `--open-noatime` | Avoid changing access time when opening files | ✅ Parity | Sender-side policy: the sender opens source files with `O_NOATIME` (Linux) when reading them for transfer, so the open/read does NOT bump the source's on-disk access time. Degrades safely when `O_NOATIME` is unavailable (not defined) or refused (`EPERM`, since it needs `CAP_FOWNER` or file ownership): the code falls back to a normal open, so the data always transfers — only the atime-bump is skipped. It does not itself capture/preserve atime; it only avoids modifying it. **Client-only, never crosses the wire.** Exposed as `file_open_for_read()` and applied to both the buffered data path and the sendfile path | | `--numeric-ids` | Do not map uid/gid by name | ✅ Parity | **A mapping modifier only:** when ownership is being applied it uses the transmitted numeric uid/gid directly, skipping the name lookup. It does **not** request ownership application on its own — combine it with `-o`/`-g`, `-a`, or an explicit map (`--chown`/`--usermap`/`--groupmap`) — and it does not need any metadata flag merely to parse. Ownership is only applied when metadata (hence the source uid/gid) is actually transmitted (see the Phase-4 identity notes) | | `--usermap=STRING` | Map usernames | ✅ Parity | Opt-in ownership application. Comma-separated `FROM:TO` rules evaluated in order, first match wins. `FROM` accepts a source-resolved user name, a name **glob** (`*`/`?`/`[...]`, expanded sender-side at CLI-parse time against the sender's passwd/group database and collapsed into numeric `LOW-HIGH` ranges, bounded by `MAX_IDENTITY_MAP`), an `@N`/bare `N` numeric id, an inclusive `LOW-HIGH` id range, `*`, or an empty field (ids with no source name). `TO` accepts a receiver-resolved **name** (protocol 2.26.0 resolves it on the receiving side against the receiver's account database, matching rsync), an `@N`/bare `N` id, or `*` (the receiving process's euid). Rules travel as resolved numeric pairs plus an optional TO name; the receiver applies a matching rule, else falls back to `--chown`, `--numeric-ids`, then a best-effort name lookup, via fd-relative `fchown`. Malformed specs are clear errors. Implies metadata; only effective where the receiver can chown (otherwise a warning) | @@ -408,7 +408,7 @@ match, exactly as prior phases did). is refused) also re-applies the incoming (or, for `-H`, the first member's) xattrs and the `--fake-super` stat, so attributes are preserved rather than silently dropped when the link fails. -- **Reserved fake-super key is receiver-only:** the `user.fastsync.stat` key is +- **Reserved fake-super key is receiver-only:** the `user.rsync.%stat` key is excluded from sender capture AND from receiver application, so it can only be written by the receiver's own `--fake-super` handling. A source file that already carries such a record is never forwarded on a plain `-X` run, so it @@ -417,15 +417,19 @@ match, exactly as prior phases did). Applying an ACL is owner-privileged: `fsetxattr` failure (e.g. non-root, unsupported filesystem) is logged (collapsed to one line per file) and never fatal. -- **`--fake-super`**: see the row above; the reserved key is `user.fastsync.stat` - with the documented `uid:gid:mode:mtime_sec:mtime_nsec` (mode octal) format. - **Replay exists**: after each stored record the receiver immediately re-applies - the recorded mode and times fd-relative (`fake_super_restore_fd`), but it - deliberately never performs a real `chown` — `--fake-super` only *records* - the resolved owner (the active `--chown`/`--usermap`/`--groupmap`/`--copy-as` - mapping when one is in effect, otherwise the source's own id) for a later - privileged restore. The recording format diverges from rsync's - `user.rsync.%stat%`; no cross-tool conversion is attempted. +- **`--fake-super`**: see the row above; the reserved key is rsync's own + `user.rsync.%stat` with rsync 3.4.1's exact ` , + :` value (mtime is not stored — the file's own + timestamp carries it, exactly as rsync does). **Replay exists**: after each + stored record the receiver immediately re-applies the recorded permission bits + fd-relative (`fake_super_restore_fd`, with the recorded special bits stripped + on disk exactly like rsync), but it deliberately never performs a real + `chown` — `--fake-super` only *records* the resolved owner (the active + `--chown`/`--usermap`/`--groupmap`/`--copy-as` mapping when one is in effect, + otherwise the source's own id) for a later privileged restore. Because the + key and grammar are rsync's, a regular-file fake-super tree is interoperable + with rsync 3.4.1 in both directions; directories and device nodes are not yet + faked. - **Chunk serialization (`-s`) incompatibility:** the per-file xattr block rides the streaming per-file frame, which `-s` replaces with a fixed buffer format, so `-X` / `-A` combined with `-s` is rejected up front on both ends (mirroring @@ -554,18 +558,22 @@ marker + rdev so `--devices/--specials` also work under `-s`. `PROTOCOL_VERSION` was bumped **2.12.0 → 2.13.0** (peers must match, exactly as prior phases did). **Privilege gating (the crux):** making a device node requires `CAP_MKNOD` (root). -CI runs the integration suite as a NON-ROOT user (via setpriv), so `mknod` fails -with `EPERM`. The receiver treats this as a graceful, logged *skip of the entry* -returned as a success/skip outcome — the whole transfer NEVER aborts just because -the environment cannot create the node. `mkfifo` (FIFOs) is unprivileged, so -`--specials` FIFO creation is a real, assertable behavior under CI. **Sockets are -recreated too** (protocol 2.23.0) with `mknodat(..., S_IFSOCK)`: Linux allows an -unprivileged `mknod` of a socket node because no live bound socket is created, -so a source socket materializes as a socket-type filesystem entry exactly as -rsync does. The "device actually created" integration assertions are guarded to -run only as root. User-facing expectation: point `--devices` at devices and a -non-root receiver will faithfully skip them while transferring everything else; -`--specials` recreates FIFOs and socket nodes for any receiver. +When the receiver attempts a device `mknod` and the kernel refuses with +`EPERM`/`EACCES`, FastSync now reports a genuine transfer error (the receiver's +outcome aggregation fails the entry), matching rsync, which logs +`mknod ... failed` and exits partial (23) whenever it attempts the node (as root +or with `--super`); with `--no-super` the device entry is pre-skipped instead. +Only `mkfifo` (FIFOs) is unprivileged, so `--specials` FIFO creation is a real, +assertable behavior under CI. **Sockets are recreated too** (protocol 2.23.0) +with `mknodat(..., S_IFSOCK)`: Linux allows an unprivileged `mknod` of a socket +node because no live bound socket is created, so a source socket materializes as +a socket-type filesystem entry exactly as rsync does. The "device actually +created" integration assertions are guarded to run only as root; a root runner +additionally drops the receiver to an unprivileged user (setpriv) to assert the +`CAP_MKNOD` failure surfaces as a failed transfer rather than a silent skip. +User-facing expectation: point `--devices` at devices and a receiver without +`CAP_MKNOD` reports the failure, while `--specials` recreates FIFOs and socket +nodes for any receiver. **Confinement & validation:** a special/device node is created with `mknodat`/`mkfifoat` on the parent directory opened fd-relative below the receive @@ -936,7 +944,7 @@ These are the last compatibility items and the closing phase toward rsync flag p | `-T` / `--timeout` | `-T` = `--temp-dir` | → `--timeout` (long-only) | | `-a` / `--archive` (= `-c -m -M`) | `-a` = `-rlptD` | → becomes **real rsync `-a`** after the renames | -**Wave B — Output & filesystem completion (✅ implemented).** `-S`/`--sparse` (`⚠️→✅`): real hole preservation — a sparse-aware writer (`write_all_sparse`) skips all-zero runs ≥ 4096 bytes with `lseek(SEEK_CUR)` and `ftruncate`s the final size, wired into both the atomic temp+rename store and `--inplace` receiver-side with **no wire change** (the full file image is already in memory; the ftruncate presize is kept). `-P` (`⚠️→✅`): interrupted-write retention — on a save failure after data reached the temp fd, `--partial` now renames the already-written temp to the destination path (best-effort; falls through to the normal unlink on failure, never retains when `--partial` is off) so a later `--append`/`--append-verify` run can resume. `--block-size=SIZE` (`⚠️→✅`): promoted after verification — `--block-size` is now an alias for `--delta-block`, both set `config->delta_block_size`, which the delta engine already honored end-to-end (`delta_signature_create_seeded` + `delta_apply`); out-of-range values keep the default. `--fake-super` (`⚠️→✅`): added `fake_super_restore_fd` to parse and re-apply the recorded `user.fastsync.stat` record fd-relative (mode/time only — protocol 2.23.0: **never a real chown**; the resolved owner is recorded for a later privileged restore); a save under `--fake-super` now re-applies the recorded attrs instead of only recording them, with the recording format unchanged. `--stderr=client` (`⚠️→❌ Divergent`): FastSync has no rsync client-message channel, and `client` is rejected at CLI parse — the rejection is the documented behavior (unit-tested). `-N`/`--crtimes` (`⚠️→❌ Divergent`): birth-times cannot be set by any portable fs call (`utimensat` sets only atime/mtime); capture/transmit stays, setting is impossible, the flag is accepted and safely inert. Review-hardening (post-eval): fake-super replay applies the mode through the shared `metadata_mode_for_policy` helper (protocol 2.23.0: exactly the source mode under `-p`, with no masking); `--sparse` takes precedence over `--preallocate` (posix_fallocate skipped so holes survive) — **reversed by the parity-completion wave: `--preallocate` now wins, matching rsync**; `--partial` retention is disabled for `--no_replace` (ignore/existing) and only marks a write-attempt after the actual write begins; `--block-size=SIZE`/`--delta-block=SIZE` inline forms are accepted. +**Wave B — Output & filesystem completion (✅ implemented).** `-S`/`--sparse` (`⚠️→✅`): real hole preservation — a sparse-aware writer (`write_all_sparse`) skips all-zero runs ≥ 4096 bytes with `lseek(SEEK_CUR)` and `ftruncate`s the final size, wired into both the atomic temp+rename store and `--inplace` receiver-side with **no wire change** (the full file image is already in memory; the ftruncate presize is kept). `-P` (`⚠️→✅`): interrupted-write retention — on a save failure after data reached the temp fd, `--partial` now renames the already-written temp to the destination path (best-effort; falls through to the normal unlink on failure, never retains when `--partial` is off) so a later `--append`/`--append-verify` run can resume. `--block-size=SIZE` (`⚠️→✅`): promoted after verification — `--block-size` is now an alias for `--delta-block`, both set `config->delta_block_size`, which the delta engine already honored end-to-end (`delta_signature_create_seeded` + `delta_apply`); out-of-range values keep the default. `--fake-super` (`⚠️→✅`): added `fake_super_restore_fd` to parse and re-apply the recorded `user.fastsync.stat` record fd-relative (mode/time only — protocol 2.23.0: **never a real chown**; the resolved owner is recorded for a later privileged restore); a save under `--fake-super` now re-applies the recorded attrs instead of only recording them. (The later fake-super xattr-interop pass replaced that native `user.fastsync.stat` format with rsync's `user.rsync.%stat` grammar — see the row and Phase-4 notes.) `--stderr=client` (`⚠️→❌ Divergent`): FastSync has no rsync client-message channel, and `client` is rejected at CLI parse — the rejection is the documented behavior (unit-tested). `-N`/`--crtimes` (`⚠️→❌ Divergent`): birth-times cannot be set by any portable fs call (`utimensat` sets only atime/mtime); capture/transmit stays, setting is impossible, the flag is accepted and safely inert. Review-hardening (post-eval): fake-super replay applies the mode through the shared `metadata_mode_for_policy` helper (protocol 2.23.0: exactly the source mode under `-p`, with no masking); `--sparse` takes precedence over `--preallocate` (posix_fallocate skipped so holes survive) — **reversed by the parity-completion wave: `--preallocate` now wins, matching rsync**; `--partial` retention is disabled for `--no_replace` (ignore/existing) and only marks a write-attempt after the actual write begins; `--block-size=SIZE`/`--delta-block=SIZE` inline forms are accepted. **Wave C — Devices & special files (finalize statuses + tests) (✅ implemented).** The four special-file rows are finalized with coverage tests. `--devices`, `--copy-devices`, and `--write-devices` are **✅ Implemented**, each with a documented, safety-driven divergence: device-node creation is privilege-gated, so a receiver without `CAP_MKNOD` skips that entry with a warning (a per-entry skip, never a transfer failure); `--copy-devices` copies a device/FIFO's reported size into an ordinary regular file (a size-bounded safe divergence from rsync's unbounded dd-like read); `--write-devices` writes only into an existing char/block node under the confined receive root and skips every unusable target rather than clobbering or aborting. `--specials` reclassified from **⛔ Impossible/Divergence** to **✅ Parity** in protocol 2.23.0: **FIFO recreation works** (unprivileged `mkfifo`) **and unix sockets are recreated** with `mknod(S_IFSOCK)`, which Linux permits unprivileged (the flag previously assumed sockets were impossible — see the `--specials` row). Tests assert FIFO recreation, socket recreation, the regular-file result of `--copy-devices`, the skipped/missing and non-device `--write-devices` targets, and (root-gated) real device-node creation; a root runner additionally drops the receiver to an unprivileged user to assert the `CAP_MKNOD` skip is graceful. (The parity-completion wave later reclassified `--devices`, `--copy-devices`, and `--write-devices` as explicit **❌ Divergent** rows, because their safe subsets are deliberately not rsync's behavior; the implementation itself is unchanged.) @@ -956,7 +964,7 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, and the triage cycle.** ✅ Parity 117 / ⚠️ Caveat 13 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--delete-before`, `--filter`, `-F`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, and the triage cycle, and the fake-super xattr-interop pass.** ✅ Parity 117 / ⚠️ Caveat 15 / ❌ Divergent 25 = 157 rows. The fake-super xattr-interop pass moved `--fake-super` and `--devices` ❌ → ⚠️ (see those rows and the Phase-4 notes): `--fake-super` now writes/reads rsync's exact `user.rsync.%stat` key and ` , :` grammar (regular files interoperate with real rsync 3.4.1 both ways), leaving only directory/device-node faking as residuals; `--devices` now surfaces a failed device `mknod` as a transfer error (rsync `--super` parity) rather than a silent non-root skip, leaving only the AUTO-vs-default-non-root difference and FastSync's general exit code 1 (vs rsync's partial 23) for receiver file errors. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--delete-before`, `--filter`, `-F`, the three basis-dir options, and `-y/--fuzzy`, `--fake-super`, and `--devices`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that receiver filter engine, flipping `--filter=RULE` back to ✅ (see above; the @@ -1057,9 +1065,11 @@ integration tests unless it is explicitly listed as a limitation. clear configuration error (matching rsync) instead of an order-dependent winner. - **`--fake-super` never real-chowns.** It records the *resolved* owner (the - active mapping, else the source id) in `user.fastsync.stat` for a later - privileged restore and replays only mode/times. Directory ownership and - directory xattrs/ACLs are preserved alongside file entries. + active mapping, else the source id) in rsync's `user.rsync.%stat` for a later + privileged restore and replays only the permission bits (mtime travels through + the normal metadata path). Directory ownership and + directory xattrs/ACLs are preserved alongside file entries, though directories + themselves are not yet given a `%stat%` record. - **`--chmod`** implements rsync's `D`/`F`/`X` selectors, `s`/`t`, append semantics, does not imply `-p`, and applies its changes without sanitization. @@ -1254,8 +1264,8 @@ These remain after the wave; the individual rows carry the precise wording. Native daemon config/auth (`--daemon`, `--config`, `--dparam`, `--password-file`, `--early-input`, `--hash-credentials`/`--iterations`), the non-interoperable batch container (`--write-batch`/`--only-write-batch`/ -`--read-batch`), `--fake-super`'s native xattr format, `-X`'s privileged -namespaces, `--devices`/`--copy-devices`/`--write-devices`'s safe subsets, +`--read-batch`), `-X`'s privileged +namespaces, `--copy-devices`/`--write-devices`'s safe subsets, `--super`/`--copy-as`'s refusal to elevate or switch credentials, and the `-s`/`--secluded-args`/`--protect-args`/`--old-args` accepted no-ops. diff --git a/src/client/usage.c b/src/client/usage.c index cd5e35c..da839d8 100644 --- a/src/client/usage.c +++ b/src/client/usage.c @@ -205,10 +205,11 @@ void print_usage(void) { printf(" -A, --acls Preserve POSIX ACLs (the system.posix_acl_* xattrs;\n"); printf(" setting an ACL the receiver is not permitted to\n"); printf(" set is warned and skipped, never fatal)\n"); - printf(" --fake-super Store the source uid/gid/mode/mtime in a reserved\n"); - printf(" user.fastsync.stat xattr on each written file and\n"); - printf(" re-apply it (fd-relative) on a privileged run; the\n"); - printf(" recording format diverges from rsync's user.rsync.%%stat%%\n"); + printf(" --fake-super Store the source mode/rdev/uid/gid in rsync's\n"); + printf(" reserved user.rsync.%%stat xattr on each written\n"); + printf(" file (interoperable with rsync); it never performs a\n"); + printf(" real chown, so an unprivileged receiver records the\n"); + printf(" privileged stat for a later restore\n"); printf(" --super Permit the receiver to attempt super-user activities\n"); printf(" (char/block device-node creation, --write-devices)\n"); printf(" within the confined receive root. Never elevates\n"); diff --git a/src/shared/config.h b/src/shared/config.h index 3596620..fd88991 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -735,11 +735,13 @@ typedef struct Config { * --copy-as) imply it. */ /* fake_super */ /* --fake-super: receiver-only. When set, each written file additionally gets - * a reserved user.fastsync.stat xattr recording the RESOLVED uid/gid (the + * rsync's reserved user.rsync.%stat xattr recording the RESOLVED uid/gid (the * source's own when no ownership request is active, else the --chown/--usermap - * result) plus mode/mtime so a later privileged restore could re-apply them. - * It NEVER real-chowns: the point is to record the source ownership on an - * unprivileged receiver. Crosses the wire. */ + * result) plus the full mode and rdev, in rsync 3.4.1's grammar, so the tree is + * interoperable and a later privileged restore could re-apply them. mtime is + * carried by the file's own timestamp, exactly as rsync does it. It NEVER + * real-chowns: the point is to record the source ownership on an unprivileged + * receiver. Crosses the wire. */ /* module */ /* Daemon module selection (Wave A, protocol 2.15.0). Client-composed from a * host::module/path destination; NULL or "" means "no module" (the ordinary diff --git a/src/shared/file.c b/src/shared/file.c index bcb46f4..2ce5b20 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -1189,14 +1189,15 @@ static void restore_extra_fd(int fd, const FileMetadata* metadata, const FileXat ownership request (--chown/--usermap/--groupmap/--copy-as or -o/-g) is active, the resolved mapping; otherwise the source's own id. The real chown is suppressed (identity_apply_ownership early-returns under - --fake-super) so recording never defeats the flag. Mode/mtime are still - replayed (policy-gated) so unprivileged --fake-super keeps working. */ + --fake-super) so recording never defeats the flag. The recorded stat is + rsync's format; the permission bits are replayed (policy-gated) so + unprivileged --fake-super keeps working while mtime comes from the + normal metadata path above. */ uint32_t store_uid; uint32_t store_gid; identity_resolve_storage_ids((int32_t)metadata->uid, (int32_t)metadata->gid, &store_uid, &store_gid); - fake_super_store_fd(fd, store_uid, store_gid, (uint32_t)metadata->mode, metadata->mtime_sec, - metadata->mtime_nsec); + fake_super_store_fd(fd, store_uid, store_gid, (uint32_t)metadata->mode, 0, 0); fake_super_restore_fd(fd, policy); } } diff --git a/src/shared/file_save.c b/src/shared/file_save.c index 4cab382..0195e6a 100644 --- a/src/shared/file_save.c +++ b/src/shared/file_save.c @@ -353,11 +353,13 @@ bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) { /* ---- Device/special node RECREATION (--devices/--specials), receiver side ---- * * Privilege gating: making a real device node requires CAP_MKNOD (root); making - * a FIFO works unprivileged (mkfifo). When the receiver lacks the capability, - * mknodat() fails with EPERM and the entry is SKIPPED with a warning -- the - * whole transfer must NOT abort just because the environment cannot make the - * node. CI runs non-root, so device creation is expected to skip there and - * only a FIFO is honestly assertable unprivileged. + * a FIFO works unprivileged (mkfifo). A device node whose mknodat() fails with + * EPERM/EACCES is a genuine transfer error (rsync parity: rsync reports the + * mknod failure and the run exits partial, code 23). Only the unprivileged + * FIFO/socket (--specials) path keeps the best-effort skip, because those are + * normally creatable without privilege and a failure there is environmental. + * CI runs non-root, so device creation is expected to fail there; only a FIFO + * is honestly assertable unprivileged. * * Confinement: the parent directory is opened fd-relative below the receive * root (file_open_secure_parent: O_NOFOLLOW, no "..", root-checked) and the @@ -495,13 +497,32 @@ static FileSaveResult file_save_special_to_disk(const char* root_directory, cons node_kind, escaped_path ? escaped_path : ""); free(escaped_path); } else if (errno == EPERM || errno == EACCES) { - /* Missing CAP_MKNOD / parent write permission: the environment cannot - create the node, so skip instead of failing the whole run. */ char* escaped_path = output_escape(file->path, log_get_8_bit_output()); + const char* shown_path = escaped_path ? escaped_path : ""; + if (is_char || is_blk) { + /* rsync parity: a device node that cannot be created (no CAP_MKNOD, or + * super-user activities not permitted) is a genuine transfer error. + * rsync reports `mknod ".../node" failed: ...` and the run exits + * partial (23); FastSync surfaces it through the outcome aggregation + * instead of silently skipping the entry. FIFO/socket creation + * (--specials) keeps the best-effort skip path below. */ + log_message(LOG_LEVEL_ERROR, + "cannot create %s %s: %s\n" + " --devices node creation needs privilege (CAP_MKNOD)", + node_kind, shown_path, strerror(errno)); + free(escaped_path); + close(parent_fd); + free(leaf); + free(destination); + return FILE_SAVE_ERROR; + } + /* Missing CAP_MKNOD / parent write permission for a FIFO/socket: the + environment cannot create the node, so skip instead of failing the + whole run. */ log_message(LOG_LEVEL_WARNING, "skipping %s: cannot create %s node (%s)\n" - " --devices/--specials node creation needs privilege (CAP_MKNOD)", - escaped_path ? escaped_path : "", node_kind, strerror(errno)); + " --specials node creation needs privilege (CAP_MKNOD)", + shown_path, node_kind, strerror(errno)); free(escaped_path); } else { char* escaped_path = output_escape(file->path, log_get_8_bit_output()); diff --git a/src/shared/xattr.c b/src/shared/xattr.c index c9c39ea..56300d7 100644 --- a/src/shared/xattr.c +++ b/src/shared/xattr.c @@ -363,16 +363,21 @@ bool xattr_apply_fd(int fd, const FileXattrList* list) { return true; } -/* ---- --fake-super: park ownership/mode/mtime in a reserved xattr ---- */ +/* ---- --fake-super: park ownership/mode/rdev in a reserved xattr ---- */ -void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, int64_t mtime_sec, - int64_t mtime_nsec) { +void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, uint32_t rdev_major, + uint32_t rdev_minor) { if (fd < 0) return; - char record[128]; - int len = - snprintf(record, sizeof(record), "%lu:%lu:%03o:%lld:%ld", (unsigned long)uid, - (unsigned long)gid, (unsigned)mode & 0777U, (long long)mtime_sec, (long)mtime_nsec); + /* rsync 3.4.1's exact grammar: " , + * :". The octal mode carries the S_IFMT bits (e.g. 0104711 for a + * setuid regular file, 020644 for a char device); the rdev pair is 0,0 for a + * non-device. No mtime field: rsync leaves the file's own timestamp in + * charge of mtime. This value is what rsync reads back to restore a + * fake-super tree, so the field order and separators must not change. */ + char record[96]; + int len = snprintf(record, sizeof(record), "%o %u,%u %u:%u", (unsigned)mode, (unsigned)rdev_major, + (unsigned)rdev_minor, (unsigned)uid, (unsigned)gid); if (len <= 0 || (size_t)len >= sizeof(record)) return; if (fsetxattr(fd, FAKESUPER_XATTR, record, (size_t)len, 0) != 0) { @@ -381,11 +386,14 @@ void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, int6 } } -/* --fake-super replay: read the freshly-stored record and re-apply mode/mtime - * fd-relative. The recorded uid/gid are retained for a later privileged - * restore but are NEVER chowned here: --fake-super only RECORDS ownership, it - * must not real-chown the recorded (resolved) owner. Mode/mtime still apply so - * unprivileged --fake-super keeps working. */ +/* --fake-super replay: read the freshly-stored record and re-apply its + * permission bits fd-relative. The recorded uid/gid are retained for a later + * privileged restore but are NEVER chowned here: --fake-super only RECORDS + * ownership, it must not real-chown the recorded (resolved) owner. The + * recorded rdev is likewise parsed for grammar compatibility but is not acted + * on (device recreation is a separate, privilege-gated path). mtime is not in + * the record: the normal metadata path applies it (policy.times), exactly as + * rsync relies on the file's own timestamp. */ bool fake_super_restore_fd(int fd, FileAttrPolicy policy) { if (fd < 0) return false; @@ -394,24 +402,25 @@ bool fake_super_restore_fd(int fd, FileAttrPolicy policy) { if (len < 0) return false; /* absent or filesystem without xattrs: silent no-op */ record[len] = '\0'; - unsigned long ul_uid, ul_gid, ul_mode; - long long mtime_sec; - long mtime_nsec; - if (sscanf(record, "%lu:%lu:%lo:%lld:%ld", &ul_uid, &ul_gid, &ul_mode, &mtime_sec, &mtime_nsec) != - 5) + unsigned ul_mode, rdev_major, rdev_minor, ul_uid, ul_gid; + if (sscanf(record, "%o %u,%u %u:%u", &ul_mode, &rdev_major, &rdev_minor, &ul_uid, &ul_gid) != 5) return false; /* malformed record: skip, never fatal */ - /* --fake-super NEVER performs a real chown: that would defeat the whole - point of the flag (record privileged ownership on an unprivileged receiver - for a later privileged restore). The uid/gid parsed above are retained in - the record for that later restore, but no ownership change happens here. */ + /* --fake-super NEVER performs a real chown: that would defeat the whole point + of the flag (record privileged ownership on an unprivileged receiver for a + later privileged restore). The uid/gid parsed above are retained in the + record for that later restore, but no ownership change happens here. The + rdev is retained for the same reason. */ + (void)rdev_major; + (void)rdev_minor; (void)ul_uid; (void)ul_gid; /* Mode is applied only when the per-attribute policy asks for it, through the - SAME shared helper the normal metadata path uses (metadata_mode_for_policy): - under --perms the recorded source mode is copied exactly, including - group/other write and setuid/setgid/sticky bits (rsync parity), and the -E - rule derives exec bits from the destination's read bits exactly like + SAME shared helper the normal metadata path uses (metadata_mode_for_policy). + The recorded special bits are stripped first: rsync's fake-super receiver + stores the full mode in the xattr but never installs setuid/setgid/sticky on + the real file, so only the 0777 permission bits may be replayed. The -E + rule then derives exec bits from the destination's read bits exactly like file_restore_metadata_fd. */ if (policy.perms || policy.executability) { struct stat cur; @@ -419,19 +428,12 @@ bool fake_super_restore_fd(int fd, FileAttrPolicy policy) { if (fstat(fd, &cur) != 0) { log_message(LOG_LEVEL_WARNING, "--fake-super: could not read destination mode: %s", strerror(errno)); - } else if (metadata_mode_for_policy((mode_t)ul_mode, cur.st_mode, policy, &want)) { + } else if (metadata_mode_for_policy((mode_t)(ul_mode & 0777U), cur.st_mode, policy, &want)) { if (fchmod(fd, want) != 0) log_message(LOG_LEVEL_WARNING, "--fake-super: could not restore mode on destination file: %s", strerror(errno)); } } - if (policy.times) { - struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, - {.tv_sec = (time_t)mtime_sec, .tv_nsec = mtime_nsec}}; - if (futimens(fd, times) != 0) - log_message(LOG_LEVEL_WARNING, - "--fake-super: could not restore mtime on destination file: %s", strerror(errno)); - } return true; } diff --git a/src/shared/xattr.h b/src/shared/xattr.h index ae8f420..d9858eb 100644 --- a/src/shared/xattr.h +++ b/src/shared/xattr.h @@ -31,11 +31,14 @@ */ /* Reserved key used by --fake-super to park the source's privileged ownership - * / mode / mtime on the destination file as an unprivileged user.* xattr, so a - * later privileged restore could re-apply them. Exact documented format: - * uid:gid:mode:mtime_sec:mtime_nsec (decimal, decimal, octal, dec, dec) - * e.g. "1000:1000:644:1765238400:0". */ -#define FAKESUPER_XATTR "user.fastsync.stat" + * / mode / rdev on the destination file as an unprivileged user.* xattr, so the + * tree is interoperable with rsync 3.4.1 and a later privileged restore can + * re-apply them. This is rsync's own key and value grammar exactly: + * , : + * e.g. "104711 0,0 1234:5678" for a setuid regular file owned by 1234:5678, + * or "20644 1,3 111:222" for a char device. mtime is deliberately NOT part of + * the record: exactly like rsync, the file's own timestamp carries it. */ +#define FAKESUPER_XATTR "user.rsync.%stat" /* --- bounds --- */ #define XATTR_NAME_MAX 255 /* xattr names are limited to 255 bytes */ @@ -91,24 +94,28 @@ FileXattrList* xattr_receive(int fd, int* ok, bool preserve_acls); * true when apply was attempted (allowing callers to treat it as best-effort). */ bool xattr_apply_fd(int fd, const FileXattrList* list); -/* --fake-super: write the source uid/gid/mode/mtime record into the reserved - * FAKESUPER_XATTR on `fd`. Best-effort (logged, never fatal). Only meaningful - * when metadata was transmitted so the values exist. */ -void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, int64_t mtime_sec, - int64_t mtime_nsec); +/* --fake-super: write the source uid/gid/mode/rdev record into the reserved + * FAKESUPER_XATTR on `fd`, using rsync 3.4.1's exact grammar (see the key + * comment above). `mode` is the full st_mode including its S_IFMT bits. + * Best-effort (logged, never fatal). Only meaningful when metadata was + * transmitted so the values exist. */ +void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, uint32_t rdev_major, + uint32_t rdev_minor); /* --fake-super replay: parse the FAKESUPER_XATTR record previously written on - * `fd` by fake_super_store_fd and re-apply mode/mtime fd-relative. The - * recorded uid/gid are deliberately NOT chowned for real: --fake-super only - * RECORDS ownership (the caller stores the resolved mapping via - * identity_resolve_storage_ids), it never performs a real chown. Best-effort: - * absence of the xattr or a malformed record is a silent no-op that never fails - * the transfer. The MODE leg is applied only when policy.perms||policy. - * executability and the MTIME leg only when policy.times, so the fake-super - * replay cannot bypass the per-attribute split; the mode follows the normal - * metadata path exactly (under --perms the source mode is copied verbatim, - * special and group/other write bits included). - * Returns true when the xattr was present and parsed. */ + * `fd` by fake_super_store_fd and re-apply the recorded permission bits + * fd-relative. The recorded uid/gid are deliberately NOT chowned for real: + * --fake-super only RECORDS ownership (the caller stores the resolved mapping + * via identity_resolve_storage_ids), it never performs a real chown. The + * recorded rdev is retained for a later privileged restore but is not acted on + * here. Best-effort: absence of the xattr or a malformed record is a silent + * no-op that never fails the transfer. The MODE leg is applied only when + * policy.perms||policy.executability, and the recorded special bits + * (setuid/setgid/sticky) are NOT applied to the real file -- exactly like + * rsync's fake-super receiver, which stores the full mode in the xattr but + * strips the special bits on disk. mtime is not part of the record; the normal + * metadata path carries it (policy.times) exactly as rsync sets the file's own + * timestamp. Returns true when the xattr was present and parsed. */ bool fake_super_restore_fd(int fd, FileAttrPolicy policy); #endif \ No newline at end of file diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index 0ac5972..fc850d0 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -183,10 +183,11 @@ class TestDeviceSpecial: ) @pytest.mark.setpriv - def test_devices_nonroot_receiver_skips_safely(self): - """A receiver without CAP_MKNOD must skip a device entry with a warning - and never abort. A root runner drops the receiver (server) to nobody - via setpriv; on a non-root runner (or without setpriv) the test skips.""" + def test_devices_nonroot_receiver_errors_like_rsync(self): + """A receiver without CAP_MKNOD must report the failed device mknod as a + transfer error (rsync parity, partial failure) instead of silently + succeeding. A root runner drops the receiver (server) to nobody via + setpriv; on a non-root runner (or without setpriv) the test skips.""" if os.geteuid() != 0 or shutil.which("setpriv") is None: pytest.skip("requires root + setpriv to run the receiver unprivileged") self._setup() @@ -202,16 +203,16 @@ class TestDeviceSpecial: flags=["--devices"], port=port) finally: out, err = _stop_captured_server(server) - assert result.returncode == 0, f"Exit {result.returncode}: {result.stderr[:300]}" - received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) - with open(os.path.join(received, "plain.txt")) as f: - assert f.read() == "regular content\n" - assert not os.path.lexists(os.path.join(received, "chardev")), ( - "a receiver without CAP_MKNOD must skip the device node, not create it" + assert result.returncode != 0, ( + f"a failed device mknod must be a transfer error like rsync (got exit 0): " + f"{(out + err)[:300]}" ) - assert ("cannot create device node" in (out + err) - or "device-node creation is not permitted" in (out + err)), ( - f"receiver did not log the documented device skip: out={out!r} err={err!r}" + received = get_dest_received_dir(DEVICE_DEST, DEVICE_SOURCE) + assert not os.path.lexists(os.path.join(received, "chardev")), ( + "a receiver without CAP_MKNOD must not create the device node" + ) + assert "cannot create device" in (out + err), ( + f"receiver did not log the device creation error: out={out!r} err={err!r}" ) @pytest.mark.skipif(os.geteuid() != 0, reason="requires root to create device nodes") @@ -6599,7 +6600,7 @@ class TestExtendedAttributes: os.getxattr(os.path.join(received, "data.txt"), "user.foo") def test_reserved_fake_super_key_not_forwarded(self, shared_server): - """A source file that already carries the reserved user.fastsync.stat + """A source file that already carries the reserved user.rsync.%stat record must NOT have it planted on the receiver during a plain -X run (it is receiver-only, so it cannot be spoofed for a later privileged restore).""" @@ -6609,7 +6610,7 @@ class TestExtendedAttributes: fh.write(b"reserved\n") if not _xattr_supported(f): pytest.skip("filesystem does not support user xattrs") - os.setxattr(f, "user.fastsync.stat", b"0:0:644:0:0") + os.setxattr(f, "user.rsync.%stat", b"100644 0,0 0:0") # A normal user.* attr still travels alongside. os.setxattr(f, "user.keep", b"yes") @@ -6619,7 +6620,7 @@ class TestExtendedAttributes: received = get_dest_received_dir(dest, source) assert os.getxattr(os.path.join(received, "data.txt"), "user.keep") == b"yes" with pytest.raises(OSError): - os.getxattr(os.path.join(received, "data.txt"), "user.fastsync.stat") + os.getxattr(os.path.join(received, "data.txt"), "user.rsync.%stat") @pytest.mark.ci def test_xattrs_multithreaded(self, shared_server): @@ -6708,10 +6709,17 @@ class TestExtendedAttributes: assert result.returncode == 0, \ f"--fake-super sync failed: {(result.stderr or result.stdout)[:300]}" received = get_dest_received_dir(dest, source) - record = os.getxattr(os.path.join(received, "data.txt"), "user.fastsync.stat").decode() - fields = record.split(":") - assert len(fields) == 5 - assert fields[0] == str(uid), f"reserved uid field {fields[0]} != source uid {uid}" + record = os.getxattr(os.path.join(received, "data.txt"), "user.rsync.%stat").decode() + # rsync 3.4.1 grammar: " , :". + fields = record.split() + assert len(fields) == 3, f"unexpected rsync fake-super record {record!r}" + mode_field, rdev_field, owner_field = fields + assert rdev_field == "0,0", f"regular file rdev must be 0,0, got {rdev_field!r}" + assert int(mode_field, 8) & 0o170000 == stat.S_IFREG, ( + f"recorded mode {mode_field!r} must carry S_IFREG" + ) + assert owner_field.split(":")[0] == str(uid), \ + f"recorded uid {owner_field!r} != source uid {uid}" @pytest.mark.ci def test_fake_super_records_resolved_chown_without_real_chown(self, shared_server): @@ -6730,12 +6738,71 @@ class TestExtendedAttributes: assert result.returncode == 0, \ f"--fake-super --chown sync failed: {(result.stderr or result.stdout)[:300]}" dst = os.path.join(get_dest_received_dir(dest, source), "data.txt") - record = os.getxattr(dst, "user.fastsync.stat").decode().split(":") - assert record[0] == "33333", f"recorded owner {record[0]} != resolved 33333" - assert record[1] == "44444", f"recorded group {record[1]} != resolved 44444" + record = os.getxattr(dst, "user.rsync.%stat").decode().split() + owner = record[2].split(":") + assert owner == ["33333", "44444"], ( + f"recorded owner {record[2]!r} != resolved 33333:44444" + ) st = os.stat(dst) assert st.st_uid != 33333, "--fake-super must not real-chown the recorded owner" + @pytest.mark.ci + def test_fake_super_rsync_interop(self, shared_server): + """A fake-super tree written by FastSync is readable by rsync 3.4.1: + rsync reads the `user.rsync.%stat` record (mode/rdev/uid:gid) and, when + it re-emits a fake-super tree, reproduces the same record. This pins + the on-disk key and value grammar against the real tool.""" + rsync = shutil.which("rsync") + if rsync is None: + pytest.skip("rsync not installed") + source, dest = self._source_and_dest("fakesuper_interop") + f = os.path.join(source, "data.txt") + with open(f, "wb") as fh: + fh.write(b"interop\n") + if not _xattr_supported(f): + pytest.skip("filesystem does not support user xattrs") + # rsync's fake-super receiver only writes a %stat% record when it has + # something to fake; a root-owned file with a matching root stat is + # a no-op. When privileged, record a non-root owner so the round-trip + # actually exercises the parser (non-root CI already has a non-zero uid). + if os.geteuid() == 0: + try: + os.chown(f, 12345, 12346) + except OSError: + pass + # A setuid bit exercises the full st_mode encoding; set it AFTER any + # chown (chown clears setuid/setgid), and note that neither tool installs + # it on the real destination file. + os.chmod(f, 0o4711) + + result, _ = run_client(source, dest, flags=["--fake-super"], + port=shared_server.port) + assert result.returncode == 0, \ + f"--fake-super sync failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + rec = os.getxattr(os.path.join(received, "data.txt"), "user.rsync.%stat").decode() + rec_fields = rec.split() + assert len(rec_fields) == 3 and rec_fields[1] == "0,0", ( + f"FastSync did not write rsync's stat grammar: {rec!r}" + ) + assert int(rec_fields[0], 8) & 0o7777 == 0o4711, ( + f"FastSync did not record the source mode in rsync's grammar: {rec!r}" + ) + + out = os.path.join(TEST_DATA_DIR, "fakesuper_interop_rsync") + clean_dir(out) + rs = subprocess.run([rsync, "-aX", "--fake-super", + received + "/", out + "/"], + capture_output=True, text=True, timeout=120) + assert rs.returncode == 0, ( + f"rsync could not read FastSync's fake-super tree: {rs.stderr[:300]}" + ) + out_rec = os.getxattr(os.path.join(out, "data.txt"), "user.rsync.%stat").decode() + assert out_rec == rec, ( + "rsync re-emitted a different fake-super record; FastSync's grammar " + f"is not interoperable: ours={rec!r} rsync={out_rec!r}" + ) + @pytest.mark.ci def test_directory_xattrs_preserved(self, shared_server): """#286.3: -aX must preserve user.* xattrs on DIRECTORIES, not just files.""" @@ -7255,9 +7322,10 @@ class TestCopyAs: ) received = get_dest_received_dir(dest, source) dst = os.path.join(received, "mixed.txt") - record = os.getxattr(dst, "user.fastsync.stat").decode().split(":") - assert (record[0], record[1]) == ("65534", "65534"), ( - f"fake-super must record the resolved copy-as ownership: {record[:2]}" + record = os.getxattr(dst, "user.rsync.%stat").decode().split() + owner = record[2].split(":") + assert owner == ["65534", "65534"], ( + f"fake-super must record the resolved copy-as ownership: {owner}" ) st = os.lstat(dst) assert (st.st_uid, st.st_gid) != (12345, 12346), ( diff --git a/tests/integration/test_preserve_attrs.py b/tests/integration/test_preserve_attrs.py index 66ab361..3b5d8a8 100644 --- a/tests/integration/test_preserve_attrs.py +++ b/tests/integration/test_preserve_attrs.py @@ -353,10 +353,10 @@ class TestOwnershipRoot: st = os.stat(dst) assert st.st_uid != 12345, \ f"--fake-super -o must NOT real-chown the source owner, got uid={st.st_uid}" - record = os.getxattr(dst, "user.fastsync.stat").decode() - fields = record.split(":") - assert fields[0] == "12345", \ - f"--fake-super must record the resolved owner, got {fields[0]}" + record = os.getxattr(dst, "user.rsync.%stat").decode() + owner = record.split()[2].split(":") + assert owner[0] == "12345", \ + f"--fake-super must record the resolved owner, got {owner[0]}" def test_o_applies_directory_owner(self, shared_server): """#286.2: -o must apply the source owner to DIRECTORIES too (the diff --git a/tests/test_xattr.c b/tests/test_xattr.c index 5762503..e659ed7 100644 --- a/tests/test_xattr.c +++ b/tests/test_xattr.c @@ -156,8 +156,8 @@ static void test_xattr_capture_and_appliable() { EXPECT_TRUE(xattr_name_appliable("user.foo", false)); EXPECT_TRUE(xattr_name_appliable("user.foo", true)); /* The reserved fake-super key is receiver-only and never forwarded/applied. */ - EXPECT_FALSE(xattr_name_appliable("user.fastsync.stat", false)); - EXPECT_FALSE(xattr_name_appliable("user.fastsync.stat", true)); + EXPECT_FALSE(xattr_name_appliable("user.rsync.%stat", false)); + EXPECT_FALSE(xattr_name_appliable("user.rsync.%stat", true)); /* B4: the ACL names require --acls; -X alone must not authorize them. */ EXPECT_FALSE(xattr_name_appliable("system.posix_acl_access", false)); EXPECT_FALSE(xattr_name_appliable("system.posix_acl_default", false)); @@ -351,9 +351,10 @@ static void test_xattr_capture_filters_acls() { } /* --fake-super replay: fake_super_store_fd records the source stat into the - * reserved xattr, and fake_super_restore_fd re-applies mode/mtime (and owner, - * when the process may) fd-relative. Restore must also be a safe no-op with no - * xattr present. Guarded on filesystem xattr support. */ + * reserved xattr, and fake_super_restore_fd re-applies the permission bits + * fd-relative (mtime travels through the normal metadata path; the owner is + * never chowned). Restore must also be a safe no-op with no xattr present. + * Guarded on filesystem xattr support. */ static void test_fake_super_restore() { const char* path = "test_fake_super_restore.txt"; unlink(path); @@ -373,7 +374,7 @@ static void test_fake_super_restore() { FileAttrPolicy policy = {true, true, false, false, true}; EXPECT_FALSE(fake_super_restore_fd(fd, policy)); - fake_super_store_fd(fd, 1001, 1002, 0751, 1700000000, 123456789); + fake_super_store_fd(fd, 1001, 1002, S_IFREG | 0751, 0, 0); EXPECT_TRUE(fake_super_restore_fd(fd, policy)); struct stat st; EXPECT_EQ_INT(fstat(fd, &st), 0); @@ -381,7 +382,7 @@ static void test_fake_super_restore() { /* Strict rsync parity: -p restores the recorded mode exactly, including group/other write (a recorded 0666 restores as 0666). */ - fake_super_store_fd(fd, 1001, 1002, 0666, 1700000000, 0); + fake_super_store_fd(fd, 1001, 1002, S_IFREG | 0666, 0, 0); EXPECT_TRUE(fake_super_restore_fd(fd, policy)); EXPECT_EQ_INT(fstat(fd, &st), 0); EXPECT_EQ_INT((int)(st.st_mode & 0777), 0666); @@ -401,6 +402,58 @@ static void test_fake_super_restore() { unlink(path); } +/* The stored record is rsync 3.4.1's exact grammar + * " , :" + * so a fake-super tree is readable by rsync. Also pins two rsync parity + * rules: the special bits are stored in the record but NOT applied to the real + * file, and a device record's rdev round-trips through the parser. Guarded on + * filesystem xattr support. */ +static void test_fake_super_rsync_format() { + const char* path = "test_fake_super_format.txt"; + unlink(path); + int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0600); + if (fd < 0) + return; + bool has_xattr = setxattr(path, "user.fastsync.xprobe", "p", 1, 0) == 0; + if (has_xattr) + removexattr(path, "user.fastsync.xprobe"); + if (!has_xattr) { + close(fd); + unlink(path); + return; /* skip silently when the filesystem has no xattr support */ + } + + /* A setuid regular file: the full st_mode (with S_IFMT + special bits) is + recorded, rdev is 0,0, and the owner is uid:gid. */ + fake_super_store_fd(fd, 1234, 5678, S_IFREG | 04711, 0, 0); + char value[128]; + ssize_t got = fgetxattr(fd, FAKESUPER_XATTR, value, sizeof(value)); + EXPECT_EQ_INT((int)got, 20); + EXPECT_TRUE(got == 20 && memcmp(value, "104711 0,0 1234:5678", 20) == 0); + + /* The special bits in the record are NOT installed on the real file. */ + FileAttrPolicy policy = {true, true, false, false, true}; + EXPECT_TRUE(fake_super_restore_fd(fd, policy)); + struct stat st; + EXPECT_EQ_INT(fstat(fd, &st), 0); + EXPECT_EQ_INT((int)(st.st_mode & 07777), 0711); + EXPECT_EQ_INT((int)(st.st_mode & (S_ISUID | S_ISGID | S_ISVTX)), 0); + + /* A device record (char 1,3, uid 111, gid 222) parses without error and + still never real-chowns or installs the device's mode bits verbatim. */ + EXPECT_EQ_INT((int)fsetxattr(fd, FAKESUPER_XATTR, "20644 1,3 111:222", 17, 0), 0); + struct stat before; + fstat(fd, &before); + EXPECT_TRUE(fake_super_restore_fd(fd, policy)); + fstat(fd, &st); + EXPECT_EQ_INT((int)(st.st_mode & 0777), 0644); + EXPECT_EQ_INT((int)st.st_uid, (int)before.st_uid); + EXPECT_EQ_INT((int)st.st_gid, (int)before.st_gid); + + close(fd); + unlink(path); +} + /* --fake-super must NEVER perform a real chown: fake_super_restore_fd applies * only mode/mtime and leaves the entry's uid/gid exactly as they were, even * when an explicit ownership policy is active and super_mode permits it. This @@ -421,7 +474,7 @@ static void test_fake_super_no_real_chown() { } struct stat before; EXPECT_EQ_INT(fstat(fd, &before), 0); - fake_super_store_fd(fd, 12345, 12346, 0755, 1700000000, 0); + fake_super_store_fd(fd, 12345, 12346, S_IFREG | 0755, 0, 0); Config* c = config_create(); FileAttrPolicy policy = {true, true, false, false, true}; @@ -600,6 +653,7 @@ void test_xattr() { test_xattr_receive_drops_acl_without_preserve_acls(); test_link_copy_fallback_preserves_xattrs(); test_fake_super_restore(); + test_fake_super_rsync_format(); test_fake_super_no_real_chown(); test_fake_super_storage_resolution(); test_file_save_directory_applies_xattrs(); -- 2.54.0 From 3ec0ffb64450d57e0137b3035dfacd01295b7ef4 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 22:05:50 +0200 Subject: [PATCH 56/68] fix(delete-before): reuse pre-scan file list in single-threaded data pass --- RSYNC_COMPAT.md | 4 +- src/client/client_scan.c | 34 ++++++-- src/client/client_send.c | 87 ++++++++++++++----- src/client/client_send_internal.h | 3 +- .../integration/test_delete_timing_parity.py | 81 ++++++++++++++++- 5 files changed, 173 insertions(+), 36 deletions(-) diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 5a6b5b8..a3a45b6 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -193,7 +193,7 @@ Every one of those has an entry below with its remaining caveats. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| | `--delete` | Delete extraneous files from dest | ✅ Parity | `use_delete` config field. Deletion is always derived from the keep-set the sender actually transmitted (the per-directory `STATUS_DELETE_PLAN` set by default, or the whole-tree manifest for the late timings — never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. **Lockstep track 6 (protocol 2.28.0): plain `--delete` with no explicit timing flag now defaults to `--delete-during`**, exactly like rsync's `--del` (the client normalizes it to the existing `delete_during` wire bool; no new wire field). This frees destination space progressively during the transfer and avoids the whole-old+new-tree peak that could `ENOSPC` a tight destination. The old late whole-tree commit is opt-in via `--delete-after` or the FastSync-only long spelling `--delete-commit`. **Abort/ordering parity (parity-2.29):** the complete per-directory plan set is transmitted before the first data frame, so a mid-transfer abort has already applied every planned removal exactly like rsync's generator (which runs ahead of its throttled sender); `-d/--dirs` uses the same per-directory plans (the generator records only the directories whose direct children it enumerated, so an untraversed subdirectory's mirror is shielded); and the sorted depth-first traversal makes the removal order — and therefore the survivor set under a partial `--max-delete` — match rsync exactly (`test_delete_boundary_parity.py`, `test_parity_order.py`). By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). Deletion is scoped to the **synchronized directories** sent on the wire (protocol 2.23.0), so a `--files-from` subset no longer deletes untransmitted paths outside the listed directory subtrees. The walk is bounded: a client `--max-delete=NUM` (or the 100000-entry server bound) makes it **partial** — entries up to the bound are removed, the rest are skipped, and the client exits **25** (`RERR_PARTIAL`), matching rsync, rather than failing the transfer. Extraneous destination symlinks are unlinked by name (never followed); a directory still holding a kept/protected entry is left behind rather than failing | -| `--delete-before` | Delete before transfer | ⚠️ Caveat | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. **Phase-0 divergence (sharpened):** rsync builds the full file list first, so a source file created after that scan is NOT transferred and its destination extra is deleted; FastSync's single-threaded data pass re-scans the source, so the late file IS transferred (a safe superset), while FastSync `--threads` pipelines the scan and matches rsync | +| `--delete-before` | Delete before transfer | ✅ Parity | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. **Phase-0 divergence closed (no-wire):** the single-threaded data pass now replays the exact file list the pre-scan built for the keep-set instead of re-reading the source, so a source file created after that scan is NOT transferred and its destination extra is deleted, exactly like rsync's single file list (and like the `--threads` path). The pre-scan captures the deferred directory times and the `--stats` directory count because no later scan runs (`test_delete_timing_parity.py::TestDeleteBeforeLateFileParity`, differential vs rsync 3.4.1) | | `--del`, `--delete-during` | Delete during transfer | ✅ Parity | Both spellings accepted; imply `--delete`, and since lockstep track 6 this is also the default timing of a plain `--delete`. **Protocol 2.24.0 implements per-directory delete plans:** as the sender reaches each source directory it streams a `STATUS_DELETE_PLAN` for that directory and the receiver removes that directory's extras (verified with a byte-slicing proxy). The one-shot per-run config block (protected prefixes, size-pruned mirrors, `--delete-missing-args` exact paths) rides a dedicated config-only carrier frame with an `apply=false` flag, so it reaches the receiver even when the scope allows no directory plan at all (a `--files-from` list of bare files synchronizes no directory). **Abort/ordering parity (parity-2.29):** the complete plan set is transmitted before the first data frame, so on a mid-transfer abort every planned extra has already been removed exactly like rsync's generator (which runs ahead of its throttled sender); `-d/--dirs` no longer falls back to the end-of-transfer commit but records only the directories whose direct children it enumerated; and the sorted depth-first traversal makes the removal order — and the partial-`--max-delete` survivor set — identical to rsync (`test_delete_boundary_parity.py`, `test_parity_order.py`). `-R` plans are scoped to the transferred prefix subtree | | `--delete-delay` | Find deletions during, delete after | ✅ Parity | Implies `--delete`. **Protocol 2.24.0 implements rsync's delete-delay timing:** the sender records each directory's delete plan while scanning and the receiver commits those removals only after the whole transfer succeeds (per plan), so an extra created in the destination after its directory's plan survives while `--delete-after` re-scans and removes it, and a failed transfer removes nothing. The **reported** deleted count advances only on an actual removal. **Fixed (no-wire):** the `--max-delete` budget is now charged on ACTUAL removals (an unlink/rmdir that succeeded), not at plan/snapshot time, and a queued directory is re-scanned at commit and removed recursively (content created after the plan included), matching rsync: a snapshotted entry that fails or is skipped consumes no budget, so a later extra rsync would delete is still deleted. The deferred snapshot list keeps an independent hard cap (`DELETE_PLAN_SERVER_LIMIT`) so it cannot grow without bound now that the budget is no longer charged while scanning. A `--max-delete=2` partial delete reports exactly 2 and exits 25 in both tools, and the refilled-directory differential (late content removed, directory removed, budget shared) now matches rsync 3.4.1 on both sides (`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`). Unit tests cover recursive removal, actual-removal charging, and the bounded deferred list. **Ordering parity (parity-2.29):** the sorted depth-first traversal plus the up-front plan set make the order in which extras are removed — and therefore the survivor set under a partial `--max-delete` — match rsync exactly (differential `test_parity_order.py::test_delete_delay_deletion_order_matches_rsync` and `::test_partial_max_delete_survivor_order_matches_rsync`) | | `--delete-after` | Delete after transfer | ✅ Parity | Implies `--delete`. Selects the late whole-tree commit: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing. Since lockstep track 6 a plain `--delete` defaults to delete-during (rsync's `--del`); `--delete-after` — or the FastSync-only `--delete-commit` spelling, which selects the identical timing — is the explicit way to keep the old commit-style behavior | @@ -956,7 +956,7 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, and the triage cycle.** ✅ Parity 117 / ⚠️ Caveat 13 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--delete-before`, `--filter`, `-F`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, the triage cycle, and the no-wire `--delete-before` phase-0 close.** ✅ Parity 118 / ⚠️ Caveat 12 / ❌ Divergent 27 = 157 rows. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The `--delete-before` close (no-wire) makes the single-threaded data pass replay the pre-scan file list, so a source file created after the scan is neither transferred nor kept, matching rsync (`--delete-before` ⚠️ → ✅). The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--filter`, `-F`, the three basis-dir options, and `-y/--fuzzy`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that receiver filter engine, flipping `--filter=RULE` back to ✅ (see above; the diff --git a/src/client/client_scan.c b/src/client/client_scan.c index 0a01530..4cc4283 100644 --- a/src/client/client_scan.c +++ b/src/client/client_scan.c @@ -381,21 +381,28 @@ bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* s paths, loading and sending nothing. --delete-before/--delete-during need the complete keep-set manifest before the first data byte, so it is built by a dedicated pre-scan pass and transmitted early; the data pass then re-scans - with a fresh scanner. A source I/O error is fatal unless the options carry - --ignore-errors, in which case the scan continues past the unreadable - directory and *io_error_out reports it (the caller still performs the - deletion but reports the run as errored). */ + with a fresh scanner. --delete-before additionally replays this very scan as + its data pass (rsync's single file list), so `chunks_out` (optional) retains + the scanned Chunk objects for the caller to send instead of destroying them; + the caller owns the list and must give it a chunk_destroy destructor. A + source I/O error is fatal unless the options carry --ignore-errors, in which + case the scan continues past the unreadable directory and *io_error_out + reports it (the caller still performs the deletion but reports the run as + errored). */ bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest, DeletePlanSender* plans, bool* io_error_out, - unsigned long long* non_dir_count_out) { + unsigned long long* non_dir_count_out, ArrayList* chunks_out, + bool emit_nonreg) { if (io_error_out) *io_error_out = false; if (non_dir_count_out) *non_dir_count_out = 0; ScannerOptions local = *options; - /* The pre-scan is a paths-only pass with no client output; it must not emit - --info=nonreg lines (the data pass does that once). */ - local.note_nonreg = false; + /* The pre-scan is normally a paths-only pass with no client output: it must + not emit --info=nonreg lines because the data pass re-scans and emits them + once. When the caller replays this scan as the data pass (--delete-before) + there is no later scan, so it opts in and the lines are emitted here. */ + local.note_nonreg = emit_nonreg && options->note_nonreg; DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local); if (!scanner) return false; @@ -430,7 +437,16 @@ bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayL break; } } - chunk_destroy(chunk); + if (chunks_out) { + /* Retain the chunk for the caller's data pass; ownership moves with it. */ + if (!array_list_add(chunks_out, chunk)) { + ok = false; + chunk_destroy(chunk); + break; + } + } else { + chunk_destroy(chunk); + } } if (ok) { /* Keep every traversed source directory, including empty ones, so a plan diff --git a/src/client/client_send.c b/src/client/client_send.c index 9bb783f..8c6212e 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1394,6 +1394,10 @@ typedef struct { Client* client; DirectoryScanner* scanner; ArrayList* manifest; + /* --delete-before: the pre-scan that built the keep-set, retained as the data + pass's file list (owning Chunk*; consumed chunks are NULLed as they are + sent). NULL in every other mode, where the data pass scans normally. */ + ArrayList* prescan_chunks; DeletePlanSender* plan_sender; ArrayList* remove_sources; ArrayList* dir_entries; @@ -1478,12 +1482,23 @@ static bool send_files_prepare_delete(Config* config, SendFilesState* state) { if (state->delete_early) { /* Pass 1: collect the complete keep-set (paths only, no data loaded) and transmit it now, before any file data. The receiver removes extras and - acks; the transfer aborts here if the deletion could not commit. */ + acks; the transfer aborts here if the deletion could not commit. The + scanned chunks are retained so the data pass can replay this exact list + instead of re-reading the source (rsync builds one file list and never + transfers a file created after it). */ ArrayList* early_manifest = array_list_create(free); - if (!early_manifest) + ArrayList* prescan_chunks = array_list_create(chunk_destroy); + if (!early_manifest || !prescan_chunks) { + array_list_delete(early_manifest); + array_list_delete(prescan_chunks); return false; + } + /* No later scan runs for --delete-before, so this pass must also capture the + deferred directory times and the --stats directory count. */ + state->prepared.options.dir_entries = state->dir_entries; + state->prepared.options.dir_count = config->stats ? &state->dir_count : NULL; bool prescan_ok = scan_paths_only(config, &state->prepared.options, early_manifest, NULL, - &state->had_scan_io, NULL); + &state->had_scan_io, NULL, prescan_chunks, true); bool early_ok = false; bool skip_delete = false; if (prescan_ok) { @@ -1514,8 +1529,13 @@ static bool send_files_prepare_delete(Config* config, SendFilesState* state) { state->prepared.options.excluded_paths = NULL; state->prepared.options.size_skipped_paths = NULL; state->prepared.options.synced_dirs = NULL; - if (!prescan_ok || (!early_ok && !skip_delete)) + if (!prescan_ok || (!early_ok && !skip_delete)) { + array_list_delete(prescan_chunks); return false; + } + /* Adopt the captured scan as the data pass's file list (including when an + I/O error suppressed only the deletion: the list is still complete). */ + state->prescan_chunks = prescan_chunks; } else if (state->delete_per_dir) { /* --delete-during/--delete-delay: build one plan per source directory from a path-only pre-scan and transmit the COMPLETE plan set now, before any data, @@ -1527,8 +1547,9 @@ static bool send_files_prepare_delete(Config* config, SendFilesState* state) { if (!state->plan_sender || !state->plan_dirs) return false; state->prepared.options.plan_dirs = state->plan_dirs; - bool prescan_ok = scan_paths_only(config, &state->prepared.options, NULL, state->plan_sender, - &state->had_scan_io, &state->per_dir_non_dir_count); + bool prescan_ok = + scan_paths_only(config, &state->prepared.options, NULL, state->plan_sender, + &state->had_scan_io, &state->per_dir_non_dir_count, NULL, false); bool plans_ok = false; bool skip_delete = false; if (prescan_ok) { @@ -1592,24 +1613,42 @@ static bool send_files_run(Config* config, SendFilesState* state) { state->stop = stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->cli.stop_at_set, config->stop_at, now_mono); state->prepared.options.stop_condition = &state->stop; - /* The early-delete pre-scan above already ran; only the data pass should feed - the directory-time list (otherwise every directory would be captured - twice). */ - state->prepared.options.dir_entries = state->dir_entries; - state->prepared.options.dir_count = config->stats ? &state->dir_count : NULL; - state->scanner = - directory_scanner_create_with_options(config->send_directory, &state->prepared.options); - if (!state->scanner) - return false; + /* --delete-before reuses the pre-scan that built the keep-set as the data + pass's file list, so a source file created after that scan is neither + transferred nor kept (rsync builds one file list). That pre-scan captured + the deferred directory times and the --stats directory count because no + later scan runs; every other mode opens a fresh data scanner here. */ + if (state->prescan_chunks == NULL) { + state->prepared.options.dir_entries = state->dir_entries; + state->prepared.options.dir_count = config->stats ? &state->dir_count : NULL; + state->scanner = + directory_scanner_create_with_options(config->send_directory, &state->prepared.options); + if (!state->scanner) + return false; + } Chunk* current_chunk; + int prescan_index = 0; memset(&state->transfer_stats, 0, sizeof(state->transfer_stats)); state->start = time(NULL); client_progress_begin(config); /* True when the stop deadline cut the scan short so the keep-set manifest is only a prefix of the source. */ bool send_failed = false; - while ((current_chunk = directory_scanner_next(state->scanner)) != NULL) { + while (true) { + if (state->prescan_chunks != NULL) { + if (prescan_index >= state->prescan_chunks->size) + break; + /* Move ownership out of the retained list so chunk_destroy below (and the + cleanup tail for an early exit) never double-frees it. */ + current_chunk = (Chunk*)state->prescan_chunks->items[prescan_index]; + state->prescan_chunks->items[prescan_index] = NULL; + prescan_index++; + } else { + current_chunk = directory_scanner_next(state->scanner); + if (current_chunk == NULL) + break; + } /* Graceful abort (Ctrl-C/SIGTERM): notify the receiver and clean up. The session is active (config_send already succeeded); a send failure here is fine because the client is exiting anyway. */ @@ -1667,10 +1706,14 @@ static bool send_files_run(Config* config, SendFilesState* state) { * Returns the rsync-compatible exit code. */ static int send_files_finalize(const Config* config, SendFilesState* state) { Client* client = state->client; - if (directory_scanner_failed(state->scanner)) - return 1; - if (directory_scanner_had_io_error(state->scanner)) - state->had_scan_io = true; + /* A --delete-before run replays the pre-scan and owns no data scanner; its + I/O-error verdict was already recorded by that pre-scan. */ + if (state->scanner != NULL) { + if (directory_scanner_failed(state->scanner)) + return 1; + if (directory_scanner_had_io_error(state->scanner)) + state->had_scan_io = true; + } /* An abort that arrived after the last chunk must still stop the completion tail (manifest/finalize) rather than let it run to success. */ if (client_abort_pending()) { @@ -1774,6 +1817,8 @@ static int send_files_finalize(const Config* config, SendFilesState* state) { static void send_files_cleanup(SendFilesState* state) { if (state->manifest) array_list_delete(state->manifest); + if (state->prescan_chunks) + array_list_delete(state->prescan_chunks); if (state->plan_sender) delete_plan_sender_destroy(state->plan_sender); if (state->excluded) @@ -1993,7 +2038,7 @@ int send_files_multithreaded(Config* config) { bool prebuilt = prepared_ok && scan_paths_only(config, &prepared.options, context->manifest, context->delete_plans, - &context->scan_had_io_error, &pre_scan_non_dir); + &context->scan_had_io_error, &pre_scan_non_dir, NULL, false); prepared_scanner_destroy(&prepared); if (per_dir && prebuilt) { const char* walk_root = delete_plan_walk_root(config, context->synced_dirs); diff --git a/src/client/client_send_internal.h b/src/client/client_send_internal.h index d82d45d..7ce71bc 100644 --- a/src/client/client_send_internal.h +++ b/src/client/client_send_internal.h @@ -44,7 +44,8 @@ const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_ bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out); bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest, DeletePlanSender* plans, bool* io_error_out, - unsigned long long* non_dir_count_out); + unsigned long long* non_dir_count_out, ArrayList* chunks_out, + bool emit_nonreg); /* client_report.c */ void log_server_rejection(const char* context); diff --git a/tests/integration/test_delete_timing_parity.py b/tests/integration/test_delete_timing_parity.py index 4f4c78d..6757f76 100644 --- a/tests/integration/test_delete_timing_parity.py +++ b/tests/integration/test_delete_timing_parity.py @@ -111,13 +111,20 @@ class _SlicingProxy: """ def __init__(self, target_port, forward_limit=None, hook=None, hook_after=0, - throttle=0.0, wait_for_reply=False): + throttle=0.0, wait_for_reply=False, hook_after_config_ack=False): self.target = ("127.0.0.1", target_port) self.forward_limit = forward_limit self.hook = hook self.hook_after = hook_after self.throttle = throttle self.wait_for_reply = wait_for_reply + # When set, the hook fires on the FIRST client->server bytes that follow + # the config-frame ack, BEFORE they are forwarded. For --delete-before + # those bytes are the keep-set manifest, so this runs the hook after the + # client's source pre-scan but before the receiver's delete ack releases + # the client into its data pass -- a deterministic late-file window. + self.hook_after_config_ack = hook_after_config_ack + self.config_acked = False self.server_replied = threading.Event() self.hook_called = threading.Event() self.listener = socket.socket(socket.AF_INET, socket.SOCK_STREAM) @@ -165,6 +172,13 @@ class _SlicingProxy: socks = [] break data = data[:room] + if (self.hook_after_config_ack and self.config_acked and self.hook is not None + and not self.hook_called.is_set()): + # The first client bytes after the config ack are the + # pre-scan keep-set manifest: run the injection before + # forwarding so it is causally after the source scan. + self.hook() + self.hook_called.set() backend.sendall(data) forwarded += len(data) self._maybe_hook(forwarded) @@ -177,6 +191,7 @@ class _SlicingProxy: client.sendall(data) # Any server reply proves the receiver consumed the # frames that precede it, so the hook barrier is met. + self.config_acked = True self.server_replied.set() self._maybe_hook(forwarded) except OSError: @@ -198,8 +213,11 @@ class _SlicingProxy: def _maybe_hook(self, forwarded): """Fire the one-shot hook once its barrier is satisfied: enough client bytes have been forwarded and, when ``wait_for_reply`` is set, the - server has sent a reply proving it processed the preceding frames.""" - if self.hook is None or self.hook_called.is_set(): + server has sent a reply proving it processed the preceding frames. + + ``hook_after_config_ack`` uses its own barrier (see ``_serve``), so the + byte/reply heuristic is bypassed entirely.""" + if self.hook is None or self.hook_called.is_set() or self.hook_after_config_ack: return if forwarded < self.hook_after: return @@ -537,3 +555,60 @@ class TestDeleteDelayMaxDeleteRefilledDir: assert os.path.isdir(later_dir), "later extra was not skipped by the budget" # The one actual removal is reported. assert _deleted_count(result.stdout) == 1, result.stdout + + +class TestDeleteBeforeLateFileParity: + """rsync builds its file list once, so a source file created after that scan + is NOT transferred and its destination extra is deleted. FastSync's + single-threaded --delete-before used to re-scan the source in its data pass + and would transfer the late file (a safe superset); it now replays the + pre-scan file list instead, matching rsync. + + The late file is injected through the config-ack barrier: the first client + bytes after the config ack are the pre-scan keep-set manifest, so the hook + runs causally after the source scan and before the receiver's delete ack + releases the client into its data pass -- deterministic, no timing guess. + """ + + @requires_rsync + def test_late_source_file_not_transferred_and_extra_deleted(self): + source = os.path.join(TEST_DATA_DIR, "dblate_src") + dest = os.path.join(TEST_DATA_DIR, "dblate_dst") + rsync_dst = os.path.join(TEST_DATA_DIR, "dblate_rsync_dst") + clean_dir(source) + clean_dir(dest) + clean_dir(rsync_dst) + _write(os.path.join(source, "d", "keep.txt"), b"kept payload\n") + # Both destinations carry the would-be late file as an extra. + for root in (dest, rsync_dst): + _write(os.path.join(get_dest_received_dir(root, source), "d", "late.txt"), + b"stale extra\n") + + # rsync reference: the same source with no late file; the extra is removed + # and nothing is transferred for the (never-scanned) late path. + rsync_result = _rsync(["-a", "--delete-before", source + "/", rsync_dst + "/"]) + assert rsync_result.returncode == 0, rsync_result.stderr + rsync_tree = _tree(rsync_dst) + assert "d/late.txt" not in rsync_tree + + received = get_dest_received_dir(dest, source) + late_source = os.path.join(source, "d", "late.txt") + + def hook(): + # Runs after the pre-scan and before the data pass begins. + _write(late_source, b"created after the scan\n") + + with ServerManager() as server: + server.start(extra_args=["--allow-delete"]) + proxy = _SlicingProxy(server.port, hook=hook, hook_after_config_ack=True) + result, _ = run_client(source, dest, flags=["--delete-before"], port=proxy.port) + proxy.finish() + assert result.returncode == 0, (result.stderr or result.stdout)[:400] + assert proxy.hook_called.is_set(), "late-file hook never fired" + assert os.path.exists(late_source), "the source late file unexpectedly vanished" + assert not os.path.exists(os.path.join(received, "d", "late.txt")), ( + "late source file was transferred: the single-threaded data pass re-scanned" + ) + assert _tree(received) == rsync_tree, ( + f"fastsync tree {_tree(received)} != rsync tree {rsync_tree}" + ) -- 2.54.0 From bb6c788cf9e97a961a0de81537ed83eb76b5fe97 Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 22:53:23 +0200 Subject: [PATCH 57/68] fix(daemon): rsync read-only module default; warn on unenforced security keys --- CHANGELOG.md | 18 ++++ README.md | 14 ++- src/shared/daemon_conf.c | 97 +++++++++++++++-- src/shared/daemon_conf.h | 28 +++-- tests/integration/test_daemon.py | 16 ++- tests/test_daemon_conf.c | 172 +++++++++++++++++++++++++++++++ 6 files changed, 322 insertions(+), 23 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index cd7502d..2853154 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -79,6 +79,24 @@ of 157 rows. - **Daemon umask no longer forced to `0`.** `daemonize()` now sets the conventional `022`, so implied parent directories created without `-p` are no longer world-writable `0777`. +- **Daemon modules are read-only by default.** A `--daemon` module is now + served read-only unless it sets `read only = no` (or rsync's `write only = + yes`), matching rsync: a real `rsyncd.conf` that omits `read only` is no + longer silently writable. A global `read only` still sets the default for + later modules, and an explicit module value wins. This is a behavior change + for existing FastSync-native configs that relied on the old writable default; + add `read only = no` to keep them writable. An rsync `write only = yes` is + mapped to writability (FastSync is push-only, so a module can never be read + from the network). +- **Accepted-but-unenforced rsync security keys now warn at startup.** The + rsync keys FastSync recognizes but does not implement — `secrets file`, + `refuse options`, `exclude`/`include`/`filter`, `max size`/`min size`, + `pre-xfer exec`/`post-xfer exec`, `incoming chmod`/`outgoing chmod`, + `name converter`, `use chroot`, `uid`/`gid`, and the rest of the + access-control set — load for migration compatibility but now emit a + `WARN` naming the key (and module) so an operator does not believe the + restriction is enforced. `auth users`/`secrets file` stay fail-closed: a + module declaring `auth users` still requires a FastSync credential store. - **Credentials and signal handling hardened.** Secret files are opened with `O_NOFOLLOW|O_NONBLOCK` (while allowing fd-backed store paths and bound-waiting a FIFO read for ~3 s so a slow process substitution works but a connected-but- diff --git a/README.md b/README.md index c40469f..897b1b5 100644 --- a/README.md +++ b/README.md @@ -768,9 +768,17 @@ and `address`, the global section accepts: - `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access patterns. -A `[module]` requires `path`, and may also set `read only`, `client owner`, -`auth users`, `max connections` (0 = unlimited; enforced per module across all -connection children), and its own `hosts allow`/`hosts deny`. +A `[module]` requires `path`, and may also set `read only`, `write only`, +`client owner`, `auth users`, `max connections` (0 = unlimited; enforced per +module across all connection children), and its own `hosts allow`/`hosts deny`. + +Like rsync, a module is **read-only by default**: a bare `[module]` with only a +`path` refuses a write transfer. Opt a module into writability explicitly with +`read only = no` or `write only = yes`; a global `read only` value in the +section before the first `[module]` sets the default for later modules, and a +module's own `read only`/`write only = yes` always wins over it. An +rsync-style `write only = yes` is mapped to writability because FastSync is +push-only (a module can never be read from the network). The per-host cap and the shared auth lockout identify a source by its numeric peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local diff --git a/src/shared/daemon_conf.c b/src/shared/daemon_conf.c index 3b0a351..5144668 100644 --- a/src/shared/daemon_conf.c +++ b/src/shared/daemon_conf.c @@ -1,5 +1,6 @@ #include "daemon_conf.h" #include "credentials.h" +#include "log.h" #include "utils.h" #include #include @@ -77,11 +78,13 @@ static const char* const kRsyncInertGlobalKeys[] = { /* rsync 3.4.1 rsyncd.conf MODULE keys accepted in a [module] section that have * no FastSync equivalent (accepted-and-documented inert). Keys with a FastSync - * meaning (`path`, `read only`, `auth users`, `max connections`, + * meaning (`path`, `read only`, `write only`, `auth users`, `max connections`, * `hosts allow`/`hosts deny`, `client owner`) are handled by apply_module_key * before this list is consulted. Security-relevant keys (`exclude`, `filter`, * `secrets file`, `refuse options`, ...) are inert, so a daemon-side filter or - * rsync secrets file is NOT enforced: see RSYNC_COMPAT.md for the residual. */ + * rsync secrets file is NOT enforced: each is loudly warned about at load time + * (see kRsyncUnenforcedModuleSecurityKeys) and documented as a residual in + * RSYNC_COMPAT.md. */ static const char* const kRsyncInertModuleKeys[] = { "comment", "use chroot", @@ -110,7 +113,6 @@ static const char* const kRsyncInertModuleKeys[] = { "numeric ids", "fake super", "munge symlinks", - "write only", "list", "dont compress", "charset", @@ -132,8 +134,57 @@ static const char* const kRsyncInertModuleKeys[] = { "ignore nonreadable", }; +/* Subset of the inert rsync keys whose intent is access control (data + * visibility, credential source, transfer hooks, daemon privilege), plus the + * global keys that shape the daemon's privilege/identity. These load for + * rsync-config compatibility, but because FastSync ignores them an operator + * migrating a hardened rsyncd.conf must not believe the restriction applies. + * The loader emits one LOG_LEVEL_WARNING per occurrence naming the key (and the + * module, for a module key). `write only` is deliberately absent: it is mapped + * onto writability instead (FastSync is push-only, so a write-only module is + * simply writable). */ +static const char* const kRsyncUnenforcedModuleSecurityKeys[] = { + "secrets file", + "auth digest", + "refuse options", + "exclude", + "include", + "exclude from", + "include from", + "filter", + "max size", + "min size", + "pre-xfer exec", + "post-xfer exec", + "incoming chmod", + "outgoing chmod", + "name converter", + "use chroot", + "daemon chroot", + "uid", + "gid", + "daemon uid", + "daemon gid", + "munge symlinks", + "fake super", + "strict modes", + "proxy protocol", + "proxy protocol hosts", +}; + +static const char* const kRsyncUnenforcedGlobalSecurityKeys[] = { + "use chroot", + "uid", + "gid", + "strict modes", +}; + #define kRsyncInertGlobalCount (sizeof(kRsyncInertGlobalKeys) / sizeof(kRsyncInertGlobalKeys[0])) #define kRsyncInertModuleCount (sizeof(kRsyncInertModuleKeys) / sizeof(kRsyncInertModuleKeys[0])) +#define kRsyncUnenforcedModuleSecurityCount \ + (sizeof(kRsyncUnenforcedModuleSecurityKeys) / sizeof(kRsyncUnenforcedModuleSecurityKeys[0])) +#define kRsyncUnenforcedGlobalSecurityCount \ + (sizeof(kRsyncUnenforcedGlobalSecurityKeys) / sizeof(kRsyncUnenforcedGlobalSecurityKeys[0])) static bool parse_bool_value(const char* value, bool* out) { if (strcasecmp(value, "yes") == 0 || strcasecmp(value, "true") == 0 || strcmp(value, "1") == 0) { @@ -352,7 +403,10 @@ DaemonConf* daemon_conf_create(void) { if (!conf) return NULL; conf->global.port = DAEMON_CONF_DEFAULT_PORT; - conf->global.read_only_default = false; + /* rsync modules are READ-ONLY unless `read only = no` (or `write only = yes`) + * is set, so FastSync must default the same way: a migrated rsyncd.conf that + * omits `read only` is served read-only, never writable. */ + conf->global.read_only_default = true; conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS; conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS; conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST; @@ -483,8 +537,14 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value, "hosts deny", NULL, replace_hosts, err, err_size); /* A recognized rsync global key with no FastSync equivalent loads inert. */ - if (key_in_list(key, kRsyncInertGlobalKeys, kRsyncInertGlobalCount)) + if (key_in_list(key, kRsyncInertGlobalKeys, kRsyncInertGlobalCount)) { + if (key_in_list(key, kRsyncUnenforcedGlobalSecurityKeys, kRsyncUnenforcedGlobalSecurityCount)) + log_message(LOG_LEVEL_WARNING, + "daemon config: global key '%s' is accepted for rsync compatibility but is NOT " + "enforced by FastSync; the restriction it expresses will not be applied", + key); return true; + } set_error(err, err_size, "unknown global key '%s'", key); return false; } @@ -516,6 +576,25 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* module->read_only_explicit = true; return true; } + /* rsync's `write only = yes` makes the module client-writable. FastSync has + * no read/pull path, so mapping it to writability is the exact + * security-relevant effect; set `read_only_explicit` so a global default + * cannot override the module's explicit choice. `write only = no` is the + * rsync default and leaves the module's read-only state untouched. */ + if (key_equals(key, "write only")) { + bool parsed; + if (!parse_bool_value(value, &parsed)) { + set_error(err, err_size, + "module '%s': 'write only' must be yes/no (or true/false/1/0), got '%s'", + module->name, value); + return false; + } + if (parsed) { + module->read_only = false; + module->read_only_explicit = true; + } + return true; + } if (key_equals(key, "client owner")) { bool parsed; if (!parse_bool_value(value, &parsed)) { @@ -584,8 +663,14 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char* return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny", module->name, false, err, err_size); /* A recognized rsync module key with no FastSync equivalent loads inert. */ - if (key_in_list(key, kRsyncInertModuleKeys, kRsyncInertModuleCount)) + if (key_in_list(key, kRsyncInertModuleKeys, kRsyncInertModuleCount)) { + if (key_in_list(key, kRsyncUnenforcedModuleSecurityKeys, kRsyncUnenforcedModuleSecurityCount)) + log_message(LOG_LEVEL_WARNING, + "daemon config: module '%s' key '%s' is accepted for rsync compatibility but is " + "NOT enforced by FastSync; the restriction it expresses will not be applied", + module->name, key); return true; + } set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name); return false; } diff --git a/src/shared/daemon_conf.h b/src/shared/daemon_conf.h index 81af6db..49e7a30 100644 --- a/src/shared/daemon_conf.h +++ b/src/shared/daemon_conf.h @@ -21,12 +21,16 @@ * rsync compatibility: to reduce the divergence from rsync 3.4.1's rsyncd.conf * grammar, the parser also ACCEPTS the common rsync GLOBAL and MODULE keys. * Keys with a FastSync equivalent are mapped onto it (the native spellings are - * unchanged). Keys with no FastSync equivalent are accepted and documented as - * inert (they load successfully but have no effect) rather than failing the - * whole config; the accepted inert set is listed in kRsyncInertGlobalKeys / - * kRsyncInertModuleKeys in daemon_conf.c and in RSYNC_COMPAT.md. A key - * outside both the FastSync-native grammar and the recognized rsync subset is - * still rejected as unknown. */ + * unchanged; `read only` defaults to yes like rsync, and `write only = yes` + * opts a module into writability). Keys with no FastSync equivalent are + * accepted and documented as inert (they load successfully but have no effect) + * rather than failing the whole config; the accepted inert set is listed in + * kRsyncInertGlobalKeys / kRsyncInertModuleKeys in daemon_conf.c and in + * RSYNC_COMPAT.md. Every inert key whose intent is access control is loudly + * warned about at load time (kRsyncUnenforced*SecurityKeys) so an operator + * migrating a hardened rsyncd.conf is never misled into believing the + * restriction is enforced. A key outside both the FastSync-native grammar and + * the recognized rsync subset is still rejected as unknown. */ /* A daemon module's configured root is used exactly like the standalone * server's --destination-root: the daemon confines every connection that @@ -55,10 +59,11 @@ typedef struct DaemonModule { char* path; /* module root (daemon-side authorized root) */ bool read_only; /* `read only = yes/no`; defaults to the global `read only` default (rsync allows it in the global section), which is - itself default no */ - bool read_only_explicit; /* set when this module set its own `read only`, so a - later global default (from a `--dparam read only=`) - does not override it */ + itself default YES (rsync modules are read-only unless + `read only = no` / `write only = yes` opts in) */ + bool read_only_explicit; /* set when this module set its own `read only` or + `write only = yes`, so a later global default (from a + `--dparam read only=`) does not override it */ bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in that lets this module's clients choose ownership (--numeric-ids/--chown/--usermap/--groupmap/--fake-super/ @@ -85,7 +90,8 @@ typedef struct DaemonConfGlobals { char* address; /* `address` (optional bind address), may be NULL */ bool read_only_default; /* global `read only` default for modules defined after it (rsync allows the module key in the - global section); default no */ + global section); default YES to match rsync's + read-only modules */ int max_connections; /* `max connections`, default DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */ int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default diff --git a/tests/integration/test_daemon.py b/tests/integration/test_daemon.py index 2b9917a..e424300 100644 --- a/tests/integration/test_daemon.py +++ b/tests/integration/test_daemon.py @@ -229,6 +229,7 @@ def daemon_env(): "\n" "[files]\n" "path = %s\n" + "read only = no\n" "\n" "[readonly]\n" "path = %s\n" @@ -236,18 +237,22 @@ def daemon_env(): "\n" "[locked]\n" "path = %s\n" + "read only = no\n" "auth users = alice\n" "\n" "[team]\n" "path = %s\n" + "read only = no\n" "auth users = alice,bob\n" "\n" "[owner]\n" "path = %s\n" + "read only = no\n" "client owner = yes\n" "\n" "[denied]\n" "path = %s\n" + "read only = no\n" "hosts deny = 127.0.0.1\n" % (config_port, FILES_MODULE, READONLY_MODULE, AUTH_MODULE, TEAM_MODULE, OWNER_MODULE, DENIED_MODULE)) @@ -266,7 +271,8 @@ def daemon_env(): global DETACH_PORT DETACH_PORT = _find_free_port() with open(DETACH_CONF, "w") as f: - f.write("port = %d\n\n[detach]\npath = %s\n" % (DETACH_PORT, DETACH_MODULE)) + f.write("port = %d\n\n[detach]\npath = %s\nread only = no\n" + % (DETACH_PORT, DETACH_MODULE)) yield _kill_by_cmdline_marker(DETACH_CONF) @@ -353,7 +359,8 @@ class TestDaemonModuleSelection: the fix regresses.""" port = _find_free_port() with open(UMASK_CONF, "w") as f: - f.write("port = %d\n\n[files]\npath = %s\n" % (port, FILES_MODULE)) + f.write("port = %d\n\n[files]\npath = %s\nread only = no\n" + % (port, FILES_MODULE)) sub = os.path.join(FILES_MODULE, "umask_check") shutil.rmtree(sub, ignore_errors=True) os.makedirs(sub, exist_ok=True) @@ -1128,7 +1135,8 @@ class TestDaemonMotd: motd_line = "motd file = %s\n" % motd_path if motd_path else "" os.makedirs(self.MOTD_MODULE, exist_ok=True) with open(self.MOTD_CONF, "w") as f: - f.write("port = %d\n%s\n[files]\npath = %s\n" % (port, motd_line, self.MOTD_MODULE)) + f.write("port = %d\n%s\n[files]\npath = %s\nread only = no\n" + % (port, motd_line, self.MOTD_MODULE)) d = DaemonManager() d.start(self.MOTD_CONF, port_override=port) return d, port @@ -1365,6 +1373,7 @@ class TestDaemonConnectionLimits: "\n" "[locked]\n" "path = %s\n" + "read only = no\n" "auth users = alice\n" % (port, AUTH_MODULE)) d = DaemonManager() @@ -1402,6 +1411,7 @@ class TestDaemonConnectionLimits: "\n" "[files]\n" "path = %s\n" + "read only = no\n" "max connections = 2\n" % (port, FILES_MODULE)) d = DaemonManager() diff --git a/tests/test_daemon_conf.c b/tests/test_daemon_conf.c index fde819c..f83465e 100644 --- a/tests/test_daemon_conf.c +++ b/tests/test_daemon_conf.c @@ -1,6 +1,7 @@ #include "test_daemon_conf.h" #include "credentials.h" #include "daemon_conf.h" +#include "log.h" #include "test_utils.h" #include #include @@ -31,6 +32,9 @@ static void test_daemon_conf_create_defaults() { EXPECT_EQ_INT(conf->global.port, DAEMON_CONF_DEFAULT_PORT); EXPECT_NULL(conf->global.motd_file); EXPECT_NULL(conf->global.address); + /* rsync modules are read-only unless they opt in, so the default must be + * true. */ + EXPECT_TRUE(conf->global.read_only_default); EXPECT_EQ_INT(conf->global.max_connections, DAEMON_CONF_DEFAULT_MAX_CONNECTIONS); EXPECT_EQ_INT(conf->global.auth_failure_delay_ms, DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS); EXPECT_EQ_INT(conf->global.max_connections_per_host, @@ -802,6 +806,172 @@ static void test_daemon_conf_dparam_rsync_keys() { daemon_conf_free(conf); } +/* rsync modules default to READ-ONLY; `read only = no` / `write only = yes` + * opt a module into writability, and an explicit module value wins over a + * global default. */ +static void test_daemon_conf_read_only_default_and_opt_in() { + char* path; + char err[256]; + DaemonConf* conf; + + /* A module that never mentions read only/write only is READ-ONLY, matching + * rsync (a migrated rsyncd.conf must not be served writable). */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_TRUE(conf->modules[0].read_only); + EXPECT_FALSE(conf->modules[0].read_only_explicit); + daemon_conf_free(conf); + + /* `read only = no` opts in to writable. */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nread only = no\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_FALSE(conf->modules[0].read_only); + EXPECT_TRUE(conf->modules[0].read_only_explicit); + daemon_conf_free(conf); + + /* `write only = yes` opts in to writable (FastSync is push-only). */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nwrite only = yes\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_FALSE(conf->modules[0].read_only); + EXPECT_TRUE(conf->modules[0].read_only_explicit); + daemon_conf_free(conf); + + /* `write only = no` is rsync's default and does not undo read-only. */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nwrite only = no\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_TRUE(conf->modules[0].read_only); + EXPECT_FALSE(conf->modules[0].read_only_explicit); + daemon_conf_free(conf); + + /* A global `read only = no` is the default for later modules; an explicit + * module `read only`/`write only = yes` still wins. */ + EXPECT_EQ_INT(write_conf("read only = no\n[a]\npath = /a\n" + "[b]\npath = /b\nread only = yes\n" + "[c]\npath = /c\nwrite only = yes\n", + &path), + 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_FALSE(conf->global.read_only_default); + EXPECT_FALSE(conf->modules[0].read_only); + EXPECT_TRUE(conf->modules[1].read_only); + EXPECT_FALSE(conf->modules[2].read_only); + daemon_conf_free(conf); + + /* A global `read only = yes` keeps modules without an explicit value + * read-only. */ + EXPECT_EQ_INT(write_conf("read only = yes\n[a]\npath = /a\n" + "[b]\npath = /b\nread only = no\n" + "[c]\npath = /c\nwrite only = yes\n", + &path), + 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NOT_NULL(conf); + EXPECT_TRUE(conf->global.read_only_default); + EXPECT_TRUE(conf->modules[0].read_only); + EXPECT_FALSE(conf->modules[1].read_only); + EXPECT_FALSE(conf->modules[2].read_only); + daemon_conf_free(conf); + + /* An invalid `write only` value is a clear parse error. */ + EXPECT_EQ_INT(write_conf("[m]\npath = /x\nwrite only = maybe\n", &path), 0); + conf = daemon_conf_load(path, err, sizeof(err)); + free(path); + EXPECT_NULL(conf); + EXPECT_TRUE(strstr(err, "write only") != NULL); +} + +/* Security-relevant rsync keys are accepted for migration but have no FastSync + * effect, so loading must warn loudly (naming the key and module) rather than + * letting an operator believe the restriction is enforced. */ +static void test_daemon_conf_unenforced_security_keys_warned() { + char* path; + char err[256]; + EXPECT_EQ_INT(write_conf("use chroot = yes\n" + "uid = nobody\n" + "[m]\n" + "path = /x\n" + "read only = no\n" + "secrets file = /etc/rsyncd.secrets\n" + "refuse options = delete\n" + "exclude = *.tmp\n" + "max size = 1M\n" + "pre-xfer exec = /bin/true\n" + "incoming chmod = F644\n" + "name converter = sh\n", + &path), + 0); + + set_log_level(LOG_LEVEL_WARNING); + FILE* log_capture = tmpfile(); + EXPECT_NOT_NULL(log_capture); + log_set_file(log_capture); + DaemonConf* conf = daemon_conf_load(path, err, sizeof(err)); + fflush(log_capture); + rewind(log_capture); + log_set_file(NULL); + free(path); + + /* Inert security keys must never fail the load. */ + EXPECT_NOT_NULL(conf); + EXPECT_FALSE(conf->modules[0].read_only); + + bool saw_secrets = false; + bool saw_refuse = false; + bool saw_filter = false; + bool saw_size = false; + bool saw_hook = false; + bool saw_chmod = false; + bool saw_converter = false; + bool saw_global_chroot = false; + bool saw_global_uid = false; + char line[512]; + while (fgets(line, sizeof(line), log_capture) != NULL) { + if (strstr(line, "NOT enforced") == NULL) + continue; + if (strstr(line, "module 'm'") && strstr(line, "secrets file")) + saw_secrets = true; + if (strstr(line, "module 'm'") && strstr(line, "refuse options")) + saw_refuse = true; + if (strstr(line, "module 'm'") && strstr(line, "exclude")) + saw_filter = true; + if (strstr(line, "module 'm'") && strstr(line, "max size")) + saw_size = true; + if (strstr(line, "module 'm'") && strstr(line, "pre-xfer exec")) + saw_hook = true; + if (strstr(line, "module 'm'") && strstr(line, "incoming chmod")) + saw_chmod = true; + if (strstr(line, "module 'm'") && strstr(line, "name converter")) + saw_converter = true; + if (strstr(line, "global key 'use chroot'")) + saw_global_chroot = true; + if (strstr(line, "global key 'uid'")) + saw_global_uid = true; + } + fclose(log_capture); + + EXPECT_TRUE(saw_secrets); + EXPECT_TRUE(saw_refuse); + EXPECT_TRUE(saw_filter); + EXPECT_TRUE(saw_size); + EXPECT_TRUE(saw_hook); + EXPECT_TRUE(saw_chmod); + EXPECT_TRUE(saw_converter); + EXPECT_TRUE(saw_global_chroot); + EXPECT_TRUE(saw_global_uid); + daemon_conf_free(conf); +} + void test_daemon_conf() { test_daemon_conf_create_defaults(); test_daemon_conf_full_parse(); @@ -826,4 +996,6 @@ void test_daemon_conf() { test_daemon_conf_rsync_unknown_keys_rejected(); test_daemon_conf_rsync_read_only_invalid(); test_daemon_conf_dparam_rsync_keys(); + test_daemon_conf_read_only_default_and_opt_in(); + test_daemon_conf_unenforced_security_keys_warned(); } \ No newline at end of file -- 2.54.0 From 47b1b9b915273ba6e66ab79905532422b85e774f Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 23:05:40 +0200 Subject: [PATCH 58/68] fix(xattr): fake-super device round-trip, --devices continue-on-error, harden stat parse --- src/server/receiver.c | 36 ++++++- src/server/receiver_pipeline.c | 8 ++ src/server/receiver_pipeline.h | 5 + src/server/server.c | 8 +- src/shared/file.c | 48 +++++---- src/shared/file.h | 13 ++- src/shared/file_save.c | 33 +++--- src/shared/file_save.h | 12 ++- src/shared/xattr.c | 53 +++++++++- tests/integration/test_temp_dir_absolute.py | 18 ++++ tests/test_file.c | 105 +++++++++++++++++++- tests/test_xattr.c | 18 ++++ 12 files changed, 310 insertions(+), 47 deletions(-) diff --git a/src/server/receiver.c b/src/server/receiver.c index 50f4e0b..765501d 100644 --- a/src/server/receiver.c +++ b/src/server/receiver.c @@ -53,7 +53,13 @@ bool receiver_send_final_success(int fd, const Config* config, const ReceiverOut return send_status(fd, final_status); size_t count = outcomes ? outcomes->count : 0; for (size_t i = 0; i < count; i++) { - Status per_file = outcomes->entries[i] == FILE_SAVE_WRITTEN ? STATUS_NEXT : STATUS_OK; + Status per_file; + if (outcomes->entries[i] == FILE_SAVE_WRITTEN) + per_file = STATUS_NEXT; + else if (outcomes->entries[i] == FILE_SAVE_FAILED) + per_file = STATUS_ERROR; + else + per_file = STATUS_OK; if (!send_status(fd, per_file)) return false; } @@ -719,6 +725,11 @@ typedef struct { ArrayList* would_delete; /* --info=del: actually-removed paths collected during the delete commit. */ ArrayList* deleted_paths; + /* Per-run count of entries that failed to materialize without aborting the + stream (currently ONLY a --devices mknod EPERM/EACCES). A nonzero count + makes the terminal frame carry a non-OK status so the client exits + non-zero, matching rsync's continue-and-exit-partial behavior. */ + size_t failed_entries; } ReceiverSaveContext; static bool receiver_save_file(File* file, void* context_pointer) { @@ -742,6 +753,11 @@ static bool receiver_save_file(File* file, void* context_pointer) { count as matched data in the end-of-transfer report. */ if (result != FILE_SAVE_ERROR && file->matched_bytes > 0) context->stats.matched_data += file->matched_bytes; + /* --devices parity: a device node the receiver could not mknod (EPERM/EACCES) + is counted per-run but does not abort the transfer. The terminal frame + turns a nonzero count into a non-OK status so the client exits non-zero. */ + if (result == FILE_SAVE_FAILED) + context->failed_entries++; /* Protocol 2.28.0: receiver-observed literal bytes and the created-entry breakdown (regular/dir/link/special) for the `--stats` report. */ if (result == FILE_SAVE_WRITTEN) @@ -775,9 +791,25 @@ static void receiver_note_delete_limit(void* context_pointer) { context->delete_limit_reached = true; } +/* Terminal status for a run. A capped --delete limit wins (rsync exit 25); + otherwise any per-entry failure (for example an unprivileged --devices + mknod) makes the terminal frame non-OK so the client exits non-zero. rsync + reports 23 here; mapping the client's exact exit code to 23 is a separate, + pre-existing concern. A clean run keeps STATUS_OK. */ +static Status receiver_final_status(bool delete_limit_reached, size_t failed_entries) { + if (delete_limit_reached) + return STATUS_DELETE_LIMIT; + return failed_entries > 0 ? STATUS_ERROR : STATUS_OK; +} + static bool receiver_send_success_frame(int fd, void* context_pointer) { ReceiverSaveContext* context = context_pointer; - Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK; + if (context->failed_entries > 0) + log_message(LOG_LEVEL_WARNING, + "%zu entr%s failed to materialize; continuing (partial transfer)", + context->failed_entries, context->failed_entries == 1 ? "y" : "ies"); + Status final_status = + receiver_final_status(context->delete_limit_reached, context->failed_entries); if (!receiver_send_stats_frame(fd, context->config, &context->stats, context->would_delete, context->deleted_paths)) return false; diff --git a/src/server/receiver_pipeline.c b/src/server/receiver_pipeline.c index 5d1066e..6feab4e 100644 --- a/src/server/receiver_pipeline.c +++ b/src/server/receiver_pipeline.c @@ -29,6 +29,7 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* context->deferred_manifest = NULL; context->deferred_plans = NULL; context->delete_limit_reached = false; + context->failed_entries = 0; memset(&context->stats, 0, sizeof(context->stats)); context->would_delete = NULL; context->deleted_paths = NULL; @@ -264,6 +265,13 @@ int write_thread(void* pipeline_context) { receiver_stats_note_saved(&context->stats, file, created, created_dirs); mtx_unlock(&context->mutex); } + /* --devices parity: a device node that could not be mknod'ed is counted + per-run but does NOT abort the transfer. */ + if (result == FILE_SAVE_FAILED) { + mtx_lock(&context->mutex); + context->failed_entries++; + mtx_unlock(&context->mutex); + } if (result == FILE_SAVE_ERROR) { file_destroy(file); pipeline_context_receiver_note_bytes_released(context, file_bytes); diff --git a/src/server/receiver_pipeline.h b/src/server/receiver_pipeline.h index 099bd9c..33e9f73 100644 --- a/src/server/receiver_pipeline.h +++ b/src/server/receiver_pipeline.h @@ -64,6 +64,11 @@ typedef struct PipelineContextReceiver { /* --info=del actually-removed path list, collected by the deferred delete commit in server.c and reported in the STATUS_STATS frame. */ struct ArrayList* deleted_paths; + /* Per-run count of entries that failed to materialize without aborting the + stream (currently ONLY a --devices mknod EPERM/EACCES). write_thread + increments it under `mutex`; server.c turns a nonzero count into a non-OK + terminal status so the client exits non-zero. */ + size_t failed_entries; } PipelineContextReceiver; PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver, diff --git a/src/server/server.c b/src/server/server.c index 5978d71..f84a616 100644 --- a/src/server/server.c +++ b/src/server/server.c @@ -1044,7 +1044,13 @@ static void server_run_mt_receiver(ServerSession* state) { dir_metadata_list_apply(&context->dir_times, config->receive_root_directory, config); } if (transfer_ok) { - Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK; + if (context->failed_entries > 0) + log_message(LOG_LEVEL_WARNING, + "%zu entr%s failed to materialize; continuing (partial transfer)", + context->failed_entries, context->failed_entries == 1 ? "y" : "ies"); + Status final_status = context->delete_limit_reached + ? STATUS_DELETE_LIMIT + : (context->failed_entries > 0 ? STATUS_ERROR : STATUS_OK); /* Emit the optional wire-stats record first (protocol 2.25.0), then the success/outcome frame, exactly like the single-threaded receiver. */ if (!receiver_send_stats_frame(state->fd, config, &context->stats, context->would_delete, diff --git a/src/shared/file.c b/src/shared/file.c index 2ce5b20..de06c09 100644 --- a/src/shared/file.c +++ b/src/shared/file.c @@ -1182,7 +1182,8 @@ int file_open_temp_dir(const char* dir_path) { * destination file) and best-effort: a per-attribute or privilege failure is * logged and skipped, never fatal. */ static void restore_extra_fd(int fd, const FileMetadata* metadata, const FileXattrList* xattrs, - bool fake_super, FileAttrPolicy policy) { + bool fake_super, FileAttrPolicy policy, uint32_t fake_super_rdev_major, + uint32_t fake_super_rdev_minor) { xattr_apply_fd(fd, xattrs); if (fake_super && metadata) { /* Record the ownership that WOULD have been applied: when an explicit @@ -1197,7 +1198,8 @@ static void restore_extra_fd(int fd, const FileMetadata* metadata, const FileXat uint32_t store_gid; identity_resolve_storage_ids((int32_t)metadata->uid, (int32_t)metadata->gid, &store_uid, &store_gid); - fake_super_store_fd(fd, store_uid, store_gid, (uint32_t)metadata->mode, 0, 0); + fake_super_store_fd(fd, store_uid, store_gid, (uint32_t)metadata->mode, fake_super_rdev_major, + fake_super_rdev_minor); fake_super_restore_fd(fd, policy); } } @@ -1207,7 +1209,8 @@ file_to_disk_secure_impl(const char* path, const void* data, unsigned long long bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata, FileAttrPolicy policy, bool update, bool no_replace, bool use_fsync, const char* temp_dir, const FileXattrList* xattrs, bool fake_super, - bool keep_partial, unsigned* dirs_created, const char* count_floor) { + bool keep_partial, unsigned* dirs_created, const char* count_floor, + uint32_t fake_super_rdev_major, uint32_t fake_super_rdev_minor) { char* leaf = NULL; int dirfd = file_open_secure_parent_counted(path, &leaf, true, dirs_created, count_floor); if (dirfd < 0) @@ -1314,7 +1317,8 @@ file_to_disk_secure_impl(const char* path, const void* data, unsigned long long } } if (ok) - restore_extra_fd(fd, metadata, xattrs, fake_super, policy); + restore_extra_fd(fd, metadata, xattrs, fake_super, policy, fake_super_rdev_major, + fake_super_rdev_minor); if (ok && use_fsync) ok = fsync(fd) == 0; } @@ -1432,7 +1436,8 @@ file_to_disk_secure_impl(const char* path, const void* data, unsigned long long } } if (ok) - restore_extra_fd(fd, metadata, xattrs, fake_super, policy); + restore_extra_fd(fd, metadata, xattrs, fake_super, policy, fake_super_rdev_major, + fake_super_rdev_minor); if (ok && use_fsync) ok = fsync(fd) == 0; } @@ -1500,7 +1505,8 @@ file_to_disk_secure_impl(const char* path, const void* data, unsigned long long "non-atomic copy into the destination directory"); return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, policy, update, no_replace, use_fsync, NULL, xattrs, fake_super, - keep_partial, dirs_created, count_floor); + keep_partial, dirs_created, count_floor, fake_super_rdev_major, + fake_super_rdev_minor); } return ok; } @@ -1510,7 +1516,7 @@ bool file_to_disk_secure(const char* path, const void* data, unsigned long long FileAttrPolicy policy, const char* temp_dir) { return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, policy, false, false, false, temp_dir, NULL, false, false, NULL, - NULL); + NULL, 0, 0); } bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size, @@ -1519,7 +1525,7 @@ bool file_to_disk_secure_update(const char* path, const void* data, unsigned lon const char* temp_dir) { return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, policy, true, false, false, temp_dir, NULL, false, false, NULL, - NULL); + NULL, 0, 0); } bool file_to_disk_secure_with_fsync(const char* path, const void* data, @@ -1528,7 +1534,7 @@ bool file_to_disk_secure_with_fsync(const char* path, const void* data, FileAttrPolicy policy, bool use_fsync, const char* temp_dir) { return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, policy, false, false, use_fsync, temp_dir, NULL, false, false, - NULL, NULL); + NULL, NULL, 0, 0); } bool file_to_disk_secure_no_replace(const char* path, const void* data, @@ -1537,7 +1543,7 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data, const char* temp_dir) { return file_to_disk_secure_impl(path, data, data_size, false, sparse, preallocate, metadata, policy, false, true, false, temp_dir, NULL, false, false, NULL, - NULL); + NULL, 0, 0); } /* Receiver write-path variant that also applies the per-file xattrs (-X/-A) @@ -1552,19 +1558,19 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long bool fake_super, bool keep_partial, const char* temp_dir) { return file_to_disk_secure_attrs_counted(path, data, data_size, inplace, sparse, preallocate, metadata, policy, update, no_replace, use_fsync, xattrs, - fake_super, keep_partial, temp_dir, NULL, NULL); + fake_super, keep_partial, temp_dir, NULL, NULL, 0, 0); } -bool file_to_disk_secure_attrs_counted(const char* path, const void* data, - unsigned long long data_size, bool inplace, bool sparse, - bool preallocate, const FileMetadata* metadata, - FileAttrPolicy policy, bool update, bool no_replace, - bool use_fsync, const FileXattrList* xattrs, bool fake_super, - bool keep_partial, const char* temp_dir, - unsigned* dirs_created, const char* count_floor) { +bool file_to_disk_secure_attrs_counted( + const char* path, const void* data, unsigned long long data_size, bool inplace, bool sparse, + bool preallocate, const FileMetadata* metadata, FileAttrPolicy policy, bool update, + bool no_replace, bool use_fsync, const FileXattrList* xattrs, bool fake_super, + bool keep_partial, const char* temp_dir, unsigned* dirs_created, const char* count_floor, + uint32_t fake_super_rdev_major, uint32_t fake_super_rdev_minor) { return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata, policy, update, no_replace, use_fsync, temp_dir, xattrs, - fake_super, keep_partial, dirs_created, count_floor); + fake_super, keep_partial, dirs_created, count_floor, + fake_super_rdev_major, fake_super_rdev_minor); } /* Atomic --link-dest install. The destination is replaced (via a temporary @@ -1694,7 +1700,7 @@ static bool file_copy_basis_stream_impl(const char* path, const char* basis_path wrote = false; } if (wrote) - restore_extra_fd(fd, metadata, xattrs, fake_super, policy); + restore_extra_fd(fd, metadata, xattrs, fake_super, policy, 0, 0); if (wrote && use_fsync) wrote = fsync(fd) == 0; if (close(fd) != 0) @@ -1843,7 +1849,7 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa return true; return file_to_disk_secure_attrs_counted( path, data, data_size, false, false, preallocate, metadata, policy, false, false, use_fsync, - xattrs, fake_super, false, temp_dir, dirs_created, count_floor); + xattrs, fake_super, false, temp_dir, dirs_created, count_floor, 0, 0); } if (scratch_dirfd >= 0) diff --git a/src/shared/file.h b/src/shared/file.h index 12654ee..dc02584 100644 --- a/src/shared/file.h +++ b/src/shared/file.h @@ -182,13 +182,12 @@ bool file_copy_basis_stream_attrs(const char* path, const char* basis_path, * confined secure walk had to create that lie strictly below `count_floor` (a * receive-root-relative prefix, or NULL for all). Used to reproduce rsync's * `Number of created files` directory count on a fresh destination. */ -bool file_to_disk_secure_attrs_counted(const char* path, const void* data, - unsigned long long data_size, bool inplace, bool sparse, - bool preallocate, const FileMetadata* metadata, - FileAttrPolicy policy, bool update, bool no_replace, - bool use_fsync, const FileXattrList* xattrs, bool fake_super, - bool keep_partial, const char* temp_dir, - unsigned* dirs_created, const char* count_floor); +bool file_to_disk_secure_attrs_counted( + const char* path, const void* data, unsigned long long data_size, bool inplace, bool sparse, + bool preallocate, const FileMetadata* metadata, FileAttrPolicy policy, bool update, + bool no_replace, bool use_fsync, const FileXattrList* xattrs, bool fake_super, + bool keep_partial, const char* temp_dir, unsigned* dirs_created, const char* count_floor, + uint32_t fake_super_rdev_major, uint32_t fake_super_rdev_minor); bool file_to_disk_secure_link_attrs_counted(const char* path, const char* basis_path, const void* data, unsigned long long data_size, bool preallocate, const FileMetadata* metadata, diff --git a/src/shared/file_save.c b/src/shared/file_save.c index 73f4748..d0d4672 100644 --- a/src/shared/file_save.c +++ b/src/shared/file_save.c @@ -401,12 +401,13 @@ bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) { * * Privilege gating: making a real device node requires CAP_MKNOD (root); making * a FIFO works unprivileged (mkfifo). A device node whose mknodat() fails with - * EPERM/EACCES is a genuine transfer error (rsync parity: rsync reports the - * mknod failure and the run exits partial, code 23). Only the unprivileged - * FIFO/socket (--specials) path keeps the best-effort skip, because those are - * normally creatable without privilege and a failure there is environmental. - * CI runs non-root, so device creation is expected to fail there; only a FIFO - * is honestly assertable unprivileged. + * EPERM/EACCES is a PER-ENTRY failure (rsync parity: rsync reports the mknod + * failure, still transfers the rest, and exits partial, code 23), reported as + * FILE_SAVE_FAILED so the receiver counts it and continues. Only the + * unprivileged FIFO/socket (--specials) path keeps the best-effort skip, + * because those are normally creatable without privilege and a failure there is + * environmental. CI runs non-root, so device creation is expected to fail + * there; only a FIFO is honestly assertable unprivileged. * * Confinement: the parent directory is opened fd-relative below the receive * root (file_open_secure_parent: O_NOFOLLOW, no "..", root-checked) and the @@ -548,10 +549,10 @@ static FileSaveResult file_save_special_to_disk(const char* root_directory, cons const char* shown_path = escaped_path ? escaped_path : ""; if (is_char || is_blk) { /* rsync parity: a device node that cannot be created (no CAP_MKNOD, or - * super-user activities not permitted) is a genuine transfer error. - * rsync reports `mknod ".../node" failed: ...` and the run exits - * partial (23); FastSync surfaces it through the outcome aggregation - * instead of silently skipping the entry. FIFO/socket creation + * super-user activities not permitted) is a per-entry failure. rsync + * logs `mknod ".../node" failed: ...`, still transfers the remaining + * files, and exits partial (23); FastSync logs it, counts it, and + * continues rather than aborting the stream. FIFO/socket creation * (--specials) keeps the best-effort skip path below. */ log_message(LOG_LEVEL_ERROR, "cannot create %s %s: %s\n" @@ -561,7 +562,7 @@ static FileSaveResult file_save_special_to_disk(const char* root_directory, cons close(parent_fd); free(leaf); free(destination); - return FILE_SAVE_ERROR; + return FILE_SAVE_FAILED; } /* Missing CAP_MKNOD / parent write permission for a FIFO/socket: the environment cannot create the node, so skip instead of failing the @@ -936,6 +937,14 @@ static bool file_save_try_special_dispatch(const FileSavePlan* plan, bool* creat /* Device/special node (--devices/--specials): recreate the node instead of writing content (privilege-gated, confined, rdev-validated). */ if (file->is_special) { + /* Under --fake-super rsync never mknod()s a device: it writes a regular + empty file and records the real rdev in user.rsync.%stat. Fall through to + the ordinary writer so the device round-trips (its S_IFMT mode bits and + rdev are parked in the record). Without --fake-super the node is + recreated (or, when privilege is refused, handled per-entry). */ + mode_t special_mode = file->metadata ? file->metadata->mode : 0; + if (config && config->fake_super && (S_ISCHR(special_mode) || S_ISBLK(special_mode))) + return false; *out = file_save_special_to_disk(plan->root_directory, file, config, created); return true; } @@ -1110,7 +1119,7 @@ static bool file_save_install_data(FileSavePlan* plan, const FileMetadata* metad config && config->preallocate, metadata, plan->policy, config && config->update, config && config->ignore_existing, config && config->use_fsync, file->xattrs, config ? config->fake_super : false, config ? config->partial : false, plan->confined_temp, - created_dirs, count_floor); + created_dirs, count_floor, (uint32_t)file->rdev_major, (uint32_t)file->rdev_minor); } free(count_floor); return ok; diff --git a/src/shared/file_save.h b/src/shared/file_save.h index d995e5f..5483c27 100644 --- a/src/shared/file_save.h +++ b/src/shared/file_save.h @@ -12,8 +12,16 @@ /* Outcome of a single file_save_to_disk operation. The receiver needs to distinguish "written" from "skipped" so --remove-source-files can be told - which sources were actually stored. */ -typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2 } FileSaveResult; + which sources were actually stored. FILE_SAVE_FAILED is a per-entry failure + (for example a device node that mknodat() refused with EPERM/EACCES): it is + logged and counted by the receiver but does NOT abort the transfer, matching + rsync's continue-and-exit-partial behavior. */ +typedef enum { + FILE_SAVE_ERROR = 0, + FILE_SAVE_WRITTEN = 1, + FILE_SAVE_SKIPPED = 2, + FILE_SAVE_FAILED = 3 +} FileSaveResult; bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode); diff --git a/src/shared/xattr.c b/src/shared/xattr.c index 56300d7..91f82c4 100644 --- a/src/shared/xattr.c +++ b/src/shared/xattr.c @@ -7,6 +7,7 @@ #include "utils.h" #include "file_types.h" #include +#include #include #include #include @@ -386,6 +387,56 @@ void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, uint } } +/* Parse rsync's `user.rsync.%stat` grammar strictly: + * " , :" + * Every field is parsed with strtoul() so an out-of-range value is a clean + * rejection rather than the undefined behavior sscanf("%u") exhibited, each + * field is range-checked against the same bounds the wire validator uses, and + * the whole record must be consumed (only trailing whitespace is tolerated) so + * trailing garbage is refused. Returns false on any malformed input. */ +static bool fake_super_parse_stat(const char* record, unsigned* mode_out, unsigned* rdev_major_out, + unsigned* rdev_minor_out, unsigned* uid_out, unsigned* gid_out) { + if (!record) + return false; + char* end = NULL; + const char* p = record; + errno = 0; + unsigned long mode = strtoul(p, &end, 8); + if (errno != 0 || end == p || mode > (unsigned long)UINT_MAX || *end != ' ') + return false; + p = end + 1; + errno = 0; + unsigned long rdev_major = strtoul(p, &end, 10); + if (errno != 0 || end == p || rdev_major > 0xffffUL || *end != ',') + return false; + p = end + 1; + errno = 0; + unsigned long rdev_minor = strtoul(p, &end, 10); + if (errno != 0 || end == p || rdev_minor > 0x00ffffffUL || *end != ' ') + return false; + p = end + 1; + errno = 0; + unsigned long uid = strtoul(p, &end, 10); + if (errno != 0 || end == p || uid > (unsigned long)UINT_MAX || *end != ':') + return false; + p = end + 1; + errno = 0; + unsigned long gid = strtoul(p, &end, 10); + if (errno != 0 || end == p || gid > (unsigned long)UINT_MAX) + return false; + p = end; + while (*p == ' ' || *p == '\t' || *p == '\n' || *p == '\r') + p++; + if (*p != '\0') + return false; + *mode_out = (unsigned)mode; + *rdev_major_out = (unsigned)rdev_major; + *rdev_minor_out = (unsigned)rdev_minor; + *uid_out = (unsigned)uid; + *gid_out = (unsigned)gid; + return true; +} + /* --fake-super replay: read the freshly-stored record and re-apply its * permission bits fd-relative. The recorded uid/gid are retained for a later * privileged restore but are NEVER chowned here: --fake-super only RECORDS @@ -403,7 +454,7 @@ bool fake_super_restore_fd(int fd, FileAttrPolicy policy) { return false; /* absent or filesystem without xattrs: silent no-op */ record[len] = '\0'; unsigned ul_mode, rdev_major, rdev_minor, ul_uid, ul_gid; - if (sscanf(record, "%o %u,%u %u:%u", &ul_mode, &rdev_major, &rdev_minor, &ul_uid, &ul_gid) != 5) + if (!fake_super_parse_stat(record, &ul_mode, &rdev_major, &rdev_minor, &ul_uid, &ul_gid)) return false; /* malformed record: skip, never fatal */ /* --fake-super NEVER performs a real chown: that would defeat the whole point diff --git a/tests/integration/test_temp_dir_absolute.py b/tests/integration/test_temp_dir_absolute.py index ff7f761..abc591e 100644 --- a/tests/integration/test_temp_dir_absolute.py +++ b/tests/integration/test_temp_dir_absolute.py @@ -64,6 +64,13 @@ def test_read_batch_absolute_temp_dir_inside_root_accepted(tmp_path): clean_dir(dest) scratch = os.path.join(dest, "scratch") os.makedirs(scratch) + # Stamp the scratch dir with an old mtime so the test can prove the receiver + # really created (and then removed) its temp file there: the directory mtime + # changes when an entry is created/removed, so a silently-ignored --temp-dir + # would leave the stamp untouched. An empty scratch dir alone does not + # distinguish "used and cleaned up" from "never used". + stale_mtime = 946684800 # 2000-01-01 + os.utime(scratch, (stale_mtime, stale_mtime)) batch = _make_batch(str(tmp_path), source) result = _run(["--read-batch", batch, dest, "--temp-dir", scratch]) @@ -73,6 +80,9 @@ def test_read_batch_absolute_temp_dir_inside_root_accepted(tmp_path): for rel, data in FILES.items(): assert _read(os.path.join(received, rel)) == data, f"content mismatch for {rel}" assert os.listdir(scratch) == [], "scratch dir was not left clean" + assert os.stat(scratch).st_mtime != stale_mtime, ( + "--temp-dir scratch dir was never written to (temp file not created there)" + ) @pytest.mark.ci @@ -98,6 +108,11 @@ def test_tcp_absolute_temp_dir_inside_root_accepted(shared_server): clean_dir(dest) scratch = os.path.join(dest, "scratch") os.makedirs(scratch) + # See test_read_batch_absolute_temp_dir_inside_root_accepted: the stale + # mtime makes actual scratch usage observable (the temp file creation and + # removal bump the directory mtime). + stale_mtime = 946684800 # 2000-01-01 + os.utime(scratch, (stale_mtime, stale_mtime)) result, _ = run_client(source, dest, flags=["--temp-dir", scratch], port=shared_server.port) assert result.returncode == 0, (result.stdout, result.stderr)[:300] @@ -105,6 +120,9 @@ def test_tcp_absolute_temp_dir_inside_root_accepted(shared_server): for rel, data in FILES.items(): assert _read(os.path.join(received, rel)) == data, f"content mismatch for {rel}" assert os.listdir(scratch) == [], "scratch dir was not left clean" + assert os.stat(scratch).st_mtime != stale_mtime, ( + "--temp-dir scratch dir was never written to (temp file not created there)" + ) def test_tcp_absolute_temp_dir_outside_root_rejected(shared_server): diff --git a/tests/test_file.c b/tests/test_file.c index 786ab4c..744ec6d 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -9,7 +9,9 @@ #include "charset.h" #include "utils.h" #include "protocol.h" +#include "xattr.h" #include "test_utils.h" +#include #include #include #include @@ -17,6 +19,7 @@ #include #include #include +#include #include #include @@ -348,7 +351,8 @@ static void test_file_save_to_disk_temp_dir_confined() { mkdir(root, 0755); mkdir("test_temp_confine_tmp/scratch", 0755); mkdir("test_temp_confine_tmp/abs_scratch", 0755); - EXPECT_NOT_NULL(realpath("test_temp_confine_tmp/abs_scratch", inside_abs)); + if (!realpath("test_temp_confine_tmp/abs_scratch", inside_abs)) + EXPECT_FAIL("realpath(abs_scratch) failed; inside_abs would be uninitialized"); mkdir(outside, 0755); File* f = file_create("file.txt"); @@ -1327,6 +1331,103 @@ static void test_special_socket_recreated() { rmdir(root); } +/* --fake-super device round-trip (rsync parity): a char/block device must be + * materialized as a REGULAR empty file whose user.rsync.%stat records the real + * rdev -- never as an mknod'ed node -- even on a privileged receiver. This is + * the non-privileged unit counterpart to the setpriv integration test (which + * the PR gate excludes). */ +static void test_fake_super_device_writes_regular_file_with_rdev() { + const char* root = "test_fake_super_dev_tmp"; + const char* node = "test_fake_super_dev_tmp/cdev"; + unlink(node); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0700), 0); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->fake_super = true; + cfg->preserve_devices = true; + cfg->preserve_perms = true; + cfg->use_metadata = true; + cfg->use_xattrs = true; + + FileMetadata meta; + memset(&meta, 0, sizeof(meta)); + meta.mode = S_IFCHR | 0644; + + File* f = file_create("cdev"); + EXPECT_NOT_NULL(f); + f->is_special = true; + f->rdev_major = 1; + f->rdev_minor = 3; + f->metadata = &meta; + + EXPECT_EQ_INT(file_save_to_disk_full(root, f, cfg), FILE_SAVE_WRITTEN); + + struct stat st; + EXPECT_EQ_INT(lstat(node, &st), 0); + EXPECT_TRUE(S_ISREG(st.st_mode)); /* never a real device node */ + EXPECT_EQ_INT((int)st.st_size, 0); + + char value[128] = {0}; + ssize_t got = getxattr(node, FAKESUPER_XATTR, value, sizeof(value) - 1); + EXPECT_TRUE(got > 0); + EXPECT_EQ_STR(value, "20644 1,3 0:0"); /* the REAL rdev, not 0,0 */ + + f->metadata = NULL; + file_destroy(f); + config_delete(cfg); + unlink(node); + rmdir(root); +} + +/* A char/block device that mknodat() refuses (EPERM/EACCES on an unprivileged + * receiver) must be a PER-ENTRY failure -- FILE_SAVE_FAILED, which the receiver + * counts and continues past -- never the fatal FILE_SAVE_ERROR that aborts the + * stream. The unit suite normally runs as root, so drop the effective uid to + * make the kernel refusal deterministic. */ +static void test_device_mknod_failure_is_per_entry() { + const char* root = "test_device_eperm_tmp"; + const char* node = "test_device_eperm_tmp/cdev"; + unlink(node); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0777), 0); + + Config* cfg = config_create(); + EXPECT_NOT_NULL(cfg); + cfg->preserve_devices = true; + cfg->use_metadata = true; + + FileMetadata meta; + memset(&meta, 0, sizeof(meta)); + meta.mode = S_IFCHR | 0644; + + File* f = file_create("cdev"); + EXPECT_NOT_NULL(f); + f->is_special = true; + f->rdev_major = 1; + f->rdev_minor = 3; + f->metadata = &meta; + + uid_t saved = geteuid(); + bool dropped = false; + if (saved == 0 && seteuid(65534) == 0) + dropped = true; + FileSaveResult result = file_save_to_disk_full(root, f, cfg); + if (dropped) + EXPECT_EQ_INT(seteuid(saved), 0); + + EXPECT_EQ_INT(result, FILE_SAVE_FAILED); + /* Nothing was created: no device node and no regular-file fallback. */ + struct stat st; + EXPECT_EQ_INT(lstat(node, &st), -1); + + f->metadata = NULL; + file_destroy(f); + config_delete(cfg); + rmdir(root); +} + static void test_inplace_overwrite_truncates_shorter_payload() { const char* root = "test_inplace_trunc_tmp"; const char* path = "test_inplace_trunc_tmp/big.txt"; @@ -2408,6 +2509,8 @@ void test_file() { test_new_file_mode_honors_source_and_umask(); test_special_fifo_mode_honors_source_and_umask(); test_special_socket_recreated(); + test_fake_super_device_writes_regular_file_with_rdev(); + test_device_mknod_failure_is_per_entry(); test_inplace_overwrite_truncates_shorter_payload(); test_inplace_refuses_fifo_destination(); test_inplace_refuses_device_destination(); diff --git a/tests/test_xattr.c b/tests/test_xattr.c index e659ed7..9a0ad1a 100644 --- a/tests/test_xattr.c +++ b/tests/test_xattr.c @@ -450,6 +450,24 @@ static void test_fake_super_rsync_format() { EXPECT_EQ_INT((int)st.st_uid, (int)before.st_uid); EXPECT_EQ_INT((int)st.st_gid, (int)before.st_gid); + /* Hardened parser: an out-of-range field (previously UB via sscanf("%u")), + a missing field, or trailing garbage is rejected cleanly instead of being + silently accepted. */ + const char* malformed[] = { + "20644 65536,3 111:222", /* major > 0xffff */ + "20644 1,16777216 111:222", /* minor > 0xffffff */ + "20644 1,3 111:222 trailing", /* trailing garbage */ + "20644 1,3 111", /* missing gid */ + "20644 1,3 4294967296:222", /* uid > UINT_MAX */ + "20644 1,3 111:4294967296", /* gid > UINT_MAX */ + "99999999999999999999 1,3 0:0", /* mode overflow */ + "", /* empty record */ + }; + for (size_t i = 0; i < sizeof(malformed) / sizeof(malformed[0]); i++) { + EXPECT_EQ_INT((int)fsetxattr(fd, FAKESUPER_XATTR, malformed[i], strlen(malformed[i]), 0), 0); + EXPECT_FALSE(fake_super_restore_fd(fd, policy)); + } + close(fd); unlink(path); } -- 2.54.0 From ee265d78ba71fa9e6f7358204fde27c73ca5dccf Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 23:19:55 +0200 Subject: [PATCH 59/68] fix(delete-before): replay pre-scan list in --threads path --- src/client/client_send.c | 55 +++++++++++++++- src/shared/multiprocessing.c | 3 + src/shared/multiprocessing.h | 7 ++ .../integration/test_delete_timing_parity.py | 64 ++++++++++++------- 4 files changed, 103 insertions(+), 26 deletions(-) diff --git a/src/client/client_send.c b/src/client/client_send.c index 8c6212e..c8dd8bd 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -1148,6 +1148,43 @@ send_fail: static int scan_directory_multithreaded(void* pipeline_context) { PipelineContextSender* context = (PipelineContextSender*)pipeline_context; protocol_session_bind(&context->allocation_session); + if (context->prescan_chunks != NULL) { + /* --delete-before replays the pre-scan that built the early keep-set as the + data pass (rsync builds one file list). Feed the retained chunks straight + into the pipeline instead of re-reading the source, so a file created + after the pre-scan is neither transferred nor kept. The chunk also + carries the directory times captured by that scan (there is no later + scan), so no scanner is created here. */ + bool failed = false; + for (int i = 0; i < context->prescan_chunks->size; i++) { + Chunk* chunk = (Chunk*)context->prescan_chunks->items[i]; + /* Move ownership out of the retained list so a cleanup here never + double-frees a chunk the queue now owns. */ + context->prescan_chunks->items[i] = NULL; + if (chunk == NULL) + continue; + if (!queue_enqueue_multithreaded_cancel( + context->queue_scanner, chunk, &context->mutex_scanner, + &context->condition_not_empty_scanner, &context->condition_not_full_scanner, + &context->cancelled)) { + chunk_destroy(chunk); + failed = true; + break; + } + } + mtx_lock(&context->mutex_scanner); + context->scanner_done = true; + cnd_broadcast(&context->condition_not_empty_scanner); + cnd_broadcast(&context->condition_not_full_scanner); + mtx_unlock(&context->mutex_scanner); + if (failed) { + pipeline_cancel(context); + protocol_session_unbind(); + return thrd_error; + } + protocol_session_unbind(); + return thrd_success; + } PreparedScanner prepared; /* -j/--threads=N sizes the parallel scanner's worker pool; 0 (bare -j) lets * the scanner apply its built-in default. */ @@ -2032,13 +2069,27 @@ int send_files_multithreaded(Config* config) { if (prepared_ok) prepared.options.plan_dirs = context->plan_dirs; } else { + /* --delete-before: retain the pre-scan chunks as the pipeline's data + pass (rsync's single file list) so a source file created after the + scan is not transferred. No later scan runs, so this pass must also + capture the deferred directory times and the --stats directory + count. */ context->manifest = array_list_create(free); - prepared_ok = prepared_ok && context->manifest != NULL; + context->prescan_chunks = array_list_create(chunk_destroy); + prepared_ok = prepared_ok && context->manifest != NULL && context->prescan_chunks != NULL; + if (prepared_ok) { + prepared.options.dir_entries = context->dir_entries; + prepared.options.dir_entries_mutex = &context->dir_entries_mutex; + prepared.options.dir_count = config->stats ? &context->dir_count : NULL; + if (!append_implied_dir_times(config, context->dir_entries)) + prepared_ok = false; + } } bool prebuilt = prepared_ok && scan_paths_only(config, &prepared.options, context->manifest, context->delete_plans, - &context->scan_had_io_error, &pre_scan_non_dir, NULL, false); + &context->scan_had_io_error, &pre_scan_non_dir, context->prescan_chunks, + context->prescan_chunks != NULL); prepared_scanner_destroy(&prepared); if (per_dir && prebuilt) { const char* walk_root = delete_plan_walk_root(config, context->synced_dirs); diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c index 60f87ee..3e54e87 100644 --- a/src/shared/multiprocessing.c +++ b/src/shared/multiprocessing.c @@ -37,6 +37,7 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que context->scan_had_io_error = false; context->remove_source_files = NULL; context->early_delete = false; + context->prescan_chunks = NULL; context->delete_plans = NULL; context->delete_suppressed = false; context->scan_stopped_early = false; @@ -194,6 +195,8 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) { if (context->manifest) { array_list_delete(context->manifest); } + if (context->prescan_chunks) + array_list_delete(context->prescan_chunks); if (context->delete_plans) delete_plan_sender_destroy(context->delete_plans); if (context->excluded_paths) diff --git a/src/shared/multiprocessing.h b/src/shared/multiprocessing.h index e53c542..92ae4cc 100644 --- a/src/shared/multiprocessing.h +++ b/src/shared/multiprocessing.h @@ -78,6 +78,13 @@ typedef struct { path-only pre-scan on the calling thread and the pipeline scanner must not append to it. Set once before the worker threads start. */ bool early_delete; + /* --delete-before: the path-only pre-scan that built the early keep-set, + retained as the pipeline's file list (owning Chunk*; consumed and NULLed by + the scanner thread) so the data pass replays rsync's single file list + instead of re-reading the source. NULL in every other mode, where the + scanner thread scans normally. Set once before the worker threads start + and freed with the context. */ + ArrayList* prescan_chunks; /* Non-NULL for --delete-during/--delete-delay: the per-directory plan set prebuilt by the path-only pre-scan on the calling thread. The sender thread transmits the root plan before any data and the remaining plans diff --git a/tests/integration/test_delete_timing_parity.py b/tests/integration/test_delete_timing_parity.py index 6757f76..a638c03 100644 --- a/tests/integration/test_delete_timing_parity.py +++ b/tests/integration/test_delete_timing_parity.py @@ -559,56 +559,72 @@ class TestDeleteDelayMaxDeleteRefilledDir: class TestDeleteBeforeLateFileParity: """rsync builds its file list once, so a source file created after that scan - is NOT transferred and its destination extra is deleted. FastSync's - single-threaded --delete-before used to re-scan the source in its data pass - and would transfer the late file (a safe superset); it now replays the - pre-scan file list instead, matching rsync. + is NOT transferred and its destination extra is deleted. FastSync used to + re-scan the source in its data pass (single-threaded) or pipeline a fresh + re-scan against the pre-scan keep-set (``--threads``) and would transfer the + late file (a safe superset); both paths now replay the pre-scan file list + instead, matching rsync. The late file is injected through the config-ack barrier: the first client bytes after the config ack are the pre-scan keep-set manifest, so the hook runs causally after the source scan and before the receiver's delete ack - releases the client into its data pass -- deterministic, no timing guess. + releases the client into its data pass. + + For ``--threads`` the pipeline scanner runs concurrently with the sender, so + the injection must land while that re-scan is still in flight to be observed + by it. The source is therefore a tree of ``_N_DIRS`` directories: the + injection writes the late file into EVERY directory, so it is enough that + any one directory is still unscanned when the hook fires. The tree is sized + so the hook (a localhost round trip) lands long before a full scan finishes; + a re-scanning pipeline then transfers the late files for the directories it + has not yet reached, which the tree comparison catches. """ + _N_DIRS = 2000 + @requires_rsync - def test_late_source_file_not_transferred_and_extra_deleted(self): - source = os.path.join(TEST_DATA_DIR, "dblate_src") - dest = os.path.join(TEST_DATA_DIR, "dblate_dst") - rsync_dst = os.path.join(TEST_DATA_DIR, "dblate_rsync_dst") + @pytest.mark.parametrize("mt", [False, True]) + def test_late_source_file_not_transferred_and_extra_deleted(self, mt): + tag = f"dblate_mt{int(mt)}" + source = os.path.join(TEST_DATA_DIR, f"{tag}_src") + dest = os.path.join(TEST_DATA_DIR, f"{tag}_dst") + rsync_dst = os.path.join(TEST_DATA_DIR, f"{tag}_rsync_dst") clean_dir(source) clean_dir(dest) clean_dir(rsync_dst) - _write(os.path.join(source, "d", "keep.txt"), b"kept payload\n") + for i in range(self._N_DIRS): + _write(os.path.join(source, f"dir{i:05d}", "keep.txt"), b"kept payload\n") # Both destinations carry the would-be late file as an extra. for root in (dest, rsync_dst): - _write(os.path.join(get_dest_received_dir(root, source), "d", "late.txt"), - b"stale extra\n") + received = get_dest_received_dir(root, source) + for i in range(self._N_DIRS): + _write(os.path.join(received, f"dir{i:05d}", "late.txt"), b"stale extra\n") - # rsync reference: the same source with no late file; the extra is removed - # and nothing is transferred for the (never-scanned) late path. + # rsync reference: the same source with no late file; the extras are + # removed and nothing is transferred for the (never-scanned) late paths. rsync_result = _rsync(["-a", "--delete-before", source + "/", rsync_dst + "/"]) assert rsync_result.returncode == 0, rsync_result.stderr rsync_tree = _tree(rsync_dst) - assert "d/late.txt" not in rsync_tree + assert "dir00000/late.txt" not in rsync_tree received = get_dest_received_dir(dest, source) - late_source = os.path.join(source, "d", "late.txt") def hook(): - # Runs after the pre-scan and before the data pass begins. - _write(late_source, b"created after the scan\n") + # Runs after the pre-scan and before the receiver's delete ack. + for i in range(self._N_DIRS): + _write(os.path.join(source, f"dir{i:05d}", "late.txt"), + b"created after the scan\n") with ServerManager() as server: server.start(extra_args=["--allow-delete"]) proxy = _SlicingProxy(server.port, hook=hook, hook_after_config_ack=True) - result, _ = run_client(source, dest, flags=["--delete-before"], port=proxy.port) + flags = ["--delete-before"] + (["--threads=4"] if mt else []) + result, _ = run_client(source, dest, flags=flags, port=proxy.port) proxy.finish() assert result.returncode == 0, (result.stderr or result.stdout)[:400] assert proxy.hook_called.is_set(), "late-file hook never fired" - assert os.path.exists(late_source), "the source late file unexpectedly vanished" - assert not os.path.exists(os.path.join(received, "d", "late.txt")), ( - "late source file was transferred: the single-threaded data pass re-scanned" - ) assert _tree(received) == rsync_tree, ( - f"fastsync tree {_tree(received)} != rsync tree {rsync_tree}" + f"late source {'multithreaded' if mt else 'single-threaded'} data pass re-scanned: " + f"{sum(1 for p in _tree(received) if p.endswith('late.txt'))} late files were " + "transferred" ) -- 2.54.0 From 4f945a8e39c10d04ea44175fdcf7108556c9a74c Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 23:27:54 +0200 Subject: [PATCH 60/68] docs: reconcile RSYNC_COMPAT for parity-next review follow-ups --- RSYNC_COMPAT.md | 14 +++++++------- 1 file changed, 7 insertions(+), 7 deletions(-) diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 986bda9..0689b9e 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -193,7 +193,7 @@ Every one of those has an entry below with its remaining caveats. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| | `--delete` | Delete extraneous files from dest | ✅ Parity | `use_delete` config field. Deletion is always derived from the keep-set the sender actually transmitted (the per-directory `STATUS_DELETE_PLAN` set by default, or the whole-tree manifest for the late timings — never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. **Lockstep track 6 (protocol 2.28.0): plain `--delete` with no explicit timing flag now defaults to `--delete-during`**, exactly like rsync's `--del` (the client normalizes it to the existing `delete_during` wire bool; no new wire field). This frees destination space progressively during the transfer and avoids the whole-old+new-tree peak that could `ENOSPC` a tight destination. The old late whole-tree commit is opt-in via `--delete-after` or the FastSync-only long spelling `--delete-commit`. **Abort/ordering parity (parity-2.29):** the complete per-directory plan set is transmitted before the first data frame, so a mid-transfer abort has already applied every planned removal exactly like rsync's generator (which runs ahead of its throttled sender); `-d/--dirs` uses the same per-directory plans (the generator records only the directories whose direct children it enumerated, so an untraversed subdirectory's mirror is shielded); and the sorted depth-first traversal makes the removal order — and therefore the survivor set under a partial `--max-delete` — match rsync exactly (`test_delete_boundary_parity.py`, `test_parity_order.py`). By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). Deletion is scoped to the **synchronized directories** sent on the wire (protocol 2.23.0), so a `--files-from` subset no longer deletes untransmitted paths outside the listed directory subtrees. The walk is bounded: a client `--max-delete=NUM` (or the 100000-entry server bound) makes it **partial** — entries up to the bound are removed, the rest are skipped, and the client exits **25** (`RERR_PARTIAL`), matching rsync, rather than failing the transfer. Extraneous destination symlinks are unlinked by name (never followed); a directory still holding a kept/protected entry is left behind rather than failing | -| `--delete-before` | Delete before transfer | ✅ Parity | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. **Phase-0 divergence closed (no-wire):** the single-threaded data pass now replays the exact file list the pre-scan built for the keep-set instead of re-reading the source, so a source file created after that scan is NOT transferred and its destination extra is deleted, exactly like rsync's single file list (and like the `--threads` path). The pre-scan captures the deferred directory times and the `--stats` directory count because no later scan runs (`test_delete_timing_parity.py::TestDeleteBeforeLateFileParity`, differential vs rsync 3.4.1) | +| `--delete-before` | Delete before transfer | ✅ Parity | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. **Phase-0 divergence closed (no-wire):** both data passes now replay the exact file list the pre-scan built for the keep-set instead of re-reading the source — the single-threaded send loop and the `--threads` pipeline (whose scanner thread feeds the retained pre-scan chunks into the pipeline rather than re-scanning) — so a source file created after that scan is NOT transferred and its destination extra is deleted, exactly like rsync's single file list. The pre-scan captures the deferred directory times and the `--stats` directory count because no later scan runs (`test_delete_timing_parity.py::TestDeleteBeforeLateFileParity`, parametrized single-threaded vs `--threads=4`, differential vs rsync 3.4.1) | | `--del`, `--delete-during` | Delete during transfer | ✅ Parity | Both spellings accepted; imply `--delete`, and since lockstep track 6 this is also the default timing of a plain `--delete`. **Protocol 2.24.0 implements per-directory delete plans:** as the sender reaches each source directory it streams a `STATUS_DELETE_PLAN` for that directory and the receiver removes that directory's extras (verified with a byte-slicing proxy). The one-shot per-run config block (protected prefixes, size-pruned mirrors, `--delete-missing-args` exact paths) rides a dedicated config-only carrier frame with an `apply=false` flag, so it reaches the receiver even when the scope allows no directory plan at all (a `--files-from` list of bare files synchronizes no directory). **Abort/ordering parity (parity-2.29):** the complete plan set is transmitted before the first data frame, so on a mid-transfer abort every planned extra has already been removed exactly like rsync's generator (which runs ahead of its throttled sender); `-d/--dirs` no longer falls back to the end-of-transfer commit but records only the directories whose direct children it enumerated; and the sorted depth-first traversal makes the removal order — and the partial-`--max-delete` survivor set — identical to rsync (`test_delete_boundary_parity.py`, `test_parity_order.py`). `-R` plans are scoped to the transferred prefix subtree | | `--delete-delay` | Find deletions during, delete after | ✅ Parity | Implies `--delete`. **Protocol 2.24.0 implements rsync's delete-delay timing:** the sender records each directory's delete plan while scanning and the receiver commits those removals only after the whole transfer succeeds (per plan), so an extra created in the destination after its directory's plan survives while `--delete-after` re-scans and removes it, and a failed transfer removes nothing. The **reported** deleted count advances only on an actual removal. **Fixed (no-wire):** the `--max-delete` budget is now charged on ACTUAL removals (an unlink/rmdir that succeeded), not at plan/snapshot time, and a queued directory is re-scanned at commit and removed recursively (content created after the plan included), matching rsync: a snapshotted entry that fails or is skipped consumes no budget, so a later extra rsync would delete is still deleted. The deferred snapshot list keeps an independent hard cap (`DELETE_PLAN_SERVER_LIMIT`) so it cannot grow without bound now that the budget is no longer charged while scanning. A `--max-delete=2` partial delete reports exactly 2 and exits 25 in both tools, and the refilled-directory differential (late content removed, directory removed, budget shared) now matches rsync 3.4.1 on both sides (`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`). Unit tests cover recursive removal, actual-removal charging, and the bounded deferred list. **Ordering parity (parity-2.29):** the sorted depth-first traversal plus the up-front plan set make the order in which extras are removed — and therefore the survivor set under a partial `--max-delete` — match rsync exactly (differential `test_parity_order.py::test_delete_delay_deletion_order_matches_rsync` and `::test_partial_max_delete_survivor_order_matches_rsync`) | | `--delete-after` | Delete after transfer | ✅ Parity | Implies `--delete`. Selects the late whole-tree commit: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing. Since lockstep track 6 a plain `--delete` defaults to delete-during (rsync's `--del`); `--delete-after` — or the FastSync-only `--delete-commit` spelling, which selects the identical timing — is the explicit way to keep the old commit-style behavior | @@ -342,7 +342,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | `-X`, `--xattrs` | Preserve extended attributes | ❌ Divergent | Deliberately restricted to unprivileged `user.*` extended attributes plus the two POSIX ACL xattrs; `security.*` (SELinux, capabilities, ...) and `trusted.*` are **never** captured or applied — a client can never force a privileged attribute onto the destination, and the receiver independently re-validates every incoming name against the whitelist. This is a security-policy divergence from rsync, which can preserve the privileged namespaces with the needed privilege; implementing them would defeat FastSync's privilege-escalation guard. `user.*` capture/apply matches rsync in a differential test. Payloads are bounded on both ends. Incompatible with `-s`. **Also divergent: symlink xattrs/ACLs are not captured or applied** — `-X`/`-A` with `-l` carries only the link's owner/times/mode, not its xattrs (the capture uses path-following `listxattr`/`getxattr`, so the link's own xattrs are never read, and the receiver's symlink write path applies no xattr block). Closing this needs a dedicated symlink-xattr wire block and a `PROTOCOL_VERSION` bump | | `-H`, `--hard-links` | Preserve hard links | ✅ Parity | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below | | `-D` | Same as --devices --specials | ✅ Parity | Implies `--devices --specials`. `-D` was unassigned in FastSync (verified: no collision), so it is free to imply both device-node and special-file preservation. As of protocol 2.23.0 `--specials` genuinely covers **both FIFOs and unix sockets**, so `-D` covers the full rsync set. See the `--devices`/`--specials` rows and the Phase-4 devices notes below | -| `--devices` | Preserve device files | ⚠️ Caveat | Recreates char/block device nodes with `mknodat` (type + rdev strictly validated, confined fd-relative below the receive root). A device whose `mknodat` fails with `EPERM`/`EACCES` (no `CAP_MKNOD`, or super-user activity forbidden) is now a **transfer error** surfaced through the receiver's outcome aggregation — rsync parity: rsync reports `mknod ... failed` and exits partial (23) when it attempts the node (as root or with `--super`). Residual: FastSync's default AUTO still *attempts* the node on a non-root receiver and therefore errors, whereas rsync without `--super` silently ignores `--devices` and skips the non-regular entry with exit 0; use `--no-super` for rsync's silent-skip behavior. (FastSync's process exit code for a receiver-side transfer error is the general error code 1, not rsync's partial 23 — a client exit-code-mapping residual that applies to every receiver file error, not just this branch.) `--specials` (FIFOs and unix sockets) keeps the unprivileged skip path and remains parity | +| `--devices` | Preserve device files | ⚠️ Caveat | Recreates char/block device nodes with `mknodat` (type + rdev strictly validated, confined fd-relative below the receive root). A device whose `mknodat` fails with `EPERM`/`EACCES` (no `CAP_MKNOD`, or super-user activity forbidden) is a **per-entry failure**: FastSync logs `cannot create device ...` (rsync logs `mknod ... failed`), counts it, **continues with the remaining files**, and ends the run with a non-OK terminal status. rsync parity: rsync likewise continues and exits partial (23). Residuals: (1) FastSync's default AUTO still *attempts* the node on a non-root receiver and therefore reports the per-entry failure, whereas rsync without `--super` silently ignores `--devices` and skips the non-regular entry with exit 0 — use `--no-super` for rsync's silent-skip behavior; (2) FastSync's process exit code for a receiver-side per-entry failure is the general error code 1, not rsync's partial 23 (a client exit-code-mapping residual that applies to every receiver file error, not just this branch); (3) with `--remove-source-files`, the non-OK terminal status means successfully transferred sources are not removed on a partial run. `--specials` (FIFOs and unix sockets) keeps the unprivileged skip path and remains parity | | `--specials` | Preserve special files | ✅ Parity | **FIFO and unix-socket recreation work** (protocol 2.23.0): FIFOs are recreated with `mkfifoat`, and sockets with `mknodat(..., S_IFSOCK)` — the latter is unprivileged on Linux because it materializes the socket *node*, not a live bound socket, so it is a real, assertable behavior under CI (it matches rsync, which also recreates a socket by `mknod`). Node creation is confined below the receive root (fd-relative parent; no `..`, no symlink follow) and type/rdev are validated strictly; a matching existing node is left in place and an unrelated entry is never replaced. Crosses the wire like `--devices` (the `STATUS_SPECIAL` frame). See the Phase-4 devices notes | | `--copy-devices` | Copy device contents as file | ❌ Divergent | Copies a device/FIFO's reported `st_size` into an ordinary regular file and never reads an unbounded pseudo-device, so `--sendfile` cannot hang and the run always succeeds. Deliberate safe divergence from rsync's dd-like unbounded device read, which can block; the dangerous behavior will not be implemented | | `--write-devices` | Write to devices as files | ❌ Divergent | Writes only into an existing char/block node under the confined receive root (`O_NOFOLLOW` + `O_NONBLOCK`); a missing, symlinked, FIFO-with-no-reader, non-device, or otherwise unusable destination is skipped with a warning rather than allowed or aborted. Deliberate confinement divergence from rsync's more permissive behavior | @@ -351,7 +351,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | `-O`, `--omit-dir-times` | Omit dirs from --times | ✅ Parity | Real modifier now that FastSync preserves directory times. With metadata on, the scanner captures every traversed source directory's mtime (and atime under `-U`) and the sender transmits them in trailing `STATUS_DIR_TIMES` frame(s) **after all file data and the optional delete manifest** (chunked at the receiver's `MAX_MANIFEST_ENTRIES` per-frame cap); a dir-time entry only RECORDS metadata and never creates the directory (an empty source directory is created by the separate `STATUS_MKDIR` entry the scanner now emits, and `-m/--prune-empty-dirs` suppresses that; the trailing dir-time simply re-applies the metadata). The receiver defers applying them until its delete / `--delay-updates` publication phases have committed, so writing or removing a child never clobbers a parent directory's mtime (rsync applies directory times at the end for exactly this reason). When `-O` is set (the boolean crosses the wire) the receiver does not apply any of them; without `-O` an `-a`/`--preserve` transfer now restores directory times (reversing the old "never preserves dir times" divergence). Wire change: the terminal `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** | | `-J`, `--omit-link-times` | Omit symlinks from --times | ✅ Parity | Real modifier now that FastSync preserves symlink times. Symlink entries already carried their metadata on `STATUS_SYMLINK`; the receiver now applies it with **no-follow primitives only** (`utimensat(..., AT_SYMLINK_NOFOLLOW)`, plus best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)`), so the link itself is stamped without ever dereferencing it, confined fd-relative below the authorized receive root. A symlink has no children, so the times are applied immediately at creation. When `-J` is set (the boolean crosses the wire) the receiver skips the timestamps (mode/ownership are unaffected); without `-J` an `-a`/`-l` transfer restores symlink mtimes. Wire change alongside `-O`: the shared `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** | | `--super` | Receiver attempts super-user activities | ❌ Divergent | Safe-subset privilege model. `--super` permits the receiver to attempt already-confined super-user activities (ownership application, char/block device-node creation, `--write-devices`); `--no-super` forbids them even for root; `auto` keeps the historical best-effort attempt. **FastSync never elevates** — no `setuid`/`seteuid`/`setgid` — and `--super` never bypasses the confinement floor, so it diverges from rsync's real elevation. A server `--no-super` veto forces it off for every connection; a privileged standalone listener defaults off without `--allow-super`; daemon modules opt in with `client owner = yes` | -| `--fake-super` | Store/recover privileged attrs via xattrs | ⚠️ Caveat | Writes rsync 3.4.1's reserved `user.rsync.%stat` xattr with rsync's exact value grammar ` , :` (e.g. `104711 0,0 1234:5678`), recording the RESOLVED owner (the `--chown`/`--usermap`/`--groupmap`/`--copy-as` mapping when active, else the source's own id) plus the full mode and rdev; it **never performs a real `chown`**. mtime is carried by the file's own timestamp, exactly as rsync does it (there is no mtime field). The receiver parses the same grammar and replays the permission bits fd-relative, stripping the recorded special bits on disk exactly like rsync's fake-super receiver. Regular files are interoperable with real rsync 3.4.1 in both directions (the differential test has rsync read a FastSync fake-super tree and re-emit the identical record). Residual: directories and device nodes are not yet faked — no `%stat` record is written for a directory, and a char/block node is still recreated/skipped rather than stored as a regular file carrying the stat. Implies metadata transmission; incompatible with `-s` | +| `--fake-super` | Store/recover privileged attrs via xattrs | ⚠️ Caveat | Writes rsync 3.4.1's reserved `user.rsync.%stat` xattr with rsync's exact value grammar ` , :` (e.g. `104711 0,0 1234:5678`), recording the RESOLVED owner (the `--chown`/`--usermap`/`--groupmap`/`--copy-as` mapping when active, else the source's own id) plus the full mode and rdev; it **never performs a real `chown`**. mtime is carried by the file's own timestamp, exactly as rsync does it (there is no mtime field). The receiver parses the same grammar and replays the permission bits fd-relative, stripping the recorded special bits on disk exactly like rsync's fake-super receiver. Regular files are interoperable with real rsync 3.4.1 in both directions (the differential test has rsync read a FastSync fake-super tree and re-emit the identical record). Char/block devices **are** faked: a device is written as a regular empty file and its `user.rsync.%stat` records the real `rdev` (e.g. `20644 1,3 0:0`), never `mknod`'d, on both privileged and unprivileged receivers, exactly as rsync does. The record parser range-checks every field (mode/rdev/uid/gid) and rejects malformed records cleanly. Residual: directories are not yet faked — no `%stat` record is written for a directory. Implies metadata transmission; incompatible with `-s` | | `--open-noatime` | Avoid changing access time when opening files | ✅ Parity | Sender-side policy: the sender opens source files with `O_NOATIME` (Linux) when reading them for transfer, so the open/read does NOT bump the source's on-disk access time. Degrades safely when `O_NOATIME` is unavailable (not defined) or refused (`EPERM`, since it needs `CAP_FOWNER` or file ownership): the code falls back to a normal open, so the data always transfers — only the atime-bump is skipped. It does not itself capture/preserve atime; it only avoids modifying it. **Client-only, never crosses the wire.** Exposed as `file_open_for_read()` and applied to both the buffered data path and the sendfile path | | `--numeric-ids` | Do not map uid/gid by name | ✅ Parity | **A mapping modifier only:** when ownership is being applied it uses the transmitted numeric uid/gid directly, skipping the name lookup. It does **not** request ownership application on its own — combine it with `-o`/`-g`, `-a`, or an explicit map (`--chown`/`--usermap`/`--groupmap`) — and it does not need any metadata flag merely to parse. Ownership is only applied when metadata (hence the source uid/gid) is actually transmitted (see the Phase-4 identity notes) | | `--usermap=STRING` | Map usernames | ✅ Parity | Opt-in ownership application. Comma-separated `FROM:TO` rules evaluated in order, first match wins. `FROM` accepts a source-resolved user name, a name **glob** (`*`/`?`/`[...]`, expanded sender-side at CLI-parse time against the sender's passwd/group database and collapsed into numeric `LOW-HIGH` ranges, bounded by `MAX_IDENTITY_MAP`), an `@N`/bare `N` numeric id, an inclusive `LOW-HIGH` id range, `*`, or an empty field (ids with no source name). `TO` accepts a receiver-resolved **name** (protocol 2.26.0 resolves it on the receiving side against the receiver's account database, matching rsync), an `@N`/bare `N` id, or `*` (the receiving process's euid). Rules travel as resolved numeric pairs plus an optional TO name; the receiver applies a matching rule, else falls back to `--chown`, `--numeric-ids`, then a best-effort name lookup, via fd-relative `fchown`. Malformed specs are clear errors. Implies metadata; only effective where the receiver can chown (otherwise a warning) | @@ -738,8 +738,8 @@ targets verbatim, matching rsync. | Flag | Rsync Description | FastSync Status | Notes | |------|-------------------|-----------------|-------| | `--daemon` | Run as rsync daemon | ❌ Divergent | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding | -| `--config=FILE` | Alternate rsyncd.conf file | ❌ Divergent | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and still strictly rejects a genuinely unknown key so a typo can never silently change what a module serves; requires `--daemon`. **rsync 3.4.1 key subset accepted:** the common rsyncd.conf GLOBAL keys (`port`, `address`, `motd file`, `max connections`, `hosts allow`/`hosts deny`, plus the inert `pid file`, `log file`, `socket options`/`sockopts`, `listen backlog`, `syslog facility`, `syslog tag`, `log format`, `use chroot`, `uid`, `gid`, `timeout`, `max verbosity`/`min verbosity`, `lock file`, `transfer logging`, `strict modes`, `reverse lookup`/`forward lookup`, `ignore errors`, `ignore nonreadable`, `dont compress`) and MODULE keys (`path`, `read only`, `max connections`, `auth users`, `hosts allow`/`hosts deny`, plus the inert `comment`, `use chroot`, `uid`/`gid`/`daemon uid`/`daemon gid`, `exclude`, `include`, `exclude from`/`include from`, `filter`, `secrets file`, `auth digest`, `max verbosity`/`min verbosity`, `lock file`, `transfer logging`, `log file`/`log format`/`syslog facility`/`syslog tag`, `timeout`, `strict modes`, `numeric ids`, `fake super`, `munge symlinks`, `write only`, `list`, `dont compress`, `charset`, `refuse options`, `incoming chmod`/`outgoing chmod`, `open noatime`, `max size`/`min size`, `temp dir`, `pre-xfer exec`/`post-xfer exec`, `name converter`, `proxy protocol`/`proxy protocol hosts`, `reverse lookup`/`forward lookup`, `ignore errors`, `ignore nonreadable`) are recognized. Keys with a FastSync equivalent map onto it (a global `read only` is honored as the default for later modules); keys with no FastSync equivalent load **inert** (no effect) rather than failing the whole config. Residual: the native grammar still differs from rsync's (no `\` line continuation, `%VAR%` expansion, `[global]` re-entry, or inline `#` comments), and the inert keys are genuinely not enforced — in particular a daemon-side `exclude`/`filter` is NOT applied and `secrets file` is NOT read (use `path`, `--password-file`, and client-side filters instead) | -| `--dparam=OVERRIDE` | Override global daemon config | ❌ Divergent | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Reuses the exact same global-key dispatch as `--config`, so it accepts the native global keys (`port`, `motd file`, `address`, `read only`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`), the recognized inert rsync global keys, and rsync's compact spellings (`motdfile`, `pidfile`, `logfile`); keys are case-insensitive. `read only` sets the global default and re-applies it to every module that did not set its own value. Genuinely unknown keys and invalid values are rejected. Requires `--daemon` | +| `--config=FILE` | Alternate rsyncd.conf file | ❌ Divergent | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and still strictly rejects a genuinely unknown key so a typo can never silently change what a module serves; requires `--daemon`. **rsync 3.4.1 key subset accepted:** the common rsyncd.conf GLOBAL keys (`port`, `address`, `motd file`, `max connections`, `hosts allow`/`hosts deny`, plus the inert `pid file`, `log file`, `socket options`/`sockopts`, `listen backlog`, `syslog facility`, `syslog tag`, `log format`, `use chroot`, `uid`, `gid`, `timeout`, `max verbosity`/`min verbosity`, `lock file`, `transfer logging`, `strict modes`, `reverse lookup`/`forward lookup`, `ignore errors`, `ignore nonreadable`, `dont compress`) and MODULE keys (`path`, `read only`, `max connections`, `auth users`, `hosts allow`/`hosts deny`, plus the inert `comment`, `use chroot`, `uid`/`gid`/`daemon uid`/`daemon gid`, `exclude`, `include`, `exclude from`/`include from`, `filter`, `secrets file`, `auth digest`, `max verbosity`/`min verbosity`, `lock file`, `transfer logging`, `log file`/`log format`/`syslog facility`/`syslog tag`, `timeout`, `strict modes`, `numeric ids`, `fake super`, `munge symlinks`, `write only`, `list`, `dont compress`, `charset`, `refuse options`, `incoming chmod`/`outgoing chmod`, `open noatime`, `max size`/`min size`, `temp dir`, `pre-xfer exec`/`post-xfer exec`, `name converter`, `proxy protocol`/`proxy protocol hosts`, `reverse lookup`/`forward lookup`, `ignore errors`, `ignore nonreadable`) are recognized. Keys with a FastSync equivalent map onto it. Modules are **read-only by default**, exactly like rsync: `read only = no` (or `write only = yes`, which FastSync maps to writability because it is push-only) opts a module in; a global `read only` sets the default for later modules, and an explicit module value always wins. Keys with no FastSync equivalent load **inert** (no effect) rather than failing the whole config, and every inert key whose intent is access control (`secrets file`, `refuse options`, `exclude`/`include`/`filter`, `max size`/`min size`, `pre-xfer exec`/`post-xfer exec`, `incoming chmod`/`outgoing chmod`, `name converter`, `use chroot`, `uid`/`gid`, ...) emits a startup **WARN** naming the key (and module), so an operator cannot mistake an unenforced restriction for an enforced one. Residual: the native grammar still differs from rsync's (no `\` line continuation, `%VAR%` expansion, `[global]` re-entry, or inline `#` comments), and the inert keys are genuinely not enforced — in particular a daemon-side `exclude`/`filter` is NOT applied and `secrets file` is NOT read (use `path`, `--password-file`, and client-side filters instead) | +| `--dparam=OVERRIDE` | Override global daemon config | ❌ Divergent | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Reuses the exact same global-key dispatch as `--config`, so it accepts the native global keys (`port`, `motd file`, `address`, `read only`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`), the recognized inert rsync global keys, and rsync's compact spellings (`motdfile`, `pidfile`, `logfile`); keys are case-insensitive. `read only` sets the global default and re-applies it to every module that did not set its own value; the default is `yes` (rsync modules are read-only unless `read only = no` / `write only = yes`), so `--dparam read only=no` is required to make modules without their own value writable, and an inert security global key (`use chroot`, `uid`, `gid`, `strict modes`) emits the same startup **WARN** as `--config`. Genuinely unknown keys and invalid values are rejected. Requires `--daemon` | | `--no-detach` | Don't detach from parent | ✅ Parity | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` | | `--password-file=FILE` | Read daemon password from file | ❌ Divergent | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. **Hardening follow-up:** the file is opened with `O_NOFOLLOW`, so a symlinked credential path fails closed (`ELOOP`) instead of being followed before the owner/mode gate; literal fd-backed paths (`/dev/fd/`, `/proc/self/fd/`, which is what a bash process substitution passes) are exempt, so process substitution still works. A FIFO/process-substitution read now waits under a bounded ~3 s deadline for its writer, so a slow producer works while a connected-but-silent FIFO fails instead of hanging. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat | | `--early-input=FILE` | Use FILE for daemon early exec | ❌ Divergent | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. Opened with the same `O_NOFOLLOW` hardening as `--password-file` (a symlinked path fails closed with `ELOOP`; fd-backed `/dev/fd/N`/`/proc/self/fd/N` process-substitution paths are exempt) and a FIFO read is bound-waited (~3 s) so a slow producer works while a writer-less FIFO cannot hang. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) | @@ -747,7 +747,7 @@ targets verbatim, matching rsync. **Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding. -- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. To reduce the rsync divergence, the parser additionally accepts the common rsync 3.4.1 GLOBAL and MODULE keys: the keys with a FastSync equivalent (`path`, `read only`, `max connections`, `auth users`, `hosts allow`/`hosts deny`, and the global `port`/`address`/`motd file`) map onto it, a global `read only` becomes the default for modules defined after it, and the keys with no FastSync equivalent (e.g. `pid file`, `log file`, `use chroot`, `uid`/`gid`, `comment`, `exclude`/`include`, `max verbosity`, `lock file`, `transfer logging`, `timeout`, `secrets file`) are recognized and loaded **inert** (accepted-but-ignored) instead of failing the whole file. `--dparam` reuses the same dispatch, so it also accepts the inert rsync global keys and the compact spellings `motdfile`/`pidfile`/`logfile`. A key outside both sets is still rejected. The inert keys are genuinely not enforced: a daemon-side `exclude`/`include`/`filter` is not applied and a `secrets file` is not read (use `--password-file`/`--early-input`), so an rsync config that relies on those must be edited rather than trusted. +- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default yes — rsync modules are read-only unless `read only = no`/`write only = yes`), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves. To reduce the rsync divergence, the parser additionally accepts the common rsync 3.4.1 GLOBAL and MODULE keys: the keys with a FastSync equivalent (`path`, `read only`, `max connections`, `auth users`, `hosts allow`/`hosts deny`, and the global `port`/`address`/`motd file`) map onto it, a global `read only` becomes the default for modules defined after it, and the keys with no FastSync equivalent (e.g. `pid file`, `log file`, `use chroot`, `uid`/`gid`, `comment`, `exclude`/`include`, `max verbosity`, `lock file`, `transfer logging`, `timeout`, `secrets file`) are recognized and loaded **inert** (accepted-but-ignored) instead of failing the whole file. `--dparam` reuses the same dispatch, so it also accepts the inert rsync global keys and the compact spellings `motdfile`/`pidfile`/`logfile`. A key outside both sets is still rejected. The inert keys are genuinely not enforced: a daemon-side `exclude`/`include`/`filter` is not applied and a `secrets file` is not read (use `--password-file`/`--early-input`), so an rsync config that relies on those must be edited rather than trusted. - **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time. - **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`) and re-derives the per-module and per-source occupancy counts from the surviving REGISTERED slots, so a child killed mid-registration cannot leak a count. The per-source table has a bounded lifetime: an entry with no live connection is reclaimed after its lockout expires or it has been idle (300 s); if the table is genuinely full the per-source cap/lockout fails open for new sources (per-module cap and ACLs still apply) with a rate-limited warning. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4), and a trusted loopback peer (127.0.0.0/8 / `::1`, `utils_fd_peer_is_local`) is exempt from the per-source cap and the auth lockout because all local clients share one address (the per-module/global caps still apply). Clients behind a shared NAT/proxy address likewise share one per-source budget and lockout counter. A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start. - **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused. @@ -964,7 +964,7 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, the triage cycle, and a later no-wire parity pass.** ✅ Parity 119 / ⚠️ Caveat 14 / ❌ Divergent 24 = 157 rows. The no-wire parity pass accepted `--inc-recursive`/`--no-inc-recursive` as inert no-ops (❌ → ✅, since FastSync's full scan is rsync's `--no-inc-recursive` and the destination is identical), narrowed the `--temp-dir` divergence by accepting an absolute path that canonicalizes inside the receive root (the row stays ❌ for out-of-root absolute paths), closed the `--delete-before` phase-0 divergence (⚠️ → ✅: the single-threaded data pass now replays the pre-scan file list, so a source file created after the scan is neither transferred nor kept, matching rsync), and moved `--fake-super` and `--devices` ❌ → ⚠️ (`--fake-super` now writes/reads rsync's exact `user.rsync.%stat` key and ` , :` grammar, interoperating with real rsync 3.4.1 for regular files; `--devices` now surfaces a failed device `mknod` as a transfer error instead of a silent non-root skip — see those rows for the remaining directory/device-faking and exit-code residuals). The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--filter`, `-F`, the three basis-dir options, `-y/--fuzzy`, `--fake-super`, and `--devices`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, the triage cycle, and a later no-wire parity pass.** ✅ Parity 119 / ⚠️ Caveat 14 / ❌ Divergent 24 = 157 rows. The no-wire parity pass accepted `--inc-recursive`/`--no-inc-recursive` as inert no-ops (❌ → ✅, since FastSync's full scan is rsync's `--no-inc-recursive` and the destination is identical), narrowed the `--temp-dir` divergence by accepting an absolute path that canonicalizes inside the receive root (the row stays ❌ for out-of-root absolute paths), closed the `--delete-before` phase-0 divergence (⚠️ → ✅: both the single-threaded and the `--threads` data passes now replay the pre-scan file list, so a source file created after the scan is neither transferred nor kept, matching rsync), and moved `--fake-super` and `--devices` ❌ → ⚠️ (`--fake-super` now writes/reads rsync's exact `user.rsync.%stat` key and ` , :` grammar, interoperating with real rsync 3.4.1 for regular files and faking char/block devices as regular files carrying the real rdev; `--devices` now logs a failed device `mknod` as a per-entry failure that continues the transfer instead of a silent non-root skip — see those rows for the remaining directory-faking and exit-code residuals). A review pass then hardened the fake-super stat parser (strict range-checked parsing), made rsync-style daemon modules read-only by default with a startup warning for accepted-but-unenforced access-control keys, and extended the `--delete-before` replay to the `--threads` path. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--filter`, `-F`, the three basis-dir options, `-y/--fuzzy`, `--fake-super`, and `--devices`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that receiver filter engine, flipping `--filter=RULE` back to ✅ (see above; the -- 2.54.0 From a07cc00bb0e6d971375790555ea021c53520a21d Mon Sep 17 00:00:00 2001 From: TapTap Date: Tue, 22 Sep 2026 23:45:41 +0200 Subject: [PATCH 61/68] test: restore dumpable flag after setuid drop so LSan can run under ASan --- tests/test_file.c | 12 +++++++++++- 1 file changed, 11 insertions(+), 1 deletion(-) diff --git a/tests/test_file.c b/tests/test_file.c index 744ec6d..7958687 100644 --- a/tests/test_file.c +++ b/tests/test_file.c @@ -21,6 +21,9 @@ #include #include #include +#ifdef __linux__ +#include +#endif #include static void test_file_create() { @@ -1414,8 +1417,15 @@ static void test_device_mknod_failure_is_per_entry() { if (saved == 0 && seteuid(65534) == 0) dropped = true; FileSaveResult result = file_save_to_disk_full(root, f, cfg); - if (dropped) + if (dropped) { EXPECT_EQ_INT(seteuid(saved), 0); +#ifdef __linux__ + /* A setuid transition clears the process dumpable flag, which makes + * LeakSanitizer's ptrace-based thread suspension fail at exit. Restore it + * so the ASan/UBSan CI jobs can still run the leak check. */ + (void)prctl(PR_SET_DUMPABLE, 1, 0, 0, 0); +#endif + } EXPECT_EQ_INT(result, FILE_SAVE_FAILED); /* Nothing was created: no device node and no regular-file fallback. */ -- 2.54.0 From 1f8d60e30d27fbe73621ba73049b91052747b454 Mon Sep 17 00:00:00 2001 From: TapTap Date: Wed, 23 Sep 2026 00:18:29 +0200 Subject: [PATCH 62/68] protocol: resolve TLS transport from the bound session, not thread-local io_ssl file_send.c chose between sendfile() and the TLS-aware buffered path by calling io_get_ssl(), which reads the thread-local io_ssl. A worker thread that bound a TLS ProtocolSession via protocol_session_bind() never ran the handshake in that thread, so io_ssl is NULL there and a TLS + --threads transfer took the raw sendfile() path on an encrypted socket. Add protocol_current_ssl(), which prefers the bound session's SSL and falls back to io_ssl on the fd-shim path, and use it in file_send.c. Un-xfail test_tls_with_multithreading. --- src/shared/file_send.c | 8 ++++++-- src/shared/protocol.c | 11 +++++++++++ src/shared/protocol.h | 9 +++++++++ tests/integration/test_tls.py | 1 - 4 files changed, 26 insertions(+), 3 deletions(-) diff --git a/src/shared/file_send.c b/src/shared/file_send.c index 9f43724..70f0e95 100644 --- a/src/shared/file_send.c +++ b/src/shared/file_send.c @@ -124,8 +124,12 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta } /* sendfile cannot encrypt TLS records. Keep the framing identical but - route encrypted transfers through the deadline-aware IO layer. */ - if (io_get_ssl() != NULL) { + route encrypted transfers through the deadline-aware IO layer. Resolve + the transport from the bound session, not the thread-local io_ssl: a + worker thread running a TLS transfer has its SSL only on the session it + bound, so io_get_ssl() would be NULL there and the raw sendfile() path + would be taken on an encrypted socket. */ + if (protocol_current_ssl() != NULL) { unsigned char buffer[64 * 1024]; unsigned long long remaining = file_size; bool ok = true; diff --git a/src/shared/protocol.c b/src/shared/protocol.c index b6f8280..eba3a61 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -270,6 +270,17 @@ SSL* io_get_ssl(void) { return io_ssl; } +SSL* protocol_current_ssl(void) { + /* The bound session is the authoritative transport for a worker thread: it + * was explicitly handed to protocol_session_bind() and carries its own SSL, + * whereas io_ssl is thread-local and NULL in a thread that never performed + * the handshake. With no session bound (the fd-shim path), fall back to the + * legacy thread-local SSL. */ + if (bound_session) + return bound_session->ssl; + return io_ssl; +} + unsigned long long protocol_bytes_written(void) { return atomic_load(&io_bytes_written); } diff --git a/src/shared/protocol.h b/src/shared/protocol.h index 0bb0b42..ff4ec0d 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -216,6 +216,15 @@ void io_set_bwlimit(unsigned long long bytes_per_sec); unsigned long long io_get_bwlimit(void); void io_set_ssl(SSL* ssl); SSL* io_get_ssl(void); +/* SSL object of the transport in effect on this thread: the currently bound + * session's SSL when a session is bound, otherwise the legacy thread-local + * io_ssl. NULL for a plaintext transport. Unlike io_get_ssl(), this resolves + * worker threads that bound a TLS session via protocol_session_set_ssl()/ + * protocol_session_bind() but never called io_set_ssl() themselves (C11 + * _Thread_local state is not inherited by a new thread). Callers that must + * choose a TLS-only code path (e.g. file_send.c's sendfile fallback) must use + * this instead of io_get_ssl(). */ +SSL* protocol_current_ssl(void); /* Process-wide wire byte counters. protocol_send_n_data/protocol_receive_n_data * update them; the zero-copy sendfile path reports through diff --git a/tests/integration/test_tls.py b/tests/integration/test_tls.py index 1e5c029..cdc3a24 100644 --- a/tests/integration/test_tls.py +++ b/tests/integration/test_tls.py @@ -146,7 +146,6 @@ class TestTLSBasic: assert not missing, f"Missing files: {missing}" assert not mismatches, f"Mismatched files: {mismatches}" - @pytest.mark.xfail(reason="TLS multithreading has architectural limitations with per-thread SSL context") def test_tls_with_multithreading(self, certs): """TLS + multithreading.""" clean_dir(DEST_DIR) -- 2.54.0 From ec6692ac411cc28a0e744139b32ec9cbd2e3843c Mon Sep 17 00:00:00 2001 From: TapTap Date: Wed, 23 Sep 2026 00:18:36 +0200 Subject: [PATCH 63/68] protocol: add transport I/O vtable over TCP/TLS primitives Introduce ProtocolIoOps (send/recv/has_pending), selected once by protocol_session_init() and protocol_session_set_ssl(), and dispatch the send, receive and status-read loops through session->ops instead of branching on session->ssl at runtime. Each op performs one transfer attempt and classifies the result (PROTOCOL_IO_RETRY/CLOSED/ERROR), preserving the WANT_READ/WANT_WRITE wait_events switching, the SSL_ERROR_SYSCALL/EINTR retry, the SSL_pending poll gating and the deadline handling. The raw read()/write() fallback lives in the plaintext ops. Add unit tests: a socketpair session with a counting ops wrapper proving the loops dispatch through the vtable, and a worker-thread test that protocol_current_ssl() resolves the bound session's SSL when io_ssl is NULL. --- src/shared/protocol.c | 228 ++++++++++++++++++++++++++++-------------- src/shared/protocol.h | 36 ++++++- tests/test_protocol.c | 129 ++++++++++++++++++++++++ 3 files changed, 314 insertions(+), 79 deletions(-) diff --git a/src/shared/protocol.c b/src/shared/protocol.c index eba3a61..69dec04 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -38,6 +38,122 @@ static atomic_ullong io_bytes_read = 0; static unsigned long long global_bwlimit(void); +/* ------------------------------------------------------------------------- * + * Transport vtable implementations. + * + * Each op performs exactly one transfer attempt. WANT_READ/WANT_WRITE and an + * EINTR-interrupted syscall are reported as PROTOCOL_IO_RETRY (with + * *wait_events set to the poll event the caller must wait on); a clean peer + * close is PROTOCOL_IO_CLOSED and anything else is PROTOCOL_IO_ERROR. This + * keeps every WANT_READ/WANT_WRITE and EINTR retry exactly where it was before + * the vtable was introduced, just moved behind the function pointer. + * ------------------------------------------------------------------------- */ + +static ssize_t plain_io_send(ProtocolSession* session, const void* data, size_t size, + short* wait_events) { + ssize_t written = write(session->write_fd, data, size); + if (written < 0) { + if (errno == EINTR) + return PROTOCOL_IO_RETRY; + return PROTOCOL_IO_ERROR; + } + if (written == 0) + return PROTOCOL_IO_ERROR; + *wait_events = POLLOUT; + return written; +} + +static ssize_t plain_io_recv(ProtocolSession* session, void* data, size_t size, + short* wait_events) { + ssize_t received = read(session->read_fd, data, size); + if (received < 0) { + if (errno == EINTR) + return PROTOCOL_IO_RETRY; + return PROTOCOL_IO_ERROR; + } + if (received == 0) + return PROTOCOL_IO_CLOSED; + *wait_events = POLLIN; + return received; +} + +static bool plain_io_has_pending(const ProtocolSession* session) { + (void)session; + return false; +} + +static ssize_t tls_io_send(ProtocolSession* session, const void* data, size_t size, + short* wait_events) { + /* SSL_write takes an int length; clamp a >INT_MAX request into chunks so the + * size_t downcast can never truncate into a negative/partial write. */ + size_t chunk = size > (size_t)INT_MAX ? (size_t)INT_MAX : size; + ssize_t written = SSL_write(session->ssl, data, (int)chunk); + if (written <= 0) { + int ssl_err = SSL_get_error(session->ssl, (int)written); + if (ssl_err == SSL_ERROR_WANT_WRITE) { + *wait_events = POLLOUT; + return PROTOCOL_IO_RETRY; + } + if (ssl_err == SSL_ERROR_WANT_READ) { + *wait_events = POLLIN; + return PROTOCOL_IO_RETRY; + } + /* A signal (e.g. Ctrl-C) interrupts the blocking TLS write: retry so the + * send loop can observe the abort flag at the next checkpoint. */ + if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) + return PROTOCOL_IO_RETRY; + return PROTOCOL_IO_ERROR; + } + *wait_events = POLLOUT; + return written; +} + +static ssize_t tls_io_recv(ProtocolSession* session, void* data, size_t size, short* wait_events) { + /* SSL_read takes an int length; clamp a >INT_MAX request into chunks + * (mirrors the send path) so the size_t downcast can never truncate into a + * negative/partial read. */ + size_t chunk = size > (size_t)INT_MAX ? (size_t)INT_MAX : size; + ssize_t received = SSL_read(session->ssl, data, (int)chunk); + if (received <= 0) { + int ssl_err = SSL_get_error(session->ssl, (int)received); + if (ssl_err == SSL_ERROR_WANT_WRITE) { + *wait_events = POLLOUT; + return PROTOCOL_IO_RETRY; + } + if (ssl_err == SSL_ERROR_WANT_READ) { + *wait_events = POLLIN; + return PROTOCOL_IO_RETRY; + } + /* A signal interrupts the blocking TLS read: retry (mirrors the send path) + * so the loop reaches its next abort/deadline checkpoint. */ + if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) + return PROTOCOL_IO_RETRY; + /* A zero-length SSL_read is the peer's clean close_notify (or EOF without + * one); report it distinctly so the caller can log it as a close. */ + if (received == 0) + return PROTOCOL_IO_CLOSED; + return PROTOCOL_IO_ERROR; + } + *wait_events = POLLIN; + return received; +} + +static bool tls_io_has_pending(const ProtocolSession* session) { + return session->ssl != NULL && SSL_pending(session->ssl) > 0; +} + +static const ProtocolIoOps plain_io_ops = { + .send = plain_io_send, + .recv = plain_io_recv, + .has_pending = plain_io_has_pending, +}; + +static const ProtocolIoOps tls_io_ops = { + .send = tls_io_send, + .recv = tls_io_recv, + .has_pending = tls_io_has_pending, +}; + static bool protocol_reserve_memory(ProtocolSession* session, size_t charge) { unsigned long long allocated = atomic_load(&session->total_allocated_bytes); while (true) { @@ -79,6 +195,7 @@ void io_set_fds(int read_fd, int write_fd) { legacy_io_session.read_fd = read_fd; legacy_io_session.write_fd = write_fd; legacy_io_session.ssl = NULL; + legacy_io_session.ops = &plain_io_ops; legacy_io_session.eight_bit_output = false; atomic_store(&legacy_io_session.total_allocated_bytes, 0); legacy_io_session.max_alloc = DEFAULT_MAX_ALLOC; @@ -91,6 +208,7 @@ void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd) memset(session, 0, sizeof(*session)); session->read_fd = read_fd; session->write_fd = write_fd; + session->ops = &plain_io_ops; session->max_alloc = DEFAULT_MAX_ALLOC; session->io_timeout_sec = RECEIVE_TIMEOUT_SEC; atomic_init(&session->total_allocated_bytes, 0); @@ -158,8 +276,12 @@ void protocol_session_unbind(void) { } void protocol_session_set_ssl(ProtocolSession* session, SSL* ssl) { - if (session) - session->ssl = ssl; + if (!session) + return; + session->ssl = ssl; + /* Select the transport dispatch once, here, instead of branching on the SSL + * pointer inside every I/O loop. */ + session->ops = ssl ? &tls_io_ops : &plain_io_ops; } static void bw_mutex_init(void) { @@ -309,6 +431,7 @@ static ProtocolSession* legacy_session(int read_fd, int write_fd) { protocol_session_set_bwlimit(&legacy_io_session, global_bwlimit()); } legacy_io_session.ssl = io_ssl; + legacy_io_session.ops = io_ssl ? &tls_io_ops : &plain_io_ops; return &legacy_io_session; } @@ -346,8 +469,9 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat if (!data && data_size != 0) return false; log_debug_message(LOG_DEBUG_IO, " Sending n Data: %zu", data_size); - if (!session) + if (!session || !session->ops) return false; + const ProtocolIoOps* ops = session->ops; /* A non-positive session timeout disables the deadline entirely (rsync's * --timeout=0 default); poll then blocks until the socket becomes writable. */ int timeout_sec = session->io_timeout_sec > 0 ? session->io_timeout_sec : 0; @@ -373,36 +497,16 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat continue; if (pfd.revents & (POLLERR | POLLNVAL)) return false; - ssize_t bytes_send; - if (session->ssl) { - /* SSL_write takes an int length; clamp a >INT_MAX request into chunks so - * the size_t downcast can never truncate into a negative/partial write. */ - size_t ssl_chunk = chunk > (size_t)INT_MAX ? (size_t)INT_MAX : chunk; - bytes_send = SSL_write(session->ssl, (const char*)data + total_bytes_send, (int)ssl_chunk); - } else { - bytes_send = write(fd, (const char*)data + total_bytes_send, chunk); - } + ssize_t bytes_send = + ops->send(session, (const char*)data + total_bytes_send, chunk, &wait_events); + if (bytes_send == PROTOCOL_IO_RETRY) + continue; if (bytes_send <= 0) { - if (session->ssl) { - int ssl_err = SSL_get_error(session->ssl, (int)bytes_send); - if (ssl_err == SSL_ERROR_WANT_WRITE || ssl_err == SSL_ERROR_WANT_READ) { - wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN; - continue; - } - /* A signal (e.g. Ctrl-C) interrupts the blocking TLS write: retry so - the send loop can observe the abort flag at the next checkpoint. */ - if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) - continue; - } else if (errno == EINTR) { - continue; - } log_message(LOG_LEVEL_ERROR, "Could not send data"); return false; } bw_throttle_session(session, (size_t)bytes_send); total_bytes_send += bytes_send; - if (session->ssl) - wait_events = POLLOUT; } log_debug_message(LOG_DEBUG_IO, " Send n Data: %zd", total_bytes_send); atomic_fetch_add(&io_bytes_written, (unsigned long long)total_bytes_send); @@ -435,14 +539,15 @@ bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_s static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, size_t data_size, const struct timespec* deadline) { log_debug_message(LOG_DEBUG_IO, " Receiving n Data: %zu", data_size); - if (!session) + if (!session || !session->ops) return false; + const ProtocolIoOps* ops = session->ops; int fd = session->read_fd; size_t total_bytes_received = 0; short wait_events = POLLIN; while (total_bytes_received < data_size) { - if (!session->ssl || SSL_pending(session->ssl) == 0) { + if (!ops->has_pending(session)) { struct pollfd pfd = {.fd = fd, .events = wait_events}; /* A NULL deadline means "wait indefinitely" (timeout disabled). */ int poll_result = poll(&pfd, 1, deadline ? deadline_remaining_ms(deadline) : -1); @@ -460,43 +565,19 @@ static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, return false; } - ssize_t bytes_received; - if (session->ssl) { - /* SSL_read takes an int length; clamp a >INT_MAX request into chunks - * (mirrors the send path) so the size_t downcast can never truncate into - * a negative/partial read. */ - size_t ssl_chunk = data_size - total_bytes_received > (size_t)INT_MAX - ? (size_t)INT_MAX - : data_size - total_bytes_received; - bytes_received = SSL_read(session->ssl, (char*)data + total_bytes_received, (int)ssl_chunk); - } else { - bytes_received = - read(fd, (char*)data + total_bytes_received, data_size - total_bytes_received); + ssize_t bytes_received = ops->recv(session, (char*)data + total_bytes_received, + data_size - total_bytes_received, &wait_events); + if (bytes_received == PROTOCOL_IO_RETRY) + continue; + if (bytes_received == PROTOCOL_IO_CLOSED) { + log_message(LOG_LEVEL_ERROR, "Connection closed while receiving data"); + return false; } if (bytes_received <= 0) { - if (session->ssl) { - int ssl_err = SSL_get_error(session->ssl, (int)bytes_received); - if (ssl_err == SSL_ERROR_WANT_WRITE || ssl_err == SSL_ERROR_WANT_READ) { - wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN; - continue; - } - /* A signal interrupts the blocking TLS read: retry (mirrors the send - path and protocol_read_status_until) so the loop reaches its next - abort/deadline checkpoint instead of failing spuriously. */ - if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) - continue; - } else if (errno == EINTR) { - continue; - } - if (bytes_received == 0) - log_message(LOG_LEVEL_ERROR, "Connection closed while receiving data"); - else - log_message(LOG_LEVEL_ERROR, "Could not receive bytes"); + log_message(LOG_LEVEL_ERROR, "Could not receive bytes"); return false; } total_bytes_received += (size_t)bytes_received; - if (session->ssl) - wait_events = POLLIN; } log_debug_message(LOG_DEBUG_IO, " Received n Data: %zu", total_bytes_received); atomic_fetch_add(&io_bytes_read, (unsigned long long)total_bytes_received); @@ -856,11 +937,14 @@ bool protocol_receive_status_timed(ProtocolSession* session, Status* status, int * reply across a frame boundary. Returns false on timeout/EOF/error. */ static bool protocol_read_status_until(ProtocolSession* session, Status* status, const struct timespec* deadline) { + if (!session || !session->ops) + return false; + const ProtocolIoOps* ops = session->ops; Status received = STATUS_ERROR; size_t got = 0; short wait_events = POLLIN; while (got < sizeof(Status)) { - if (!session->ssl || SSL_pending(session->ssl) == 0) { + if (!ops->has_pending(session)) { int remaining_ms = deadline ? deadline_remaining_ms(deadline) : -1; if (remaining_ms == 0) { log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status"); @@ -880,21 +964,11 @@ static bool protocol_read_status_until(ProtocolSession* session, Status* status, if (pfd.revents & (POLLERR | POLLNVAL)) return false; } - ssize_t bytes_received; - if (session->ssl) - bytes_received = SSL_read(session->ssl, (char*)&received + got, sizeof(Status) - got); - else - bytes_received = read(session->read_fd, (char*)&received + got, sizeof(Status) - got); + ssize_t bytes_received = + ops->recv(session, (char*)&received + got, sizeof(Status) - got, &wait_events); + if (bytes_received == PROTOCOL_IO_RETRY) + continue; if (bytes_received <= 0) { - if (session->ssl) { - int ssl_err = SSL_get_error(session->ssl, (int)bytes_received); - if (ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) { - wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN; - continue; - } - } - if (bytes_received < 0 && errno == EINTR) - continue; log_message(LOG_LEVEL_ERROR, "Connection closed while receiving status"); return false; } @@ -923,7 +997,7 @@ bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, while (true) { if (abort_check && abort_check()) return false; - if (!session->ssl || SSL_pending(session->ssl) == 0) { + if (!session->ops || !session->ops->has_pending(session)) { int remaining_ms = deadline_remaining_ms(&deadline); if (remaining_ms <= 0) { log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", timeout_sec); diff --git a/src/shared/protocol.h b/src/shared/protocol.h index ff4ec0d..bd6c752 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -50,16 +50,48 @@ typedef struct ssl_st SSL; +typedef struct ProtocolSession ProtocolSession; + +/* + * Transport vtable: the per-session set of I/O primitives the three protocol + * loops (send, receive, status-read) dispatch through. The ops are selected + * once, when the session is initialized or its SSL is installed, so the loops + * never branch on the transport at runtime. A plaintext session uses the + * read()/write() ops; a TLS session uses the SSL_read()/SSL_write() ops. + * + * `send`/`recv` attempt exactly one transfer and return: + * > 0 bytes transferred, + * PROTOCOL_IO_RETRY no progress; poll on *wait_events and retry, + * PROTOCOL_IO_CLOSED peer closed the stream, + * PROTOCOL_IO_ERROR fatal transport error. + * `has_pending` reports bytes already buffered by the transport (a TLS record + * residue); the receive loops skip the poll() gate when it is true. + */ +typedef struct ProtocolIoOps { + ssize_t (*send)(ProtocolSession* session, const void* data, size_t size, short* wait_events); + ssize_t (*recv)(ProtocolSession* session, void* data, size_t size, short* wait_events); + bool (*has_pending)(const ProtocolSession* session); +} ProtocolIoOps; + +/* Negative sentinels returned by ProtocolIoOps.send/recv (see above). */ +enum { + PROTOCOL_IO_RETRY = -1, + PROTOCOL_IO_CLOSED = -2, + PROTOCOL_IO_ERROR = -3, +}; + /* * Explicit owner of protocol I/O. A session does not own the descriptors or * SSL object; it only describes the transport used by a transfer. This makes * it safe to pass the transport to a worker without relying on inherited * thread-local state. */ -typedef struct ProtocolSession { +struct ProtocolSession { int read_fd; int write_fd; SSL* ssl; + /* Transport dispatch selected by protocol_session_init()/set_ssl(). */ + const ProtocolIoOps* ops; unsigned long long bwlimit; long long bw_tokens; long long bw_last_refill_sec; @@ -75,7 +107,7 @@ typedef struct ProtocolSession { * SO_RCVTIMEO/SO_SNDTIMEO. The server does not propagate a client 0 here: it * installs protocol_server_io_timeout_sec() so its sessions keep a floor. */ int io_timeout_sec; -} ProtocolSession; +}; typedef int Status; enum NET_STATUS { diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 3992256..365c82d 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -1,9 +1,13 @@ #include "protocol.h" #include "test_utils.h" +#include #include #include +#include +#include #include #include +#include #include #include #include @@ -786,6 +790,129 @@ static void test_protocol_throttle_bytes_legacy_same_session() { io_set_fds(-1, -1); } +/* ------------------------------------------------------------------------- * + * Transport-vtable dispatch tests. + * ------------------------------------------------------------------------- */ + +static int dispatch_send_calls; +static int dispatch_recv_calls; + +static ssize_t counting_send(ProtocolSession* session, const void* data, size_t size, + short* wait_events) { + dispatch_send_calls++; + ssize_t written = write(session->write_fd, data, size); + if (written < 0) + return errno == EINTR ? PROTOCOL_IO_RETRY : PROTOCOL_IO_ERROR; + if (written == 0) + return PROTOCOL_IO_ERROR; + *wait_events = POLLOUT; + return written; +} + +static ssize_t counting_recv(ProtocolSession* session, void* data, size_t size, + short* wait_events) { + dispatch_recv_calls++; + ssize_t received = read(session->read_fd, data, size); + if (received < 0) + return errno == EINTR ? PROTOCOL_IO_RETRY : PROTOCOL_IO_ERROR; + if (received == 0) + return PROTOCOL_IO_CLOSED; + *wait_events = POLLIN; + return received; +} + +static bool counting_has_pending(const ProtocolSession* session) { + (void)session; + return false; +} + +static const ProtocolIoOps counting_ops = { + .send = counting_send, + .recv = counting_recv, + .has_pending = counting_has_pending, +}; + +/* A plain-TCP socketpair session must route every byte through the ops table: + * installing a counting ops wrapper proves the send/receive loops dispatch via + * session->ops instead of branching on session->ssl. */ +static void test_protocol_dispatch_via_ops() { + int sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); + + ProtocolSession sender; + ProtocolSession receiver; + protocol_session_init(&sender, sv[0], sv[0]); + protocol_session_set_bwlimit(&sender, 0); + protocol_session_init(&receiver, sv[1], sv[1]); + protocol_session_set_bwlimit(&receiver, 0); + EXPECT_NOT_NULL(sender.ops); + EXPECT_NOT_NULL(receiver.ops); + + dispatch_send_calls = 0; + dispatch_recv_calls = 0; + sender.ops = &counting_ops; + receiver.ops = &counting_ops; + + const char payload[] = "dispatch-through-vtable"; + EXPECT_TRUE(protocol_send_n_data(&sender, payload, sizeof(payload))); + char received[sizeof(payload)] = {0}; + EXPECT_TRUE(protocol_receive_n_data(&receiver, received, sizeof(received))); + EXPECT_EQ_INT(memcmp(payload, received, sizeof(payload)), 0); + EXPECT_TRUE(dispatch_send_calls > 0); + EXPECT_TRUE(dispatch_recv_calls > 0); + + close(sv[0]); + close(sv[1]); +} + +typedef struct { + ProtocolSession* session; + SSL* expected_ssl; + SSL* resolved_ssl; + SSL* thread_local_ssl; +} SslResolverWorkerArg; + +static int ssl_resolver_worker(void* arg) { + SslResolverWorkerArg* worker = arg; + protocol_session_bind(worker->session); + worker->resolved_ssl = protocol_current_ssl(); + worker->thread_local_ssl = io_get_ssl(); + protocol_session_unbind(); + return thrd_success; +} + +/* The worker-thread bug fix: a thread that bound a TLS session but never ran + * the handshake has io_ssl == NULL, yet protocol_current_ssl() must return the + * session's SSL so callers pick the TLS path. */ +static void test_protocol_current_ssl_prefers_bound_session() { + SSL_CTX* ctx = SSL_CTX_new(TLS_method()); + EXPECT_NOT_NULL(ctx); + SSL* ssl = SSL_new(ctx); + EXPECT_NOT_NULL(ssl); + + int sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); + ProtocolSession session; + protocol_session_init(&session, sv[0], sv[0]); + protocol_session_set_ssl(&session, ssl); + + /* Clear the calling thread's legacy SSL: only the bound session carries it. */ + io_set_fds(-1, -1); + + SslResolverWorkerArg arg = { + .session = &session, .expected_ssl = ssl, .resolved_ssl = NULL, .thread_local_ssl = ssl}; + thrd_t worker; + EXPECT_EQ_INT(thrd_create(&worker, ssl_resolver_worker, &arg), thrd_success); + EXPECT_EQ_INT(thrd_join(worker, NULL), thrd_success); + EXPECT_TRUE(arg.resolved_ssl == ssl); + EXPECT_NULL(arg.thread_local_ssl); + + close(sv[0]); + close(sv[1]); + SSL_free(ssl); + SSL_CTX_free(ctx); +} + void test_protocol() { test_send_receive_n_data(); test_send_receive_n_data_zero(); @@ -819,4 +946,6 @@ void test_protocol() { test_protocol_throttle_bytes_paces(); test_protocol_throttle_bytes_unlimited(); test_protocol_throttle_bytes_legacy_same_session(); + test_protocol_dispatch_via_ops(); + test_protocol_current_ssl_prefers_bound_session(); } -- 2.54.0 From 492ce0ce89c83e823da53aae9cb4fd8adc0db6d6 Mon Sep 17 00:00:00 2001 From: TapTap Date: Wed, 23 Sep 2026 00:34:39 +0200 Subject: [PATCH 64/68] feat(xattr): carry and apply symlink xattrs (protocol 2.29.0) --- README.md | 2 +- RSYNC_COMPAT.md | 4 +- src/client/client_send.c | 6 +- src/client/scanner_filter.c | 8 +- src/client/scanner_parallel.c | 4 +- src/shared/config.h | 17 +- src/shared/file_receive.c | 8 + src/shared/file_save.c | 13 ++ src/shared/xattr.c | 68 +++++++- src/shared/xattr.h | 31 +++- tests/integration/test_fault_injection.py | 2 +- tests/integration/test_features.py | 50 ++++++ tests/integration/test_preflight.py | 4 +- tests/test_client_cli.c | 6 +- tests/test_config.c | 13 +- tests/test_xattr.c | 183 ++++++++++++++++++++++ 16 files changed, 392 insertions(+), 27 deletions(-) diff --git a/README.md b/README.md index 897b1b5..67d548c 100644 --- a/README.md +++ b/README.md @@ -832,7 +832,7 @@ before the module list, before authentication, and the connecting peer address ## Protocol and Security -FastSync protocol version `2.28.0` is shared by the client and server. The +FastSync protocol version `2.29.0` is shared by the client and server. The current protocol is sender-driven and includes configuration negotiation, including the maximum allocation limit, incremental checks, checksums, manifests, keep-alives, abort handling, per-file remove-source results, and diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 0689b9e..902ecef 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -339,7 +339,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`. | `-E`, `--executability` | Preserve executability | ✅ Parity | Preserves executable permission bits (implies metadata preservation) | | `--chmod=CHMOD` | Affect file permissions | ✅ Parity | Faithful port of rsync 3.4.1's `parse_chmod`/`tweak_mode`: numeric octal and symbolic `ugo`/`rwx` changes, `D`/`F` directory/file selectors, `X` (execute only on directories or already-executable files), `s`/`t` setuid/setgid/sticky, and append semantics — repeated clauses and repeated `--chmod` options accumulate in order (joined with commas). The changes are applied to the new mode **without sanitization** (matching rsync), except that setuid/setgid/sticky are masked when the connection forbids super-user activities (audit-cycle fix, see `-p`), and `--chmod` does **not** imply `-p` (rsync parity). Applied to files and directories on the receiver | | `-A`, `--acls` | Preserve ACLs | ✅ Parity | Implemented on Linux via the POSIX-ACL xattr representation: the sender captures the `system.posix_acl_access` / `system.posix_acl_default` xattrs and the receiver re-applies them fd-relative. A differential test with `setfacl` confirms the complete access and default ACL sets (including `mask`) are identical to rsync's on a directory. libacl is not required; a `fsetxattr` an unprivileged receiver may not perform is logged and skipped, never fatal. Only the `system.posix_acl_*` namespaces plus `user.*` are ever applied; privileged namespaces are never applied. Implies metadata transmission | -| `-X`, `--xattrs` | Preserve extended attributes | ❌ Divergent | Deliberately restricted to unprivileged `user.*` extended attributes plus the two POSIX ACL xattrs; `security.*` (SELinux, capabilities, ...) and `trusted.*` are **never** captured or applied — a client can never force a privileged attribute onto the destination, and the receiver independently re-validates every incoming name against the whitelist. This is a security-policy divergence from rsync, which can preserve the privileged namespaces with the needed privilege; implementing them would defeat FastSync's privilege-escalation guard. `user.*` capture/apply matches rsync in a differential test. Payloads are bounded on both ends. Incompatible with `-s`. **Also divergent: symlink xattrs/ACLs are not captured or applied** — `-X`/`-A` with `-l` carries only the link's owner/times/mode, not its xattrs (the capture uses path-following `listxattr`/`getxattr`, so the link's own xattrs are never read, and the receiver's symlink write path applies no xattr block). Closing this needs a dedicated symlink-xattr wire block and a `PROTOCOL_VERSION` bump | +| `-X`, `--xattrs` | Preserve extended attributes | ❌ Divergent | Deliberately restricted to unprivileged `user.*` extended attributes plus the two POSIX ACL xattrs; `security.*` (SELinux, capabilities, ...) and `trusted.*` are **never** captured or applied — a client can never force a privileged attribute onto the destination, and the receiver independently re-validates every incoming name against the whitelist. This is a security-policy divergence from rsync, which can preserve the privileged namespaces with the needed privilege; implementing them would defeat FastSync's privilege-escalation guard. `user.*` capture/apply matches rsync in a differential test. Payloads are bounded on both ends. Incompatible with `-s`. **Symlink xattrs are now carried (protocol 2.29.0):** a symlink entry appends the same bounded trailing xattr block to its `STATUS_SYMLINK` frame as every other entry kind, captured with `llistxattr`/`lgetxattr` so the link's OWN attributes are read and never the referent's, and re-applied no-follow with `lsetxattr` through the already-confined parent directory (`fsetxattr` cannot target a symlink: there is no `*at` xattr syscall and an `O_PATH` fd is rejected). On Linux the VFS refuses to associate xattrs with a symlink at all — every `lsetxattr` on a link fails with `EPERM` for `user.*`, `trusted.*` and `security.*`, even as root, verified in the CI container — so on FastSync's supported platforms the captured block is always empty and the apply is a no-op; the wire block is present for correctness and for a filesystem/platform that does support symlink xattrs. rsync 3.4.1's `--fake-super` is not a counterexample: it stores a symlink as a regular file whose `user.rsync.%stat` records the `S_IFLNK` mode bits, not an xattr on a real symlink. The row stays divergent only for the never-preserved privileged namespaces above | | `-H`, `--hard-links` | Preserve hard links | ✅ Parity | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below | | `-D` | Same as --devices --specials | ✅ Parity | Implies `--devices --specials`. `-D` was unassigned in FastSync (verified: no collision), so it is free to imply both device-node and special-file preservation. As of protocol 2.23.0 `--specials` genuinely covers **both FIFOs and unix sockets**, so `-D` covers the full rsync set. See the `--devices`/`--specials` rows and the Phase-4 devices notes below | | `--devices` | Preserve device files | ⚠️ Caveat | Recreates char/block device nodes with `mknodat` (type + rdev strictly validated, confined fd-relative below the receive root). A device whose `mknodat` fails with `EPERM`/`EACCES` (no `CAP_MKNOD`, or super-user activity forbidden) is a **per-entry failure**: FastSync logs `cannot create device ...` (rsync logs `mknod ... failed`), counts it, **continues with the remaining files**, and ends the run with a non-OK terminal status. rsync parity: rsync likewise continues and exits partial (23). Residuals: (1) FastSync's default AUTO still *attempts* the node on a non-root receiver and therefore reports the per-entry failure, whereas rsync without `--super` silently ignores `--devices` and skips the non-regular entry with exit 0 — use `--no-super` for rsync's silent-skip behavior; (2) FastSync's process exit code for a receiver-side per-entry failure is the general error code 1, not rsync's partial 23 (a client exit-code-mapping residual that applies to every receiver file error, not just this branch); (3) with `--remove-source-files`, the non-OK terminal status means successfully transferred sources are not removed on a partial run. `--specials` (FIFOs and unix sockets) keeps the unprivileged skip path and remains parity | @@ -964,7 +964,7 @@ These are the last compatibility items and the closing phase toward rsync flag p **Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity. -**Honest status after the parity 2.29 cycle (protocol 2.28.0, no wire change), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, the triage cycle, and a later no-wire parity pass.** ✅ Parity 119 / ⚠️ Caveat 14 / ❌ Divergent 24 = 157 rows. The no-wire parity pass accepted `--inc-recursive`/`--no-inc-recursive` as inert no-ops (❌ → ✅, since FastSync's full scan is rsync's `--no-inc-recursive` and the destination is identical), narrowed the `--temp-dir` divergence by accepting an absolute path that canonicalizes inside the receive root (the row stays ❌ for out-of-root absolute paths), closed the `--delete-before` phase-0 divergence (⚠️ → ✅: both the single-threaded and the `--threads` data passes now replay the pre-scan file list, so a source file created after the scan is neither transferred nor kept, matching rsync), and moved `--fake-super` and `--devices` ❌ → ⚠️ (`--fake-super` now writes/reads rsync's exact `user.rsync.%stat` key and ` , :` grammar, interoperating with real rsync 3.4.1 for regular files and faking char/block devices as regular files carrying the real rdev; `--devices` now logs a failed device `mknod` as a per-entry failure that continues the transfer instead of a silent non-root skip — see those rows for the remaining directory-faking and exit-code residuals). A review pass then hardened the fake-super stat parser (strict range-checked parsing), made rsync-style daemon modules read-only by default with a startup warning for accepted-but-unenforced access-control keys, and extended the `--delete-before` replay to the `--threads` path. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--filter`, `-F`, the three basis-dir options, `-y/--fuzzy`, `--fake-super`, and `--devices`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP +**Honest status after the parity 2.29 cycle (protocol 2.29.0 since the symlink-xattr wire wave, which adds no config-frame field and leaves this matrix unchanged), updated by the parity cycle 2.29 pass, the audit-cycle follow-ups, the triage cycle, and a later no-wire parity pass.** ✅ Parity 119 / ⚠️ Caveat 14 / ❌ Divergent 24 = 157 rows. The no-wire parity pass accepted `--inc-recursive`/`--no-inc-recursive` as inert no-ops (❌ → ✅, since FastSync's full scan is rsync's `--no-inc-recursive` and the destination is identical), narrowed the `--temp-dir` divergence by accepting an absolute path that canonicalizes inside the receive root (the row stays ❌ for out-of-root absolute paths), closed the `--delete-before` phase-0 divergence (⚠️ → ✅: both the single-threaded and the `--threads` data passes now replay the pre-scan file list, so a source file created after the scan is neither transferred nor kept, matching rsync), and moved `--fake-super` and `--devices` ❌ → ⚠️ (`--fake-super` now writes/reads rsync's exact `user.rsync.%stat` key and ` , :` grammar, interoperating with real rsync 3.4.1 for regular files and faking char/block devices as regular files carrying the real rdev; `--devices` now logs a failed device `mknod` as a per-entry failure that continues the transfer instead of a silent non-root skip — see those rows for the remaining directory-faking and exit-code residuals). A review pass then hardened the fake-super stat parser (strict range-checked parsing), made rsync-style daemon modules read-only by default with a startup warning for accepted-but-unenforced access-control keys, and extended the `--delete-before` replay to the `--threads` path. The 2.29 cycle closed the scanner-order, delete-timing, relative-basis, and fuzzy-eligibility residuals (moving `-n`/`--delete`/`--del`/`--delete-delay` to ✅) and improved the `--info`/`--stats`/`--debug` partial rows; the triage cycle moved `-F` and `-i`/`--itemize-changes` ✅ → ⚠️ for their documented residuals. The remaining ⚠️ rows are `--info`, `--debug`, `--msgs2stderr`, `--stats`, `--progress`, `-i`, `--filter`, `-F`, the three basis-dir options, `-y/--fuzzy`, `--fake-super`, and `--devices`. Earlier: **Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel / receiver filter engine); the wire parity-track-4a pass later added that receiver filter engine, flipping `--filter=RULE` back to ✅ (see above; the diff --git a/src/client/client_send.c b/src/client/client_send.c index c8dd8bd..16e386c 100644 --- a/src/client/client_send.c +++ b/src/client/client_send.c @@ -654,7 +654,11 @@ static bool send_symlink_entry(const Client* client, File* file, const Config* c if (!send_status(fd, STATUS_SYMLINK) || !send_wire_str(fd, file_wire_path(file)) || !send_wire_str(fd, file->symlink_target)) return false; - return !config->use_metadata || metadata_send(fd, file->metadata); + if (config->use_metadata && !metadata_send(fd, file->metadata)) + return false; + /* Symlink xattrs/ACLs (-X/-A) ride the same trailing block as regular files + and directories when the xattr transport was negotiated. */ + return !config->use_xattrs || xattr_send(fd, file->xattrs); } // Send a single file directly via sendfile (non-incremental path). diff --git a/src/client/scanner_filter.c b/src/client/scanner_filter.c index 225d9f2..374aa45 100644 --- a/src/client/scanner_filter.c +++ b/src/client/scanner_filter.c @@ -282,11 +282,15 @@ bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* } /* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to - * read xattrs is non-fatal: the file is transferred without them. */ + * read xattrs is non-fatal: the file is transferred without them. A symlink + * entry reads the LINK's own xattrs (never the referent's) with the no-follow + * variant; on Linux the VFS refuses xattrs on symlinks, so that yields NULL. */ void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) { if (!scanner || !file || !(scanner->options.preserve_xattrs || scanner->options.preserve_acls)) return; - file->xattrs = xattr_capture_path(file->path, scanner->options.preserve_acls); + file->xattrs = file->is_symlink + ? xattr_capture_path_nofollow(file->path, scanner->options.preserve_acls) + : xattr_capture_path(file->path, scanner->options.preserve_acls); } /* Apply --hard-links (-H) detection to one regular File. On a sibling (a diff --git a/src/client/scanner_parallel.c b/src/client/scanner_parallel.c index 9fa5151..f82eccb 100644 --- a/src/client/scanner_parallel.c +++ b/src/client/scanner_parallel.c @@ -428,7 +428,9 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo } if ((options->preserve_xattrs || options->preserve_acls) && !(file->link_group != 0 && !file->link_first)) - file->xattrs = xattr_capture_path(file->path, options->preserve_acls); + file->xattrs = file->is_symlink + ? xattr_capture_path_nofollow(file->path, options->preserve_acls) + : xattr_capture_path(file->path, options->preserve_acls); if (!array_list_add(root_files, file)) { free(rel); file_destroy(file); diff --git a/src/shared/config.h b/src/shared/config.h index fd88991..a7e5bf7 100644 --- a/src/shared/config.h +++ b/src/shared/config.h @@ -83,7 +83,7 @@ typedef struct { typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF = 2 } SuperMode; /* =========================================================================== - * Config wire-field table (single source of truth for protocol 2.28.0). + * Config wire-field table (single source of truth for protocol 2.29.0). * * Every field below crosses the wire. The table is the ONLY place a * serialized field is named: config.h expands CONFIG_WIRE_FIELDS() to declare @@ -1071,7 +1071,20 @@ typedef struct Config { * filter rules so the receiver can protect DESTINATION-ONLY entries from * --delete with `protect`/`risk` rules (rsync parity). The block appends after * compression_algo; see CONFIG_WIRE_PROTECT_FIELDS. */ -#define PROTOCOL_VERSION "2.28.0" +/* (10) Symlink xattrs/ACLs (protocol 2.29.0): the config-frame LAYOUT is + * unchanged (the derived use_xattrs bit already crosses the wire), but the + * STATUS_SYMLINK frame BODY grows a trailing bounded xattr block when -X/-A is + * negotiated -- exactly the block STATUS_MKDIR, STATUS_DIR_TIMES and regular + * files already carry. The sender captures the symlink's OWN xattrs with + * llistxattr/lgetxattr (so it can never attach the REFERENT's attributes to the + * link) and the receiver re-applies them to the link itself with lsetxattr on a + * confined /proc/self/fd// path (there is no *at xattr syscall and + * fsetxattr cannot target a symlink). A 2.28 peer that does not consume the new + * trailing block would desynchronize after every symlink, so the protocol + * version must bump; the strict same-version handshake (config_receive rejects a + * mismatched version before parsing anything else) keeps a 2.29 client and a + * 2.28 server from ever reaching that state. */ +#define PROTOCOL_VERSION "2.29.0" #define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024) /* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */ #define MAX_BASIS_DIRS 64 diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c index a9e02ab..be49127 100644 --- a/src/shared/file_receive.c +++ b/src/shared/file_receive.c @@ -482,6 +482,14 @@ File* file_receive_symlink(int file_descriptor, const Config* config) { return NULL; } } + /* Symlink xattrs/ACLs (-X/-A) arrive in the same trailing block as the other + entry kinds; the block is present iff use_xattrs (which itself implies + use_metadata, so the metadata frame above is always consumed first). */ + if (config && !receive_file_xattrs(file, file_descriptor, config)) { + file_destroy(file); + free(target); + return NULL; + } file->is_symlink = true; file->symlink_target = target; return file; diff --git a/src/shared/file_save.c b/src/shared/file_save.c index d0d4672..3c49ebd 100644 --- a/src/shared/file_save.c +++ b/src/shared/file_save.c @@ -923,6 +923,19 @@ static FileSaveResult file_save_symlink_to_disk(const FileSavePlan* plan, bool* ok = file_restore_symlink_metadata(link_path, file->metadata, link_policy, config->omit_link_times); } + /* -X/-A: apply the symlink's OWN xattrs with a no-follow primitive. The + confined parent directory is the anchor and the final component is applied + with lsetxattr, so the referent is never touched. Best-effort: on Linux + the VFS refuses xattrs on symlinks, so this is normally a no-op. */ + if (ok && config && config->use_xattrs && file->xattrs) { + char* leaf = NULL; + int parent_fd = file_open_secure_parent(link_path, &leaf, false); + if (parent_fd >= 0) { + xattr_apply_path_nofollow(parent_fd, leaf, file->xattrs); + close(parent_fd); + } + free(leaf); + } if (ok && created && !link_existed) *created = true; free(link_path); diff --git a/src/shared/xattr.c b/src/shared/xattr.c index 91f82c4..e1ca859 100644 --- a/src/shared/xattr.c +++ b/src/shared/xattr.c @@ -134,16 +134,23 @@ static bool xattr_name_is_posix_acl(const char* name) { /* ---- SENDER: capture ---- */ -FileXattrList* xattr_capture_path(const char* path, bool preserve_acls) { +/* The two syscall families differ only in whether the FINAL component is + * followed (`listxattr`/`getxattr` follow; `llistxattr`/`lgetxattr` do not), so + * one common implementation backs both public entry points. */ +typedef ssize_t (*XattrListFn)(const char* path, char* list, size_t size); +typedef ssize_t (*XattrGetFn)(const char* path, const char* name, void* value, size_t size); + +static FileXattrList* xattr_capture_common(const char* path, bool preserve_acls, + XattrListFn list_fn, XattrGetFn get_fn) { if (!path) return NULL; - ssize_t list_size = listxattr(path, NULL, 0); + ssize_t list_size = list_fn(path, NULL, 0); if (list_size <= 0) return NULL; /* no xattrs, ENOTSUP, or error: nothing appliable */ char* names = malloc((size_t)list_size); if (!names) return NULL; - ssize_t got = listxattr(path, names, (size_t)list_size); + ssize_t got = list_fn(path, names, (size_t)list_size); if (got < 0) { free(names); return NULL; @@ -166,7 +173,7 @@ FileXattrList* xattr_capture_path(const char* path, bool preserve_acls) { negotiated. Without it a plain -X capture never carries an ACL. */ if (!xattr_name_appliable(name, preserve_acls)) continue; - ssize_t value_size = getxattr(path, name, NULL, 0); + ssize_t value_size = get_fn(path, name, NULL, 0); if (value_size < 0) continue; if (value_size > XATTR_VALUE_MAX) @@ -179,7 +186,7 @@ FileXattrList* xattr_capture_path(const char* path, bool preserve_acls) { free(names); return NULL; } - ssize_t read_len = getxattr(path, name, buffer, (size_t)value_size); + ssize_t read_len = get_fn(path, name, buffer, (size_t)value_size); if (read_len < 0 || read_len != value_size) { free(buffer); continue; @@ -201,6 +208,14 @@ FileXattrList* xattr_capture_path(const char* path, bool preserve_acls) { return list; } +FileXattrList* xattr_capture_path(const char* path, bool preserve_acls) { + return xattr_capture_common(path, preserve_acls, listxattr, getxattr); +} + +FileXattrList* xattr_capture_path_nofollow(const char* path, bool preserve_acls) { + return xattr_capture_common(path, preserve_acls, llistxattr, lgetxattr); +} + /* ---- WIRE ---- */ bool xattr_send(int fd, const FileXattrList* list) { @@ -364,6 +379,49 @@ bool xattr_apply_fd(int fd, const FileXattrList* list) { return true; } +/* Symlink counterpart of xattr_apply_fd(): target the link ITSELF, never its + * referent. fsetxattr cannot be used (no *at xattr syscall exists, and the + * kernel rejects xattr syscalls on an O_PATH descriptor), so the already-open, + * confinement-checked parent directory is addressed through /proc/self/fd and + * the final component is applied with lsetxattr, which does not follow it. */ +bool xattr_apply_path_nofollow(int parent_fd, const char* leaf, const FileXattrList* list) { + if (parent_fd < 0 || !leaf || leaf[0] == '\0' || strchr(leaf, '/') != NULL || !list) + return false; + if (list->count == 0) + return true; + char prefix[64]; + int prefix_len = snprintf(prefix, sizeof(prefix), "/proc/self/fd/%d/", parent_fd); + if (prefix_len < 0 || (size_t)prefix_len >= sizeof(prefix)) + return false; + size_t leaf_len = strlen(leaf); + char* path = malloc((size_t)prefix_len + leaf_len + 1); + if (!path) + return false; + memcpy(path, prefix, (size_t)prefix_len); + memcpy(path + prefix_len, leaf, leaf_len + 1); + bool warned = false; + int first_errno = 0; + for (int i = 0; i < list->count; i++) { + const FileXattr* xa = &list->items[i]; + /* Defense in depth: even a hand-crafted list can never apply the reserved + --fake-super key (only fake_super_store_fd may write it). */ + if (strcmp(xa->name, FAKESUPER_XATTR) == 0) + continue; + if (lsetxattr(path, xa->name, xa->value, xa->value_len, 0) != 0) { + if (!warned) { + warned = true; + first_errno = errno; + } + } + } + if (warned) + log_message(LOG_LEVEL_WARNING, + "could not set one or more xattrs on the destination symlink: %s", + strerror(first_errno)); + free(path); + return true; +} + /* ---- --fake-super: park ownership/mode/rdev in a reserved xattr ---- */ void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, uint32_t rdev_major, diff --git a/src/shared/xattr.h b/src/shared/xattr.h index d9858eb..6628aeb 100644 --- a/src/shared/xattr.h +++ b/src/shared/xattr.h @@ -26,8 +26,11 @@ * and total bytes) on BOTH ends to prevent OOM/memory abuse; an oversized * or malformed frame is a clean protocol rejection, never an allocation * blowup. - * * Application is confined to the exact destination file descriptor - * (fsetxattr on the just-written fd), never a caller-controlled path. + * * Application is confined to the exact destination entry: fsetxattr on the + * just-written fd for regular files/directories, and for a symlink an + * lsetxattr on "/proc/self/fd//" reached through the + * already-opened, confinement-checked parent directory -- never a + * caller-controlled path, and never following the link. */ /* Reserved key used by --fake-super to park the source's privileged ownership @@ -79,6 +82,17 @@ bool xattr_name_appliable(const char* name, bool preserve_acls); * distinct from NULL. */ FileXattrList* xattr_capture_path(const char* path, bool preserve_acls); +/* Sender: like xattr_capture_path() but reads the xattrs of `path` ITSELF, + * never following a final symlink (llistxattr/lgetxattr). A symlink entry must + * use this so the scanner never captures the REFERENT's attributes onto the + * link (the path-following variant would). On Linux the VFS refuses to + * associate xattrs with symlinks at all, so this normally returns NULL; it is + * still correct and portable for a filesystem/platform that supports them. + * The same whitelist/bounds as xattr_capture_path() apply. Returns NULL when + * the link has no appliable xattrs (or the filesystem does not support them); + * an empty-but-valid list is never returned distinct from NULL. */ +FileXattrList* xattr_capture_path_nofollow(const char* path, bool preserve_acls); + /* Wire: bounded serialization. xattr_send returns false on write failure; an * empty/NULL list transmits a zero-count block. xattr_receive returns NULL and * sets *ok = 0 on any malformed / oversized / non-whitelisted entry. When @@ -94,6 +108,19 @@ FileXattrList* xattr_receive(int fd, int* ok, bool preserve_acls); * true when apply was attempted (allowing callers to treat it as best-effort). */ bool xattr_apply_fd(int fd, const FileXattrList* list); +/* Receiver: apply every entry to the symlink named by (parent_fd, leaf) WITHOUT + * following it, via lsetxattr() on the confined path + * "/proc/self/fd//". A symlink cannot be targeted by the + * fd-relative fsetxattr() path: there is no *at() xattr syscall and the kernel + * rejects xattr syscalls on an O_PATH descriptor, so the already-opened, + * confinement-checked parent directory is the anchor and only the final + * component is the (no-follow) link. `leaf` must be a single path component. + * Best-effort exactly like xattr_apply_fd(): a per-attribute failure (on Linux + * every set on a symlink fails with EPERM) is logged once and skipped, never + * fatal. Returns false only for an invalid anchor/list; true when an apply was + * attempted. The reserved --fake-super key is never applied. */ +bool xattr_apply_path_nofollow(int parent_fd, const char* leaf, const FileXattrList* list); + /* --fake-super: write the source uid/gid/mode/rdev record into the reserved * FAKESUPER_XATTR on `fd`, using rsync 3.4.1's exact grammar (see the key * comment above). `mode` is the full st_mode including its S_IFMT bits. diff --git a/tests/integration/test_fault_injection.py b/tests/integration/test_fault_injection.py index 7b8eb46..8c67f6d 100644 --- a/tests/integration/test_fault_injection.py +++ b/tests/integration/test_fault_injection.py @@ -36,7 +36,7 @@ from common import ( # noqa: E402 verify_transfer, ) -PROTOCOL_VERSION = b"2.28.0" +PROTOCOL_VERSION = b"2.29.0" STATUS_MANIFEST = 5 STATUS_OK = 0 diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index fc850d0..d14f58a 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -6637,6 +6637,56 @@ class TestExtendedAttributes: received = get_dest_received_dir(dest, source) assert os.getxattr(os.path.join(received, "data.txt"), "user.k") == b"v" + @pytest.mark.ci + def test_symlink_own_xattrs_never_referent(self, shared_server): + """Protocol 2.29.0: a symlink's STATUS_SYMLINK frame carries a trailing + xattr block captured with llistxattr/lgetxattr (no follow) and applied + with lsetxattr on the link itself. Linux's VFS refuses to associate + xattrs with a symlink at all, so the portable guarantee asserted here is + the no-follow one: a referent that carries user.* must NOT have those + attributes appear on the destination symlink entry (the old + path-following capture would have copied the referent's attrs onto the + link). On a platform/filesystem that does support symlink xattrs the + full round-trip of the link's own attribute is asserted too.""" + source, dest = self._source_and_dest("symlink_xattr") + target = os.path.join(source, "target.txt") + with open(target, "wb") as fh: + fh.write(b"referent payload\n") + if not _xattr_supported(target): + pytest.skip("filesystem does not support user xattrs") + os.setxattr(target, "user.referent-only", b"referent-value") + + link = os.path.join(source, "link") + os.symlink("target.txt", link) + link_xattr_supported = False + try: + os.setxattr(link, "user.link-own", b"link-value", follow_symlinks=False) + link_xattr_supported = os.getxattr( + link, "user.link-own", follow_symlinks=False + ) == b"link-value" + except (OSError, AttributeError, NotImplementedError): + link_xattr_supported = False + + result, _ = run_client(source, dest, flags=["-aX"], port=shared_server.port) + assert result.returncode == 0, \ + f"-aX symlink sync failed: {(result.stderr or result.stdout)[:300]}" + received = get_dest_received_dir(dest, source) + dst_link = os.path.join(received, "link") + assert os.path.islink(dst_link), "destination link entry is not a symlink" + assert os.readlink(dst_link) == "target.txt" + + # The no-follow guarantee: the referent's attribute must never leak onto + # the symlink entry. + link_names = os.listxattr(dst_link, follow_symlinks=False) + assert "user.referent-only" not in link_names, ( + "the destination symlink captured its REFERENT's xattr " + "(path-following capture bug)" + ) + if link_xattr_supported: + assert os.getxattr( + dst_link, "user.link-own", follow_symlinks=False + ) == b"link-value", "the symlink's own xattr did not round-trip" + @pytest.mark.ci def test_acls_via_posix_acl_xattr(self, shared_server): source, dest = self._source_and_dest("acl") diff --git a/tests/integration/test_preflight.py b/tests/integration/test_preflight.py index ab62da7..7a47661 100644 --- a/tests/integration/test_preflight.py +++ b/tests/integration/test_preflight.py @@ -133,14 +133,14 @@ def _seed_protocol_source(source): class TestProtocol: @pytest.mark.ci def test_protocol_current_version_accepted(self, shared_server): - """--protocol=2.28.0 (the current PROTOCOL_VERSION) is accepted and the + """--protocol=2.29.0 (the current PROTOCOL_VERSION) is accepted and the transfer completes normally.""" source = os.path.join(TEST_DATA_DIR, "proto_ok_src") dest = os.path.join(TEST_DATA_DIR, "proto_ok_dst") shutil.rmtree(dest, ignore_errors=True) os.makedirs(dest) _seed_protocol_source(source) - result, _ = run_client(source, dest, flags=["--protocol=2.28.0"], + result, _ = run_client(source, dest, flags=["--protocol=2.29.0"], port=shared_server.port) assert result.returncode == 0, \ f"--protocol current run failed: {(result.stderr or result.stdout)[:400]}" diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c index 64f9610..d43a9a1 100644 --- a/tests/test_client_cli.c +++ b/tests/test_client_cli.c @@ -352,7 +352,7 @@ static void test_parse_args_protocol_accept_current() { Config* cfg = valid_client_config(); EXPECT_NOT_NULL(cfg); char* argv_equals[] = {"fastsync", "--source-dir", "/src", - "--dest-dir", "/dst", "--protocol=2.28.0"}; + "--dest-dir", "/dst", "--protocol=2.29.0"}; int positional_args[2]; int positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 6, argv_equals, positional_args, &positional_count), 0); @@ -362,7 +362,7 @@ static void test_parse_args_protocol_accept_current() { cfg = valid_client_config(); EXPECT_NOT_NULL(cfg); char* argv_space[] = {"fastsync", "--source-dir", "/src", "--dest-dir", - "/dst", "--protocol", "2.28.0"}; + "/dst", "--protocol", "2.29.0"}; positional_count = 0; EXPECT_EQ_INT(parse_args(cfg, 7, argv_space, positional_args, &positional_count), 0); EXPECT_EQ_STR(cfg->version, PROTOCOL_VERSION); @@ -375,7 +375,7 @@ static void test_parse_args_protocol_rejects_other_versions() { static const char* const bad_versions[] = {"2.17", "2.16", "2.15.0", "2.16.0", "2.17.0", "2.18.0", "2.19.0", "2.20.0", "2.21.0", "2.22.0", "2.23.0", "2.24.0", "2.25.0", "2.26.0", "2.27.0", - "216", "31", "abc", ""}; + "2.28.0", "216", "31", "abc", ""}; for (size_t i = 0; i < sizeof(bad_versions) / sizeof(bad_versions[0]); i++) { Config* cfg = valid_client_config(); EXPECT_NOT_NULL(cfg); diff --git a/tests/test_config.c b/tests/test_config.c index 4c87b34..890de8e 100644 --- a/tests/test_config.c +++ b/tests/test_config.c @@ -2924,7 +2924,7 @@ static void golden_config_populate(Config* c) { array_list_add(c->filters, str_dup("- /sub/dir/")); } -/* The pinned golden frame (protocol 2.28.0). The values below are the only +/* The pinned golden frame (protocol 2.29.0). The values below are the only * thing that ties the generated table to the historical wire format; update * them ONLY with a PROTOCOL_VERSION bump and a documented reason. The 2.24.0 * delete-plan wave changed only the version string; 2.25.0 appended the @@ -2933,10 +2933,13 @@ static void golden_config_populate(Config* c) { * appended the receiver-side delete-protection rule block (the STATUS_STATS * body also grew, but that is not part of this frame). Track 5a appends the * FastSync-only verify_basis bool to the basis block WITHOUT a version bump - * (project decision), so the frame grew by one int to 886 bytes. The - * byte-exact values are recomputed for the merged layout. */ + * (project decision), so the frame grew by one int to 886 bytes. The 2.29.0 + * symlink-xattr wave changes only the version string: the config-frame layout + * is unchanged (use_xattrs already crosses the wire); the STATUS_SYMLINK frame + * body grows instead. The byte-exact values are recomputed for the merged + * layout. */ #define GOLDEN_WIRE_LEN 886 -#define GOLDEN_WIRE_HASH 5809509022716816757ULL +#define GOLDEN_WIRE_HASH 17827864270611927842ULL static unsigned long long fnv1a_64(const unsigned char* buf, size_t len) { unsigned long long h = 1469598103934665603ULL; @@ -3018,7 +3021,7 @@ static unsigned long long capture_wire_hash(const Config* cfg, size_t* out_len) return h; } -/* Byte-for-byte wire compatibility guard (protocol 2.28.0). The expected hash +/* Byte-for-byte wire compatibility guard (protocol 2.29.0). The expected hash * pins the pre-X-macro byte stream; the refactor MUST NOT change it. */ static void test_config_wire_golden() { if (is_running_under_valgrind()) diff --git a/tests/test_xattr.c b/tests/test_xattr.c index 9a0ad1a..8fc680a 100644 --- a/tests/test_xattr.c +++ b/tests/test_xattr.c @@ -1,9 +1,12 @@ #include "test_xattr.h" #include "xattr.h" +#include "charset.h" #include "config.h" #include "file.h" +#include "file_receive.h" #include "file_save.h" #include "identity.h" +#include "metadata.h" #include "protocol.h" #include "test_utils.h" #include @@ -11,6 +14,7 @@ #include #include #include +#include #include #include #include @@ -660,8 +664,187 @@ static void test_file_save_directory_applies_xattrs() { rmdir(root); } +/* Symlink xattrs (protocol 2.29.0): the no-follow capture must read the LINK's + * OWN attributes and never the REFERENT's. Linux's VFS refuses to associate + * xattrs with a symlink at all, so the nofollow capture returns NULL while the + * path-following capture sees the referent's attribute -- which is exactly the + * bug the no-follow variant exists to prevent (a symlink entry must not carry + * its target's attributes). Guarded on filesystem xattr support. */ +static void test_xattr_capture_symlink_nofollow() { + const char* target = "test_symlink_xattr_capture_target"; + const char* link = "test_symlink_xattr_capture_link"; + unlink(link); + unlink(target); + int fd = open(target, O_WRONLY | O_CREAT | O_TRUNC, 0600); + if (fd < 0) + return; + bool has_xattr = setxattr(target, "user.symref", "referent", 8, 0) == 0; + close(fd); + if (!has_xattr) { + unlink(target); + return; /* filesystem without xattr support */ + } + if (symlink(target, link) != 0) { + unlink(target); + return; + } + + /* The no-follow capture must never pick up the referent's attributes. */ + FileXattrList* nofollow = xattr_capture_path_nofollow(link, false); + EXPECT_NULL(nofollow); + + /* The path-following capture does, proving the referent really carries one + and that the no-follow variant differs. */ + FileXattrList* follow = xattr_capture_path(link, false); + bool saw = false; + for (int i = 0; follow && i < follow->count; i++) { + if (strcmp(follow->items[i].name, "user.symref") == 0) + saw = true; + } + EXPECT_TRUE(saw); + xattr_list_free(follow); + xattr_list_free(nofollow); + unlink(link); + unlink(target); +} + +/* Symlink xattrs (protocol 2.29.0): the no-follow apply must target the LINK, + * never its referent. On Linux the LSETXATTR is refused (the VFS does not + * allow symlink xattrs), but the critical guarantee is observable: the + * referent's attributes are UNCHANGED. A regression from lsetxattr to the + * path-following setxattr would rewrite the referent here and fail this test. */ +static void test_xattr_apply_path_nofollow_does_not_follow() { + const char* root = "test_symlink_xattr_apply_tmp"; + const char* target = "test_symlink_xattr_apply_tmp/target"; + const char* link = "test_symlink_xattr_apply_tmp/link"; + unlink(link); + unlink(target); + rmdir(root); + EXPECT_EQ_INT(mkdir(root, 0700), 0); + int tfd = open(target, O_WRONLY | O_CREAT | O_TRUNC, 0600); + if (tfd < 0) { + rmdir(root); + return; + } + bool has_xattr = setxattr(target, "user.orig", "orig", 4, 0) == 0; + close(tfd); + if (!has_xattr) { + unlink(target); + rmdir(root); + return; /* filesystem without xattr support */ + } + EXPECT_EQ_INT(symlink("target", link), 0); + + FileXattrList* list = xattr_list_new(); + EXPECT_NOT_NULL(list); + EXPECT_TRUE(xattr_list_append(list, "user.orig", "hacked", 6)); + EXPECT_TRUE(xattr_list_append(list, "user.added", "x", 1)); + + /* Invalid anchors are refused before any syscall (no fd/leaf/list). */ + EXPECT_FALSE(xattr_apply_path_nofollow(-1, "link", list)); + EXPECT_FALSE(xattr_apply_path_nofollow(0, "", list)); + EXPECT_FALSE(xattr_apply_path_nofollow(0, "a/b", list)); + EXPECT_FALSE(xattr_apply_path_nofollow(0, "link", NULL)); + + /* The confined parent directory is the anchor; the final component is the + link. Best-effort: returns true even when the kernel refuses. */ + int dir_fd = open(root, O_RDONLY | O_DIRECTORY); + EXPECT_TRUE(dir_fd >= 0); + EXPECT_TRUE(xattr_apply_path_nofollow(dir_fd, "link", list)); + close(dir_fd); + + /* The referent must be untouched: a following apply would have set user.orig + to "hacked" and created user.added on the target. */ + char buf[16]; + ssize_t got = getxattr(target, "user.orig", buf, sizeof(buf)); + EXPECT_EQ_INT(4, (int)got); + if (got == 4) + EXPECT_TRUE(memcmp(buf, "orig", 4) == 0); + EXPECT_TRUE(getxattr(target, "user.added", buf, sizeof(buf)) < 0); + + /* If the platform DOES support symlink xattrs, they must have landed on the + link itself; on Linux the VFS refuses them, so the link stays empty. */ + if (llistxattr(link, NULL, 0) > 0) { + ssize_t n = lgetxattr(link, "user.added", buf, sizeof(buf)); + EXPECT_EQ_INT(1, (int)n); + if (n == 1) + EXPECT_TRUE(buf[0] == 'x'); + } + + xattr_list_free(list); + unlink(link); + unlink(target); + rmdir(root); +} + +/* Protocol 2.29.0: a STATUS_SYMLINK frame followed by an -X/-A xattr block is + * decoded by file_receive_symlink() with the block attached to the File. This + * is the wire round-trip for the new trailing symlink xattr block. */ +static void run_recv_symlink_with_xattrs(int fd) { + /* Stack-allocated so the forked child leaks nothing at _exit() (a + config_create() in the parent would be inherited and never freed here). */ + Config config; + memset(&config, 0, sizeof(config)); + config.use_metadata = true; + config.use_xattrs = true; + config.preserve_xattrs = true; + File* file = file_receive_symlink(fd, &config); + if (!file) + _exit(1); + bool ok = file->is_symlink && file->symlink_target != NULL && + strcmp(file->symlink_target, "target") == 0; + ok = ok && file->xattrs != NULL && file->xattrs->count == 1 && + strcmp(file->xattrs->items[0].name, "user.sym") == 0 && + file->xattrs->items[0].value_len == 3 && memcmp(file->xattrs->items[0].value, "sym", 3) == 0; + file_destroy(file); + _exit(ok ? 0 : 1); +} + +static void test_symlink_frame_carries_xattrs() { + int p[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0); + io_set_fds(p[0], p[1]); + io_set_bwlimit(0); + + pid_t pid = fork(); + if (pid == 0) { + close(p[1]); + io_set_fds(p[0], p[0]); + io_set_bwlimit(0); + run_recv_symlink_with_xattrs(p[0]); + } + close(p[0]); + io_set_fds(p[1], p[1]); + io_set_bwlimit(0); + + FileMetadata m; + memset(&m, 0, sizeof(m)); + m.mode = S_IFLNK | 0777; + m.uid = (uint32_t)geteuid(); + m.gid = (uint32_t)getegid(); + m.mtime_sec = 1700000000; + + FileXattrList* list = xattr_list_new(); + EXPECT_NOT_NULL(list); + EXPECT_TRUE(xattr_list_append(list, "user.sym", "sym", 3)); + + /* Exactly the sender's order: path, target, metadata, xattr block. */ + bool wrote = send_wire_str(p[1], "link") && send_wire_str(p[1], "target") && + metadata_send(p[1], &m) && xattr_send(p[1], list); + xattr_list_free(list); + close(p[1]); + + int status = 0; + waitpid(pid, &status, 0); + EXPECT_TRUE(wrote); + EXPECT_TRUE(WIFEXITED(status) && WEXITSTATUS(status) == 0); +} + void test_xattr() { test_xattr_list_clone(); + test_xattr_capture_symlink_nofollow(); + test_xattr_apply_path_nofollow_does_not_follow(); + test_symlink_frame_carries_xattrs(); test_xattr_wire_roundtrip(); test_xattr_reject_privileged_namespace(); test_xattr_reject_oversized_value(); -- 2.54.0 From b88acdbd3c03882af40009c622db850d30f9b1f1 Mon Sep 17 00:00:00 2001 From: TapTap Date: Wed, 23 Sep 2026 01:01:18 +0200 Subject: [PATCH 65/68] fix(xattr): whitelist path-based symlink apply; strengthen symlink xattr tests --- src/shared/file_save.c | 8 ++-- src/shared/xattr.c | 20 +++++++--- src/shared/xattr.h | 32 +++++++++++---- tests/integration/test_features.py | 16 +++++++- tests/test_xattr.c | 63 +++++++++++++++++++++++++++--- 5 files changed, 116 insertions(+), 23 deletions(-) diff --git a/src/shared/file_save.c b/src/shared/file_save.c index 3c49ebd..f51215b 100644 --- a/src/shared/file_save.c +++ b/src/shared/file_save.c @@ -926,12 +926,14 @@ static FileSaveResult file_save_symlink_to_disk(const FileSavePlan* plan, bool* /* -X/-A: apply the symlink's OWN xattrs with a no-follow primitive. The confined parent directory is the anchor and the final component is applied with lsetxattr, so the referent is never touched. Best-effort: on Linux - the VFS refuses xattrs on symlinks, so this is normally a no-op. */ - if (ok && config && config->use_xattrs && file->xattrs) { + the VFS refuses xattrs on symlinks, so this is normally a no-op. Hoist the + empty-list check so the common Linux case (NULL/empty xattrs) does not pay + an open/close of the parent per symlink. */ + if (ok && config && config->use_xattrs && file->xattrs && file->xattrs->count > 0) { char* leaf = NULL; int parent_fd = file_open_secure_parent(link_path, &leaf, false); if (parent_fd >= 0) { - xattr_apply_path_nofollow(parent_fd, leaf, file->xattrs); + xattr_apply_path_nofollow(parent_fd, leaf, file->xattrs, config->preserve_acls); close(parent_fd); } free(leaf); diff --git a/src/shared/xattr.c b/src/shared/xattr.c index e1ca859..aa0e747 100644 --- a/src/shared/xattr.c +++ b/src/shared/xattr.c @@ -383,8 +383,17 @@ bool xattr_apply_fd(int fd, const FileXattrList* list) { * referent. fsetxattr cannot be used (no *at xattr syscall exists, and the * kernel rejects xattr syscalls on an O_PATH descriptor), so the already-open, * confinement-checked parent directory is addressed through /proc/self/fd and - * the final component is applied with lsetxattr, which does not follow it. */ -bool xattr_apply_path_nofollow(int parent_fd, const char* leaf, const FileXattrList* list) { + * the final component is applied with lsetxattr, which does not follow it. + * + * The list is trusted to come from xattr_receive() (already whitelisted), but + * every name is re-validated here so this path-based primitive is confined on + * its own -- this is the only apply primitive that addresses a path, and the + * header promises a whitelisted apply. The apply is best-effort: if /proc is + * not mounted (the anchor cannot be formed) or the kernel refuses the set, the + * failure is skipped and never fails the transfer. See xattr.h for the bounded + * residual TOCTOU between link creation and lsetxattr. */ +bool xattr_apply_path_nofollow(int parent_fd, const char* leaf, const FileXattrList* list, + bool preserve_acls) { if (parent_fd < 0 || !leaf || leaf[0] == '\0' || strchr(leaf, '/') != NULL || !list) return false; if (list->count == 0) @@ -403,9 +412,10 @@ bool xattr_apply_path_nofollow(int parent_fd, const char* leaf, const FileXattrL int first_errno = 0; for (int i = 0; i < list->count; i++) { const FileXattr* xa = &list->items[i]; - /* Defense in depth: even a hand-crafted list can never apply the reserved - --fake-super key (only fake_super_store_fd may write it). */ - if (strcmp(xa->name, FAKESUPER_XATTR) == 0) + /* Defense in depth: re-validate against the receiver's full whitelist, so a + hand-crafted list can never apply a privileged namespace or the reserved + --fake-super key through this path-based primitive. */ + if (!xattr_name_appliable(xa->name, preserve_acls)) continue; if (lsetxattr(path, xa->name, xa->value, xa->value_len, 0) != 0) { if (!warned) { diff --git a/src/shared/xattr.h b/src/shared/xattr.h index 6628aeb..bbaa045 100644 --- a/src/shared/xattr.h +++ b/src/shared/xattr.h @@ -110,16 +110,32 @@ bool xattr_apply_fd(int fd, const FileXattrList* list); /* Receiver: apply every entry to the symlink named by (parent_fd, leaf) WITHOUT * following it, via lsetxattr() on the confined path - * "/proc/self/fd//". A symlink cannot be targeted by the - * fd-relative fsetxattr() path: there is no *at() xattr syscall and the kernel - * rejects xattr syscalls on an O_PATH descriptor, so the already-opened, + * "/proc/self/fd//". Every incoming name is independently + * re-validated against xattr_name_appliable() with `preserve_acls`, exactly like + * xattr_apply_fd(): a non-whitelisted namespace (including the reserved + * --fake-super key) is skipped, so this primitive stays confined even if handed + * a hand-crafted list. A symlink cannot be targeted by the fd-relative + * fsetxattr() path: there is no *at() xattr syscall and the kernel rejects + * xattr syscalls on an O_PATH descriptor, so the already-opened, * confinement-checked parent directory is the anchor and only the final * component is the (no-follow) link. `leaf` must be a single path component. - * Best-effort exactly like xattr_apply_fd(): a per-attribute failure (on Linux - * every set on a symlink fails with EPERM) is logged once and skipped, never - * fatal. Returns false only for an invalid anchor/list; true when an apply was - * attempted. The reserved --fake-super key is never applied. */ -bool xattr_apply_path_nofollow(int parent_fd, const char* leaf, const FileXattrList* list); + * + * Portability: the "/proc/self/fd/" anchor requires a mounted /proc. + * Where /proc is unavailable (or the fd cannot be addressed that way) the + * lsetxattr simply fails and is skipped -- the apply is best-effort exactly like + * xattr_apply_fd(), so no error is propagated and the transfer continues. A + * per-attribute failure (on Linux every set on a symlink fails with EPERM) is + * logged once and skipped, never fatal. Returns false only for an invalid + * anchor/list; true when an apply was attempted. + * + * Residual TOCTOU: `leaf` is a caller-supplied name resolved by path in the + * parent, so a local writer could replace the just-created symlink between its + * creation and lsetxattr(). This is bounded: it requires write access to the + * confinement-checked destination directory (already trusted), can only install + * a whitelisted user namespace or POSIX-ACL name, and never follows the link (a + * replacement symlink is still applied to as the final, no-follow component). */ +bool xattr_apply_path_nofollow(int parent_fd, const char* leaf, const FileXattrList* list, + bool preserve_acls); /* --fake-super: write the source uid/gid/mode/rdev record into the reserved * FAKESUPER_XATTR on `fd`, using rsync 3.4.1's exact grammar (see the key diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py index d14f58a..5730af0 100644 --- a/tests/integration/test_features.py +++ b/tests/integration/test_features.py @@ -6675,13 +6675,25 @@ class TestExtendedAttributes: assert os.path.islink(dst_link), "destination link entry is not a symlink" assert os.readlink(dst_link) == "target.txt" - # The no-follow guarantee: the referent's attribute must never leak onto - # the symlink entry. + # The no-follow guarantee. Checking only the link's own xattr list is + # vacuous on Linux (lsetxattr on a symlink always fails EPERM), so also + # prove the apply never followed the link: the destination REFERENT must + # keep its own user.* value untouched. + dst_target = os.path.join(received, "target.txt") + assert os.getxattr(dst_target, "user.referent-only") == b"referent-value", ( + "the destination symlink apply followed the link and rewrote the " + "referent's xattr" + ) link_names = os.listxattr(dst_link, follow_symlinks=False) assert "user.referent-only" not in link_names, ( "the destination symlink captured its REFERENT's xattr " "(path-following capture bug)" ) + if sys.platform.startswith("linux"): + assert link_names == [], ( + "Linux associates no xattrs with a symlink; the link entry must " + "carry none" + ) if link_xattr_supported: assert os.getxattr( dst_link, "user.link-own", follow_symlinks=False diff --git a/tests/test_xattr.c b/tests/test_xattr.c index 8fc680a..314fbaf 100644 --- a/tests/test_xattr.c +++ b/tests/test_xattr.c @@ -8,6 +8,7 @@ #include "identity.h" #include "metadata.h" #include "protocol.h" +#include "scanner_internal.h" #include "test_utils.h" #include #include @@ -741,16 +742,16 @@ static void test_xattr_apply_path_nofollow_does_not_follow() { EXPECT_TRUE(xattr_list_append(list, "user.added", "x", 1)); /* Invalid anchors are refused before any syscall (no fd/leaf/list). */ - EXPECT_FALSE(xattr_apply_path_nofollow(-1, "link", list)); - EXPECT_FALSE(xattr_apply_path_nofollow(0, "", list)); - EXPECT_FALSE(xattr_apply_path_nofollow(0, "a/b", list)); - EXPECT_FALSE(xattr_apply_path_nofollow(0, "link", NULL)); + EXPECT_FALSE(xattr_apply_path_nofollow(-1, "link", list, false)); + EXPECT_FALSE(xattr_apply_path_nofollow(0, "", list, false)); + EXPECT_FALSE(xattr_apply_path_nofollow(0, "a/b", list, false)); + EXPECT_FALSE(xattr_apply_path_nofollow(0, "link", NULL, false)); /* The confined parent directory is the anchor; the final component is the link. Best-effort: returns true even when the kernel refuses. */ int dir_fd = open(root, O_RDONLY | O_DIRECTORY); EXPECT_TRUE(dir_fd >= 0); - EXPECT_TRUE(xattr_apply_path_nofollow(dir_fd, "link", list)); + EXPECT_TRUE(xattr_apply_path_nofollow(dir_fd, "link", list, false)); close(dir_fd); /* The referent must be untouched: a following apply would have set user.orig @@ -777,6 +778,57 @@ static void test_xattr_apply_path_nofollow_does_not_follow() { rmdir(root); } +/* Protocol 2.29.0 scanner wiring: scanner_capture_xattrs() must choose the + * NO-FOLLOW capture for a symlink entry, so the link's FileXattrList never + * carries the REFERENT's user.* attributes. xattr_capture_path_nofollow() is + * already covered directly above; this exercises the scanner CALL SITE, which is + * what makes the no-follow variant actually reach symlink entries. If the + * scanner regressed to the path-following capture, file->xattrs would contain + * user.symref and this test fails. Guarded on filesystem xattr support. */ +static void test_scanner_symlink_capture_is_nofollow() { + const char* target = "test_scanner_symlink_xattr_target"; + const char* link = "test_scanner_symlink_xattr_link"; + unlink(link); + unlink(target); + int fd = open(target, O_WRONLY | O_CREAT | O_TRUNC, 0600); + if (fd < 0) + return; + bool has_xattr = setxattr(target, "user.symref", "referent", 8, 0) == 0; + close(fd); + if (!has_xattr) { + unlink(target); + return; /* filesystem without xattr support */ + } + if (symlink(target, link) != 0) { + unlink(target); + return; + } + + DirectoryScanner scanner; + memset(&scanner, 0, sizeof(scanner)); + scanner.options.preserve_xattrs = true; + File* file = file_create(link); + EXPECT_NOT_NULL(file); + file->is_symlink = true; + + scanner_capture_xattrs(&scanner, file); + + /* The referent's attribute must not appear on the symlink's captured list. */ + bool leaked = false; + for (int i = 0; file->xattrs && i < file->xattrs->count; i++) { + if (strcmp(file->xattrs->items[i].name, "user.symref") == 0) + leaked = true; + } + EXPECT_FALSE(leaked); + /* On Linux the VFS associates no xattrs with a symlink, so the capture is + NULL (never an empty-but-valid list). */ + EXPECT_NULL(file->xattrs); + + file_destroy(file); + unlink(link); + unlink(target); +} + /* Protocol 2.29.0: a STATUS_SYMLINK frame followed by an -X/-A xattr block is * decoded by file_receive_symlink() with the block attached to the File. This * is the wire round-trip for the new trailing symlink xattr block. */ @@ -843,6 +895,7 @@ static void test_symlink_frame_carries_xattrs() { void test_xattr() { test_xattr_list_clone(); test_xattr_capture_symlink_nofollow(); + test_scanner_symlink_capture_is_nofollow(); test_xattr_apply_path_nofollow_does_not_follow(); test_symlink_frame_carries_xattrs(); test_xattr_wire_roundtrip(); -- 2.54.0 From 46dcefe2180fd7a47d8cb67f3a3c64bc24682a47 Mon Sep 17 00:00:00 2001 From: TapTap Date: Wed, 23 Sep 2026 01:36:08 +0200 Subject: [PATCH 66/68] fix(protocol): classify TLS EOF before EINTR retry; harden current-ssl resolver; real TLS regression test --- src/shared/protocol.c | 24 ++- src/shared/protocol.h | 10 +- tests/integration/test_tls.py | 12 +- tests/test_protocol.c | 330 ++++++++++++++++++++++++++++++++++ 4 files changed, 363 insertions(+), 13 deletions(-) diff --git a/src/shared/protocol.c b/src/shared/protocol.c index 69dec04..6bce72d 100644 --- a/src/shared/protocol.c +++ b/src/shared/protocol.c @@ -99,8 +99,10 @@ static ssize_t tls_io_send(ProtocolSession* session, const void* data, size_t si return PROTOCOL_IO_RETRY; } /* A signal (e.g. Ctrl-C) interrupts the blocking TLS write: retry so the - * send loop can observe the abort flag at the next checkpoint. */ - if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) + * send loop can observe the abort flag at the next checkpoint. Only an + * actual negative return is an interrupted syscall; a 0-byte SSL_write is + * not a valid EINTR retry. */ + if (written < 0 && ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) return PROTOCOL_IO_RETRY; return PROTOCOL_IO_ERROR; } @@ -125,8 +127,13 @@ static ssize_t tls_io_recv(ProtocolSession* session, void* data, size_t size, sh return PROTOCOL_IO_RETRY; } /* A signal interrupts the blocking TLS read: retry (mirrors the send path) - * so the loop reaches its next abort/deadline checkpoint. */ - if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) + * so the loop reaches its next abort/deadline checkpoint. Only an actual + * negative return is an interrupted syscall: a 0-byte SSL_read is an + * unexpected EOF (the peer closed without close_notify), which OpenSSL also + * reports as SSL_ERROR_SYSCALL with errno possibly still EINTR from an + * earlier interrupted poll/read. Retrying that would busy-spin the + * status-read loop until its deadline, so classify it as closed instead. */ + if (received < 0 && ssl_err == SSL_ERROR_SYSCALL && errno == EINTR) return PROTOCOL_IO_RETRY; /* A zero-length SSL_read is the peer's clean close_notify (or EOF without * one); report it distinctly so the caller can log it as a close. */ @@ -396,9 +403,12 @@ SSL* protocol_current_ssl(void) { /* The bound session is the authoritative transport for a worker thread: it * was explicitly handed to protocol_session_bind() and carries its own SSL, * whereas io_ssl is thread-local and NULL in a thread that never performed - * the handshake. With no session bound (the fd-shim path), fall back to the - * legacy thread-local SSL. */ - if (bound_session) + * the handshake. Only a session whose selected dispatch is TLS may supply + * the SSL: a bound plaintext session has ssl == NULL and must not shadow a + * live thread-local io_ssl, or file_send.c would take the raw sendfile(2) + * path on a socket this thread is driving with TLS. With no TLS session + * bound (plaintext session, or the fd-shim path), fall back to io_ssl. */ + if (bound_session && bound_session->ops == &tls_io_ops && bound_session->ssl) return bound_session->ssl; return io_ssl; } diff --git a/src/shared/protocol.h b/src/shared/protocol.h index bd6c752..d852fb1 100644 --- a/src/shared/protocol.h +++ b/src/shared/protocol.h @@ -249,13 +249,15 @@ unsigned long long io_get_bwlimit(void); void io_set_ssl(SSL* ssl); SSL* io_get_ssl(void); /* SSL object of the transport in effect on this thread: the currently bound - * session's SSL when a session is bound, otherwise the legacy thread-local + * session's SSL when a TLS session is bound, otherwise the legacy thread-local * io_ssl. NULL for a plaintext transport. Unlike io_get_ssl(), this resolves * worker threads that bound a TLS session via protocol_session_set_ssl()/ * protocol_session_bind() but never called io_set_ssl() themselves (C11 - * _Thread_local state is not inherited by a new thread). Callers that must - * choose a TLS-only code path (e.g. file_send.c's sendfile fallback) must use - * this instead of io_get_ssl(). */ + * _Thread_local state is not inherited by a new thread). A bound session only + * wins when its selected dispatch is TLS; a bound plaintext session (ssl == + * NULL) falls back to io_ssl so it can never mask a live encrypted transport. + * Callers that must choose a TLS-only code path (e.g. file_send.c's sendfile + * fallback) must use this instead of io_get_ssl(). */ SSL* protocol_current_ssl(void); /* Process-wide wire byte counters. protocol_send_n_data/protocol_receive_n_data diff --git a/tests/integration/test_tls.py b/tests/integration/test_tls.py index cdc3a24..6ead299 100644 --- a/tests/integration/test_tls.py +++ b/tests/integration/test_tls.py @@ -146,8 +146,16 @@ class TestTLSBasic: assert not missing, f"Missing files: {missing}" assert not mismatches, f"Mismatched files: {mismatches}" + @pytest.mark.ci def test_tls_with_multithreading(self, certs): - """TLS + multithreading.""" + """TLS + multithreading + --sendfile. + + Exercises the TLS/sendfile interaction end to end: with --sendfile the + sender must route the file body through the buffered TLS path rather + than raw sendfile(2) on the encrypted socket. The focused decision + guard lives in tests/test_protocol.c + (test_tls_sendfile_decision_uses_buffered_path). + """ clean_dir(DEST_DIR) with ServerManager() as server: server.start(extra_args=[ @@ -156,7 +164,7 @@ class TestTLSBasic: ]) result, dur = run_client( SOURCE_DIR, DEST_DIR, - flags=["--threads", "--tls", + flags=["--threads", "--sendfile", "--tls", "--cert", certs["client_cert"], "--key", certs["client_key"], "--ca", certs["ca"]], port=server.port, diff --git a/tests/test_protocol.c b/tests/test_protocol.c index 365c82d..5852e3f 100644 --- a/tests/test_protocol.c +++ b/tests/test_protocol.c @@ -1,9 +1,13 @@ #include "protocol.h" +#include "file.h" #include "test_utils.h" +#include "utils.h" #include #include #include +#include #include +#include #include #include #include @@ -865,6 +869,106 @@ static void test_protocol_dispatch_via_ops() { close(sv[1]); } +/* Retry-contract tests: an op that reports PROTOCOL_IO_RETRY once (and hands the + * loop a switched wait event) must be retried rather than treated as a fatal + * error or a close. The send/receive loops had no unit coverage for this path + * even though every TLS WANT_READ/WANT_WRITE and EINTR retry relies on it. */ +static int retry_send_calls; +static short retry_send_last_wait; +static int retry_recv_calls; +static short retry_recv_last_wait; + +static ssize_t retry_once_send(ProtocolSession* session, const void* data, size_t size, + short* wait_events) { + retry_send_calls++; + if (retry_send_calls == 1) { + /* Simulate a WANT_READ-style retry: switch the poll event and make no + * progress. The send loop must consume this and retry. */ + *wait_events = POLLIN; + return PROTOCOL_IO_RETRY; + } + ssize_t written = write(session->write_fd, data, size); + if (written < 0) + return PROTOCOL_IO_ERROR; + if (written == 0) + return PROTOCOL_IO_ERROR; + *wait_events = POLLOUT; + retry_send_last_wait = *wait_events; + return written; +} + +static ssize_t retry_once_recv(ProtocolSession* session, void* data, size_t size, + short* wait_events) { + retry_recv_calls++; + if (retry_recv_calls == 1) { + *wait_events = POLLOUT; + return PROTOCOL_IO_RETRY; + } + ssize_t received = read(session->read_fd, data, size); + if (received < 0) + return PROTOCOL_IO_ERROR; + if (received == 0) + return PROTOCOL_IO_CLOSED; + *wait_events = POLLIN; + retry_recv_last_wait = *wait_events; + return received; +} + +static const ProtocolIoOps retry_send_ops = { + .send = retry_once_send, + .recv = counting_recv, + .has_pending = counting_has_pending, +}; + +static const ProtocolIoOps retry_recv_ops = { + .send = counting_send, + .recv = retry_once_recv, + .has_pending = counting_has_pending, +}; + +static void test_protocol_io_retry_contract() { + const char payload[] = "retry-contract"; + + /* The send loop: the first attempt reports RETRY and switches the poll event + * to POLLIN. A pre-seeded readable byte on the *opposite* end of the + * socketpair keeps that poll immediately satisfiable, so the retry is the + * only thing under test. */ + int send_sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, send_sv), 0); + char seed = 'x'; + EXPECT_EQ_INT(write(send_sv[1], &seed, 1), 1); + ProtocolSession sender; + protocol_session_init(&sender, send_sv[0], send_sv[0]); + protocol_session_set_bwlimit(&sender, 0); + sender.ops = &retry_send_ops; + retry_send_calls = 0; + retry_send_last_wait = 0; + EXPECT_TRUE(protocol_send_n_data(&sender, payload, sizeof(payload))); + EXPECT_EQ_INT(retry_send_calls, 2); + EXPECT_EQ_INT(retry_send_last_wait, POLLOUT); + close(send_sv[0]); + close(send_sv[1]); + + /* The receive loop: the first attempt reports RETRY and switches the poll + * event to POLLOUT, which a socketpair read fd satisfies immediately. */ + int recv_sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, recv_sv), 0); + EXPECT_EQ_INT((int)write(recv_sv[0], payload, sizeof(payload)), (int)sizeof(payload)); + ProtocolSession receiver; + protocol_session_init(&receiver, recv_sv[1], recv_sv[1]); + protocol_session_set_bwlimit(&receiver, 0); + receiver.ops = &retry_recv_ops; + retry_recv_calls = 0; + retry_recv_last_wait = 0; + char received[sizeof(payload)] = {0}; + EXPECT_TRUE(protocol_receive_n_data(&receiver, received, sizeof(received))); + EXPECT_EQ_INT(memcmp(payload, received, sizeof(payload)), 0); + EXPECT_EQ_INT(retry_recv_calls, 2); + EXPECT_EQ_INT(retry_recv_last_wait, POLLIN); + close(recv_sv[0]); + close(recv_sv[1]); +} + typedef struct { ProtocolSession* session; SSL* expected_ssl; @@ -894,7 +998,14 @@ static void test_protocol_current_ssl_prefers_bound_session() { EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); ProtocolSession session; protocol_session_init(&session, sv[0], sv[0]); + const ProtocolIoOps* plain_ops = session.ops; protocol_session_set_ssl(&session, ssl); + /* set_ssl must select a distinct (TLS) dispatch table; protocol_current_ssl + * only returns a bound session's SSL for TLS ops, so arg.resolved_ssl == ssl + * below also proves the bound session's ops are the TLS ops. */ + EXPECT_NOT_NULL(plain_ops); + EXPECT_TRUE(session.ops != plain_ops); + EXPECT_TRUE(session.ssl == ssl); /* Clear the calling thread's legacy SSL: only the bound session carries it. */ io_set_fds(-1, -1); @@ -904,6 +1015,7 @@ static void test_protocol_current_ssl_prefers_bound_session() { thrd_t worker; EXPECT_EQ_INT(thrd_create(&worker, ssl_resolver_worker, &arg), thrd_success); EXPECT_EQ_INT(thrd_join(worker, NULL), thrd_success); + EXPECT_TRUE(arg.resolved_ssl == arg.expected_ssl); EXPECT_TRUE(arg.resolved_ssl == ssl); EXPECT_NULL(arg.thread_local_ssl); @@ -913,6 +1025,221 @@ static void test_protocol_current_ssl_prefers_bound_session() { SSL_CTX_free(ctx); } +/* A bound plaintext session must NOT mask a live thread-local TLS transport: + * protocol_current_ssl() only trusts a bound session whose dispatch is TLS, so + * it falls back to io_ssl here. This is the safe direction for the sendfile + * decision -- returning NULL would let file_send.c take raw sendfile(2) on a + * socket this thread is encrypting. */ +static void test_protocol_current_ssl_plaintext_bound_falls_back() { + SSL_CTX* ctx = SSL_CTX_new(TLS_method()); + EXPECT_NOT_NULL(ctx); + SSL* ssl = SSL_new(ctx); + EXPECT_NOT_NULL(ssl); + + int sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); + + /* Live thread-local TLS, then a bound plaintext session: the plaintext + * session's NULL ssl must not shadow the encrypted transport. */ + io_set_ssl(ssl); + ProtocolSession plain; + protocol_session_init(&plain, sv[0], sv[0]); + protocol_session_bind(&plain); + EXPECT_TRUE(protocol_current_ssl() == ssl); + protocol_session_unbind(); + + /* A bound TLS session still wins over a different thread-local TLS object. */ + SSL* other = SSL_new(ctx); + EXPECT_NOT_NULL(other); + io_set_ssl(other); + ProtocolSession tls; + protocol_session_init(&tls, sv[0], sv[0]); + protocol_session_set_ssl(&tls, ssl); + protocol_session_bind(&tls); + EXPECT_TRUE(protocol_current_ssl() == ssl); + EXPECT_TRUE(protocol_current_ssl() != other); + protocol_session_unbind(); + + io_set_fds(-1, -1); + close(sv[0]); + close(sv[1]); + SSL_free(other); + SSL_free(ssl); + SSL_CTX_free(ctx); +} + +/* ------------------------------------------------------------------------- * + * Genuine TLS + sendfile regression test. + * + * file_send_sendfile_with_skip() must route a TLS transfer through the + * buffered SSL path, resolved from the bound session, even in a worker thread + * whose thread-local io_ssl was never installed. This drives a real TLS + * handshake between two in-memory endpoints and calls the production + * file_send entry from a worker that bound a TLS session only: if the sendfile + * decision regresses to io_get_ssl() it sees NULL, takes raw sendfile(2), and + * copies the file's plaintext into the encrypted stream, so the peer's final + * SSL_read here fails. A tautology-free end-to-end decision guard. + * ------------------------------------------------------------------------- */ + +static void test_set_fd_nonblocking(int fd) { + int flags = fcntl(fd, F_GETFL, 0); + if (flags != -1) + fcntl(fd, F_SETFL, flags | O_NONBLOCK); +} + +static SSL_CTX* test_tls_context_with_self_signed_cert(void) { + EVP_PKEY* key = EVP_PKEY_new(); + EVP_PKEY_CTX* key_ctx = EVP_PKEY_CTX_new_id(EVP_PKEY_RSA, NULL); + if (!key || !key_ctx) { + EVP_PKEY_free(key); + EVP_PKEY_CTX_free(key_ctx); + return NULL; + } + bool key_ok = EVP_PKEY_keygen_init(key_ctx) == 1 && + EVP_PKEY_CTX_set_rsa_keygen_bits(key_ctx, 2048) == 1 && + EVP_PKEY_keygen(key_ctx, &key) == 1; + EVP_PKEY_CTX_free(key_ctx); + + X509* cert = key_ok ? X509_new() : NULL; + bool cert_ok = cert != NULL && X509_set_version(cert, 2) == 1 && + ASN1_INTEGER_set(X509_get_serialNumber(cert), 1) == 1 && + X509_gmtime_adj(X509_getm_notBefore(cert), 0) != NULL && + X509_gmtime_adj(X509_getm_notAfter(cert), 3600) != NULL && + X509_set_pubkey(cert, key) == 1; + if (cert_ok) { + X509_NAME* name = X509_get_subject_name(cert); + cert_ok = X509_NAME_add_entry_by_txt(name, "CN", MBSTRING_ASC, (unsigned char*)"localhost", -1, + -1, 0) == 1 && + X509_set_issuer_name(cert, name) == 1 && X509_sign(cert, key, EVP_sha256()) > 0; + } + + SSL_CTX* ctx = cert_ok ? SSL_CTX_new(TLS_method()) : NULL; + bool installed = ctx != NULL && SSL_CTX_use_certificate(ctx, cert) == 1 && + SSL_CTX_use_PrivateKey(ctx, key) == 1; + if (ctx && !installed) { + SSL_CTX_free(ctx); + ctx = NULL; + } + if (ctx) + SSL_CTX_set_verify(ctx, SSL_VERIFY_NONE, NULL); + + X509_free(cert); + EVP_PKEY_free(key); + return ctx; +} + +static bool test_tls_pump_handshake(SSL* ssl, int* done) { + int result = SSL_do_handshake(ssl); + if (result == 1) { + *done = 1; + return true; + } + int err = SSL_get_error(ssl, result); + return err == SSL_ERROR_WANT_READ || err == SSL_ERROR_WANT_WRITE; +} + +static bool test_tls_read_exact(SSL* ssl, void* out, size_t size) { + char* bytes = out; + size_t got = 0; + while (got < size) { + int result = SSL_read(ssl, bytes + got, (int)(size - got)); + if (result > 0) { + got += (size_t)result; + continue; + } + int err = SSL_get_error(ssl, result); + if (err != SSL_ERROR_WANT_READ && err != SSL_ERROR_WANT_WRITE) + return false; + struct pollfd pfd = {.fd = SSL_get_fd(ssl), + .events = err == SSL_ERROR_WANT_READ ? POLLIN : POLLOUT}; + if (poll(&pfd, 1, 5000) <= 0) + return false; + } + return true; +} + +typedef struct { + ProtocolSession* session; + File* file; + int fd; + bool ok; +} TlsSendfileWorkerArg; + +static int tls_sendfile_worker(void* arg) { + TlsSendfileWorkerArg* worker = arg; + /* Deliberately never call io_set_ssl(): the bound session is the only + * transport this thread has, exactly like a worker in the -m pipeline. */ + protocol_session_bind(worker->session); + worker->ok = + file_send_sendfile_with_skip(worker->file, worker->fd, false, 0, false, NULL, 0, 0, false); + protocol_session_unbind(); + return thrd_success; +} + +static void test_tls_sendfile_decision_uses_buffered_path() { + const char content[] = "tls-sendfile-regression-payload"; + const char* path = "test_tls_sendfile_regression.bin"; + EXPECT_TRUE(file_write_to_disk(path, content, sizeof(content), false, false)); + + SSL_CTX* ctx = test_tls_context_with_self_signed_cert(); + EXPECT_NOT_NULL(ctx); + SSL* server_ssl = SSL_new(ctx); + SSL* client_ssl = SSL_new(ctx); + EXPECT_NOT_NULL(server_ssl); + EXPECT_NOT_NULL(client_ssl); + + int sv[2]; + EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, sv), 0); + test_set_fd_nonblocking(sv[0]); + test_set_fd_nonblocking(sv[1]); + EXPECT_EQ_INT(SSL_set_fd(server_ssl, sv[0]), 1); + EXPECT_EQ_INT(SSL_set_fd(client_ssl, sv[1]), 1); + SSL_set_accept_state(server_ssl); + SSL_set_connect_state(client_ssl); + + int server_done = 0; + int client_done = 0; + for (int i = 0; i < 1000 && !(server_done && client_done); i++) { + bool server_ok = server_done || test_tls_pump_handshake(server_ssl, &server_done); + bool client_ok = client_done || test_tls_pump_handshake(client_ssl, &client_done); + if (!server_ok || !client_ok) + break; + } + EXPECT_TRUE(server_done && client_done); + + File* file = file_create(path); + EXPECT_NOT_NULL(file); + file->data->size = sizeof(content); + + ProtocolSession session; + protocol_session_init(&session, sv[0], sv[0]); + protocol_session_set_bwlimit(&session, 0); + protocol_session_set_ssl(&session, server_ssl); + + TlsSendfileWorkerArg arg = {.session = &session, .file = file, .fd = sv[0], .ok = false}; + thrd_t worker; + EXPECT_EQ_INT(thrd_create(&worker, tls_sendfile_worker, &arg), thrd_success); + EXPECT_EQ_INT(thrd_join(worker, NULL), thrd_success); + EXPECT_TRUE(arg.ok); + + /* The peer must be able to decrypt the whole framing: size header and the + * file body, both produced through the TLS transport. */ + unsigned long long wire_size = 0; + EXPECT_TRUE(test_tls_read_exact(client_ssl, &wire_size, sizeof(wire_size))); + EXPECT_EQ_INT((int)wire_size, (int)sizeof(content)); + char received[sizeof(content)] = {0}; + EXPECT_TRUE(test_tls_read_exact(client_ssl, received, sizeof(received))); + EXPECT_EQ_INT(memcmp(received, content, sizeof(content)), 0); + + file_destroy(file); + close(sv[0]); + close(sv[1]); + SSL_free(server_ssl); + SSL_free(client_ssl); + SSL_CTX_free(ctx); + unlink(path); +} + void test_protocol() { test_send_receive_n_data(); test_send_receive_n_data_zero(); @@ -947,5 +1274,8 @@ void test_protocol() { test_protocol_throttle_bytes_unlimited(); test_protocol_throttle_bytes_legacy_same_session(); test_protocol_dispatch_via_ops(); + test_protocol_io_retry_contract(); test_protocol_current_ssl_prefers_bound_session(); + test_protocol_current_ssl_plaintext_bound_falls_back(); + test_tls_sendfile_decision_uses_buffered_path(); } -- 2.54.0 From f3d76726940c577499c254550ee7eae1319aa273 Mon Sep 17 00:00:00 2001 From: TapTap Date: Wed, 23 Sep 2026 01:45:30 +0200 Subject: [PATCH 67/68] docs: bump release version to 2.29.0 and reconcile protocol docs --- .opencode/skills/release/SKILL.md | 2 +- CHANGELOG.md | 22 ++++++++++++++++++++++ CMakeLists.txt | 2 +- RSYNC_COMPAT.md | 2 +- tests/integration/test_preflight.py | 2 +- 5 files changed, 26 insertions(+), 4 deletions(-) diff --git a/.opencode/skills/release/SKILL.md b/.opencode/skills/release/SKILL.md index fefc22a..2866aa5 100644 --- a/.opencode/skills/release/SKILL.md +++ b/.opencode/skills/release/SKILL.md @@ -16,7 +16,7 @@ Ask the user or determine from context: - **Minor** (x.Y.0) — new features, backward compatible - **Patch** (x.y.Z) — bug fixes, no protocol changes -Current version: `PROTOCOL_VERSION "2.26.0"` in `src/shared/config.h` +Current version: `PROTOCOL_VERSION "2.29.0"` in `src/shared/config.h` ### Step 2: Check Protocol Version diff --git a/CHANGELOG.md b/CHANGELOG.md index 2853154..3f299f7 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -25,6 +25,28 @@ remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ / fixes** below) moves `-F` and `-i` to ⚠️, for a final **117 ✅ / 13 ⚠️ / 27 ❌** of 157 rows. +A no-wire parity burn-down cycle follows on 2.28.0: it accepts +`--inc-recursive`/`--no-inc-recursive` as inert no-ops, accepts an absolute +`--temp-dir` that canonicalizes inside the receive root, closes the +`--delete-before` phase-0 divergence (both the single-threaded and `--threads` +data passes replay the pre-scan list), makes `--fake-super` interoperable with +rsync's `user.rsync.%stat` key/grammar (regular files and char/block devices +faked as regular files), turns a failed device `mknod` into a continuing +per-entry failure, and accepts a practical subset of rsync's `rsyncd.conf` +grammar (modules are read-only by default, and accepted-but-unenforced +access-control keys emit a startup warning). The matrix moves to **119 ✅ / +14 ⚠️ / 24 ❌** of 157 rows. + +A structural cycle then lands a transport I/O vtable over TCP/TLS (fixing the +TLS-multithreaded sendfile path and making the per-thread SSL resolution +explicit) and bumps the wire to **2.29.0**: the `STATUS_SYMLINK` frame grows an +optional symlink-xattr block (captured no-follow with `llistxattr`/`lgetxattr`, +applied no-follow with `lsetxattr`). Because the handshake is strict, 2.28.0 and +2.29.0 peers are incompatible. Note: Linux refuses to associate xattrs with a +symlink at all, so the symlink-xattr block is a no-op on Linux and is carried +for correctness on platforms/filesystems that do support it; the config-frame +layout is unchanged (golden length still 886). + ### Changed - **rsync-exact traversal order.** The sequential scanner now walks each diff --git a/CMakeLists.txt b/CMakeLists.txt index 8f5dbf2..308fa82 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -1,6 +1,6 @@ cmake_minimum_required(VERSION 3.22) -project(FastFileTransfer VERSION 2.28.0) +project(FastFileTransfer VERSION 2.29.0) set(CMAKE_EXPORT_COMPILE_COMMANDS ON) set(CMAKE_C_STANDARD 11) diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md index 902ecef..031c4f6 100644 --- a/RSYNC_COMPAT.md +++ b/RSYNC_COMPAT.md @@ -804,7 +804,7 @@ modes or links. | `--stop-after=MINS` | Stop after N minutes | ✅ Parity | Client-only sender stop deadline (Phase 6): computing `--stop-after=MINS` (a positive minute count; 0/negative/garbage rejected) and `--stop-at=TIME` (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`; a past time stops immediately). The transfer stops ELEGANTLY at the next chunk boundary: everything already fully sent is kept and applied, the run returns 0, and --delete (late/delete-after timing) does NOT wipe the destination — when the scan is cut short the partial keep-set manifest is suppressed with a warning (the delete walk is skipped rather than acting on an incomplete keep-set, so unscanned source mirrors survive). `--delete-before`/`--delete-during` still run their complete pre-scan (which ignores the deadline). Local client-only fields: never serialized into the wire config frame, so no PROTOCOL_VERSION bump. `--stop-after` uses CLOCK_MONOTONIC; `--stop-at` uses the wall clock. Works single-threaded and under `-j`/`--threads` (multithreaded). Divergence: rsync computes `--stop-after` from the run start; FastSync likewise. When both are given, the earlier of the two deadlines wins (checked per iteration). See the Phase-6 stop notes below | | `--stop-at=TIME` | Stop at specified time | ✅ Parity | Deadline transfer stop (client-only, never serialized). Protocol 2.26.0 accepts rsync's full date/time grammar (`2030-12-31T23:59`, `2030/12/31T23:59`, `2030-12-31`, `12-31`, `14:00`, `:59`, `1`) in addition to FastSync's `HH:MM[:SS]` and `now+N[smhd]`; a past time stops immediately. Everything already transferred is kept and an early stop suppresses the late `--delete` keep-set so unscanned source mirrors survive. Works single-threaded and under `-j`/`--threads` | | `--fsync` | Fsync every written file before publication | ✅ Parity | | -| `--protocol=NUM` | Force older protocol version | ❌ Divergent | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.28.0) with no downgrade/backward-compat code paths, so `--protocol=2.28.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.27.0`/`2.26.0`/`2.25.0`/`2.24.0`/`2.23.0`/`2.22.0`/`2.21.0`/`2.20.0`/`2.19.0`/`2.18.0`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below | +| `--protocol=NUM` | Force older protocol version | ❌ Divergent | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.29.0) with no downgrade/backward-compat code paths, so `--protocol=2.29.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.28.0`/`2.27.0`/`2.26.0`/`2.25.0`/`2.24.0`/`2.23.0`/`2.22.0`/`2.21.0`/`2.20.0`/`2.19.0`/`2.18.0`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below | | `--iconv=CONVERT_SPEC` | Charset conversion | ✅ Parity | Charset conversion of FILE NAMES (not content) at the protocol boundary via iconv(3): `--iconv=LOCAL[,REMOTE]` — the sender converts each local filename LOCAL→REMOTE before transmitting, matching rsync's rule that the spec "stays the same whether you're pushing or pulling": on a PUSH the destination end's charset is the spec's REMOTE half, so the default receiver writes the wire bytes verbatim, and only a server started with its own `--iconv` (the daemon `charset` analog) declares a different destination charset and converts REMOTE→that LOCAL (rsync push parity, differential-tested with and without a server `--iconv`). The full CONVERT_SPEC is serialized into the config frame as a new trailing string field so the peer knows the wire charset; **PROTOCOL_VERSION bumped 2.15.0 → 2.16.0**. `LOCAL[,REMOTE]` parse: single charset ⇒ LOCAL==REMOTE (identity both ways); garbage rejected up front; protocol 2.26.0 additionally accepts `--iconv=.` (the locale's default charset for both directions), `--iconv=-` and `--no-iconv` (disable conversion). Validation probes BOTH directions (a spec that only opens one way is refused, as is a NUL-emitting target charset like utf-16/utf-32/ucs-2, since filenames cannot contain NUL). An unrepresentable name (EILSEQ/EINVAL) fails that path cleanly with a logged `--iconv: cannot convert file name ...` and is never written mangled/truncated. Conversion is applied at EVERY wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest, the incremental-check path, and the `-s`/`chunk_serialize` embedded blob path), on both client and server (`--iconv` is also a server/daemon option). Zero overhead when unset. See the Phase-6 iconv notes below | | `--checksum-seed=NUM` | Set checksum seed | ✅ Parity | Sets the seed for FastSync's whole-file xxHash digest (full 64-bit seed) and for the delta path's per-block xxHash32 strong checksum (low 32 bits of the seed). **As of protocol 2.23.0 a seed of `0` — the default when the flag is unset — is randomized per transfer and the chosen seed is sent to the receiver**, exactly like rsync, so two runs against different content do not share a predictable seed; an explicit non-zero seed is used verbatim, so an explicit seed deterministically reproduces every computed digest on BOTH endpoints (the seed crosses in the config frame). `--checksum-choice=md5` has no seed and ignores it (documented). The value is a strict decimal 0..2⁶⁴-1 (blank, signed, or non-numeric values are rejected). Like rsync, a seed only matters where a digest is actually computed (`--checksum` or a basis-dir run, or a delta transfer); it does not by itself enable `--checksum`/`--delta` | | `--secluded-args`, `-s` | Use protocol to send args | ❌ Divergent | Accepted for CLI compatibility (including the rsync short `-s`, Phase 7 Wave A) but a documented **no-op / divergence**. rsync's `-s` protects arguments from shell expansion by shipping them over the protocol; FastSync never passes remote arguments through a shell expansion boundary in the first place — its SSH transport builds the remote argv as **single-quote-escaped shell words** (`ssh_build_remote_command`), so the injection/leak that `-s` guards against does not exist and there is nothing to "seclude". Implementing a true arg-send protocol would mean replacing the argv-based SSH launch with an in-band argument channel, a large redesign of the transport that buys no security here. Chunk serialization remains the long-only `--chunk-serialization`. | diff --git a/tests/integration/test_preflight.py b/tests/integration/test_preflight.py index 7a47661..1763800 100644 --- a/tests/integration/test_preflight.py +++ b/tests/integration/test_preflight.py @@ -157,7 +157,7 @@ class TestProtocol: shutil.rmtree(dest, ignore_errors=True) os.makedirs(dest) _seed_protocol_source(source) - for bad in ("2.27.0", "2.26.0", "2.25.0", "2.24.0", "2.23.0", "2.22.0", "2.21.0", "2.20.0", + for bad in ("2.28.0", "2.27.0", "2.26.0", "2.25.0", "2.24.0", "2.23.0", "2.22.0", "2.21.0", "2.20.0", "2.19.0", "2.18.0", "2.17.0", "2.15.0", "2.16.0", "216", "31"): result, _ = run_client(source, dest, flags=[f"--protocol={bad}"], port=shared_server.port) -- 2.54.0 From 00197102bfe367d152977bbdfffb781c13b32ac5 Mon Sep 17 00:00:00 2001 From: TapTap Date: Wed, 23 Sep 2026 01:59:44 +0200 Subject: [PATCH 68/68] Release v2.29.0 - Transport I/O vtable over TCP/TLS; TLS multithreaded sendfile fixed - Symlink-xattr wire block (protocol 2.29.0; config frame unchanged) - No-wire parity burn-down: --inc-recursive, in-root --temp-dir, --delete-before phase-0, rsync-interoperable --fake-super, --devices per-entry failure, rsyncd.conf key subset (read-only default) - Protocol version: 2.29.0 - Tested: unit, integration, ASan, UBSan, valgrind, differential parity --- CHANGELOG.md | 2 ++ 1 file changed, 2 insertions(+) diff --git a/CHANGELOG.md b/CHANGELOG.md index 3f299f7..d2292e3 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -6,6 +6,8 @@ run the same version because the handshake is strict. ## [Unreleased] +## [2.29.0] - 2026-09-23 + The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0). `RSYNC_COMPAT.md` moves from **116 ✅ / 14 ⚠️ / 27 ❌** to **120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows. -- 2.54.0