diff --git a/.gitea/workflows/ci.yaml b/.gitea/workflows/ci.yaml
index 464511c..6cbc4cc 100644
--- a/.gitea/workflows/ci.yaml
+++ b/.gitea/workflows/ci.yaml
@@ -49,6 +49,47 @@ jobs:
if: github.event_name == 'push'
run: python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" --durations=25 --tb=short -q
+ # Differential rsync-parity gate: runs real rsync 3.4.1 and FastSync over the
+ # same corpora and compares destinations + normalized output. The fast subset
+ # guards the ✅ surface on every PR; the full set (with FASTSYNC_PARITY_STRICT
+ # so a fixed caveat must be removed from the allowlist) burns the documented
+ # ⚠️/❌ residuals down on push. See tests/integration/README.md.
+ parity-fast:
+ runs-on: ubuntu-latest
+ container: gitea.tap-tap.win/taptap/fastsync-ci:v11
+ needs: lint
+ if: github.event_name == 'pull_request'
+ steps:
+ - name: Checkout
+ uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
+
+ - name: Configure
+ run: cmake -B build -S . -DSTRICT_WARNINGS=ON
+
+ - name: Build
+ run: cmake --build build -j$(nproc)
+
+ - name: Differential parity (fast subset)
+ run: python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci -q
+
+ parity-full:
+ runs-on: ubuntu-latest
+ container: gitea.tap-tap.win/taptap/fastsync-ci:v11
+ needs: lint
+ if: github.event_name == 'push'
+ steps:
+ - name: Checkout
+ uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
+
+ - name: Configure
+ run: cmake -B build -S . -DSTRICT_WARNINGS=ON
+
+ - name: Build
+ run: cmake --build build -j$(nproc)
+
+ - name: Differential parity (full set)
+ run: FASTSYNC_PARITY_STRICT=1 python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity -q
+
sanitizers:
runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
diff --git a/AGENTS.md b/AGENTS.md
index f25379b..5ece0ec 100644
--- a/AGENTS.md
+++ b/AGENTS.md
@@ -56,6 +56,21 @@ cmake -B build -S . && cmake --build build -j$(nproc)
./build/tests # unit tests
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # full integration suite (CI excludes env-dependent privilege tests)
python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset only
+
+# Differential rsync-parity gate (real rsync 3.4.1 vs FastSync)
+python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci # fast PR subset
+python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity # full set
+```
+
+See `tests/integration/README.md` for the differential parity gate and its
+`parity_caveats.py` allowlist (the residual burn-down mechanism).
+
+Unit tests under valgrind must set `FASTSYNC_UNDER_VALGRIND=1` (CI does): the
+tests use it to skip fork-based tests, because valgrind 3.22 does not expose
+`vgpreload` in the guest's `/proc/self/maps`.
+
+```bash
+FASTSYNC_UNDER_VALGRIND=1 valgrind --leak-check=full --show-leak-kinds=definite --error-exitcode=1 ./build/tests
```
## CI Workflow — Waiting for Results
diff --git a/CHANGELOG.md b/CHANGELOG.md
index 713bd8e..7195a19 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -4,8 +4,91 @@ All notable changes to FastSync are documented here. Versions match
`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must
run the same version because the handshake is strict.
+## [2.28.0] - 2026-09-20
+
+The rsync-parity cycle. `PROTOCOL_VERSION` moves `2.26.0 → 2.27.0 → 2.28.0`;
+client and server must run the same version (the handshake is strict). See
+`RSYNC_COMPAT.md` for the per-option matrix, now **116 ✅ / 14 ⚠️ / 27 ❌** of
+157 rows.
+
+### Added
+
+- **Differential rsync 3.4.1 parity gate** (`tests/integration/
+ test_differential_parity.py`, `parity_harness.py`, `parity_caveats.py`): runs
+ real `rsync` and FastSync over generated corpora and diffs the destination
+ tree, normalized stdout and exit code. A fast subset runs on pull requests and
+ the full strict set on push; the residual allowlist is empty.
+- FastSync-only long option **`--verify-basis`**: require a
+ `--compare-dest`/`--copy-dest`/`--link-dest` hit to match the source by
+ whole-file digest instead of trusting the size+mtime quick-check.
+- FastSync-only long option **`--delete-commit`** (implies `--delete`): the old
+ atomic late whole-tree commit.
+- `--bwlimit` now parses rsync's units exactly and paces like rsync's leaky
+ bucket; `--ignore-errors` reproduces rsync's skip-unreadable-subdir and
+ IO-error-suppressed deletion (exit 23).
+- `--info=name/flist/del/remove/nonreg/progress` emit rsync's line format,
+ including real-run `deleting`/`*deleting` lines carried by a new
+ `report_deletes` wire bool.
+- Receiver-observed `--stats` counters: `Number of created files` now carries
+ rsync's `(reg/dir/link/special)` breakdown and `Literal data` is exact for a
+ delta transfer (extended `STATUS_STATS`).
+- `--progress` uses an opt-in paths-only pre-count so the `to-chk` denominator
+ counts every entry like rsync, and emits per-directory/symlink/special names.
+- Receiver-side `protect`/`risk` filter engine (new bounded filter-rule wire
+ block): `--filter='P ...'` now shields a destination-only entry like rsync.
+- `auto` for `--compress-choice`/`--checksum-choice` honors
+ `RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST`, and per-codec compression-level
+ defaults match rsync.
+- Empty source directories are recreated recursively; `-R --no-implied-dirs
+ --files-from` places listed files under missing implied parents; `--iconv`
+ matches rsync's push direction; `--delete-delay` reports actual removals and
+ recursively removes a refilled deferred directory.
+
+### Changed
+
+- **`--delete` now defaults to delete-during (rsync `--del`) timing.** With no
+ explicit timing flag, a plain `--delete` removes each directory's extras as
+ that directory is processed instead of committing one whole-tree deletion only
+ after the entire transfer succeeds. This matches rsync, frees destination
+ space progressively, and avoids the whole-old+new-tree peak that could
+ `ENOSPC` a tight destination. The client maps the default onto the existing
+ `delete_during` wire boolean, so `PROTOCOL_VERSION` stays `2.28.0`.
+- Basis directories (`--compare-dest`/`--copy-dest`/`--link-dest`) now default
+ to rsync's metadata quick-check (equal size and mtime; `--size-only` drops the
+ mtime leg) instead of FastSync's historical always-verify content hash.
+ `--copy-dest` re-applies the source attributes, and basis materialization is
+ streamed so the 256 MiB whole-file cap no longer applies to a basis hit.
+- The per-directory `STATUS_DELETE_PLAN` frame gained a one-int `apply` flag:
+ the one-shot per-run config block (protected prefixes, size-pruned mirrors,
+ `--delete-missing-args` exact paths) is now always transmitted first on a
+ config-only carrier (`apply=false`), fixing a latent bug where a
+ `--delete-missing-args` run whose `--files-from` list synchronized no directory
+ never sent its exact deletions.
+
+### Notes
+
+- `--delete`/`--delete-during` remain caveats for the mid-transfer abort
+ boundary (rsync's generator removes all planned extras ahead of its throttled
+ sender; FastSync removes only reached directories — final trees agree).
+ `--delete-before`, `--progress`, `--stats`, `--fuzzy` and the basis rows keep
+ their documented residuals in `RSYNC_COMPAT.md`; `--filter` and
+ `--delete-excluded` are now parity, including protection of a destination-only
+ excluded entry under default `--delete`.
+
+### Migration
+
+- Scripts that relied on plain `--delete` deleting nothing until the transfer
+ fully succeeded must pass **`--delete-commit`** (or `--delete-after`) to keep
+ that behavior. Plain `--delete` now removes reached directories' extras during
+ the transfer, exactly like rsync's default; on a completed run the final tree
+ is unchanged.
+- Deployments that relied on FastSync's stricter basis verification should pass
+ **`--verify-basis`**; the default now trusts the size+mtime quick-check like
+ rsync.
+
## [2.26.0] - 2026-09-17
+
### Added
- **Parity-completion wave.** Closed the remaining rsync-parity gaps against
diff --git a/CMakeLists.txt b/CMakeLists.txt
index 37620b2..905daec 100644
--- a/CMakeLists.txt
+++ b/CMakeLists.txt
@@ -1,6 +1,6 @@
cmake_minimum_required(VERSION 3.22)
-project(FastFileTransfer VERSION 2.26.0)
+project(FastFileTransfer VERSION 2.28.0)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_C_STANDARD 11)
@@ -221,6 +221,7 @@ set(TEST_SRCS
tests/test_daemon_limits.c
tests/test_data.c
tests/test_delay_updates.c
+ tests/test_delete_plan.c
tests/test_delta.c
tests/test_file.c
tests/test_file_list.c
diff --git a/HANDOFF.md b/HANDOFF.md
index 381438a..43b23f0 100644
--- a/HANDOFF.md
+++ b/HANDOFF.md
@@ -1,16 +1,19 @@
-# FastSync — Session Handoff (2026-09-17)
+# FastSync — Session Handoff (2026-09-19)
## Current status
-- **Release `v2.21.0`** tagged (`919a729`, "Release v2.21.0"); full CI green
- (run 552: lint, build-and-test, ASan, UBSan, fuzz-build, coverage, valgrind).
- `dev` has the release commit plus later doc-only merges (a README refresh and
- this handoff).
-- **Release PR #284 (`dev` -> `main`)** open, CI green (run 553).
- `main` is protected: it needs review/approval to merge.
- https://gitea.tap-tap.win/TapTap/FastSync/pulls/284
-- **`PROTOCOL_VERSION` = `"2.26.0"`** (`src/shared/config.h`); CMake
- `project(FastFileTransfer VERSION 2.26.0)`.
-- Working tree clean; no wave worktrees remain.
+- **rsync-parity tracks 1-6 landed on `dev`** via **PR #303** (`10159dc`,
+ "feat(parity): rsync parity tracks 1-6 (protocol 2.28.0)"). Dev push CI run
+ **581** fully green: lint, build-and-test, parity-full, ASan, UBSan,
+ fuzz-build, coverage, valgrind.
+- **`PROTOCOL_VERSION` = `"2.28.0"`** (`src/shared/config.h`); CMake
+ `project(FastFileTransfer VERSION 2.28.0)`. The cycle batched all wire
+ changes (stats counters, filter-rule block, `--verify-basis`) under the one
+ bump.
+- **`main` = `ef76c90`** (tag `v2.26.0`); the 2.27.0/2.28.0 work is on `dev`
+ and not yet released. A `dev -> main` v2.28.0 release PR is the next step.
+- Parity matrix: **116 ✅ / 14 ⚠️ / 27 ❌ = 157** (was 111/13/33 at cycle start).
+- Working tree clean; feature branch deleted; no scratch trees or worktrees.
+
## What landed this session
1. **Wave 8 (refactors):** Config X-macro wire table; single-owner `authorized_root`;
@@ -39,7 +42,158 @@
docs state push-only / remote-source unsupported.
5. **Preserve-attribute split (protocol 2.22.0)** landed on `feat/preserve-attr-split`: per-attribute `-p/-t/-o/-g` + `--no-*` negations, `-a` = `-rlptgoD`, and the 2.21.0 → 2.22.0 wire bump.
6. **Rsync-parity wave (protocol 2.23.0)** on `feat/rsync-parity`: rsync short options/clustering/attached values (`-r`/`-b`/`-L`/`-B`, `-av`, `-aAX`, `-B1000`, `-essh`, `-MOPT`), `-c` checksum quick-check, `--checksum-choice`/`--compress-choice` validation and seed randomization, rsync timeout/max-alloc defaults, temp-dir confinement + `EXDEV` fallback, ownership/mapping parity (numeric-ids modifier, map ranges/`*`/empty-FROM, `--chown`+map conflicts, fake-super resolved-owner record), verbatim symlink storage with rsync `--safe-links`/`--munge-links`, socket recreation under `--specials`, `--chmod` 3.4.1 semantics, and delete scoping + `--max-delete` partial/exit-25. Wire: appended delete-manifest synchronized-directory section and `STATUS_DELETE_LIMIT`.
-7. **Parity-completion wave (protocol 2.24.0 → 2.26.0)** on `feat/parity-completion`: per-directory delete plans (`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`; receiver `STATUS_STATS` counters feeding `--stats`/`--progress` and `--out-format %b/%c/%C`, plus `-n --delete` lines; `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/`none` checksums with `auto` negotiation (default `xxh128`/`zstd`); general `-R`/`--no-implied-dirs`/`-d`; the full filter grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` + modifiers) and corrected `-F`/`-FF`; receiver-side `--chown`/map TO-name resolution; absolute basis dirs + `--link-dest` relink; receiver-side `--ignore-existing` short-circuit; `--preallocate` over `--sparse` via `fallocate(2)`; `--iconv=.`/`-`/`--no-iconv`; lone `-h` help; aliases `--ignore-non-existing`/`--protect-args`/`--msgs2stderr`; and the full `--info`/`--debug` vocabulary. `RSYNC_COMPAT.md` reclassifies the matrix to 106 ✅ / 27 ⚠️ / 23 ❌.
+7. **Parity-completion wave (protocol 2.24.0 → 2.26.0)** on `feat/parity-completion`: per-directory delete plans (`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`; receiver `STATUS_STATS` counters feeding `--stats`/`--progress` and `--out-format %b/%c/%C`, plus `-n --delete` lines; `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/`none` checksums with `auto` negotiation (default `xxh128`/`zstd`); general `-R`/`--no-implied-dirs`/`-d`; the full filter grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` + modifiers) and corrected `-F`/`-FF`; receiver-side `--chown`/map TO-name resolution; absolute basis dirs + `--link-dest` relink; receiver-side `--ignore-existing` short-circuit; `--preallocate` over `--sparse` via `fallocate(2)`; `--iconv=.`/`-`/`--no-iconv`; lone `-h` help; aliases `--ignore-non-existing`/`--protect-args`/`--msgs2stderr`; and the full `--info`/`--debug` vocabulary. `RSYNC_COMPAT.md` reclassifies the matrix to 106 ✅ / 27 ⚠️ / 23 ❌; the later rsync-parity-stats pass (`fix/parity-stats`) moves it to 107 ✅ / 25 ⚠️ / 24 ❌ (see item 8).
+8. **rsync-parity-stats pass** on `fix/parity-stats` (no wire change, `PROTOCOL_VERSION` stays `2.26.0`): `--delete-delay` now reports only entries it actually removes, while the `--max-delete` budget is charged at plan/snapshot time (`planned`, via `defer_add`) to bound the deferred list (a refilled deferred directory that survives `ENOTEMPTY` is not reported but still consumes budget); `--stats` gained the `(reg/dir/link/special)` `Number of files` breakdown and now counts only regular files actually stored for `Number of regular files transferred`/transferred size/literal data (up-to-date re-runs report 0); `Total file size` includes symlink target lengths; `--progress` prints the leading `./` root line and counts it in `to-chk` so a single-file transfer matches rsync; and `%C` uses the selected transfer checksum with `checksum_digest_file` supporting md4/sha1/none, byte-identical to rsync for every algorithm. `--out-format` reclassified ❌ (`%b`/delta-`%c` are protocol-specific). Differential + regression tests added; full suite + ASan + clang-format + cppcheck clean.
+9. **Option-parity wave (protocol 2.26.0 → 2.27.0, on `fix/parity-options`):**
+ `--bwlimit` now ports rsync 3.4.1's units/quantization and paces like its
+ leaky bucket; `--ignore-errors` reproduces rsync's default (an I/O error
+ skips deletion unless the flag is set; the readable tree still transfers and
+ the run exits 23) across every delete timing; the `--info` categories with a
+ FastSync event (`name`/`flist`/`del`/`remove`/`nonreg`/`progress`) emit
+ rsync's line format, with real-run `deleting`/`*deleting` lines carried over
+ the new trailing config bool `report_deletes` (golden wire updated by
+ `tests/test_config.c`). Two residuals were reclassified **divergent**: `-M`
+ over daemon/TCP (no argv channel in FastSync's binary config handshake;
+ rsync-daemon differential pins the rsync behavior) and receiver-side
+ `protect`/`risk` re-derivation for destination-only entries (would need a
+ receiver filter engine; differential pins the divergence — **reversed by
+ track 4a below**, which adds that engine). The options pass
+ stands at **110 ✅ / 21 ⚠️ / 26 ❌**. New `tests/integration/test_option_parity.py`
+ holds the rsync differentials (bwlimit parse+rate, info lines, real-setpriv
+ `--ignore-errors`, rsync-daemon `-M`, filter-protect pin).
+
+10. **rsync-parity-fs pass** on `fix/parity-fs` (no wire change of its own; integrated
+ on top of the 2.27.0 options wave): recursive transfers now recreate empty source directories (and
+ `-m/--prune-empty-dirs` still suppresses them), a directory entry replaces a
+ blocking destination regular file, and `-R --no-implied-dirs --files-from`
+ places a listed file under a missing implied parent with default attributes
+ instead of refusing (real rsync 3.4.1 parity, differential-tested). `--iconv`
+ now reproduces rsync's push direction (destination charset = the spec's REMOTE
+ half; a server `--iconv` overrides), and `-T/--temp-dir` relative semantics are
+ confirmed identical while the absolute-path confinement is a deliberate
+ divergence. The basis-dir options, `--delay-updates` and `--dry-run` were
+ reclassified to ❌ after a differential test reproduced each exact residual
+ (basis content verification, fixed staging-name collision, and dry-run
+ would-delete over-report). `--fuzzy` was also reclassified to ❌ (deterministic
+ heuristic with a 10× size window, not rsync's matcher), but its residual is the
+ candidate-selection heuristic itself: the final tree is byte-exact by design, so
+ it is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync
+ differential. (Track 5b later found the name heuristic is rsync's own and moved
+ the row ❌ → ⚠️, leaving only the narrower delta size window; see entry 15.) The parity-review pass then moved `--delete-delay` to ⚠️ (the
+ plan-time `--max-delete` charge and non-recursive deferred removal differ from
+ rsync when a snapshotted entry fails removal). Differential-gate allowlist
+ entries `min_size`/`empty_dirs_recursive`/`dirs_plain` were removed. The
+ integrated stats+options+fs branch stands at **111 ✅ / 13 ⚠️ / 33 ❌ = 157**;
+ full suite + ASan + clang-format + cppcheck clean.
+
+11. **No-wire parity track 1** on `feat/parity-2.28` (no protocol change):
+ `-n --delete` now sends the same filter-excluded + size-pruned protected
+ prefixes and synchronized-directory scope as a real run (dry-run would-delete
+ matches rsync for source-derived protections; the destination-only exclude
+ residual was later closed by track 4a, readdir ordering remains);
+ `--delete-delay` now charges
+ `--max-delete` on actual removals and re-scans a queued directory at commit
+ to remove content created after the plan, with an independent deferred-list
+ cap (only partial-delete ordering remains); and `--info=name2` emits `NAME is
+ uptodate` plus the leading `./` root name line for `--info=name` (only the
+ root-line trigger condition and receiver-side `skip` wording remain). Matrix
+ now **111 ✅ / 14 ⚠️ / 32 ❌ = 157**; differential + unit tests added in
+ `test_features.py`, `test_option_parity.py`, `test_delete_plan.c`,
+ `test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`.
+12. **No-wire parity track 2b** on `feat/parity-2.28` (no protocol change):
+ `--progress`/`-P`/`--info=progress` (when not `--quiet`) now run an opt-in
+ paths-only metadata pre-count (no file reads/hashing) that supplies rsync's
+ full file-list total for the `to-chk` denominator and the directory names,
+ and emits per-directory/symlink/special name lines, in both the sequential
+ and `--threads` paths. `--delete-during`/`--delete-delay` reuse their
+ keep-set pre-scan instead of a second walk; non-progress runs are
+ unaffected. Differential tests (`progress`/`progress_threads` over a new
+ `multidir` corpus) match rsync's name set and `to-chk` denominator on a
+ fresh transfer, and the single-file byte-identical test still passes;
+ emission order (rsync's sorted depth-first vs FastSync's readdir/BFS stream)
+ plus re-run over-naming (unconditional `./`, ancestor dirs named with a
+ transferred child, and no quick-check for symlinks/empty dirs) remain the
+ caveats, so the row stays ⚠️ and the matrix is unchanged at
+ **111 ✅ / 14 ⚠️ / 32 ❌ = 157**.
+
+13. **Wire parity track 4a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
+ `2.28.0`): the receiver now has a delete-time filter engine. The sender
+ compiles its root-level selection rules exactly as the scanner does
+ (`filter_base_build`) and streams them as one bounded, self-describing
+ config-frame block (action, sides, anchored, dir-only, negate, owner,
+ pattern; bounded rule count and pattern bytes, unknown action/sides is a
+ protocol error). The receiver reconstructs `protect_rules` and applies them
+ first-match-wins to each extraneous destination path in every delete timing
+ (the whole-tree commit walker, the `--delete-during`/`--delete-delay`
+ per-directory plans, and the `-n` would-delete enumeration), so a
+ `P *.log` rule protects a destination-only `extra.log` like rsync (with
+ `risk` cancelling); the sender-derived protected-prefix behavior is
+ preserved when no rules are sent and `--delete-excluded` semantics are
+ unchanged. Per-directory merge (`:`/`.`) receiver re-derivation remains the
+ residual. `TestFilterProtect` (real + dry-run) plus differential cases
+ `filter_protect`, `filter_protect_during`, `filter_protect_delay` added and
+ the `--filter=RULE` row moves ❌ → ✅: matrix now
+ **115 ✅ / 11 ⚠️ / 31 ❌ = 157**; unit tests, the three named integration
+ files, clang-format and cppcheck clean.
+
+14. **Wire parity track 5a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
+ `2.28.0` by project decision): the three basis-dir options now default to
+ rsync's metadata quick-check (equal size + equal mtime, or size alone under
+ `--size-only`; `-I` disables matching) instead of FastSync's historical
+ xxHash64 content equality, so a same-size/different-content basis is trusted
+ exactly as rsync trusts it. A new FastSync-only, long-only `--verify-basis`
+ flag restores the strict whole-file content equality; its bool is appended to
+ the basis block of the config frame (golden wire frame 882 → 886 bytes).
+ `--verify-basis` streams the confined basis descriptor to hash it, and a
+ basis hit is no longer capped at the 256 MiB whole-file payload bound:
+ `--copy-dest` streams the basis through a bounded buffer and `--link-dest`'s
+ copy fallback streams from the basis, so an over-limit hit materializes (a
+ basis MISS still falls back to the normal transfer and keeps its own bound).
+ A `--copy-dest` hit re-applies the SOURCE attributes (the sender transmits
+ the source metadata with the basis check frame), matching rsync's
+ "copy then fix attributes"; a `--link-dest` success keeps the shared inode's
+ attributes (writing through it would mutate the basis). Differential cases
+ `copy_dest` and `verify_basis` added; `test_basis_dir_size_only_content_residual`
+ converted to a passing parity assertion; `TestBasisDestDirs` updated for the
+ new default + `--verify-basis`; unit tests cover the quick-check/verify
+ decision and the same-size/different-content handshake. The
+ `--compare-dest`/`--copy-dest`/`--link-dest` rows move ❌ → ⚠️ (relative-DIR
+ resolution base and over-limit MISS refusal): matrix now
+ **116 ✅ / 13 ⚠️ / 28 ❌ = 157**.
+
+15. **No-wire parity track 5b** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
+ `2.28.0` by project decision): `-y`/`--fuzzy` reclassified ❌ → ⚠️. A probe
+ against real rsync 3.4.1 (pinned `-B8192`, repeated-content 64 KiB corpus)
+ showed the name heuristic is already rsync's (`util1.c fuzzy_distance` /
+ `find_filename_suffix` + the exact size+mtime pass) and the output is always
+ byte-exact; the only residual is candidate ELIGIBILITY, because FastSync's
+ `delta_should_attempt` gate caps the size ratio at 10× and requires both
+ files ≥ 16 KiB while rsync will reuse a basis from 0.25× to 10000× and below
+ 16 KiB. The choice is observable only as `--stats` bandwidth counters. Added
+ differential case `fuzzy_basis` (same-suffix sibling, one name edit,
+ identical content, block size pinned) asserting tree **and** normalized
+ `--stats` parity where the choices coincide, plus `TestFuzzy` pinning the
+ window boundary on both sides (>10× and <16 KiB siblings declined by
+ FastSync while rsync uses them, both trees byte-identical). Matrix now
+ **116 ✅ / 14 ⚠️ / 27 ❌ = 157**.
+
+16. **Lockstep delete-default track 6** on `feat/parity-2.28` (`PROTOCOL_VERSION`
+ stays `2.28.0`): plain `--delete` now defaults to rsync's delete-during
+ (`--del`) timing, normalized on the client onto the existing `delete_during`
+ wire bool. The old late whole-tree commit is opt-in via `--delete-after` or
+ the FastSync-only long `--delete-commit` (identical `delete_after` timing).
+ `-d/--dirs` still falls back to the end commit, `--delay-updates` still
+ deletes before publication, and `--files-from`/`-R` scope is unchanged. The
+ `STATUS_DELETE_PLAN` frame gained a one-int `apply` flag so the per-run
+ config block (including `--delete-missing-args` exact paths) is always
+ transmitted, on a config-only carrier when the scope allows no directory
+ plan — fixing a latent bug with a file-only `--files-from` list. Differential
+ cases `delete`/`delete_commit`/`filter_protect_after` plus the extended
+ `test_delete_timing_parity.py` (plain `--delete` mid-abort removes reached
+ extras, `--delete-commit` defers) pass; full `-m "not setpriv"` suite,
+ clang-format and cppcheck clean. Matrix unchanged at
+ **116 ✅ / 14 ⚠️ / 27 ❌ = 157** (the `--delete`/`--delete-during` rows stay
+ ⚠️ for the abort boundary; `--delete-after` stays ✅).
## Next steps
1. **Merge PR #284** (`dev` -> `main`) once reviewed (protected branch).
diff --git a/README.md b/README.md
index fe616ff..e7ae011 100644
--- a/README.md
+++ b/README.md
@@ -113,13 +113,18 @@ matrix is classified as parity, caveat, or divergent in
(`-B1000`, `-essh`, `-MOPT`, `--opt=value`) are accepted, matching rsync.
- `-r`, `-b`, `-L`, and `-B` are parsed with the rsync short names.
- `--stats` prints the counters FastSync can observe plus the receiver-only
- counters (`Matched data`, deleted files) reported over the wire; rsync's
- per-type `Number of files` breakdown is not reproduced. `--progress` prints
- rsync-style per-file blocks (without rsync's leading `./` line).
+ counters reported over the wire (`Matched data`, deleted files, and the
+ created/literal counters); `Number of files` and `Number of created files`
+ carry rsync's per-type breakdown. `--progress` prints rsync-style per-file
+ blocks including the leading `./` line, and (when progress is requested) a
+ paths-only pre-count supplies rsync's `to-chk` denominator.
- Codecs match rsync 3.4.1: `zstd`/`lz4`/`zlib`/`zlibx` compression and
- `xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1`/`none` checksums, negotiated with
- `auto`; `zlibx` behaves as `zlib`, and the transfer checksum is not separately
- selectable.
+ `xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1`/`none` checksums. `auto` honors
+ `RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST` and otherwise follows rsync's
+ compiled-in order. An omitted `--compress-level` uses the codec's rsync
+ default (zstd 3, zlib/zlibx 6, lz4 ignored); `zlib`/`zlibx` share the
+ literal-only zlib path (rsync's zlibx semantics), and the transfer checksum is
+ not separately selectable.
The detailed flag matrix is maintained in
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It reports each row as **parity**,
@@ -202,11 +207,13 @@ This produces `./build/client` and `./build/server`. `compile_commands.json` is
| `--compare-dest
` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`) |
| `--copy-dest ` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination |
| `--link-dest ` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win) |
-| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded). Scoped to the synchronized directories, so `--files-from` subsets are safe |
+| `--verify-basis` | FastSync-only: require a basis hit (`--compare-dest`/`--copy-dest`/`--link-dest`) to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync) |
+| `--delete` | Delete files on receiver not present in source (default timing: delete-during, matching rsync, so destination space is freed progressively). Scoped to the synchronized directories, so `--files-from` subsets are safe |
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) |
| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) |
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) |
| `--delete-after` | Explicit delete-after timing (implies `--delete`) |
+| `--delete-commit` | FastSync-only: keep the pre-2.28 atomic timing — delete only after the whole transfer succeeded (identical timing to `--delete-after`) |
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) |
| `--max-delete ` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync |
| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication) |
@@ -505,8 +512,8 @@ features without changing the meaning of ordinary compatibility options.
|---|---|
| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. |
| `-z [level]`, `--compress [level]` | Enable streaming compression (default `zstd`), levels 1-22. |
-| `--compress-level ` | Set the compression level. |
-| `--zc ` | Alias for `--compress-choice`. FastSync supports `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, and `auto`; `zlibx` behaves as `zlib`. |
+| `--compress-level ` | Set the compression level (1-22). Omitted, each codec uses its rsync default: zstd 3, zlib/zlibx 6, lz4 ignored. |
+| `--zc ` | Alias for `--compress-choice`. FastSync supports `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, and `auto`; `zlib`/`zlibx` share the same literal-only zlib path. |
| `--zl ` | Alias for `--compress-level`. |
| `--skip-compress ` | Skip compression for `/`- or `,`-separated suffixes; defaults to rsync 3.4.1's built-in list. Incompatible with `--chunk-serialization`. |
| `--compress-threads ` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. |
@@ -564,6 +571,7 @@ remote SSH argv is already built injection-safe.
| `--compare-dest ` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`). |
| `--copy-dest ` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination. |
| `--link-dest ` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win). |
+| `--verify-basis` | FastSync-only: require a basis hit to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync). |
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk). |
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`). |
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch). |
@@ -571,6 +579,7 @@ remote SSH argv is already built injection-safe.
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). |
| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). |
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). |
+| `--delete-commit` | FastSync-only: atomic delete-after timing (only after the whole transfer succeeded). |
| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). |
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected). |
| `--max-delete ` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync. |
@@ -789,7 +798,7 @@ before the module list, before authentication, and the connecting peer address
## Protocol and Security
-FastSync protocol version `2.26.0` is shared by the client and server. The
+FastSync protocol version `2.28.0` is shared by the client and server. The
current protocol is sender-driven and includes configuration negotiation,
including the maximum allocation limit, incremental checks, checksums,
manifests, keep-alives, abort handling, per-file remove-source results, and
diff --git a/RSYNC_COMPAT.md b/RSYNC_COMPAT.md
index a75a51c..8701c30 100644
--- a/RSYNC_COMPAT.md
+++ b/RSYNC_COMPAT.md
@@ -6,10 +6,10 @@ This document maps rsync's full feature set to FastSync's current implementation
| Status | Count | Description |
|--------|-------|-------------|
-| ✅ Parity | 106 | Reproduces rsync's semantics for this option's scope |
-| ⚠️ Caveat | 27 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) |
-| ❌ Divergent | 23 | Rejected, an accepted no-op, deliberately non-rsync (native config/auth/batch, privileged namespaces, safe-subset privilege), or impossible on any portable filesystem call |
-| **Total** | **156** | One row per rsync option/feature group; a row may name several spellings |
+| ✅ Parity | 116 | Reproduces rsync's semantics for this option's scope |
+| ⚠️ Caveat | 14 | Wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) |
+| ❌ Divergent | 27 | Rejected, an accepted no-op, deliberately non-rsync (native config/auth/batch, privileged namespaces, safe-subset privilege), or impossible on any portable filesystem call |
+| **Total** | **157** | One row per rsync option/feature group; a row may name several spellings |
This matrix reports honest rsync parity, not "implemented" as a synonym for
"parsed". A ✅ row matches rsync for the option's scope. A ⚠️ row is real and
@@ -21,6 +21,40 @@ the batch container, `--fake-super`'s xattr format, `--copy-as` credential
switching), or impossible (`-N`/`--crtimes`). The counts are derived from the
rows below; update them together with the table.
+**Option wave (protocol 2.26.0 → 2.27.0).** A differential pass against rsync
+3.4.1 over the remaining option caveats. `--bwlimit` now parses rsync's units
+exactly and paces like rsync's leaky bucket; `--ignore-errors` reproduces
+rsync's default (an I/O error skips deletion unless the flag is set, while the
+readable tree still transfers and the run exits 23); the `--info` categories
+that map to a FastSync event (`name`, `flist`, `del`, `remove`, `nonreg`,
+`progress`) now emit rsync's line format, including real-run `deleting PATH` /
+`*deleting` lines carried over a new `report_deletes` wire bool; and two
+genuinely non-interoperable residuals are reclassified divergent (`-M` over a
+daemon/TCP connection, which FastSync's binary config handshake has no argv
+channel for, and receiver-side `protect`/`risk` re-derivation for
+destination-only entries, which would need a receiver filter engine; the
+latter was later implemented in parity track 4a below, so only `-M` over a
+daemon remains divergent). After the
+combined `fix/parity-stats` + `fix/parity-options` + `fix/parity-fs` passes (and
+the later `fix/parity-review` correction that moved `--delete-delay` to ⚠️) the
+matrix is **111 ✅ / 13 ⚠️ / 33 ❌ = 157**.
+
+**Parity track 1 (no-wire, on `feat/parity-2.28`).** Three residual burn-downs that required no protocol change: (1) `-n --delete` now propagates the filter-excluded and size-pruned protected prefixes (and the synchronized-directory scope) into the dry-run keep-set manifest, so its would-delete report matches rsync for source-derived protections and no longer lists the file merely being updated (the destination-only exclude/filter residual was closed by track 4a below; readdir ordering remains, leaving the row ⚠️); (2) `--delete-delay` now charges `--max-delete` on actual removals and re-scans a queued directory at commit to remove content created after the plan, with an independent cap bounding the deferred list and the row's prior caveat closed (ordering of partial removals remains, leaving the row ⚠️); and (3) `--info=name2` now emits `NAME is uptodate` and `--info=name` emits the leading `./` root name line (only the root-line trigger condition and the receiver-side `skip` wording remain, leaving the row ⚠️). The matrix is now **111 ✅ / 14 ⚠️ / 32 ❌ = 157**.
+
+**Parity track 2b (no-wire, on `feat/parity-2.28`).** An opt-in, paths-only metadata pre-count for `--progress`/`-P`/`--info=progress` (only when not `--quiet`) now gives the `to-chk` denominator rsync's full file-list total (every regular file, directory, symlink and special plus the transfer root) and emits the per-directory/symlink/special name lines, in both the sequential and `--threads` paths; `--delete-during`/`--delete-delay` reuse their keep-set pre-scan so no second walk happens, and non-progress runs are unaffected. A fresh multi-directory differential against rsync 3.4.1 matches the name set and the `to-chk` denominator, while a single-file transfer stays byte-identical; the remaining caveats are emission order (rsync's sorted depth-first vs FastSync's readdir/BFS stream, so the `to-chk` numerator and interleaving differ) and re-run/receiver-state-driven over-naming (unconditional `./`, ancestor dirs emitted with a transferred child, and no quick-check for symlinks/empty dirs), keeping the row ⚠️. The matrix is unchanged at **111 ✅ / 14 ⚠️ / 32 ❌ = 157**.
+
+**Parity track 3a (no-wire, on `feat/parity-2.28`).** Three codec refinements, none of which change the wire layout (`PROTOCOL_VERSION` stays 2.28.0): (1) an omitted `--compress-level` now resolves to rsync 3.4.1's per-codec default (zstd 3, zlib/zlibx 6, lz4 ignored) and an explicit level is clamped per codec (zstd 1-22, zlib/zlibx 1-9), verified against `rsync --debug=NSTR1`; (2) `auto` now honors `RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST` (rsync's whitespace-separated preference syntax, unknown names skipped, first supported wins, all-unknown is exit 4) before the compiled-in order, with an explicit `--zc`/`--cc` still winning; and (3) `--compress-choice=zlibx` is reclassified out of the caveat list — FastSync's zlib stream already carries only the literal/delta bytes (rsync's zlibx semantics) and is observably identical to `--zc=zlib`, so the zlib/zlibx aliasing is only an implementation detail. The matrix is now **113 ✅ / 12 ⚠️ / 32 ❌ = 157**.
+
+**Parity track 3b (no-wire, on `feat/parity-2.28`).** `--checksum-choice`/`--cc` is reclassified out of the caveat list: its only documented residual was the delta BLOCK strong checksum (rsync applies the negotiated algorithm there; FastSync keeps a fixed xxHash32, `DeltaBlockSig`). A pre-seeded delta-transfer differential against rsync 3.4.1 (probe: `--no-whole-file -B8192 --stats --out-format=%c|%C %n` for rsync vs `--incremental --delta` for FastSync) shows the choice is **not observable** in the surface the parity gate compares — across `xxh64`/`xxh128`/`xxh3`/`md5`/`md4`/`sha1` and both two-name orders the destination tree is byte-identical, the `Matched data`/`Literal data`/`Total transferred file size` counters are unchanged (and, with the block size pinned, equal to rsync's), the `%c` block-checksum token is invariant, and the exit code is 0. The negotiated algorithm is only visible in `%C`, which renders the whole-file transfer digest and is already byte-identical to rsync. No wire field is added (`PROTOCOL_VERSION` stays 2.28.0). The matrix is now **114 ✅ / 11 ⚠️ / 32 ❌ = 157**.
+
+**Parity track 4a (wire, on `feat/parity-2.28`; `PROTOCOL_VERSION` stays 2.28.0).** The receiver now has a filter engine for deletion: the sender compiles its root-level selection rules exactly as the scanner does (`filter_base_build`, covering `--filter`/`-f`, `--exclude`/`--include`, `-C` and the `protect`/`risk`/`hide`/`show` words) and streams them as one bounded, self-describing block appended to the config frame (per rule: action, sides, anchored, dir-only, negate, owner, pattern; strictly validated with a bounded rule count and total pattern+owner bytes, and an unknown action/sides is a protocol error). The receiver reconstructs `protect_rules` and evaluates them first-match-wins against each extraneous destination path in every delete timing — the whole-tree commit walker (plain `--delete`/`--delete-before`/`--delete-after`), the `--delete-during`/`--delete-delay` per-directory plans, and the `-n` would-delete enumeration — so a `protect`/`P` rule now shields a DESTINATION-ONLY entry that never appeared on the sender, with `risk`/`R` cancelling. The existing sender-derived protected-prefix behavior is preserved when no rules are sent (and for source-derived protections), and `--delete-excluded` semantics are unchanged. Per-directory merge (`:`/`.`) receiver-side re-derivation is the remaining residual: those rules are still enforced only through the sender-derived protected prefixes, so a destination-only entry matching ONLY a per-directory merge rule is not yet shielded (the `-F` row keeps this documented). Differential-tested against rsync 3.4.1: `filter_protect`, `filter_protect_during`, `filter_protect_delay` (`-a --delete[-during|-delay] --filter='P *.log'` over seeded destination-only `.log` extras) plus a FastSync regression test for the dry-run would-delete enumeration (`TestFilterProtect`). The `--filter=RULE` row moves ❌ → ✅. The matrix is now **115 ✅ / 11 ⚠️ / 31 ❌ = 157**.
+
+**Parity track 5a (wire, on `feat/parity-2.28`; `PROTOCOL_VERSION` stays 2.28.0 by project decision).** The three basis-dir options (`--compare-dest`/`--copy-dest`/`--link-dest`) now default to rsync's metadata quick-check instead of FastSync's historical xxHash64 content equality: a basis hit is accepted on equal size plus equal mtime (or size alone under `--size-only`; `-I` disables matching), so a same-size/different-content basis is trusted exactly as rsync trusts it. A new FastSync-only, long-only `--verify-basis` flag restores the stricter whole-file content equality, and its bool is appended to the basis block of the config frame (the golden wire frame grew by one int to 886 bytes; still 2.28.0). `--verify-basis` hashes the basis by streaming its confined descriptor, so an arbitrarily large basis is verified without buffering. Basis materialization is also no longer capped at the 256 MiB whole-file payload bound: a `--copy-dest` hit streams the basis through a bounded buffer, a `--link-dest` copy fallback streams from the basis, and a hit of any size is materialized (a basis MISS still falls back to the normal transfer, which keeps its own bound). A `--copy-dest` hit re-applies the SOURCE attributes (the sender now transmits the source metadata with the basis check frame), matching rsync's "copy then fix attributes"; a `--link-dest` success keeps the shared inode's own attributes exactly as before (writing through the shared inode would mutate the basis). The `--compare-dest`/`--copy-dest`/`--link-dest` rows move ❌ → ⚠️ (residuals: the relative-DIR resolution base and the over-limit MISS refusal). The matrix is now **116 ✅ / 13 ⚠️ / 28 ❌ = 157**.
+
+**Parity track 5b (no-wire, on `feat/parity-2.28`; `PROTOCOL_VERSION` stays 2.28.0 by project decision).** `-y`/`--fuzzy` is reclassified ❌ → ⚠️: the receiver-side similar-file basis is an internal bandwidth optimization (the config `fuzzy` bool over the existing receiver-driven delta handshake, unchanged since 2.9.0), and the transferred tree is byte-exact by design regardless of which basis — or no basis — is chosen. A probe against real rsync 3.4.1 showed the remaining difference is not the name rule (FastSync already ports `util1.c fuzzy_distance`/`find_filename_suffix` plus the exact size+mtime pass) but candidate ELIGIBILITY: rsync will pick a fuzzy basis whose size ratio to the source is unrestricted (empirically from 0.25× to 10000×, and for files as small as 300 B), while FastSync's `delta_should_attempt` gate caps the ratio at 10× and requires both files ≥ 16 KiB, so an out-of-window sibling is declined and the file is sent whole. The choice is observable only as bandwidth (`Matched data`/`Literal data`/`Total transferred file size` in `--stats`); the destination tree and exit code are identical either way. A new differential case (`fuzzy_basis`: same-suffix sibling one name-edit away, content identical, block size pinned to 8192) asserts tree **and** normalized `--stats` parity where the two tools' choices coincide; `TestFuzzy` pins the window boundary on both sides (a >10× and a <16 KiB sibling are declined by FastSync while rsync uses them, both trees byte-identical). No wire field changed. The matrix is now **116 ✅ / 14 ⚠️ / 27 ❌ = 157**.
+
+**Lockstep track 6 (delete default; `PROTOCOL_VERSION` stays 2.28.0).** Plain `--delete` with no explicit timing flag now defaults to rsync's delete-during (`--del`) timing: the client normalizes it onto the existing `delete_during` wire bool in `cli_finalize_config`, so no config-frame field was added, and a tight destination no longer has to hold the whole old+new tree at once (the old atomic commit could hit `ENOSPC`). The old late whole-tree commit is opt-in via `--delete-after` or the FastSync-only long spelling `--delete-commit`, which selects the identical `delete_after` timing (documented equivalence). Precedence is unchanged and order-independent: each timing flag implies `--delete`, at most one timing flag may be given, and a timing flag with `--no-delete` is rejected. `-d/--dirs` still falls back to the end commit; `--delay-updates` still deletes genuine extras before publication (the per-directory skip list protects the staging dir); `--files-from`/`-R` scope is unchanged. The per-directory `STATUS_DELETE_PLAN` frame gained a one-int `apply` flag (still 2.28.0): the one-shot per-run config block (protected prefixes, size-pruned mirrors, `--delete-missing-args` exact paths) is now always sent first on a config-only carrier with `apply=false`, fixing a latent bug where a `--delete-missing-args` run whose `--files-from` list synchronized no directory never transmitted its exact deletions. Differential evidence: `delete` (plain, vs rsync's default), `delete_commit` (FastSync `--delete-commit` vs rsync `--delete-after`), and `filter_protect_after` (whole-tree protect) cases; `TestDeleteTimingFinalStateParity` compares plain `--delete`/`--delete-commit` against rsync on completed runs, and `TestDeleteTimingFailure` proves plain `--delete` removes reached extras on a mid-transfer abort while `--delete-commit` removes nothing. The matrix is unchanged at **116 ✅ / 14 ⚠️ / 27 ❌ = 157** (the `--delete`/`--delete-during` rows stay ⚠️ for the abort boundary; `--delete-after` stays ✅).
+
**Parity completion wave (protocol 2.23.0 → 2.26.0).** This wave closed the
remaining gaps the rsync-parity wave left open (delete timing, wire counters and
output, codec breadth, general `-R`/`-d`, the full filter grammar, receiver-side
@@ -51,7 +85,7 @@ Every one of those has an entry below with its remaining caveats.
| `-q`, `--quiet` | Suppress non-error messages | ✅ Parity | Suppresses client output while preserving errors |
| `--help` | Show help | ✅ Parity | Prints usage and exits. A lone `-h` with no other transfer arguments also prints help (protocol 2.26.0), matching the rsync idiom; `-h` alongside a transfer keeps its rsync meaning of `--human-readable` (see that row) |
| `-V`, `--version` | Print version | ✅ Parity | |
-| `--info=FLAGS` | Fine-grained info verbosity | ⚠️ Caveat | Protocol 2.26.0 accepts rsync 3.4.1's full `--info` vocabulary — `backup`, `copy`, `del`, `flist`, `misc`, `mount`, `name`, `nonreg`, `progress`, `remove`, `skip`, `stats`, `symsafe`, `all`, `none` — with optional level suffixes (`--info=stats2`), so a valid rsync invocation is never rejected up front. The categories that map to a FastSync channel emit (`copy`, `name`, `misc`, `skip`, `stats`); the remaining rsync categories are accepted silently, with no output. `none` suppresses info output, explicit flags override `--verbose`, and a genuinely unknown name is still rejected by name (matching rsync). **Caveat:** many accepted rsync categories produce no output (e.g. `del`, `flist`, `remove`, `progress`, `symsafe`, `mount`, `nonreg`, `backup`), so e.g. `--info=progress` is accepted for CLI compatibility only; `name` maps to the `copy` channel rather than rsync's per-file name output |
+| `--info=FLAGS` | Fine-grained info verbosity | ⚠️ Caveat | Accepts rsync 3.4.1's full `--info` vocabulary — `backup`, `copy`, `del`, `flist`, `misc`, `mount`, `name`, `nonreg`, `progress`, `remove`, `skip`, `stats`, `symsafe`, `all`, `none` — with optional level suffixes (`--info=stats2`), so a valid rsync invocation is never rejected up front. Protocol 2.27.0 wires the categories that map to a real FastSync event, matching rsync's line format: `name` prints the updated entry names (with the ` -> target` link suffix), `flist` prints `sending incremental file list`, `del` prints `deleting PATH` (or `*deleting PATH` under `-i`/`--out-format`) for both dry-run would-delete and real deletions (real runs carry the removed paths over the new `report_deletes` wire bool), `remove` prints `sender removed PATH`, `nonreg` prints `skipping non-regular file "NAME"`, `progress` drives the per-file progress output, and `copy`/`misc`/`skip`/`stats` keep their existing channels. `none` suppresses info output, explicit flags override `--verbose`, and a genuinely unknown name is still rejected by name (matching rsync). **Fixed (no-wire):** `--info=name2` (and higher) also prints rsync's `NAME is uptodate` lines for entries the receiver already has, and `--info=name` emits the leading transfer-root `./` name line before the first transferred entry (the marker rides in the existing `info_level` bitset; differential tests vs rsync 3.4.1). **Caveat:** the root `./` line is emitted before the first transferred name rather than keyed off rsync's root-attribute-change decision, so a pre-existing root that rsync leaves untouched can differ; the categories with no client-observable event stay accepted-but-silent — `symsafe`, `mount`, and `backup` (the backup happens on the receiver, which FastSync's protocol does not echo back); and `skip` maps to FastSync's sender-side skip logging rather than rsync's receiver-side "not creating new file" lines |
| `--debug=FLAGS` | Fine-grained debug verbosity | ⚠️ Caveat | Protocol 2.26.0 accepts rsync 3.4.1's full `--debug` vocabulary with optional level suffixes. FastSync emits for its own channels (`io`, `proto`, `pack`, `util`, plus the aliases `hl`/`owner`); the rsync-only categories (`acl`, `filter`, `send`, ...) are accepted silently. `--debug=help` lists the flags; a genuinely unknown name is rejected by name. **Caveat:** most accepted rsync categories produce no output (e.g. `acl`, `filter`, `send`, `flist`, `del`, `deltasum`, `hash`, `recv`, `time`), so they are accepted for CLI compatibility only |
| `--stderr=MODE` | Change stderr output mode | ❌ Divergent | `errors` (default) and `all` are supported; `client` is rejected with a clear error (`--stderr=client is not supported`) because FastSync has no rsync client-message channel — the rejection itself is the documented behavior (Phase 7 Wave B decision). The modes that exist work; the missing rsync channel cannot be emulated without a wire change |
| `--msgs2stderr`, `--no-msgs2stderr` | Deprecated `--stderr` aliases | ⚠️ Caveat | `--msgs2stderr` maps to `--stderr=all` (supported, matching rsync). `--no-msgs2stderr` is rsync's spelling of `--stderr=client`, which FastSync has no client-message channel for, so it maps to the errors-only default instead of reproducing rsync's client mode. See `--stderr=MODE` |
@@ -64,12 +98,12 @@ Every one of those has an entry below with its remaining caveats.
| Flag | Rsync Description | FastSync Status | Notes |
|------|-------------------|-----------------|-------|
-| `--stats` | Give transfer stats | ⚠️ Caveat | Prints transfer statistics. Protocol 2.25.0 populates the receiver-only counters the sender cannot observe: `Matched data` (a delta basis's reused bytes) and `Number of deleted files` come from the receiver's `STATUS_STATS` report, and the protocol-independent lines (regular files transferred, total/transferred file size, literal data, matched data, deleted files, file-list size) match rsync exactly in both the sequential and `--threads` paths. **Remaining divergence:** rsync prints `Number of files` and `Number of created files` with a per-type breakdown (`(reg: X, dir: Y, link: Z)`); FastSync prints the bare transferred-entry count because its scanner does not put directory entries in the transfer list and the sender cannot tell which entries the receiver newly created. `Total bytes sent`/`received` are FastSync wire bytes and are not numerically comparable to rsync's |
+| `--stats` | Give transfer stats | ⚠️ Caveat | Prints transfer statistics. Protocol 2.25.0 populates the receiver-only counters the sender cannot observe (`Matched data`, `Number of deleted files`) from the receiver's `STATUS_STATS` report; the sender tracks the scanned file list per type so `Number of files` carries rsync's `(reg: X, dir: Y, link: Z, special: W)` breakdown (directories come from the scanner's captured directory list, present for `-a`/`-t`/`-p`), `Number of regular files transferred` excludes symlinks/specials and up-to-date files, `Total file size` includes symlink target lengths, and `Total transferred file size` counts only transferred files. **Protocol 2.28.0 extends `STATUS_STATS`** with receiver-observed `literal_bytes` and the four `created_*` counters: `Number of created files` now carries rsync's `(reg/dir/link/special)` breakdown (the receiver reports which destination entries it newly created, including implicitly-created parent directories below the transfer root) and `Literal data` is exact for a delta transfer (the receiver counts the literal fragments it stored, not the whole source size) — all differential-tested in the sequential and `--threads` paths against rsync 3.4.1 for fresh-create, update and delta shapes. **Remaining divergences:** a recursive scan that preserves no directory attribute (`-r` without `-t`/`-p`) captures no directory entries, so the `dir:` category is then omitted from `Number of files`; rsync's per-type breakdown on `Number of deleted files` is not reproduced; and `Total bytes sent`/`received` are FastSync wire bytes framed differently from rsync's, so they are not numerically comparable |
| `-h`, `--human-readable` | Human-readable numbers | ✅ Parity | Formats transfer byte and rate counts using rsync's **decimal** (base-1000) units, matching rsync `-h` (e.g. `1.23M`), not binary units. **A lone `-h` with no transfer arguments prints help instead** (protocol 2.26.0), matching the rsync idiom; `-h` alongside a transfer remains human-readable |
| `-i`, `--itemize-changes` | Per-file change summary | ✅ Parity | Prints rsync-style `>f+++++++++` lines to stdout only for files actually sent (also under `-j`/`--threads`); unchanged files print nothing, matching single-`-i` behavior |
-| `--progress` | Show progress | ⚠️ Caveat | Protocol 2.25.0 prints rsync-style per-file progress blocks (percent, transferred/total bytes, rate, elapsed, `(xfr#N, to-chk=M/T)`) fed by the receiver's `STATUS_STATS`, in both the sequential and `--threads` send paths; the first frame for a sub-32 KiB file is byte-identical to rsync. **Remaining divergences:** FastSync does not print rsync's leading `./` whole-transfer line, its `to-chk` total differs by the source-root entry (the scanner does not emit the root directory as a transfer entry), and the rate/ETA are wall-clock dependent, so only the first frame is pinned against rsync |
+| `--progress` | Show progress | ⚠️ Caveat | Protocol 2.25.0 prints rsync-style per-file progress blocks (percent, transferred/total bytes, rate, elapsed, `(xfr#N, to-chk=M/T)`) fed by the receiver's `STATUS_STATS`, in both the sequential and `--threads` send paths. FastSync also prints rsync's leading `./` transfer-root line and, when progress is requested (`--progress`/`-P`/`--info=progress`) and not `--quiet`, runs a **paths-only metadata pre-scan** (no file reads, no hashing) that supplies rsync's file-list total `T` for the `to-chk` denominator and the directory names; `--delete-during`/`--delete-delay` reuse their existing keep-set pre-scan instead of walking twice, and non-progress runs are untouched. Per-directory name lines are emitted (trailing `/`), and symlink (` -> target`) and special entries are named too, so a **fresh multi-directory tree's name set and `to-chk` denominator match rsync 3.4.1** (differential test, sequential and `--threads`) and a **single-file transfer's name lines and deterministic frames remain byte-identical** to rsync. **Remaining divergences:** rsync emits entries in sorted depth-first order while FastSync streams them in the scanner's readdir/BFS order, so the interleaving and the `to-chk` numerator differ (the denominator matches); the leading `./` root line is emitted unconditionally rather than keyed off rsync's root-attribute-change decision, and an ancestor directory line is emitted whenever a child transfers (rsync suppresses it when the directory itself is unchanged); on a re-run, entries without a quick-check (symlinks, empty directories) are still named where rsync stays silent; and the rate/ETA are wall-clock dependent |
| `-P` | Same as --partial --progress | ✅ Parity | Parses to `--partial` + `--progress`. The independent `--partial` retention semantics are rsync parity: an interrupted write retains the already-written temp at the destination (best-effort) so a later `--append`/`--append-verify` can resume. Progress presentation is owned by the `--progress` row; there is no separate `-P` divergence |
-| `--out-format=FORMAT` | Custom output format | ⚠️ Caveat | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%c` `%C` `%i` `%M` `%o` `%U` `%G` `%t` `%%`. Protocol 2.25.0 adds the wire counters: `%C` is the whole-file digest (default `xxh128`, seed 0), so `%C %l %n` matches rsync byte-for-byte for a whole-file transfer. **Remaining divergences:** `%b` counts FastSync's own wire bytes (framing and checksum trailer), not rsync's protocol-specific count, so the two are not numerically equal; `%c` matches rsync's 16-byte block-sum header for whole-file transfers but differs in delta mode (each counts its own handshake bytes). **Also:** when `--checksum-choice=xxh64` is selected explicitly, `%C` still prints an xxh128 digest rather than the selected xxh64 (`change_list.c:208-220`) |
+| `--out-format=FORMAT` | Custom output format | ❌ Divergent | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%c` `%C` `%i` `%M` `%o` `%U` `%G` `%t` `%%`. `%C` now uses the negotiated transfer algorithm (`--checksum-choice`, default `xxh128`, seed 0) and renders every algorithm exactly like rsync — xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, `none` a blank 2-char column — differential-tested across all algorithms. `%f`/`%n`/`%l`/`%i`/`%M`/`%U`/`%G`/`%B` also match. **Reclassified because `%b`/`%c` are protocol-specific and cannot match:** a differential against rsync 3.4.1 shows whole-file `%c = 16` for both, but rsync whole-file `%b = filesize + 27 + transfer-digest-bytes` (39 for a 0-byte file; 43/35/47 for xxh128/xxh64/sha1 on a 12-byte file) while FastSync `%b` counts its own framing; in delta mode rsync `%c = 16 + 6·ceil(filesize/block_size)` (verified at block sizes 512/700/1024/2048) while FastSync counts its own signature handshake, and rsync `%b` is its token stream. FastSync's wire bytes are a different quantity, so exact `%b`/delta-`%c` equality is impossible |
| `--log-file=FILE` | Log to file | ✅ Parity | `log_file` config field |
| `--log-file-format=FMT` | Log format | ✅ Parity | Requires `--log-file`; writes one template line per transferred file using the same token set as `--out-format` (including `%b` as the wire byte count) |
| `--8-bit-output`, `-8` | Leave high-bit chars unescaped | ✅ Parity | Applies to displayed paths and protocol debug output |
@@ -81,19 +115,19 @@ Every one of those has an entry below with its remaining caveats.
|------|-------------------|-----------------|-------|
| `--exclude-from=FILE` | Read exclude patterns from file | ✅ Parity | Reads patterns from file |
| `--include-from=FILE` | Read include patterns from file | ✅ Parity | Reads patterns from file |
-| `--filter=RULE` | Add file-filtering rule | ⚠️ Caveat | The short `-f` **is** bound to `--filter` (the old FastSync sendfile conflict is gone; sendfile is long-only `--sendfile`), and `-f RULE`, `-f=RULE`, `--filter=RULE` and the two-argument form all parse. Protocol 2.26.0 implements rsync's filter grammar: `+`/`-`, `include`/`exclude`, a leading `/` anchor (to the transfer root or a `.rsync-filter` file's directory), a trailing `/` dir-only rule, and the `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R` and `clear`/`!` words, including the `:`/`.` modifiers. First match wins; the filter layer is independent of `--exclude`/`--include`. **Remaining divergence:** the receiver-mirror protection a `protect`/`risk` rule produces is derived from the sender's source traversal, so a rule that would match only a destination-only entry is not re-derived on the receiver; destination-only deletion protection continues to come from the ordinary sender-derived protected-prefix mechanism |
+| `--filter=RULE` | Add file-filtering rule | ✅ Parity | The short `-f` **is** bound to `--filter` (the old FastSync sendfile conflict is gone; sendfile is long-only `--sendfile`), and `-f RULE`, `-f=RULE`, `--filter=RULE` and the two-argument form all parse. Protocol 2.26.0 implements rsync's filter grammar: `+`/`-`, `include`/`exclude`, a leading `/` anchor (to the transfer root or a `.rsync-filter` file's directory), a trailing `/` dir-only rule, and the `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R` and `clear`/`!` words, including the `:`/`.` modifiers. First match wins; the filter layer is independent of `--exclude`/`--include`. **Track 4a (protocol 2.28.0) adds the receiver filter engine:** the sender compiles its root-level rules exactly as the scanner does (`filter_base_build`) and streams them as one bounded, self-describing config-frame block; the receiver reconstructs them and re-applies first-match-wins to every extraneous destination path during deletion, so a `P *.log` rule protects a destination-only `extra.log` (differential `filter_protect`/`filter_protect_during`/`filter_protect_delay` vs rsync 3.4.1, plus the `-n` would-delete enumeration) — matching rsync's dual-sided engine for the command-line rule set. **Remaining residual:** per-directory merge (`:`/`.`, and therefore `-F`) is not yet re-derived on the receiver; a destination-only entry that matches ONLY a per-directory merge rule is still protected only through the sender-derived source-mirror prefixes, not by the received base rule list |
| `--files-from=FILE` | Read source file list from file | ✅ Parity | Entries are paths relative to the source root (leading `./` stripped, `..`/absolute rejected at parse time, blank lines ignored; NUL-delimited with `-0`). A listed regular file is transferred; a listed directory transfers its whole subtree (FastSync recursion is always on). Non-listed paths are pruned by the scanner; the delete manifest is scoped to the listed directory subtrees. A listed entry that does not exist is a hard error unless `--ignore-missing-args`/`--delete-missing-args` is given. **An empty list is a zero-transfer success (exit 0), matching rsync 3.4.1** — the earlier claim that rsync reports "no source files specified" was wrong. Scalability note: `file_list_affects` is O(list size) per scanned entry, so a very large list against a huge tree is quadratic (the documented bound) |
| `-0`, `--from0` | Delimit *-from files with NULs | ✅ Parity | `--files-from` entries become NUL-delimited; the flag may appear before or after `--files-from` on the command line. NUL mode preserves entry bytes exactly (trailing CR/LF are part of the name; only newline mode trims them) |
| `--max-size=SIZE` | Skip files larger than SIZE | ✅ Parity | `max_size` in scanner |
| `--min-size=SIZE` | Skip files smaller than SIZE | ✅ Parity | `min_size` in scanner |
| `-I`, `--ignore-times` | Don't skip files matching size+time | ✅ Parity | `ignore_times` config field (crosses the wire). Disables the size+mtime quick-check in the `--incremental` per-file handshake and the basis-dir quick-match, forcing the file to be transferred rather than skipped as unchanged. Receiver-side policy: `match_by_metadata` (file_receive.c) is bypassed, so the receiver never replies `STATUS_OK` for a matching size+mtime. Requires `--incremental` to have the handshake to act on (rsync does its quick check by default; FastSync's `-I`/`--size-only`/`--modify-window` only take effect under `--incremental`, exactly like they take effect through the basis check) |
-| `--size-only` | Skip based on size only | ✅ Parity | With `--incremental`, ignores mtime |
+| `--size-only` | Skip based on size only | ✅ Parity | With `--incremental`, ignores mtime. Applies identically to the basis-dir quick-check (track 5a): a same-size basis is a hit on size alone, and under the FastSync-only `--verify-basis` the content digest is still required (size-only drops the mtime leg, never the explicit verify) |
| `-@`, `--modify-window=NUM` | Mod-time comparison accuracy | ✅ Parity | Whole-second tolerance with nanosecond-aware comparisons |
| `--existing` | Skip creating new files on receiver | ✅ Parity | Existing destination files continue through normal update handling. The rsync man-page alias `--ignore-non-existing` sets the same flag |
| `--ignore-existing` | Skip updating existing files | ✅ Parity | `ignore_existing` config field (crosses the wire; receiver-side policy). Protocol 2.26.0 short-circuits in the per-file check **before any payload**: when the destination entry already exists, the receiver answers the skip during the incremental handshake instead of letting the sender stream data that would be discarded, so an existing 4 MiB destination costs only the config/check frames (verified with a counting proxy, matching rsync). The write-time paths (regular, delay-updates-staged, hardlink-sibling, special/device) still return `FILE_SAVE_SKIPPED` without overwriting, and `--backup` is disabled for skipped files. Like rsync, it does not apply to directories/symlinks. Combines with `-j`/`--threads` and `--delay-updates` |
| `--remove-source-files` | Sender removes regular files after confirmed transfer | ✅ Parity | |
| `-x`, `--one-file-system` | Do not cross filesystem boundaries | ✅ Parity | Sender scanner captures the root device and does not descend into mount-point crossings (`st_dev` differs). **Protocol 2.23.0 matches rsync's entry emission:** the mount-point directory itself is emitted as a payload-less directory entry (so the destination gets an empty directory) while its contents are skipped; previously the crossing subdirectory was dropped entirely |
-| `-F` | Add the default `.rsync-filter` rules | ✅ Parity | Reads one filter rule per line from each directory's `.rsync-filter` file during traversal and applies it to that directory's subtree; the current directory's rules are evaluated before its ancestors', so deeper files override shallower ones and per-directory files override the command-line `--filter`/`-C` base by default (first match wins). **A single `-F` transfers the `.rsync-filter` files themselves, matching rsync; a repeated `-FF` additionally excludes them** (rsync 3.4.1's `-F`/`-FF` are exactly these two rules, with no `.cvsignore` branch). Unsupported/unparseable rules inside a per-directory file fail the scan with a clear error |
+| `-F` | Add the default `.rsync-filter` rules | ✅ Parity | Reads one filter rule per line from each directory's `.rsync-filter` file during traversal and applies it to that directory's subtree; the current directory's rules are evaluated before its ancestors', so deeper files override shallower ones and per-directory files override the command-line `--filter`/`-C` base by default (first match wins). **A single `-F` transfers the `.rsync-filter` files themselves, matching rsync; a repeated `-FF` additionally excludes them** (rsync 3.4.1's `-F`/`-FF` are exactly these two rules, with no `.cvsignore` branch). Unsupported/unparseable rules inside a per-directory file fail the scan with a clear error. **Residual (track 4a):** per-directory rules are still enforced receiver-side only through the sender-derived source-mirror protected prefixes; the base-rule receiver filter engine does not carry per-directory rules, so a destination-only entry matching ONLY a `.rsync-filter` rule is not yet shielded from `--delete` |
## 4. Directory Options
@@ -101,8 +135,8 @@ Every one of those has an entry below with its remaining caveats.
|------|-------------------|-----------------|-------|
| `-r`, `--recursive` | Recurse into directories | ✅ Parity | Default behavior |
| `-R`, `--relative` | Use relative path names | ✅ Parity | Protocol 2.26.0 implements rsync's general `-R` path semantics: without a cut the source argument is mirrored in full below the destination root; a `/./` cut in the source argument (`src/./foo`) makes everything after the cut the destination prefix, so the layout matches rsync's relative reconstruction; and `--files-from` entries land under their bare relative path. The delete manifest derives from the sent (relative) paths and is scoped to the transferred prefix subtree, so `--delete` cannot remove destination content outside that prefix (a blocker fix). Works single-threaded and under `-j`/`--threads` |
-| `--no-implied-dirs` | Don't send implied dirs with -R | ✅ Parity | With `-R`, rsync creates the ancestor directories implied by a listed path and, with `--no-implied-dirs`, omits them from the transfer so the destination directories keep the destination's own mode/mtime. Protocol 2.26.0 matches this: the implied-dir walk applies the transfer's per-attribute metadata only to explicitly transferred directories, and a differential test verifies the modes and mtimes of the implied parents against rsync with and without the flag. Works single-threaded and under `-j`/`--threads` |
-| `-d`, `--dirs`, `--old-dirs`, `--old-d` | Transfer dirs without recursing | ⚠️ Caveat | Protocol 2.26.0 implements rsync's one-level `-d` listing for `dir`, `dir/` and `.`: the source's immediate contents are transferred (files with content, directories as explicit entries), matching rsync's destination tree in a differential test. `--dirs --files-from` transfers exactly the listed items — a listed directory is created empty and a listed file with content — under the same `-R` layout rules. Directory entries cross as `STATUS_MKDIR` and appear in the delete manifest, so `--delete` prunes correctly and an empty listed directory survives. Directory times are applied at the end of the transfer; modes/ownership follow the per-attribute policy. **Remaining divergence:** a plain recursive `-a` scan still does not create empty source directories (directory entries are record-only unless `-d`/`--files-from` explicitly lists a directory), and under `--delay-updates` directories are created immediately while only regular files are staged (see the recursive-empty-directory residual in the completion-wave section) |
+| `--no-implied-dirs` | Don't send implied dirs with -R | ✅ Parity | With `-R`, rsync creates the ancestor directories implied by a listed path and, with `--no-implied-dirs`, omits their attributes from the transfer so they keep the destination's own state (or are created with default attributes when absent). Protocol 2.26.0 matches this: without `--files-from` the implied-dir walk applies per-attribute metadata only to explicitly transferred directories, and with `-R --files-from` a listed file whose parent is not itself listed is placed normally — the missing implied parent is created with default attributes (not the source's) and the file transfers with `rc 0`, exactly like rsync 3.4.1 (a differential test verifies the modes and mtimes with and without the flag). Works single-threaded and under `-j`/`--threads` |
+| `-d`, `--dirs`, `--old-dirs`, `--old-d` | Transfer dirs without recursing | ✅ Parity | Protocol 2.26.0 implements rsync's one-level `-d` listing for `dir`, `dir/` and `.`: the source's immediate contents are transferred (files with content, directories as explicit entries), matching rsync's destination tree in a differential test. `--dirs --files-from` transfers exactly the listed items — a listed directory is created empty and a listed file with content — under the same `-R` layout rules. A plain recursive scan also recreates empty source directories now: the scanner emits a payload-less directory entry (with metadata) for every traversed directory that produced no transferred or descended child, unless `-m/--prune-empty-dirs` suppresses it or the run is `--files-from`/`--list-only` (a directory emptied by filtering is recreated too, matching rsync). Directory entries cross as `STATUS_MKDIR` and appear in the delete manifest, so `--delete` prunes correctly and an empty listed directory survives; an incoming directory replaces a destination regular file (rsync removes the non-directory and creates the directory), verified differentially. Directory times are applied at the end of the transfer; modes/ownership follow the per-attribute policy. Under `--delay-updates` directories are created immediately while only regular files are staged, exactly as rsync does |
| `--mkpath` | Create missing path components | ✅ Parity | Wire option (client → server). At connection start the server creates the client's destination root directory (and any missing leading components below its own authorized root) when `--mkpath` is set, failing the connection cleanly if it cannot. Without `--mkpath` a destination root that does not exist yet is rejected up front (rsync semantics), so the flag is the only way to transfer into a not-yet-created destination directory. Creation is confined by the same secure mkdir walk as file writes (`O_NOFOLLOW`, no `..`) |
| `--inc-recursive`, `--no-inc-recursive` | Incremental recursion mode | ❌ Divergent | rsync's man-page-only scanning-mode switch (and its short aliases). FastSync always performs a single full recursive scan, so both spellings are rejected as unknown options rather than accepted as a no-op; there is no incremental-recursion engine to toggle. A genuine implementation would be a scan-architecture change with no benefit for FastSync's push model |
@@ -122,12 +156,12 @@ Every one of those has an entry below with its remaining caveats.
| Flag | Rsync Description | FastSync Status | Notes |
|------|-------------------|-----------------|-------|
-| `-n`, `--dry-run` | Trial run with no changes | ⚠️ Caveat | Server-contacting since protocol 2.21.0. The routing predicate `dry_run_targets_server()` selects the server-contacting path for any target a real run would reach over the wire (SSH, daemon `host::module`, explicit `--server-host`/`--server-port`, TLS, source-bind `--address`); the client handshakes with the receiver, which runs the normal read-only per-file check and answers `STATUS_DRY_RUN_TRANSFER`/`STATUS_OK` without mutating anything. Protocol 2.25.0 also reports would-delete lines: with `--delete` the receiver's `STATUS_STATS` carries the extras it would have removed and the client prints rsync-style `*deleting` lines (sequential and `--threads`; control bytes escaped). Dry-run never deletes. **Remaining divergences:** the `*deleting` line ordering can differ from rsync's delete-during walk, and a filtered dry-run can over-report what the real commit would remove |
+| `-n`, `--dry-run` | Trial run with no changes | ⚠️ Caveat | Server-contacting since protocol 2.21.0. The routing predicate `dry_run_targets_server()` selects the server-contacting path for any target a real run would reach over the wire (SSH, daemon `host::module`, explicit `--server-host`/`--server-port`, TLS, source-bind `--address`); the client handshakes with the receiver, which runs the normal read-only per-file check and answers `STATUS_DRY_RUN_TRANSFER`/`STATUS_OK` without mutating anything. Protocol 2.25.0 also reports would-delete lines: with `--delete` the receiver's `STATUS_STATS` carries the extras it would have removed and the client prints rsync-style `*deleting` lines (sequential and `--threads`; control bytes escaped). Dry-run never deletes. **Fixed (no-wire):** the dry-run keep-set manifest now carries the same filter-excluded and size-pruned protected prefixes and synchronized-directory scope a real run sends, and the would-be-transferred entries are in the keep-set, so `-n --delete` lists exactly rsync's extras for source-derived protections — the file merely being updated is kept and an excluded/pruned source entry is protected (`test_dry_run_delete_lines_match_rsync`, differential vs rsync 3.4.1). **Track 4a (protocol 2.28.0):** the received receiver-side `protect`/`risk` rules are also applied to the would-delete enumeration, so a destination-only entry matching a `P` rule is no longer reported (or removed in a real run) — matching rsync (`TestFilterProtect::test_protect_dest_only_dry_run_enumeration`). **Remaining caveat:** the would-delete line ordering follows the destination readdir order rather than rsync's reverse-sorted walk (the differential compares the sorted set) |
| `-b`, `--backup` | Make backups of overwritten files | ✅ Parity | Backup before overwrite |
| `--backup-dir=DIR` | Backup directory hierarchy | ✅ Parity | `backup_dir` config field |
| `--suffix=SUFFIX` | Backup suffix (default ~) | ✅ Parity | `suffix` config field |
-| `--delay-updates` | Put updated files in place at end | ⚠️ Caveat | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication, and **`--force` is honored at publication** (protocol 2.23.0): a staged regular file or symlink may replace a destination directory that blocks it. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). Works in single-threaded and `-j`/`--threads` modes |
-| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ⚠️ Caveat | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). **Protocol 2.23.0 receiver policy: the scratch dir is confined to the receive root — a relative dir is resolved below it, and an absolute path or one containing `..` is rejected by the receiver** (an absolute/foreign-filesystem scratch dir was the divergence; rsync's standalone mode would follow an absolute `--temp-dir`, while its daemon also confines). Temp copies use a unique name there and are atomically renamed into place. **On `EXDEV` (scratch dir and destination on different filesystems) the receiver falls back to a non-atomic copy instead of aborting the transfer**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir |
+| `--delay-updates` | Put updated files in place at end | ❌ Divergent | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication, and **`--force` is honored at publication** (protocol 2.23.0): a staged regular file or symlink may replace a destination directory that blocks it. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). **Reclassified Divergent (differential evidence):** the staging name is fixed and a delayed run wipes a pre-existing destination tree of that name at start even without `--delete`, whereas rsync uses its own internal temp name and leaves a genuine destination entry named `.fastsync-stage` untouched (`test_delay_updates_staging_name_collision_residual`); deletion also runs before publication while rsync's `--delay-updates` implies `--delete-after`. Works in single-threaded and `-j`/`--threads` modes |
+| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ❌ Divergent | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). A **relative** dir matches rsync exactly: it is resolved below the receive/destination root and must already exist (differentially verified: `rsync -a --temp-dir=scratch src/ dst/` and FastSync produce identical trees and an empty scratch dir). **Reclassified as a deliberate divergence because an absolute `--temp-dir` is rejected by the receiver** — it is resolved verbatim by rsync standalone (which will use `/tmp` or any other absolute directory, including one outside the destination), but FastSync's security-reviewed receiver confines the scratch dir to the authorized receive root and rejects any absolute path or one containing `..`. A differential test confirms rsync exits 0 using an absolute scratch dir while FastSync refuses before writing anything into it (the scratch dir stays empty). Its daemon mode also confines relative to the module, but standalone rsync's absolute-temp-dir behavior is not reproduced because it would let a client place receiver scratch files outside the sandbox. Temp copies use a unique name in the scratch dir and are atomically renamed into place; **on `EXDEV` (scratch dir and destination on different filesystems, reachable via a confined relative symlink) the receiver falls back to a non-atomic copy instead of aborting**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir |
| `--partial` | Keep partially transferred files | ✅ Parity | On a failed/interrupted write the already-written temp file is retained at the destination path (best-effort rename instead of unlink) so a later `--append`/`--append-verify` run can resume it. Retention never runs when no data was actually written or under `--ignore-existing`/`--existing` (the destination is not ours to overwrite), and it only ever renames the already-written temp. A failed rename falls back to the normal unlink |
| `--partial-dir=DIR` | Keep partial files in DIR | ✅ Parity | With `--partial`, the working file is written under the confined partial directory (a relative dir below the receive root) and atomically renamed into place once complete, so an interrupted transfer leaves a resumable copy there and completed transfers do not linger under it. `--inplace` bypasses the partial dir (rsync parity). Requires `--partial` |
@@ -135,16 +169,16 @@ Every one of those has an entry below with its remaining caveats.
| Flag | Rsync Description | FastSync Status | Notes |
|------|-------------------|-----------------|-------|
-| `--delete` | Delete extraneous files from dest | ⚠️ Caveat | `use_delete` config field. Deletion is always derived from the transmitted keep-set manifest of the paths the sender sent/keeps (never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. FastSync's default timing when no timing flag is given is **delete-after** (extras are removed only once the whole transfer succeeded) — intentionally NOT rsync's `--del`/delete-during default, to preserve FastSync's commit-style safety. By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). Deletion is scoped to the **synchronized directories** sent in the manifest (protocol 2.23.0), so a `--files-from` subset no longer deletes untransmitted paths outside the listed directory subtrees. The walk is bounded: a client `--max-delete=NUM` (or the 100000-entry server bound) makes it **partial** — entries up to the bound are removed, the rest are skipped, and the client exits **25** (`RERR_PARTIAL`), matching rsync, rather than failing the transfer. Extraneous destination symlinks are unlinked by name (never followed); a directory still holding a kept/protected entry is left behind rather than failing |
-| `--delete-before` | Delete before transfer | ⚠️ Caveat | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. Divergence: the keep-set is the pre-scan snapshot, so a file that appears on the source between the pre-scan and the data pass is still transferred but was not protected from deletion |
-| `--del`, `--delete-during` | Delete during transfer | ⚠️ Caveat | Both spellings accepted; imply `--delete`. **Protocol 2.24.0 implements per-directory delete plans:** as the sender finishes each source directory it streams a `STATUS_DELETE_PLAN` for that directory and the receiver removes that directory's extras before applying the next directory's data, so a mid-transfer failure has already removed the extras of the directories reached (verified with a byte-slicing proxy). **Remaining divergence:** the exact abort boundary and the progressive ordering of removals versus rsync's generator can differ, and `-d`/`--dirs` (no descent) falls back to the end-of-transfer commit. `-R` plans are scoped to the transferred prefix subtree |
-| `--delete-delay` | Find deletions during, delete after | ⚠️ Caveat | Implies `--delete`. **Protocol 2.24.0 implements rsync's delete-delay timing:** the sender records each directory's delete plan while scanning and the receiver commits those removals only after the whole transfer succeeds (per plan), so an extra created in the destination after its directory's plan survives while `--delete-after` re-scans and removes it, and a failed transfer removes nothing. **Remaining divergence:** exact ordering/abort boundaries can differ from rsync's generator, and `-d` falls back to the end commit. **Also:** the reported "Number of deleted files" can be inflated because a directory snapshotted into the delete plan that later fails to delete (ENOTEMPTY) is still counted (`delete_plan.c:583-593`, `866`, `891`) |
-| `--delete-after` | Delete after transfer | ✅ Parity | Implies `--delete`. The delete-after timing is also what plain `--delete` does: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing |
-| `--delete-excluded` | Also delete excluded files | ⚠️ Caveat | `delete_excluded` config field. Under `--delete` FastSync protects (rsync's default) the destination mirror of paths the sender's source scan pruned by the user-selection rules — the `--filter`/`-F`/`-C` layer and the legacy `--exclude`/`--include` layer. The sender transmits those concrete pruned paths as **protected prefixes** in the delete-manifest frame (see the Phase-3 notes below); the walker never descends into or removes them. `--delete-excluded` opts back in: the sender sends an empty protected list, so the excluded destination mirrors become ordinary extras and are removed. **`--max-size`/`--min-size` pruned mirrors are a separate, always-on protection** (protocol 2.23.0, rsync parity): size-pruned source mirrors survive `--delete` even with `--delete-excluded`. Divergences (documented): protection is derived only from what the source scan actually pruned — a stray destination-only file that happens to match an exclude rule is not protected (FastSync never re-applies rules to the destination, keeping deletion sender-derived) |
+| `--delete` | Delete extraneous files from dest | ⚠️ Caveat | `use_delete` config field. Deletion is always derived from the keep-set the sender actually transmitted (the per-directory `STATUS_DELETE_PLAN` set by default, or the whole-tree manifest for the late timings — never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. **Lockstep track 6 (protocol 2.28.0): plain `--delete` with no explicit timing flag now defaults to `--delete-during`**, exactly like rsync's `--del` (the client normalizes it to the existing `delete_during` wire bool; no new wire field). This frees destination space progressively during the transfer and avoids the whole-old+new-tree peak that could `ENOSPC` a tight destination. The old late whole-tree commit is opt-in via `--delete-after` or the FastSync-only long spelling `--delete-commit`. **Caveat (shared with `--delete-during`):** the exact mid-transfer abort boundary can differ from rsync's generator (rsync removes all extras ahead of its throttled sender; FastSync removes only the directories it has reached), `-d/--dirs` falls back to the end-of-transfer commit, and the surviving-set ordering under a partial `--max-delete` can differ — on a completed run the trees agree. By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). Deletion is scoped to the **synchronized directories** sent on the wire (protocol 2.23.0), so a `--files-from` subset no longer deletes untransmitted paths outside the listed directory subtrees. The walk is bounded: a client `--max-delete=NUM` (or the 100000-entry server bound) makes it **partial** — entries up to the bound are removed, the rest are skipped, and the client exits **25** (`RERR_PARTIAL`), matching rsync, rather than failing the transfer. Extraneous destination symlinks are unlinked by name (never followed); a directory still holding a kept/protected entry is left behind rather than failing |
+| `--delete-before` | Delete before transfer | ⚠️ Caveat | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. **Phase-0 divergence (sharpened):** rsync builds the full file list first, so a source file created after that scan is NOT transferred and its destination extra is deleted; FastSync's single-threaded data pass re-scans the source, so the late file IS transferred (a safe superset), while FastSync `--threads` pipelines the scan and matches rsync |
+| `--del`, `--delete-during` | Delete during transfer | ⚠️ Caveat | Both spellings accepted; imply `--delete`, and since lockstep track 6 this is also the default timing of a plain `--delete`. **Protocol 2.24.0 implements per-directory delete plans:** as the sender reaches each source directory it streams a `STATUS_DELETE_PLAN` for that directory and the receiver removes that directory's extras (verified with a byte-slicing proxy). The one-shot per-run config block (protected prefixes, size-pruned mirrors, `--delete-missing-args` exact paths) rides a dedicated config-only carrier frame with an `apply=false` flag, so it reaches the receiver even when the scope allows no directory plan at all (a `--files-from` list of bare files synchronizes no directory). **Phase-0 divergence (sharpened):** on a mid-transfer abort rsync's generator (which runs ahead of its throttled sender) has already removed ALL the extras it planned, whereas FastSync has removed only the directories it actually reached; on a completed run both agree. Also `-d`/`--dirs` (no descent) falls back to the end-of-transfer whole-tree commit, and the surviving-set ordering under a partial `--max-delete` can differ. `-R` plans are scoped to the transferred prefix subtree |
+| `--delete-delay` | Find deletions during, delete after | ⚠️ Caveat | Implies `--delete`. **Protocol 2.24.0 implements rsync's delete-delay timing:** the sender records each directory's delete plan while scanning and the receiver commits those removals only after the whole transfer succeeds (per plan), so an extra created in the destination after its directory's plan survives while `--delete-after` re-scans and removes it, and a failed transfer removes nothing. The **reported** deleted count advances only on an actual removal. **Fixed (no-wire):** the `--max-delete` budget is now charged on ACTUAL removals (an unlink/rmdir that succeeded), not at plan/snapshot time, and a queued directory is re-scanned at commit and removed recursively (content created after the plan included), matching rsync: a snapshotted entry that fails or is skipped consumes no budget, so a later extra rsync would delete is still deleted. The deferred snapshot list keeps an independent hard cap (`DELETE_PLAN_SERVER_LIMIT`) so it cannot grow without bound now that the budget is no longer charged while scanning. A `--max-delete=2` partial delete reports exactly 2 and exits 25 in both tools, and the refilled-directory differential (late content removed, directory removed, budget shared) now matches rsync 3.4.1 on both sides (`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`). Unit tests cover recursive removal, actual-removal charging, and the bounded deferred list. **Caveat:** the exact ordering of which extras are removed first under a partial `--max-delete` can still differ from rsync's generator (the survivor set is compared by count, not identity) |
+| `--delete-after` | Delete after transfer | ✅ Parity | Implies `--delete`. Selects the late whole-tree commit: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing. Since lockstep track 6 a plain `--delete` defaults to delete-during (rsync's `--del`); `--delete-after` — or the FastSync-only `--delete-commit` spelling, which selects the identical timing — is the explicit way to keep the old commit-style behavior |
+| `--delete-excluded` | Also delete excluded files | ✅ Parity | `delete_excluded` config field. Under `--delete` FastSync protects (rsync's default) the destination mirror of paths the sender's source scan pruned by the user-selection rules — the `--filter`/`-F`/`-C` layer and the legacy `--exclude`/`--include` layer. The sender transmits those concrete pruned paths as **protected prefixes** in the delete-manifest frame (see the Phase-3 notes below); the walker never descends into or removes them. `--delete-excluded` opts back in: the sender sends an empty protected list, so the excluded destination mirrors become ordinary extras and are removed. **`--max-size`/`--min-size` pruned mirrors are a separate, always-on protection** (protocol 2.23.0, rsync parity): size-pruned source mirrors survive `--delete` even with `--delete-excluded`. Track 4a (protocol 2.28.0) additionally re-applies the received `protect`/`risk` rules on the receiver, so a destination-only entry matching an exclude rule is protected (or left at risk) exactly like rsync; the remaining sender-derived `--delete-excluded` behavior (an unqualified rule becomes sender-only, so its source mirror and matching destination-only extras are deleted) is unchanged |
| `--max-delete=NUM` | Max files to delete | ✅ Parity | `max_delete` config field (default -1 = no client limit; 0 = delete nothing). **Protocol 2.23.0 matches rsync's partial semantics:** the receiver deletes up to NUM entries (regular files, symlinks and empty directories; each directory removal counts as one) and then **stops deleting, skips the rest, and reports the run as partial**. The client prints a "deletions stopped due to `--max-delete` limit" message and exits **25** (rsync's `RERR_PARTIAL`), not a hard failure — the transfer itself succeeded. NUM only applies together with `--delete` (it is inert otherwise, matching rsync). A client NUM below the server hard bound `MAX_SERVER_DELETE_COUNT` (100000) replaces it; a NUM above it never raises that cap. Deleting an entire destination with no limit is still bounded by the server's 100000-entry ceiling. `--delete-missing-args` exact-path deletions and the ordinary extras walk draw from the same budget, matching rsync |
-| `--ignore-errors` | Delete even with I/O errors | ⚠️ Caveat | Sender-side, client-only config field. rsync suppresses `--delete` when the transfer had I/O errors; FastSync's equivalent is a source-scan I/O error (an unreadable directory, e.g. EACCES). By default the scan aborts the run so no deletion happens. With `--ignore-errors` the scan continues past the unreadable directory, the readable tree is transferred, deletion still runs (the unreadable directory's mirror is treated as an extra), and the run exits 23 (`RERR_PARTIAL`), matching rsync. **Remaining divergence / uncertainty:** the EACCES differential is not exercised in CI because the runner is root (mode 000 is still readable), so this rests on source inspection plus the setpriv integration test |
+| `--ignore-errors` | Delete even with I/O errors | ✅ Parity | Sender-side, client-only config field. Matches rsync's semantics exactly: an unreadable source subdirectory is always skipped so the readable tree transfers (the transfer root itself stays fatal), and the run reports rsync's partial-transfer exit **23**. Deletion policy follows rsync: by default an I/O error suppresses deletion (`IO error encountered -- skipping file deletion`), while `--ignore-errors` lets the deletion commit. The decision applies to every timing (`--delete`, `--delete-before`, `--delete-during`, `--delete-delay`, `--delete-after`) in both the sequential and `--threads` send paths. Differential-tested against rsync 3.4.1 with both tools run as an unprivileged user (mode-000 source directory); the reference build's root-only gate still excludes the EACCES differential, but the setpriv differential test exercises it. The piece that stays FastSync-specific is documented under the recursive-empty-directory residual: FastSync never emits an unreadable (or empty) directory entry, so that mirror is an extra that a run with `--ignore-errors` removes, where rsync emits the directory and keeps its mirror |
| `--force` | Force deletion of non-empty dirs | ✅ Parity | `force_delete` receiver config field (crosses the wire). rsync's `--force` lets an incoming non-directory replace a destination directory; FastSync implements exactly that: when a regular file (or symlink) is written to a path that is currently a (possibly non-empty) destination directory, `--force` removes that directory tree first — confined to the receive root and symlink-safe (O_NOFOLLOW fd walk, symlinks removed by name, never followed) — so the install can place the file. **Protocol 2.23.0 honors `--force` on the `--delay-updates` publication path too**, not only the immediate-install path. Without `--force` such a write fails and the run aborts. Gated by the server `--allow-delete` policy (a client cannot use `--force` to remove a destination tree on a server that forbids deletion) |
-| `-m`, `--prune-empty-dirs` | Prune empty dir chains | ✅ Parity | `-m`/`--prune-empty-dirs` (Phase 7 Wave A freed the rsync short `-m`; FastSync multithreading is now `-j`/`--threads`). FastSync's recursive transfer records directory times but never CREATES an empty directory (a `STATUS_DIR_TIMES` entry is record-only, and `--dirs` empty entries are pruned by this flag), so empty directories are inherently never transferred (which is rsync's `-m` behavior) and truly-empty destination directory chains are removed by `--delete` regardless of this flag. The flag's additional real effect is on the `--dirs` explicit directory-entry generator: a plain `-d ` run omits the empty source directory's entry, so nothing is created at the destination (no `STATUS_MKDIR`, no `-i`/`--out-format` change line, and an existing empty mirror becomes an extra that `--delete` prunes). Explicitly `--files-from`-listed directories always pass through (documented `--files-from` behavior). A directory that still holds an excluded-but-protected file survives, matching the `--delete-excluded` default |
+| `-m`, `--prune-empty-dirs` | Prune empty dir chains | ✅ Parity | `-m`/`--prune-empty-dirs` (Phase 7 Wave A freed the rsync short `-m`; FastSync multithreading is now `-j`/`--threads`). A recursive transfer now recreates empty source directories by default (rsync parity); this flag suppresses that emission, so an empty directory (physically empty, or emptied by filtering) is not created and a true empty-directory chain is removed by `--delete`, matching rsync's `-m`. It also affects the `--dirs` explicit directory-entry generator: a plain `-d ` run omits the empty source directory's entry, so nothing is created at the destination (no `STATUS_MKDIR`, no `-i`/`--out-format` change line, and an existing empty mirror becomes an extra that `--delete` prunes). Explicitly `--files-from`-listed directories always pass through (documented `--files-from` behavior; `--files-from` runs never emit implicit empty directories). A directory that still holds an excluded-but-protected file survives, matching the `--delete-excluded` default |
**Deletion-timing implementation notes (Phase 3):** the delete flags above are
real. Two new config booleans (`delete_during`, `delete_delay`) join the already
@@ -154,8 +188,8 @@ The `STATUS_MANIFEST` frame is count-delimited and position-independent: the
receiver commits the deletion either when the manifest arrives (early modes:
`--delete-before`/`--delete-during`, which additionally acknowledge with
`STATUS_OK` before data flows) or after the terminal `STATUS_FINISHED` proves
-the whole transfer succeeded (commit modes: plain `--delete`/`--delete-after`/
-`--delete-delay`). Timing is chosen purely from the config, so server policy
+the whole transfer succeeded (commit modes: `--delete-after`/`--delete-delay`;
+plain `--delete` joined the per-directory plans in track 6). Timing is chosen purely from the config, so server policy
(`--allow-delete` off) still disables deletion without deadlocking the early
manifest ack. `--delete-delay` and `--delete-during` are each implemented as
the closest safe approximation their engine mode allows; the divergences are
@@ -237,7 +271,10 @@ Flag-conflict policy: unlike rsync's last-one-wins behaviour, every deletion
timing flag implies `--delete`, and combining a timing flag with `--no-delete`
(in either argument order) — or more than one timing flag — is rejected as a
configuration error rather than silently resolved. Note the check is
-order-independent because it runs over the fully parsed config. The deletion
+order-independent because it runs over the fully parsed config. The
+FastSync-only `--delete-commit` is a late-timing spelling (it maps onto
+`--delete-after`), so it conflicts with any different timing flag exactly like
+`--delete-after` does. The deletion
POLICY flags (`--delete-excluded`, `--max-delete`, `--ignore-errors`, `--force`)
do NOT imply `--delete`; without `--delete` they are inert (matching rsync).
@@ -288,7 +325,7 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`.
| `--write-devices` | Write to devices as files | ❌ Divergent | Writes only into an existing char/block node under the confined receive root (`O_NOFOLLOW` + `O_NONBLOCK`); a missing, symlinked, FIFO-with-no-reader, non-device, or otherwise unusable destination is skipped with a warning rather than allowed or aborted. Deliberate confinement divergence from rsync's more permissive behavior |
| `-U`, `--atimes` | Preserve access times | ✅ Parity | Captures the source access time (from the scanner's pre-read stat, so it is not clobbered by reading the file for transfer) and transmits it over the wire; the receiver restores it together with the mtime via `futimens`/`utimensat`. Implies metadata transmission (the times travel inside the shared metadata payload), but does not enable ownership application (that stays opt-in via the identity flags). Wire: `atime` fields on the metadata frame + a `preserve_atimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** |
| `-N`, `--crtimes` | Preserve create times | ❌ Divergent | Birth-times cannot be set by any portable filesystem call (`utimensat`/`futimens` only set atime/mtime), so this row is an explicit **Divergent** entry (Phase 7 Wave B). Capture + transmit stays: `statx(STATX_BTIME)` on Linux records the source birth time as a wire field; the receiver logs a debug note that it cannot be applied and continues — never failing the transfer and never pretending it worked. On platforms without `statx` it parses as a documented no-op (flag accepted; nothing is captured). Implies metadata transmission. Wire: new `crtime` fields + a `preserve_crtimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** (see the Phase-4 metadata-time notes) |
-| `-O`, `--omit-dir-times` | Omit dirs from --times | ✅ Parity | Real modifier now that FastSync preserves directory times. With metadata on, the scanner captures every traversed source directory's mtime (and atime under `-U`) and the sender transmits them in trailing `STATUS_DIR_TIMES` frame(s) **after all file data and the optional delete manifest** (chunked at the receiver's `MAX_MANIFEST_ENTRIES` per-frame cap); a dir-time entry only RECORDS metadata and never creates the directory, so empty source directories stay untransferred. The receiver defers applying them until its delete / `--delay-updates` publication phases have committed, so writing or removing a child never clobbers a parent directory's mtime (rsync applies directory times at the end for exactly this reason). When `-O` is set (the boolean crosses the wire) the receiver does not apply any of them; without `-O` an `-a`/`--preserve` transfer now restores directory times (reversing the old "never preserves dir times" divergence). Wire change: the terminal `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** |
+| `-O`, `--omit-dir-times` | Omit dirs from --times | ✅ Parity | Real modifier now that FastSync preserves directory times. With metadata on, the scanner captures every traversed source directory's mtime (and atime under `-U`) and the sender transmits them in trailing `STATUS_DIR_TIMES` frame(s) **after all file data and the optional delete manifest** (chunked at the receiver's `MAX_MANIFEST_ENTRIES` per-frame cap); a dir-time entry only RECORDS metadata and never creates the directory (an empty source directory is created by the separate `STATUS_MKDIR` entry the scanner now emits, and `-m/--prune-empty-dirs` suppresses that; the trailing dir-time simply re-applies the metadata). The receiver defers applying them until its delete / `--delay-updates` publication phases have committed, so writing or removing a child never clobbers a parent directory's mtime (rsync applies directory times at the end for exactly this reason). When `-O` is set (the boolean crosses the wire) the receiver does not apply any of them; without `-O` an `-a`/`--preserve` transfer now restores directory times (reversing the old "never preserves dir times" divergence). Wire change: the terminal `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** |
| `-J`, `--omit-link-times` | Omit symlinks from --times | ✅ Parity | Real modifier now that FastSync preserves symlink times. Symlink entries already carried their metadata on `STATUS_SYMLINK`; the receiver now applies it with **no-follow primitives only** (`utimensat(..., AT_SYMLINK_NOFOLLOW)`, plus best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)`), so the link itself is stamped without ever dereferencing it, confined fd-relative below the authorized receive root. A symlink has no children, so the times are applied immediately at creation. When `-J` is set (the boolean crosses the wire) the receiver skips the timestamps (mode/ownership are unaffected); without `-J` an `-a`/`-l` transfer restores symlink mtimes. Wire change alongside `-O`: the shared `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** |
| `--super` | Receiver attempts super-user activities | ❌ Divergent | Safe-subset privilege model. `--super` permits the receiver to attempt already-confined super-user activities (ownership application, char/block device-node creation, `--write-devices`); `--no-super` forbids them even for root; `auto` keeps the historical best-effort attempt. **FastSync never elevates** — no `setuid`/`seteuid`/`setgid` — and `--super` never bypasses the confinement floor, so it diverges from rsync's real elevation. A server `--no-super` veto forces it off for every connection; a privileged standalone listener defaults off without `--allow-super`; daemon modules opt in with `client owner = yes` |
| `--fake-super` | Store/recover privileged attrs via xattrs | ❌ Divergent | Records the resolved `uid:gid:mode:mtime_sec:mtime_nsec` in a reserved `user.fastsync.stat` xattr and immediately replays mode/times fd-relative, but **never performs a real `chown`** (the owner is recorded for a later privileged restore). The on-disk key and format are FastSync-native, not rsync's `user.rsync.%stat%`, so recordings are not interoperable with rsync — the same class as the native auth and batch formats. Implies metadata transmission; incompatible with `-s` |
@@ -629,19 +666,19 @@ targets verbatim, matching rsync.
| Flag | Rsync Description | FastSync Status | Notes |
|------|-------------------|-----------------|-------|
| `--checksum` | Skip based on checksum | ✅ Parity | `-c`/`--checksum` compares per-file whole-file content digests to skip unchanged files. **As of protocol 2.23.0 the short `-c` implies the checksum quick-check**, so a plain `-c` run verifies content rather than only affecting the `--incremental` handshake. The digest algorithm is `xxh128` by default (protocol 2.26.0's negotiated default) and is selectable via `--checksum-choice`/`--cc` (`xxh128`/`xxh3`/`xxh64`/`xxhash`/`md5`/`md4`/`sha1`/`none`/`auto`, plus rsync's two-name form) and `--checksum-seed=NUM` (see those rows) |
-| `--checksum-choice=STR`, `--cc=STR` | Choose checksum algorithm | ⚠️ Caveat | Real algorithm selection for the per-file whole-file digest used by the `--incremental`/`--checksum` handshake and basis-dir verification. **Protocol 2.26.0 accepts rsync 3.4.1's full set** — `xxh128` (the negotiated default), `xxh3`, `xxh64`, `xxhash`, `md5`, `md4`, `sha1`, `none`, `auto`, and the two-name `transfer,pre-transfer` form — with rsync's exit-4 rejection of an unknown name and of `none` on the transfer side when `--checksum` is on. `--cc=ALG` and space forms both parse. The algorithm id and seed cross the wire; the receiver hashes its old file with the same algorithm+seed and the per-file `STATUS_CHECK` handshake carries a bounded digest pinned to the negotiated length. **Remaining divergences:** rsync uses this choice for the transfer checksum on the wire as well, while FastSync selects only the whole-file comparison digest and keeps the delta BLOCK strong checksum at xxHash32; the `RSYNC_CHECKSUM_LIST` environment variable is not consulted; and `auto` always resolves deterministically to the first supported entry in rsync's preference order rather than probing the peer |
-| `--compare-dest=DIR` | Compare dest files relative to DIR | ⚠️ Caveat | DIR is a receiver-side basis; protocol 2.26.0 uses an absolute path verbatim (rsync semantics) and resolves a relative path below the destination root (`..` components are rejected, `//` collapsed and trailing `/` dropped). On the receiver's per-file check (implies `--incremental`) an exact match = same size + mtime (unless `--size-only`; `-I` disables matching) **and** equal xxHash64 of the sender's file; a match suppresses the data transfer. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Divergences: when the destination already holds a *different* version rsync deletes it but FastSync instead transfers the data (keeps the mirror complete; never deletes without `--delete`); attribute-only differences on a match are not re-applied (data is skipped so the sender never sends metadata); content is verified by xxHash64, stricter than rsync's default quick check. Sizing: FastSync's whole-file payload limit is 256 MiB on **every** transfer path (not basis-specific); rsync applies basis dirs to arbitrary sizes, so FastSync refuses a basis run whose source contains a larger file up front with a clear error before any transfer. Wire: a basis-count field is always present on the config frame (protocol 2.9.0, so clients and servers must both be 2.9.0) |
-| `--copy-dest=DIR` | Include copies of unchanged files | ⚠️ Caveat | Same basis rules as `--compare-dest`, but an exact match materializes a **local copy** of the DIR file into the destination (via the normal atomic temp+rename store path, so `--existing`/`--ignore-existing`/`--update`/`--backup`/`--delay-updates` all still apply) instead of transferring data. Repeatable; command-line order = priority. Content is xxHash64-verified before the copy. Divergences: a basis-hit destination keeps the basis file's own mode/uid/gid and mtime (the sender sends no metadata on a skip), so with `--size-only` its mtime can differ from the source and attribute-only differences are copied with the basis attributes rather than rsync's "copy + fix attributes". Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
-| `--link-dest=DIR` | Hardlink to files when unchanged | ⚠️ Caveat | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy, never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Content is xxHash64-verified before linking. Divergences and caveats: protocol 2.26.0 re-links an already up-to-date destination file to the basis (the relink path installs the hard link when the content matches); a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); with `--size-only` the linked mtime can differ from the source; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
-| `-y`, `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ⚠️ Caveat | `-y/--fuzzy` is a pure bandwidth optimization on the existing receiver-driven delta path: when a file must be transferred and the destination holds no usable content at the exact path (file absent, or the destination file is outside the delta engine's size bounds), the receiver searches the SAME destination directory for an existing regular file whose basename is similar to the incoming name and uses it as the delta basis, so the sender transmits only the differences instead of the whole file. The output is always byte-exact regardless of which (or whether any) basis is chosen. Decision location: the receiver performs the candidate search inside `receive_incremental_check` and sends the normal `STATUS_DELTA_SIGNATURE`; the sender never learns the basis was a different file, so no new frame type or sender logic was needed — only the config frame grew a `fuzzy` boolean, so `PROTOCOL_VERSION` was bumped **2.8.0 → 2.9.0** (peers must match). Similarity heuristic (deterministic, simpler than rsync's deliberately-fuzzy matching, and documented precisely): candidates are the target's sibling entries in its destination directory, opened `O_NOFOLLOW`/`AT_SYMLINK_NOFOLLOW` under the confined root (symlinks never followed; nothing outside the destination root is ever read or hashed); dotfiles, directories, the target's own name, and the `.fastsync-stage`/temp scratch names are excluded; like the ordinary delta path, the block signature the receiver transmits is derived from on-disk content it may not otherwise send, so a negotiated `--fuzzy` run exposes the destination's sibling files (at block granularity) to the sender as a known-plaintext oracle — the same information class as the normal delta handshake over the file being replaced; the size gate is the delta engine's own bounds (both files ≥ 16 KiB, ≤ `--delta-max`, ratio ≤ 10×) rather than rsync's ~1.5× size window; protocol 2.26.0 uses a name-distance/suffix heuristic modelled on rsync's plus an exact size+mtime pass, and reads a single best candidate; the exact tie-break order can still differ from rsync's; the directory scan is capped at 4096 entries so a pathological directory cannot stall a transfer. When fuzzy applies: only to files the receiver would otherwise send whole — the destination's own file is always preferred as the delta basis when it exists and fits the delta size bounds, so fuzzy does NOT replace an existing-but-different destination basis; FastSync's 10× delta size-ratio bound means an existing destination file that is too far away in size still lets the fuzzy search run. When no similar candidate exists the transfer falls back to the normal whole-file transfer. rsync-divergence note: rsync's own matching uses a fuzzy name/size rule set; FastSync implements the closest safe deterministic approximation above. Because FastSync's delta machinery is off by default (rsync's is on), `--fuzzy` implies `--incremental` + `--delta` (unless `--whole-file`/`-W` or an explicit `--no-delta` switched delta off, in which case fuzzy is inert — matching rsync where `--whole-file` makes fuzzy irrelevant). Unlike the basis-dir options, `--fuzzy` honors an explicit `--no-incremental` (it does not force the handshake back on); an explicit `--no-incremental` also suppresses the delta implication so no invalid `--delta requires --incremental` config results. `--no-fuzzy` negates it. All surrounding semantics are untouched: a fuzzy-reconstructed file is stored as a normal file, so `--remove-source-files`, itemize/`-i`, `--stats`, `--backup`, `--delay-updates`, `--existing`/`--ignore-existing`/`--update` behave exactly as for a whole-file transfer (the fuzzy delta does not skip the file) |
+| `--checksum-choice=STR`, `--cc=STR` | Choose checksum algorithm | ✅ Parity | Real algorithm selection for the per-file whole-file digest used by the `--incremental`/`--checksum` handshake and basis-dir verification. **Protocol 2.26.0 accepts rsync 3.4.1's full set** — `xxh128` (the negotiated default), `xxh3`, `xxh64`, `xxhash`, `md5`, `md4`, `sha1`, `none`, `auto`, and the two-name `transfer,pre-transfer` form — with rsync's exit-4 rejection of an unknown name and of `none` on the transfer side when `--checksum` is on. `--cc=ALG` and space forms both parse. The algorithm id and seed cross the wire; the receiver hashes its old file with the same algorithm+seed and the per-file `STATUS_CHECK` handshake carries a bounded digest pinned to the negotiated length. `checksum_digest_file` now streams **every** supported algorithm (md4 via the self-contained RFC 1320 code, sha1/md5 via EVP, none as an empty digest), so the streaming path matches its contract, and `--out-format %C` uses the selected **transfer** half of a two-name choice and renders each algorithm byte-for-byte like rsync (xxh128 high-then-low, xxh64/xxh3 big-endian, md5/md4/sha1 standard hex, none a blank 2-char column) — differential-tested across all algorithms. **Track-3b finding — the block-checksum residual is not observable.** rsync applies the choice to the block checksum on its wire too, while FastSync selects only the whole-file comparison digest and keeps the delta BLOCK strong checksum fixed at xxHash32 (`DeltaBlockSig`, `delta_signature_create_seeded`). Because FastSync does not interoperate with rsync on the wire, only the compared surface matters, and a pre-seeded delta differential against rsync 3.4.1 (`--no-whole-file -B8192 --stats --out-format=%c|%C %n` vs `--incremental --delta`) shows the choice does not move it: across `xxh64`, `xxh128`, `xxh3`, `md5`, `md4`, `sha1` and both two-name orders the destination tree is byte-identical, `Matched data`/`Literal data`/`Total transferred file size` are unchanged (and equal to rsync's with the block size pinned), the `%c` block-checksum token is invariant (rsync `16 + 6·ceil(size/block)`, already documented under `--out-format`; FastSync its own basis-read counter), and the exit code is 0. The negotiated algorithm is visible only in `%C`, which applies it to the whole-file transfer digest and matches rsync byte-for-byte. A false block match would require the 4-byte adler32 AND the 4-byte xxHash32 to collide; at the 256 MiB maximum with the 1 KiB minimum block size the expected false matches are ≤2⁻¹⁸, and the choice cannot change this because FastSync's block strong sum is fixed. `auto` now consults `RSYNC_CHECKSUM_LIST` (rsync's whitespace-separated preference list; unknown names skipped, first supported wins, all-unknown is exit 4) before the compiled-in order; because both peers run the identical build this deterministic resolution needs no rsync peer probe, and an explicit `--cc` still wins. The list is differential-tested through `--out-format %C` (byte-identical digests to rsync for md5/sha1/xxh3) |
+| `--compare-dest=DIR` | Compare dest files relative to DIR | ⚠️ Caveat | DIR is a receiver-side basis; protocol 2.26.0 uses an absolute path verbatim (rsync semantics) and resolves a relative path below the destination root (`..` components are rejected, `//` collapsed and trailing `/` dropped) — note rsync resolves a relative DIR against the destination directory while FastSync resolves it below the receive root and appends the mirrored source path, so the same relative spelling addresses a different tree (use an absolute DIR for exact parity). On the receiver's per-file check (implies `--incremental`) an exact match is rsync's metadata quick-check: same size and mtime (unless `--size-only`; `-I` disables matching), with NO content digest required by default (track 5a). A match suppresses the data transfer. The FastSync-only `--verify-basis` restores the stricter whole-file content equality. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Differential-tested against rsync 3.4.1 (`compare_dest`, and `test_verify_basis_restores_strict_content`). Residual: a basis MISS above the 256 MiB whole-file payload bound is refused up front (FastSync's general whole-file limit, not basis-specific); rsync applies basis dirs to arbitrary sizes. Wire: a basis-count field plus the `verify_basis` bool are present on the config frame (protocol 2.9.0/2.28.0) |
+| `--copy-dest=DIR` | Include copies of unchanged files | ⚠️ Caveat | Same basis rules as `--compare-dest`, but an exact match materializes a **local copy** of the DIR file into the destination (via the atomic temp+rename store path, so `--existing`/`--ignore-existing`/`--update`/`--backup`/`--delay-updates` all still apply) instead of transferring data. Track 5a re-applies the SOURCE attributes on the copy (rsync's "copy then fix attributes"): the sender transmits the source metadata with the basis check frame, so the copy's mode/uid/gid/mtime match the source rather than the basis inode (differential `copy_dest` compares modes). The copy streams the basis file through a bounded buffer, so a basis larger than the whole-file payload bound still materializes. Repeatable; command-line order = priority. Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
+| `--link-dest=DIR` | Hardlink to files when unchanged | ⚠️ Caveat | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy (streamed from the basis, so an over-limit basis still works), never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Differential-tested against rsync 3.4.1 (`link_dest`). Inherent shared-inode semantics (identical to rsync): a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); protocol 2.26.0 re-links an already up-to-date destination file to the basis; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. Residual: a basis MISS above the 256 MiB whole-file payload bound is refused (FastSync's general whole-file limit). Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
+| `-y`, `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ⚠️ Caveat | `-y/--fuzzy` is a pure bandwidth optimization on the existing receiver-driven delta path: when a file must be transferred and the destination holds no usable content at the exact path (file absent, or the destination file is outside the delta engine's size bounds), the receiver searches the SAME destination directory for an existing regular file whose basename is similar to the incoming name and uses it as the delta basis, so the sender transmits only the differences instead of the whole file. The output is always byte-exact regardless of which (or whether any) basis is chosen. Decision location: the receiver performs the candidate search inside `receive_incremental_check` and sends the normal `STATUS_DELTA_SIGNATURE`; the sender never learns the basis was a different file, so no new frame type or sender logic was needed — only the config frame grew a `fuzzy` boolean, so `PROTOCOL_VERSION` was bumped **2.8.0 → 2.9.0** (peers must match). Similarity heuristic (a deterministic port of rsync 3.4.1's matcher — `util1.c` `fuzzy_distance`/`find_filename_suffix` plus `generator.c find_fuzzy`'s exact size+mtime pass — documented precisely): candidates are the target's sibling entries in its destination directory, opened `O_NOFOLLOW`/`AT_SYMLINK_NOFOLLOW` under the confined root (symlinks never followed; nothing outside the destination root is ever read or hashed); dotfiles, directories, the target's own name, and the `.fastsync-stage`/temp scratch names are excluded; like the ordinary delta path, the block signature the receiver transmits is derived from on-disk content it may not otherwise send, so a negotiated `--fuzzy` run exposes the destination's sibling files (at block granularity) to the sender as a known-plaintext oracle — the same information class as the normal delta handshake over the file being replaced; the size gate is the delta engine's own bounds (both files ≥ 16 KiB, ≤ `--delta-max`, ratio ≤ 10×); rsync's fuzzy matcher is not tied to a delta size bound and empirically reuses a basis well outside FastSync's window (a 64 KiB source against a repeated-content sibling from 0.25× to 10000×, and files as small as 300 B), so candidate ELIGIBILITY — and hence the chosen basis — can differ even though the name heuristic is the same; protocol 2.26.0 uses rsync's weighted-Levenshtein name/suffix distance plus an exact size+mtime pass and reads a single best candidate; the tie-break (smallest size gap, then lexical name) is deterministic where rsync leaves equal distances to its file-list order; the directory scan is capped at 4096 entries so a pathological directory cannot stall a transfer. When fuzzy applies: only to files the receiver would otherwise send whole — the destination's own file is always preferred as the delta basis when it exists and fits the delta size bounds, so fuzzy does NOT replace an existing-but-different destination basis; FastSync's 10× delta size-ratio bound means an existing destination file that is too far away in size still lets the fuzzy search run. When no similar candidate exists the transfer falls back to the normal whole-file transfer. rsync-divergence note: the name matching is rsync's own rule; the residual is eligibility bounded by FastSync's delta engine, so no name-matcher port can widen it. Because FastSync's delta machinery is off by default (rsync's is on), `--fuzzy` implies `--incremental` + `--delta` (unless `--whole-file`/`-W` or an explicit `--no-delta` switched delta off, in which case fuzzy is inert — matching rsync where `--whole-file` makes fuzzy irrelevant). Unlike the basis-dir options, `--fuzzy` honors an explicit `--no-incremental` (it does not force the handshake back on); an explicit `--no-incremental` also suppresses the delta implication so no invalid `--delta requires --incremental` config results. `--no-fuzzy` negates it. All surrounding semantics are untouched: a fuzzy-reconstructed file is stored as a normal file, so `--remove-source-files`, itemize/`-i`, `--stats`, `--backup`, `--delay-updates`, `--existing`/`--ignore-existing`/`--update` behave exactly as for a whole-file transfer (the fuzzy delta does not skip the file). **Reclassified Caveat (track 5b):** the output is always byte-exact regardless of the basis, and the name rule is rsync's, so the residual is candidate ELIGIBILITY: FastSync's 10× ratio / 16 KiB delta gates make its size window strictly narrower than rsync's, and when eligibility differs the chosen basis — and therefore the `--stats` `Matched data`/`Literal data`/`Total transferred file size` counters — can differ even though the tree cannot. Where the two tools' choices coincide and the block size is pinned, both the tree and the counters match rsync (differential `fuzzy_basis`); `TestFuzzy` pins the window boundary on both sides (a >10× sibling and a <16 KiB sibling are declined, with a byte-exact whole-file fallback). A wider window would require loosening the delta engine's bounds, not changing the name heuristic. |
## 12. Compression
| Flag | Rsync Description | FastSync Status | Notes |
|------|-------------------|-----------------|-------|
-| `-z`, `--compress` | Compress file data | ⚠️ Caveat | Streaming compression. **Protocol 2.26.0 implements rsync 3.4.1's codec set** (`zstd` default, `lz4`, `zlib`, `zlibx`, `none`), selectable via `--compress-choice`/`--zc` and negotiated with `auto`. `-z` is the compression short form; `-c` is rsync's `--checksum`. `--skip-compress` applies rsync 3.4.1's default suffix list when no list is given. **Remaining codec divergence:** `zlibx` is treated as `zlib`, per-codec level defaults are not mirrored, and `auto` does not probe the peer |
-| `--compress-choice=STR`, `--zc=STR` | Choose compression algorithm | ⚠️ Caveat | Protocol 2.26.0 accepts rsync 3.4.1's compiled-in choices — `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, `auto` — and rejects an unknown name with exit 4 like rsync. The negotiated codec id crosses the wire (`compression_algo`), so the receiver decodes with the sender's codec. `--zc` is the alias. **Remaining divergences:** `zlibx` behaves as `zlib` (there is no separate zlibx path), the per-codec compression-level defaults differ from rsync's, and `auto` always resolves to the first supported entry in rsync's preference order rather than probing the peer (no `RSYNC_COMPRESS_LIST` handling) |
-| `--compress-level=NUM`, `--zl=NUM` | Set compression level | ✅ Parity | 1-22, default 5 |
+| `-z`, `--compress` | Compress file data | ✅ Parity | Streaming compression. **Protocol 2.26.0 implements rsync 3.4.1's codec set** (`zstd` default, `lz4`, `zlib`, `zlibx`, `none`), selectable via `--compress-choice`/`--zc` and negotiated with `auto`. `-z` is the compression short form; `-c` is rsync's `--checksum`. `--skip-compress` applies rsync 3.4.1's default suffix list when no list is given. **Track 3a closes the codec caveats:** `zlibx` is no longer a divergence — FastSync's zlib stream already carries only the delta/token (literal) bytes, which is exactly rsync's zlibx semantics, so `--zc=zlib` and `--zc=zlibx` land the same tree/stdout/exit (differential `test_compress_codec_matches_rsync_bytes`) and the zlib/zlibx aliasing is only an implementation detail. Each codec now uses rsync's own default `--compress-level` (zstd 3, zlib/zlibx 6, lz4 ignored) and `auto` consults `RSYNC_COMPRESS_LIST` before the compiled-in order; the deterministic same-build resolution needs no peer probe |
+| `--compress-choice=STR`, `--zc=STR` | Choose compression algorithm | ✅ Parity | Protocol 2.26.0 accepts rsync 3.4.1's compiled-in choices — `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, `auto` — and rejects an unknown name with exit 4 like rsync. The negotiated codec id crosses the wire (`compression_algo`), so the receiver decodes with the sender's codec. `--zc` is the alias. `auto` now resolves through `RSYNC_COMPRESS_LIST` (whitespace-separated; unknown names skipped, first supported wins, all-unknown is exit 4) and then the compiled-in order, and an explicit `--zc` wins; the deterministic same-build resolution needs no peer probe. `zlib`/`zlibx` share FastSync's literal-only zlib path, which is rsync's zlibx behavior and is observably identical for both, so `zlibx` is not a divergence (the aliasing is an implementation detail) |
+| `--compress-level=NUM`, `--zl=NUM` | Set compression level | ✅ Parity | Accepted range 1-22. When omitted, rsync 3.4.1's **per-codec default** applies: zstd 3 (`ZSTD_CLEVEL_DEFAULT`), zlib/zlibx 6 (`Z_DEFAULT_COMPRESSION` resolved), lz4 ignored (no tunable level; FastSync keeps a positive gate value and `lz4_compress` ignores it, so the bytes match rsync). An explicit level is clamped per codec like rsync's `init_compression_level()`: zstd 1-22, zlib/zlibx 1-9, lz4 ignored. Verified against `rsync --debug=NSTR1`, which reports the same effective level per codec |
| `--compress-threads=NUM` | Set compression threads | ✅ Parity | `compression_threads` config field (client-only; does not cross the wire). Sets the number of worker threads used by the zstd compression pool to NUM (1..64; 0/garbage/oversized rejected up front). Accepted in both `--compress-threads=NUM` and two-argument `--compress-threads NUM` forms. Composes with `-z`/compression; under the `-j`/`--threads` multithreaded pipeline it parallelizes compressed chunk encoding. See test_tcp.py `-z --compress-threads=2` and test_client_cli.c |
| `--skip-compress=LIST` | Skip compress for suffixes | ✅ Parity | Comma-separated (or `/`-separated, as in rsync) case-insensitive suffix list; a leading dot is optional; an empty list skips none. **When the option is omitted, rsync 3.4.1's built-in default suffix list applies** (`3g2 3gp 7z aac … zip zst`); an explicit list replaces that default entirely, matching rsync. A user-supplied list is a client-side compression choice; incompatible with FastSync chunk serialization (`-s`) |
@@ -659,8 +696,8 @@ targets verbatim, matching rsync.
| `--address=ADDRESS` | Bind address for outgoing socket | ✅ Parity | Binds the outgoing client socket to a local source address before `connect()` (resolved with the same `-4`/`-6` family hints as the destination). Local socket concern: never crosses the wire |
| `-4`, `--ipv4` | Prefer IPv4 | ✅ Parity | Forces `AF_INET` in the `getaddrinfo` hints for client destination/source resolution and the server bind (see the Phase 5, Wave B note). Mutually exclusive with `-6` |
| `-6`, `--ipv6` | Prefer IPv6 | ✅ Parity | Forces `AF_INET6` in the `getaddrinfo` hints for client destination/source resolution and the server bind. Mutually exclusive with `-4` |
-| `--remote-option=OPT`, `-M` | Send an option only to the remote side | ⚠️ Caveat | Each value is appended to the remote server invocation over SSH as an individually single-quote-escaped shell word in `ssh_build_remote_command()`. Values are validated (non-empty, no control characters) and shell metacharacters cannot break out of the quoting (`;`, `&`, `\|`, `, `$`, `(`, `)`, quotes are neutralized), so a value cannot inject an arbitrary remote command and a subsequent `--` on the client line cannot be turned into one. The short `-M` form (`-M OPT`, `-M=OPT`, and rsync-style attached `-MOPT`) is available, matching rsync; metadata mode moved to long-only `--preserve`. **Divergence:** `-M` is only meaningful for the SSH transport (`user@host:path`); a daemon (`host::module/path`) or local TCP destination **rejects** it (there is no remote command line to append to), whereas rsync applies it to its own remote process on every transport. The options never cross the binary config frame |
-| `--bwlimit=RATE` | Limit I/O bandwidth | ⚠️ Caveat | Token-bucket throttling of the transfer I/O (Kibibytes/second). **Divergence:** FastSync accepts only a positive integer; rsync additionally accepts `0` (no limit) and decimal/suffixed rates (`1.5`, `1.5m`, `100K`), so those rsync spellings are rejected. The limit is a local I/O concern and is not negotiated on the wire |
+| `--remote-option=OPT`, `-M` | Send an option only to the remote side | ❌ Divergent | Each value is appended to the remote server invocation over SSH as an individually single-quote-escaped shell word in `ssh_build_remote_command()`. Values are validated (non-empty, no control characters) and shell metacharacters cannot break out of the quoting (`;`, `&`, `\|`, `, `$`, `(`, `)`, quotes are neutralized), so a value cannot inject an arbitrary remote command and a subsequent `--` on the client line cannot be turned into one. The short `-M` form (`-M OPT`, `-M=OPT`, and rsync-style attached `-MOPT`) is available, matching rsync; metadata mode moved to long-only `--preserve`. **Reclassified because the daemon/TCP case cannot be reproduced:** `-M` is only meaningful for the SSH transport (`user@host:path`); a daemon (`host::module/path`) or local TCP destination **rejects** it, whereas rsync forwards it to its own remote process on every transport. A differential test starts a real rsync daemon and shows `-M--totally-bogus` reaching the remote parser (`unknown option`) while a valid `-M--safe-links` is accepted. FastSync's daemon handshake is a fixed binary config frame with no per-connection argv channel; adding one would let a client set arbitrary server-side options (the same class of divergence as the native daemon config/auth), so the safe subset stays SSH-only |
+| `--bwlimit=RATE` | Limit I/O bandwidth | ✅ Parity | A faithful port of rsync 3.4.1's `parse_size_arg(bwlimit_arg, 'K', "bwlimit", 512, -1, True)`: a bare value is KiB/s, `K`/`M`/`G`/`T`/`P` are binary suffixes, `KB`/`MB` are decimal, `KiB`/`MiB` are binary, decimals are accepted and quantized to whole KiB exactly like rsync's `(size + 512) / 1024`, `0` (or an empty value) means "no limit", and any other value below the 512-byte floor is rejected. The token bucket's burst capacity is ~100 ms of bandwidth, matching the point at which rsync's leaky bucket starts sleeping, so a throttled transfer paces like rsync (4 MiB at `--bwlimit=1024`/`2048` matches rsync within ~4%). Differential-tested: the accept/reject matrix and the wall-clock rate both match rsync 3.4.1. The limit is a local I/O concern and is not negotiated on the wire |
## 14. Daemon Mode
@@ -733,8 +770,8 @@ modes or links.
| `--stop-after=MINS` | Stop after N minutes | ✅ Parity | Client-only sender stop deadline (Phase 6): computing `--stop-after=MINS` (a positive minute count; 0/negative/garbage rejected) and `--stop-at=TIME` (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`; a past time stops immediately). The transfer stops ELEGANTLY at the next chunk boundary: everything already fully sent is kept and applied, the run returns 0, and --delete (late/delete-after timing) does NOT wipe the destination — when the scan is cut short the partial keep-set manifest is suppressed with a warning (the delete walk is skipped rather than acting on an incomplete keep-set, so unscanned source mirrors survive). `--delete-before`/`--delete-during` still run their complete pre-scan (which ignores the deadline). Local client-only fields: never serialized into the wire config frame, so no PROTOCOL_VERSION bump. `--stop-after` uses CLOCK_MONOTONIC; `--stop-at` uses the wall clock. Works single-threaded and under `-j`/`--threads` (multithreaded). Divergence: rsync computes `--stop-after` from the run start; FastSync likewise. When both are given, the earlier of the two deadlines wins (checked per iteration). See the Phase-6 stop notes below |
| `--stop-at=TIME` | Stop at specified time | ✅ Parity | Deadline transfer stop (client-only, never serialized). Protocol 2.26.0 accepts rsync's full date/time grammar (`2030-12-31T23:59`, `2030/12/31T23:59`, `2030-12-31`, `12-31`, `14:00`, `:59`, `1`) in addition to FastSync's `HH:MM[:SS]` and `now+N[smhd]`; a past time stops immediately. Everything already transferred is kept and an early stop suppresses the late `--delete` keep-set so unscanned source mirrors survive. Works single-threaded and under `-j`/`--threads` |
| `--fsync` | Fsync every written file before publication | ✅ Parity | |
-| `--protocol=NUM` | Force older protocol version | ❌ Divergent | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.26.0) with no downgrade/backward-compat code paths, so `--protocol=2.26.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.25.0`/`2.24.0`/`2.23.0`/`2.22.0`/`2.21.0`/`2.20.0`/`2.19.0`/`2.18.0`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below |
-| `--iconv=CONVERT_SPEC` | Charset conversion | ⚠️ Caveat | Charset conversion of FILE NAMES (not content) at the protocol boundary via iconv(3): `--iconv=LOCAL[,REMOTE]` — the sender converts each local filename LOCAL→REMOTE before transmitting, and the receiver converts each wire filename REMOTE→LOCAL before creating/writing. The full CONVERT_SPEC is serialized into the config frame as a new trailing string field so the peer knows the wire charset; **PROTOCOL_VERSION bumped 2.15.0 → 2.16.0**. `LOCAL[,REMOTE]` parse: single charset ⇒ LOCAL==REMOTE (identity both ways); garbage rejected up front; protocol 2.26.0 additionally accepts `--iconv=.` (the locale's default charset for both directions), `--iconv=-` and `--no-iconv` (disable conversion). Validation probes BOTH directions (a spec that only opens one way is refused, as is a NUL-emitting target charset like utf-16/utf-32/ucs-2, since filenames cannot contain NUL). An unrepresentable name (EILSEQ/EINVAL) fails that path cleanly with a logged `--iconv: cannot convert file name ...` and is never written mangled/truncated. Conversion is applied at EVERY wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest, the incremental-check path, and the `-s`/`chunk_serialize` embedded blob path), on both client and server (`--iconv` is also a server/daemon option). Zero overhead when unset. See the Phase-6 iconv notes below |
+| `--protocol=NUM` | Force older protocol version | ❌ Divergent | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.28.0) with no downgrade/backward-compat code paths, so `--protocol=2.28.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.27.0`/`2.26.0`/`2.25.0`/`2.24.0`/`2.23.0`/`2.22.0`/`2.21.0`/`2.20.0`/`2.19.0`/`2.18.0`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below |
+| `--iconv=CONVERT_SPEC` | Charset conversion | ✅ Parity | Charset conversion of FILE NAMES (not content) at the protocol boundary via iconv(3): `--iconv=LOCAL[,REMOTE]` — the sender converts each local filename LOCAL→REMOTE before transmitting, matching rsync's rule that the spec "stays the same whether you're pushing or pulling": on a PUSH the destination end's charset is the spec's REMOTE half, so the default receiver writes the wire bytes verbatim, and only a server started with its own `--iconv` (the daemon `charset` analog) declares a different destination charset and converts REMOTE→that LOCAL (rsync push parity, differential-tested with and without a server `--iconv`). The full CONVERT_SPEC is serialized into the config frame as a new trailing string field so the peer knows the wire charset; **PROTOCOL_VERSION bumped 2.15.0 → 2.16.0**. `LOCAL[,REMOTE]` parse: single charset ⇒ LOCAL==REMOTE (identity both ways); garbage rejected up front; protocol 2.26.0 additionally accepts `--iconv=.` (the locale's default charset for both directions), `--iconv=-` and `--no-iconv` (disable conversion). Validation probes BOTH directions (a spec that only opens one way is refused, as is a NUL-emitting target charset like utf-16/utf-32/ucs-2, since filenames cannot contain NUL). An unrepresentable name (EILSEQ/EINVAL) fails that path cleanly with a logged `--iconv: cannot convert file name ...` and is never written mangled/truncated. Conversion is applied at EVERY wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest, the incremental-check path, and the `-s`/`chunk_serialize` embedded blob path), on both client and server (`--iconv` is also a server/daemon option). Zero overhead when unset. See the Phase-6 iconv notes below |
| `--checksum-seed=NUM` | Set checksum seed | ✅ Parity | Sets the seed for FastSync's whole-file xxHash digest (full 64-bit seed) and for the delta path's per-block xxHash32 strong checksum (low 32 bits of the seed). **As of protocol 2.23.0 a seed of `0` — the default when the flag is unset — is randomized per transfer and the chosen seed is sent to the receiver**, exactly like rsync, so two runs against different content do not share a predictable seed; an explicit non-zero seed is used verbatim, so an explicit seed deterministically reproduces every computed digest on BOTH endpoints (the seed crosses in the config frame). `--checksum-choice=md5` has no seed and ignores it (documented). The value is a strict decimal 0..2⁶⁴-1 (blank, signed, or non-numeric values are rejected). Like rsync, a seed only matters where a digest is actually computed (`--checksum` or a basis-dir run, or a delta transfer); it does not by itself enable `--checksum`/`--delta` |
| `--secluded-args`, `-s` | Use protocol to send args | ❌ Divergent | Accepted for CLI compatibility (including the rsync short `-s`, Phase 7 Wave A) but a documented **no-op / divergence**. rsync's `-s` protects arguments from shell expansion by shipping them over the protocol; FastSync never passes remote arguments through a shell expansion boundary in the first place — its SSH transport builds the remote argv as **single-quote-escaped shell words** (`ssh_build_remote_command`), so the injection/leak that `-s` guards against does not exist and there is nothing to "seclude". Implementing a true arg-send protocol would mean replacing the argv-based SSH launch with an in-band argument channel, a large redesign of the transport that buys no security here. Chunk serialization remains the long-only `--chunk-serialization`. |
| `--protect-args` | Old name of --secluded-args | ❌ Divergent | Accepted for CLI compatibility as a documented no-op; the same rationale as `--secluded-args`/`-s` (FastSync's remote SSH argv is already built injection-safe, so there is no argument-leak to close) |
@@ -848,9 +885,9 @@ These are the hardest compatibility items because they require durable formats o
**Phase 6, Wave A (stop deadline) shipping note:** `--stop-after=MINS` and `--stop-at=TIME` are client-only sender stop deadlines. `--stop-after` takes a positive minute count (0/negative/garbage rejected); `--stop-at` takes `HH:MM`, `HH:MM:SS`, or `now+N[smhd]` (a past time stops immediately, a garbage spec is rejected at parse time). The deadline is computed once at the start of the transfer (CLOCK_MONOTONIC for `--stop-after`, wall clock via `time()` for `--stop-at`) and checked at every chunk boundary in both the single-threaded `send_files` loop and the multithreaded `send_chunks_multithreaded` path, and inside the scanner loops so a busy scan itself stops. When it fires, the transfer stops ELEGANTLY: the in-flight chunk completes, the existing completion tail runs (summary, `disconnect`), and the run returns 0 — exactly like rsync's clean early stop. Because the deadline is client-only and never crosses the wire config frame, no PROTOCOL_VERSION bump is required. The safety-critical interaction is with `--delete`: FastSync streams while scanning, so a deadline can cut the source scan short and yield a PARTIAL keep-set manifest; committing that would make the receiver delete destination mirrors of source files not yet scanned. So the sender tracks `scan_stopped_early` and, when it is true on the late/delete-after (`--delete`/`--delete-after`/`--delete-delay`) path, SUPPRESSES the keep-set manifest (logs a warning) so no deletion happens from an incomplete set — this is the safe direction (preserves data; the delete simply does not run). `--delete-before`/`--delete-during` are unaffected: their complete pre-scan runs before any data and ignores the deadline (a stop can be exceeded by that pre-scan). Under `-j`/`--threads` the stop is symmetric and the scanner thread's still-in-progress manifest appends can never race the tail because the tail does not read the manifest on the early-stop path.
-**Phase 6, Wave B (iconv) shipping note (PROTOCOL 2.15.0 → 2.16.0):** `--iconv=LOCAL[,REMOTE]` converts file NAMES at the wire boundary (never content). The full CONVERT_SPEC is serialized into the config frame as a new trailing string field (empty→NULL canonicalized), so both ends share the same wire charset interpretation; this required the PROTOCOL bump because the frame is a strict ordered sequence and a peer that does not parse the new trailing field would desynchronize. Each end derives LOCAL (its own charset) and REMOTE (the wire charset): the sender opens LOCAL→REMOTE and converts every transmitted filename; the receiver opens REMOTE→LOCAL and converts every received filename before creating/writing. Conversion is applied at every wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest keep/protected/missing entries, the incremental-check path, and the embedded `-s`/chunk-blob path). A name it cannot convert (EILSEQ/EINVAL) is failed cleanly with a logged `--iconv: cannot convert file name ...` and is never written truncated/mangled. Validation probes both directions up front (both the sender local→remote and the receiver remote→local, and, for a server/daemon with its own `--iconv`, the client-REMOTE→server-LOCAL pair) so an unusable spec is rejected before the connection rather than mid-transfer, and NUL-emitting target charsets (utf-16/utf-32/ucs-2) are refused because filenames cannot contain NUL. Divergence documented upstream: the receiver does NOT half-swap; the wire charset always comes from the sender's REMOTE half, so a server whose local charset differs from the client's LOCAL must declare it with its own `--iconv`. Conversion is process-global and runs on a single thread per process (sender thread / receiver-loop thread), initialized before worker threads start and freed after they join.
+**Phase 6, Wave B (iconv) shipping note (PROTOCOL 2.15.0 → 2.16.0):** `--iconv=LOCAL[,REMOTE]` converts file NAMES at the wire boundary (never content). The full CONVERT_SPEC is serialized into the config frame as a new trailing string field (empty→NULL canonicalized), so both ends share the same wire charset interpretation; this required the PROTOCOL bump because the frame is a strict ordered sequence and a peer that does not parse the new trailing field would desynchronize. Each end derives its charset and the wire charset: the sender opens LOCAL→REMOTE and converts every transmitted filename; on a push the receiver's destination charset is the spec's REMOTE half, so it writes the wire bytes verbatim, unless the server was started with its own `--iconv` naming a different LOCAL charset (then it opens REMOTE→that LOCAL). Conversion is applied at every wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest keep/protected/missing entries, the incremental-check path, and the embedded `-s`/chunk-blob path). A name it cannot convert (EILSEQ/EINVAL) is failed cleanly with a logged `--iconv: cannot convert file name ...` and is never written truncated/mangled. Validation probes both directions up front (both the sender local→remote and the receiver remote→destination, and, for a server/daemon with its own `--iconv`, the client-REMOTE→server-LOCAL pair) so an unusable spec is rejected before the connection rather than mid-transfer, and NUL-emitting target charsets (utf-16/utf-32/ucs-2) are refused because filenames cannot contain NUL. The wire charset always comes from the sender's REMOTE half; a server whose local charset differs from the client's REMOTE must declare it with its own `--iconv` (the daemon `charset` analog). Conversion is process-global and runs on a single thread per process (sender thread / receiver-loop thread), initialized before worker threads start and freed after they join.
-**Phase 6, Wave C (protocol-version) shipping note (no PROTOCOL_VERSION change):** `--protocol=NUM` lets the client force the wire protocol version for a transfer. FastSync's protocol is a single lockstep format: the config frame is a strict ordered sequence and the server requires the client's version string to equal `PROTOCOL_VERSION` exactly (`config_receive_with_validate`, src/shared/config.c) — there are no older-format code paths and no downgrade/negotiation machinery, so a lower/higher/virtual version can never be spoken. The honest contract is therefore: the current `PROTOCOL_VERSION` (2.26.0 as of the parity-completion wave) is accepted and stored into the client's `version` claim (which `config_send` already transmits), and every other value — `2.25.0`, `2.24.0`, `2.23.0`, `2.22.0`, `2.21.0`, `2.20.0`, `2.19.0`, `2.18.0`, `2.18`, `2.17.0`, `2.16.0`, `2.15.0`, `3.0.0`, rsync-integer spellings like `216`/`31`, garbage, empty — is rejected up front in `validate_config()` before any connection, with a clear error that FastSync supports only its current wire protocol and cannot speak an older or virtual one. Implementation is client-only: a server-side `--protocol` is intentionally not added because the server has no negotiation (it only enforces exact match), and it could only ever be the current version. This preserves (and slightly tightens) existing validation: the client now also refuses to launch with a version it cannot actually speak, rather than only the server rejecting it later. A genuine downgrade would require a per-version compatibility layer for every frame/feature added since (append 2.10, preallocate 2.11, hardlinks 2.12, devices/specials/symlink-trust/xattr 2.13, remote-option 2.14, daemon module/auth 2.15, iconv 2.16, dir/symlink times 2.17, privilege flags --super/--copy-as 2.18, SCRAM daemon auth 2.19, packed metadata 2.20, error-detail/dry-run 2.21, preserve-attribute split 2.22, rsync-parity wave 2.23) and is intentionally out of scope — documented divergences from rsync's integer-negotiated downgrade remain.
+**Phase 6, Wave C (protocol-version) shipping note (no PROTOCOL_VERSION change):** `--protocol=NUM` lets the client force the wire protocol version for a transfer. FastSync's protocol is a single lockstep format: the config frame is a strict ordered sequence and the server requires the client's version string to equal `PROTOCOL_VERSION` exactly (`config_receive_with_validate`, src/shared/config.c) — there are no older-format code paths and no downgrade/negotiation machinery, so a lower/higher/virtual version can never be spoken. The honest contract is therefore: the current `PROTOCOL_VERSION` (2.28.0 as of the parity 2.28.0 cycle) is accepted and stored into the client's `version` claim (which `config_send` already transmits), and every other value — `2.25.0`, `2.24.0`, `2.23.0`, `2.22.0`, `2.21.0`, `2.20.0`, `2.19.0`, `2.18.0`, `2.18`, `2.17.0`, `2.16.0`, `2.15.0`, `3.0.0`, rsync-integer spellings like `216`/`31`, garbage, empty — is rejected up front in `validate_config()` before any connection, with a clear error that FastSync supports only its current wire protocol and cannot speak an older or virtual one. Implementation is client-only: a server-side `--protocol` is intentionally not added because the server has no negotiation (it only enforces exact match), and it could only ever be the current version. This preserves (and slightly tightens) existing validation: the client now also refuses to launch with a version it cannot actually speak, rather than only the server rejecting it later. A genuine downgrade would require a per-version compatibility layer for every frame/feature added since (append 2.10, preallocate 2.11, hardlinks 2.12, devices/specials/symlink-trust/xattr 2.13, remote-option 2.14, daemon module/auth 2.15, iconv 2.16, dir/symlink times 2.17, privilege flags --super/--copy-as 2.18, SCRAM daemon auth 2.19, packed metadata 2.20, error-detail/dry-run 2.21, preserve-attribute split 2.22, rsync-parity wave 2.23) and is intentionally out of scope — documented divergences from rsync's integer-negotiated downgrade remain.
**Phase-1/2 selection-and-update status correction (docs):** `-I/--ignore-times`, `--size-only`, `-@/--modify-window`, `--existing`, `--ignore-existing`, `-u/--update`, `-W/--whole-file`, and `--compress-threads` were previously listed as not-implemented in this document but are in fact fully implemented and tested on `dev`. This pass corrects the matrix to match the code. The realistic model of these is that FastSync is a *sender-driven* whole-tree copy, so the size+mtime quick-check and all three receiver-policy skips (`--existing`, `--ignore-existing`, `-u`) are evaluated against the **destination** on the receiver side, and their booleans cross the wire in the config frame. `-I`/`--size-only`/`--modify-window` modify the `--incremental` per-file `STATUS_CHECK` handshake's match predicate (`-I` disables the mtime leg and forces transfer; `--size-only` drops only the mtime leg; `--modify-window` adds tolerance to `metadata_mtime_matches`); they require `--incremental` (or a basis dir) to have a handshake to affect, mirroring how they only matter where a quick-check exists in rsync. `--existing`/`--ignore-existing`/`-u` are receiver write-time policies (skipping the write / newer-destination guard) applied across the regular-file, `--delay-updates`-staged, hardlink-sibling, and special/device paths; `-u` implies `-M` metadata and uses a second-then-nanosecond strict `>` newer check; both correctly influence `--remove-source-files` (a skipped source is not removed). `-W/--whole-file` disables block-level delta (opt-in via `--delta`), folded into the wire `use_delta` so no protocol bump was needed, and makes `--fuzzy` inert; `--append`/`--append-verify` are rejected with `-W`. `--compress-threads=NUM` (1..64, client-only, never crosses the wire) sizes the zstd compression worker pool. No code was changed by this correction; the implementation had landed in earlier merge waves (feat/ignore-times, feat/ignore-existing via the newer `file_to_disk_secure_no_replace`/`linkat EEXIST` path, feat/size-only, feat/modify-window, feat/whole-file, feat/update, compression-threads).
@@ -879,7 +916,7 @@ These are the last compatibility items and the closing phase toward rsync flag p
**Wave D — Times superstructure & arg-protection no-ops (✅ implemented, `--secluded-args` ❌).** `-O`/`--omit-dir-times` and `-J`/`--omit-link-times` are now **real modifiers** (both `🔄 → ✅ Implemented`), reversing the old "never preserves directory/symlink times" divergence:
-- **Directory times.** The recursive scanner captures every traversed source directory's metadata (mtime, plus atime under `-U`) into a per-transfer list — two paths are covered: the sequential `DirectoryScanner` captures each opened directory (including the transfer root), and the parallel scanner captures both the root in `parallel_scanner_create_with_options` and each worker's subdirectories in `open_next_directory` (appends are guarded by a mutex shared with the sender's pipeline context). The sender transmits them in trailing `STATUS_DIR_TIMES` frames (each: int count + count × (wire path, metadata) pairs) sent **after all file data and after the optional delete manifest**, just before `STATUS_FINISHED`. A tree larger than `MAX_MANIFEST_ENTRIES` (1 048 576) directories is chunked into repeated frames, each within the receiver's per-frame bound. A dir-time entry is RECORD-ONLY (`file->dir_time_only`): `file_save_to_disk_full` returns `FILE_SAVE_SKIPPED` without creating anything, so a source directory that was empty (or pruned by `-m/--prune-empty-dirs`) is never resurrected. The receiver accumulates received directory metadata in a `DirTimeList` and applies it only at the very end — after the entire stream, after the commit-style `--delete` deletion, and after `--delay-updates` publication — because creating or removing a child bumps the parent's mtime. Application is fd-relative/walk-confined (`file_open_secure_parent` + `utimensat(..., AT_SYMLINK_NOFOLLOW)`) and best-effort per entry: an absent path (an intentionally uncreated empty dir) is skipped QUIETLY and only a real existing directory is stamped. `-O` (config boolean, already on the wire) makes the receiver skip the whole set. The single-threaded sink applies in `receiver_send_success_frame`; the `-j`/`--threads` sink accumulates in `write_thread` and server.c applies after both threads join and the deletion commits.
+- **Directory times.** The recursive scanner captures every traversed source directory's metadata (mtime, plus atime under `-U`) into a per-transfer list — two paths are covered: the sequential `DirectoryScanner` captures each opened directory (including the transfer root), and the parallel scanner captures both the root in `parallel_scanner_create_with_options` and each worker's subdirectories in `open_next_directory` (appends are guarded by a mutex shared with the sender's pipeline context). The sender transmits them in trailing `STATUS_DIR_TIMES` frames (each: int count + count × (wire path, metadata) pairs) sent **after all file data and after the optional delete manifest**, just before `STATUS_FINISHED`. A tree larger than `MAX_MANIFEST_ENTRIES` (1 048 576) directories is chunked into repeated frames, each within the receiver's per-frame bound. A dir-time entry is RECORD-ONLY (`file->dir_time_only`): `file_save_to_disk_full` returns `FILE_SAVE_SKIPPED` without creating anything (the directory's creation, when it is empty, is now carried by a separate `STATUS_MKDIR` entry the scanner emits for every directory that produced no transferred child, and `-m/--prune-empty-dirs` suppresses that). The receiver accumulates received directory metadata in a `DirTimeList` and applies it only at the very end — after the entire stream, after the commit-style `--delete` deletion, and after `--delay-updates` publication — because creating or removing a child bumps the parent's mtime. Application is fd-relative/walk-confined (`file_open_secure_parent` + `utimensat(..., AT_SYMLINK_NOFOLLOW)`) and best-effort per entry: an absent path (an intentionally uncreated empty dir) is skipped QUIETLY and only a real existing directory is stamped. `-O` (config boolean, already on the wire) makes the receiver skip the whole set. The single-threaded sink applies in `receiver_send_success_frame`; the `-j`/`--threads` sink accumulates in `write_thread` and server.c applies after both threads join and the deletion commits.
- **Symlink times/owner/mode.** `STATUS_SYMLINK` already carried metadata; the receiver now applies it with no-follow primitives only: `utimensat(..., AT_SYMLINK_NOFOLLOW)`, best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` (honest no-op where unsupported, e.g. Linux), and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)` via a new `identity_apply_ownership_link` that shares the identity resolver with the fd path. `-J` suppresses only the timestamps; ownership stays governed by the identity opt-in (`--numeric-ids`/`--usermap`/`--groupmap`/`--chown`) exactly like regular files. A symlink has no children, so this is applied immediately at creation.
- **Wire:** the shared `STATUS_DIR_TIMES` frame (and metadata on `STATUS_MKDIR` for `--dirs` entries) is a frame-sequence change, so `PROTOCOL_VERSION` was bumped **2.16.0 → 2.17.0**; every version-sensitive test (`--protocol` accepted/rejected values) was updated. The config-frame layout itself is unchanged (the omit booleans already crossed). Non-metadata and `--no-preserve` transfers send no `STATUS_DIR_TIMES` frame and no directory metadata, keeping them byte-identical.
@@ -893,7 +930,10 @@ These are the last compatibility items and the closing phase toward rsync flag p
**Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity.
-**Honest status after the parity-completion wave (protocol 2.26.0).** ✅ Parity 106 / ⚠️ Caveat 27 / ❌ Divergent 23 = 156 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The remaining ⚠️ rows are the ones with a documented residual (see the row notes and the **Parity Completion Wave (protocol 2.26.0)** section below).
+**Honest status after the parity 2.28.0 cycle (protocol 2.28.0), updated by the rsync-parity-stats, rsync-parity-options, rsync-parity-fs, parity-review, no-wire parity-track-1/2b and wire parity-track-4a/5a passes.** ✅ Parity 116 / ⚠️ Caveat 14 / ❌ Divergent 27 = 157 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. This reclassification makes every difference explicit. The completion wave closed 23 previously-caveated rows (9 that triage showed were already parity, plus 14 genuine fixes) and turned the 17 inherently non-rsync rows — native daemon config/auth, the FastSync batch container, the safe-subset device/privilege flags, `-X`'s privileged namespaces, `--fake-super`'s native xattr format, and the `--old-args` no-op — into explicit ❌ divergences. The stats pass flipped `--delete-delay` to ✅ (actual-removal accounting), but the parity-review pass moved it back to ⚠️ because FastSync charged the `--max-delete` budget at plan/snapshot time and left a refilled snapshotted directory in place, whereas rsync charges on actual removals and recursively removes a queued directory (including content created after its plan). The no-wire parity-track-1 pass fixed both (actual-removal charging plus recursive deferred removal with an independent deferred-list cap), narrowing the caveat to the partial-delete ordering. The stats pass also reclassified `--out-format` to ❌ (protocol-specific `%b`/delta-`%c`), and sharpened the `--stats`/`--progress`/`--checksum-choice` residuals. The options pass flipped `--bwlimit` and `--ignore-errors` to ✅ (rsync-exact size parsing and ~100 ms leaky-bucket throttling, and rsync's skip-unreadable-subdir plus IO-error-suppressed deletion with exit 23) and emits rsync-format `--info=name/flist/del/remove/nonreg/progress` lines (real-run `deleting`/`*deleting` carried over a new trailing `report_deletes` wire bool, `PROTOCOL_VERSION` 2.26.0 → 2.27.0), while reclassifying `-M` over daemon/TCP
+and receiver-side `protect`/`risk` re-derivation to ❌ (no argv channel /
+receiver filter engine); the wire parity-track-4a pass later added that
+receiver filter engine, flipping `--filter=RULE` back to ✅ (see above). The fs pass flips `-d/--dirs` and `--iconv` to ✅ — recursive transfers now recreate empty source directories (and replace a blocking destination non-directory with an incoming directory); `-R --no-implied-dirs --files-from` places a listed file under a missing implied parent with default attributes instead of refusing; and `--iconv` now reproduces rsync's push direction (destination charset = the spec's REMOTE half) — and reclassified six rows to ❌ after reproducing their exact residual with differential tests: `--temp-dir` (the receiver confines the scratch dir to the receive root, so an absolute temp dir is deliberately rejected although standalone rsync follows it), the three basis-dir options (FastSync xxHash-verifies a basis hit while rsync's `--size-only` quick check installs the wrong basis content), `--delay-updates` (fixed staging name wipes an unrelated destination entry of that name), and `--dry-run` (would-delete report over-reports). `--fuzzy` was also reclassified to ❌ (deterministic heuristic with a 10× size window, not rsync's matcher), but its residual is the candidate-selection heuristic itself: the final tree is byte-exact by design, so no destination differential can expose it and the row is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync differential. (Track 5b later showed the name heuristic is in fact rsync's own and moved the row ❌ → ⚠️, leaving only the narrower delta size window as the residual; see the track 5b paragraph above.) The remaining ⚠️ rows are the ones with a documented residual (see the row notes and the **Parity Completion Wave (protocol 2.26.0)** section below).
**Preserve-attribute split (protocol 2.21.0 → 2.22.0) — ✅ implemented.** FastSync splits the former single metadata bundle into four independent, rsync-compatible per-attribute flags — `-p/--perms`, `-t/--times`, `-o/--owner`, `-g/--group` — each with a negation (`--no-perms`/`--no-times`/`--no-owner`/`--no-group`, short `--no-p`/`--no-t`/`--no-o`/`--no-g`), plus `--no-preserve` clearing all four. `-a/--archive` is now full rsync `-rlptgoD` (owner and group included, though their application stays privilege-gated), `-A/--acls` implies `-p`, `-X/--xattrs` does not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply `-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the user explicitly negated them. Wire: the binary config frame gains four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/`preserve_group`) after `omit_link_times`, so `PROTOCOL_VERSION` is bumped **2.21.0 → 2.22.0**; the fixed-width `FileMetadata` layout is unchanged and the receiver gates the metadata frame on a derived `use_metadata`. Receiver behavior: each attribute is applied independently, directory modes are applied under `-p` (at the end of the transfer, alongside dir times), symlink mode under `-p`, and `-O/--omit-dir-times` suppresses directory times only. Documented divergences as of 2.22.0, **all but (d)/(e) removed by the rsync-parity wave (protocol 2.23.0)**: (a) the mode-masking divergence is **gone** — under `-p` the source mode is now copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky; (b) a brand-new file without `-p` still gets `source_mode & ~umask` when metadata is present (else the historical fixed `0644`), and a new *directory* without `-p` still uses FastSync's `0755` default; (c) the `--chmod`-implies-`-p` divergence is **gone** — `--chmod` no longer implies `-p` (rsync parity); (d) `-o`/`-g` map by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); (e) a daemon module without `client owner = yes` does not refuse a plain `-a`/`-o`/`-g` — it forces super off, applies no ownership, and logs a warning, while explicit `--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused.
@@ -1091,10 +1131,11 @@ wire protocol three times (full rationale in `src/shared/config.h`):
`sha1`, `none`, `auto`, plus the two-name `transfer,pre-transfer` form.
Unknown names and `none` on the transfer side under `--checksum` exit 4 like
rsync.
-- **Remaining codec residuals:** `zlibx` behaves as `zlib`; the transfer
- checksum is not independently selectable (only the whole-file comparison
- digest is); per-codec level defaults differ; and `RSYNC_CHECKSUM_LIST`/
- `RSYNC_COMPRESS_LIST` are not consulted.
+- **Codec residuals:** the transfer checksum is not independently selectable
+ (only the whole-file comparison digest is); `zlib`/`zlibx` share FastSync's
+ literal-only zlib path (rsync's zlibx behavior, observably identical for both,
+ so `zlibx` is not a divergence). Per-codec compression-level defaults and
+ `RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST` are implemented (track 3a).
### Selection, paths, and filters
@@ -1123,26 +1164,40 @@ wire protocol three times (full rationale in `src/shared/config.h`):
These remain after the wave; the individual rows carry the precise wording.
-- **`--stats`** lacks rsync's `(reg/dir/link)` breakdown on `Number of files`
- and `Number of created files`; **`--progress`** omits the leading `./` line
- and its `to-chk` total differs by the root entry; **`--out-format`** `%b`/`%c`
- count FastSync wire bytes.
-- **`-n --delete`** ordering can differ from rsync's delete-during walk and a
- filtered dry-run can over-report.
-- **Delete timing:** the default `--delete` remains delete-after rather than
- rsync's delete-during; per-directory plans have generator-order/abort-boundary
- differences; `--delete-before` keeps its pre-scan snapshot race; and
- destination-only files matching an exclude are still removed (protection is
- sender-derived). `--ignore-errors` exits 23 but its EACCES differential is not
+- **`--stats`** now reproduces rsync's `(reg/dir/link/special)` breakdown on
+ `Number of files`, the regular-transferred count and the size totals, but
+ `Number of created files` is the transferred-regular count without a type
+ breakdown and wire-byte totals differ; **`--progress`** now prints the leading
+ `./` line and includes the root in `to-chk` (single-file output is
+ byte-identical), but a multi-directory `to-chk` denominator and per-directory
+ name lines still differ; **`--out-format`** `%C` matches for every algorithm,
+ but `%b`/`%c` count FastSync wire bytes (protocol-specific, hence ❌).
+- **`-n --delete`** ordering can differ from rsync's delete-during walk (the
+ differential compares the sorted would-delete set).
+- **Delete timing:** plain `--delete` now defaults to rsync's delete-during
+ (lockstep track 6); the residual is the mid-transfer abort boundary, where
+ rsync's generator removes all extras ahead of its throttled sender while
+ FastSync removes only the reached directories (completed runs agree).
+ `--delete-before` keeps its pre-scan snapshot race; and while
+ base-rule `protect`/`risk` rules are now re-applied on the receiver (track 4a),
+ per-directory merge (`.rsync-filter`) protection is still sender-derived, so a
+ destination-only entry matching only a per-directory rule is not re-derived.
+ `--ignore-errors` exits 23 but its EACCES differential is not
exercised in CI.
- **`--delay-updates`** uses a fixed staging name with an advisory lock and
- deletes before publication; **`--temp-dir`** rejects absolute/foreign paths;
- **`--remote-option`** is SSH-only; **`--iconv`** keeps the receiver
- half-swap/charset-declaration caveat.
-- **Basis dirs** do not re-apply attributes on a match, keep the
- `--size-only` mtime caveat, and share the 256 MiB whole-file cap; **`--fuzzy`**
- has a different tie-break order; recursive transfers still do not create empty
- directories; and **`--bwlimit`** rejects rsync's `0`/decimal/suffixed rates.
+ deletes before publication; **`--temp-dir`** rejects absolute/foreign paths
+ (deliberately confined, see the row); **`--remote-option`** is SSH-only.
+ **`--iconv`** now matches rsync's push direction (destination charset = the
+ spec's REMOTE half; a server `--iconv` overrides it).
+- **Basis dirs** now use rsync's metadata quick-check by default (track 5a) and
+ stream a hit of any size; the FastSync-only `--verify-basis` restores the
+ stricter content equality. Remaining residuals: the relative-DIR resolution
+ base and the over-limit basis-MISS refusal. **`--fuzzy`** uses rsync's
+ name heuristic, but its candidate eligibility is bounded by the delta
+ engine (both files ≥ 16 KiB, size ratio ≤ 10×), a narrower window than
+ rsync's, so the selected basis — and the `--stats` bandwidth counters —
+ can differ while the tree stays byte-exact; and **`--bwlimit`** rejects rsync's
+ `0`/decimal/suffixed rates.
- **`--inc-recursive`/`--no-inc-recursive`** are not implemented (rejected).
### Intentional divergences (explicit ❌ rows)
@@ -1216,5 +1271,7 @@ Ranked by user demand, implementation complexity, and interoperability impact (_
| `--tls` | TLS encryption (mutual auth) |
| `--fastsync-server-path` | Path to fastsync-server binary |
| `--server-host` / `--server-port` | Direct TCP connection |
+| `--verify-basis` | FastSync-only (long form, not in rsync): require a `--compare-dest`/`--copy-dest`/`--link-dest` basis hit to match the source by whole-file digest instead of trusting rsync's size+mtime (or `--size-only`) quick-check. Off by default (the default matches rsync). Crosses the wire (protocol 2.28.0) |
+| `--delete-commit` | FastSync-only (long form, not in rsync): restore the late whole-tree delete commit — the keep-set manifest is committed only after the entire transfer succeeded. Identical timing to `--delete-after`, and implemented as the same `delete_after` wire bool (no new field); it exists because plain `--delete` now defaults to `--delete-during` (lockstep track 6). Implies `--delete` and conflicts with any different timing flag |
| Incremental sync | Skip unchanged files (size+mtime) |
| Delta transfer | Block-level delta for changed files |
diff --git a/pytest.ini b/pytest.ini
index 9cd9e47..a8b9327 100644
--- a/pytest.ini
+++ b/pytest.ini
@@ -8,3 +8,6 @@ markers =
daemon_detach: real double-fork backgrounding path (--daemon without
--no-detach); slower/fragile, so it runs in the full suite but not the
fast PR gate
+ parity: differential rsync-parity case (full set; runs on push to
+ dev/main)
+ parity_ci: fast differential rsync-parity subset (runs on the PR gate)
diff --git a/src/client/change_list.c b/src/client/change_list.c
index a14739c..9f3d728 100644
--- a/src/client/change_list.c
+++ b/src/client/change_list.c
@@ -1,5 +1,6 @@
#include "change_list.h"
#include "checksum.h"
+#include "log.h"
#include "utils.h"
#include
#include
@@ -70,7 +71,16 @@ static bool strbuf_append(StrBuf* buf, const char* text) {
bool change_list_enabled(const Config* config) {
return config != NULL && (config->itemize_changes || config->out_format != NULL ||
- (config->log_file != NULL && config->log_file_format != NULL));
+ (config->log_file != NULL && config->log_file_format != NULL) ||
+ (config->info_level & LOG_INFO_NAME) != 0);
+}
+
+/* Emitted once, lazily, ahead of the first --info=name entry: rsync prints the
+ * transfer-root `./` name line when the root directory is (re)created. */
+static bool name_root_printed = false;
+
+void change_reset_name_root(void) {
+ name_root_printed = false;
}
/* ---- Itemize code ---- */
@@ -192,27 +202,55 @@ char* change_render_itemize(const Config* config, const ChangeEvent* event) {
return line.data;
}
-/* ---- --out-format / --log-file-format ---- */
-
-/* rsync 3.4.1's `%C` uses the negotiated transfer checksum; with the default
- * "auto" choice on both ends that is xxh128. FastSync's internal XXH64 default
- * is not an rsync algorithm, so map it to xxh128 for parity. */
-static ChecksumAlgo out_format_checksum_algo(const Config* config) {
- switch ((ChecksumAlgo)config->checksum_algo) {
- case CHECKSUM_ALGO_MD5:
- return CHECKSUM_ALGO_MD5;
- case CHECKSUM_ALGO_XXH3:
- return CHECKSUM_ALGO_XXH3;
- case CHECKSUM_ALGO_XXH128:
- return CHECKSUM_ALGO_XXH128;
- case CHECKSUM_ALGO_XXH64:
- default:
- return CHECKSUM_ALGO_XXH128;
+/* rsync's `--info=name` line for an updated entry: the transfer-relative name
+ * (trailing slash for directories) plus the ` -> target` / ` => target` link
+ * suffix. `--info=name` does not alter an itemize/out-format run. */
+static char* change_render_name(const ChangeEvent* event) {
+ StrBuf line = {0};
+ bool ok = append_name(&line, event) && append_link_suffix(&line, event);
+ if (!ok) {
+ strbuf_free(&line);
+ return NULL;
}
+ if (line.data == NULL) {
+ line.data = str_dup("");
+ if (!line.data)
+ return NULL;
+ }
+ return line.data;
}
-/* Render a digest as rsync's sum_as_hex: for xxh128 the HIGH 64-bit half is
- * printed before the low half; every other algorithm prints its bytes in order. */
+/* rsync's `--info=name2` line for an unchanged entry: `NAME is uptodate`. */
+static char* change_render_name_uptodate(const ChangeEvent* event) {
+ char* name = change_render_name(event);
+ if (name == NULL)
+ return NULL;
+ size_t length = strlen(name);
+ char* line = malloc(length + sizeof(" is uptodate"));
+ if (line == NULL) {
+ free(name);
+ return NULL;
+ }
+ memcpy(line, name, length);
+ memcpy(line + length, " is uptodate", sizeof(" is uptodate"));
+ free(name);
+ return line;
+}
+
+/* ---- --out-format / --log-file-format ---- */
+
+/* rsync 3.4.1's `%C` uses the negotiated TRANSFER checksum (the first name of a
+ * two-name "transfer,pre-transfer" --checksum-choice), not the pre-transfer
+ * whole-file digest FastSync compares against on the wire. The default "auto"
+ * resolves to xxh128, so an explicit selection and the default both render the
+ * selected algorithm's digest. */
+static ChecksumAlgo out_format_checksum_algo(const Config* config) {
+ return (ChecksumAlgo)config->checksum_transfer_algo;
+}
+
+/* Render a digest as rsync's sum_as_hex: xxh128 prints the HIGH 64-bit half
+ * before the low half, and xxh64/xxh3 print their 64-bit value big-endian; every
+ * other algorithm prints its bytes in order. */
static void digest_to_hex(ChecksumAlgo algo, const uint8_t* digest, size_t len, char* out) {
if (algo == CHECKSUM_ALGO_XXH128 && len == 16) {
uint64_t low = 0;
@@ -222,6 +260,12 @@ static void digest_to_hex(ChecksumAlgo algo, const uint8_t* digest, size_t len,
snprintf(out, len * 2 + 1, "%016llx%016llx", (unsigned long long)high, (unsigned long long)low);
return;
}
+ if ((algo == CHECKSUM_ALGO_XXH64 || algo == CHECKSUM_ALGO_XXH3) && len == 8) {
+ uint64_t value = 0;
+ memcpy(&value, digest, sizeof(value));
+ snprintf(out, len * 2 + 1, "%016llx", (unsigned long long)value);
+ return;
+ }
static const char hex[] = "0123456789abcdef";
for (size_t i = 0; i < len; i++) {
out[i * 2] = hex[(digest[i] >> 4) & 0xf];
@@ -260,6 +304,9 @@ static void fill_event_checksum(const Config* config, const File* file, ChangeEv
if (file->path == NULL)
return;
ChecksumAlgo algo = out_format_checksum_algo(config);
+ /* rsync renders `--checksum-choice=none` as a blank 2-character column. */
+ if (algo == CHECKSUM_ALGO_NONE)
+ return;
uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
size_t len = 0;
/* rsync's %C is the transfer checksum, which is always seeded with 0 (it is
@@ -329,9 +376,10 @@ char* change_render_format(const char* format, const Config* config, const Chang
if (event->checksum_known) {
ok = strbuf_append(&line, event->checksum);
} else {
- /* rsync pads a non-regular / untransferred entry with spaces. */
+ /* rsync pads a non-regular / untransferred / `none` entry with spaces;
+ `none` renders as a blank 2-character column. */
ChecksumAlgo algo = out_format_checksum_algo(config);
- int width = checksum_digest_len(algo) * 2;
+ int width = algo == CHECKSUM_ALGO_NONE ? 2 : checksum_digest_len(algo) * 2;
for (int i = 0; i < width && ok; i++)
ok = strbuf_append_char(&line, ' ');
}
@@ -438,10 +486,24 @@ static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_ou
void change_emit(const Config* config, const ChangeEvent* event) {
if (event == NULL || !change_list_enabled(config))
return;
- if (event->decision == CHANGE_UP_TO_DATE)
- return;
bool to_stdout = config->itemize_changes || config->out_format != NULL;
bool to_log = config->log_file != NULL && config->log_file_format != NULL;
+ bool progress_active = config->show_progress || (config->info_level & LOG_INFO_PROGRESS);
+ if (event->decision == CHANGE_UP_TO_DATE) {
+ /* --info=name2 prints `NAME is uptodate` for entries the receiver already
+ had. An itemize/out-format run reports them through its own format (or
+ not at all), the progress stream has no frame for them, and neither the
+ itemize nor the log-file stream previously reported an up-to-date entry,
+ so nothing else here changes. */
+ if (!to_stdout && (config->info_level & LOG_INFO_NAME_UPTODATE) != 0 && !progress_active) {
+ char* line = change_render_name_uptodate(event);
+ if (line != NULL) {
+ print_escaped_line(stdout, line, config->eight_bit_output);
+ free(line);
+ }
+ }
+ return;
+ }
if (to_stdout) {
char* line = config->out_format != NULL
? change_render_format(config->out_format, config, event)
@@ -450,6 +512,20 @@ void change_emit(const Config* config, const ChangeEvent* event) {
print_escaped_line(stdout, line, config->eight_bit_output);
free(line);
}
+ } else if ((config->info_level & LOG_INFO_NAME) != 0 && !progress_active) {
+ /* --info=name without -i/--out-format: print the updated entry's name. The
+ --progress path owns the name line when progress output is active (it
+ emits the same names before the progress frames), so do not duplicate.
+ The transfer-root `./` line precedes the first such name. */
+ if (!name_root_printed) {
+ name_root_printed = true;
+ fputs("./\n", stdout);
+ }
+ char* line = change_render_name(event);
+ if (line != NULL) {
+ print_escaped_line(stdout, line, config->eight_bit_output);
+ free(line);
+ }
}
if (to_log) {
char* line = change_render_format(config->log_file_format, config, event);
@@ -610,6 +686,29 @@ void change_emit_file_sent(const Config* config, const File* file) {
change_emit_file_sent_bytes(config, file, payload, 0);
}
+void change_emit_file_uptodate(const Config* config, const File* file) {
+ if (file == NULL || !change_list_enabled(config))
+ return;
+ ChangeEvent event;
+ memset(&event, 0, sizeof(event));
+ event.decision = CHANGE_UP_TO_DATE;
+ event.is_directory = false;
+ event.is_symlink = file->is_symlink;
+ event.is_special = file->is_special;
+ event.is_hardlink = file->link_group != 0 && !file->link_first;
+ event.symlink_target = file->symlink_target;
+ event.hardlink_target = file->hardlink_target;
+ event.size = file->data != NULL ? file->data->size : 0;
+ event.dest = file->dest_state;
+ char* name = NULL;
+ char* path = NULL;
+ fill_event_from_file(config, file, &event, &name, &path);
+ if (name != NULL && path != NULL)
+ change_emit(config, &event);
+ free(name);
+ free(path);
+}
+
void change_emit_dir_sent(const Config* config, const File* file) {
if (file == NULL || !change_list_enabled(config))
return;
diff --git a/src/client/change_list.h b/src/client/change_list.h
index d8ea32c..a57e405 100644
--- a/src/client/change_list.h
+++ b/src/client/change_list.h
@@ -102,4 +102,13 @@ void change_emit_file_sent(const Config* config, const File* file);
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
void change_emit_dir_sent(const Config* config, const File* file);
+/* Build and emit a CHANGE_UP_TO_DATE event for a file the receiver already had.
+ * With --info=name2 it renders rsync's "NAME is uptodate" line (no output
+ * otherwise). */
+void change_emit_file_uptodate(const Config* config, const File* file);
+
+/* Reset the lazy transfer-root `./` line emitted ahead of the first
+ * --info=name entry. Call once at the start of a transfer. */
+void change_reset_name_root(void);
+
#endif
diff --git a/src/client/client_cli.c b/src/client/client_cli.c
index 6680bcb..c830ca9 100644
--- a/src/client/client_cli.c
+++ b/src/client/client_cli.c
@@ -23,6 +23,7 @@
#include
#include
#include
+#include
#include
#include
#include
@@ -160,10 +161,16 @@ static int set_compression_choice(Config* config, const char* value) {
return -1;
}
int algo;
- if (strcasecmp(value, "auto") == 0)
- algo = (int)compression_negotiate_default();
- else
+ if (strcasecmp(value, "auto") == 0) {
+ algo = compression_choice_resolve();
+ if (algo < 0) {
+ log_message(LOG_LEVEL_ERROR, "RSYNC_COMPRESS_LIST names no supported compression algorithm");
+ config->cli_exit_code = 4;
+ return -1;
+ }
+ } else {
algo = compression_algo_from_name(value);
+ }
if (algo < 0) {
log_message(LOG_LEVEL_ERROR,
"--compress-choice '%s' is not a supported algorithm; FastSync supports zstd, "
@@ -227,16 +234,25 @@ static int set_checksum_choice(Config* config, const char* value) {
config->cli_exit_code = 4;
return -1;
}
- ChecksumAlgo negotiated = checksum_negotiate_default();
+ int negotiated = -1;
+ if (rc1 == 1 || rc2 == 1) {
+ negotiated = checksum_choice_resolve();
+ if (negotiated < 0) {
+ log_message(LOG_LEVEL_ERROR, "RSYNC_CHECKSUM_LIST names no supported checksum algorithm");
+ config->cli_exit_code = 4;
+ return -1;
+ }
+ }
if (rc1 == 1)
- transfer = (int)negotiated;
+ transfer = negotiated;
if (!name2)
pre = transfer;
else if (rc2 == 1)
- pre = (int)negotiated;
+ pre = negotiated;
config->checksum_algo = pre;
config->checksum_transfer_algo = transfer;
+ config->checksum_choice_set = true;
/* rsync: "none" for the transfer checksum forces --whole-file. */
if (transfer == (int)CHECKSUM_ALGO_NONE)
config->whole_file = true;
@@ -501,7 +517,10 @@ static bool is_accepted_debug_category(const char* name) {
static bool is_accepted_info_category(const char* name) {
static const char* const categories[] = {
- "backup", "del", "flist", "mount", "nonreg", "progress", "remove", "syms", "symsafe",
+ "backup",
+ "mount",
+ "syms",
+ "symsafe",
};
for (size_t i = 0; i < sizeof(categories) / sizeof(categories[0]); i++) {
if (strcmp(name, categories[i]) == 0)
@@ -608,14 +627,36 @@ static int parse_info_flags(const char* value, Config* config) {
free(flags);
return 1;
}
- if (strcmp(name, "copy") == 0 || strcmp(name, "name") == 0)
+ if (strcmp(name, "copy") == 0)
flag = LOG_INFO_COPY;
- else if (strcmp(name, "misc") == 0)
+ else if (strcmp(name, "name") == 0) {
+ /* name level 2 adds rsync's "is uptodate" lines. */
+ if (level == 0)
+ parsed &= ~(uint32_t)(LOG_INFO_NAME | LOG_INFO_NAME_UPTODATE);
+ else {
+ parsed |= LOG_INFO_NAME;
+ if (level >= 2)
+ parsed |= LOG_INFO_NAME_UPTODATE;
+ else
+ parsed &= ~(uint32_t)LOG_INFO_NAME_UPTODATE;
+ }
+ continue;
+ } else if (strcmp(name, "misc") == 0)
flag = LOG_INFO_MISC;
else if (strcmp(name, "skip") == 0)
flag = LOG_INFO_SKIP;
else if (strcmp(name, "stats") == 0)
flag = LOG_INFO_STATS;
+ else if (strcmp(name, "del") == 0)
+ flag = LOG_INFO_DEL;
+ else if (strcmp(name, "remove") == 0)
+ flag = LOG_INFO_REMOVE;
+ else if (strcmp(name, "flist") == 0)
+ flag = LOG_INFO_FLIST;
+ else if (strcmp(name, "nonreg") == 0)
+ flag = LOG_INFO_NONREG;
+ else if (strcmp(name, "progress") == 0)
+ flag = LOG_INFO_PROGRESS;
else if (is_accepted_info_category(name))
continue;
else {
@@ -956,6 +997,12 @@ static const OptionEntry OPTION_TABLE[] = {
{"--delete-during", "--del", OPT_FLAG, offsetof(Config, delete_during)},
{"--delete-delay", NULL, OPT_FLAG, offsetof(Config, delete_delay)},
{"--delete-after", NULL, OPT_FLAG, offsetof(Config, delete_after)},
+ /* FastSync-only long spelling of the late whole-tree commit, which selects
+ the same timing as rsync's --delete-after in FastSync (the whole-tree
+ keep-set manifest is committed only after the entire transfer succeeded).
+ Plain --delete now defaults to delete-during, so this restores the old
+ FastSync behavior; it maps onto the same delete_after wire field. */
+ {"--delete-commit", NULL, OPT_FLAG, offsetof(Config, delete_after)},
{"--delete-excluded", NULL, OPT_FLAG, offsetof(Config, delete_excluded)},
{"--max-delete", NULL, OPT_SIGNED_INT, offsetof(Config, max_delete)},
{"--ignore-errors", NULL, OPT_FLAG, offsetof(Config, ignore_errors)},
@@ -1013,6 +1060,11 @@ static const OptionEntry OPTION_TABLE[] = {
* --remote-option is parsed. --trust-sender is a local receiver policy and
* never travels to the remote peer. */
{"--trust-sender", NULL, OPT_FLAG, offsetof(Config, trust_sender)},
+ /* FastSync-only (not an rsync option): require a basis-hit's content to
+ * match the source by whole-file digest instead of trusting rsync's
+ * size+mtime quick-check. Long-only; crosses the wire so the receiver
+ * performs the extra read/hash. */
+ {"--verify-basis", NULL, OPT_FLAG, offsetof(Config, verify_basis)},
};
/* Only boolean options with no required argument are safe to negate. */
@@ -1055,6 +1107,7 @@ static const NegatableOption NEGATABLE_OPTIONS[] = {
{"xattrs", "X", offsetof(Config, preserve_xattrs)},
{"acls", "A", offsetof(Config, preserve_acls)},
{"fake-super", NULL, offsetof(Config, fake_super)},
+ {"verify-basis", NULL, offsetof(Config, verify_basis)},
};
static bool opt_is(const char* arg, const char* name, const char* alias) {
@@ -1438,6 +1491,8 @@ static bool cli_handle_table_option(CliParseCtx* ctx) {
ctx->exit_code = -1;
return true;
}
+ if (entry->offset == offsetof(Config, compression_level))
+ config->compression_level_set = true;
if (entry->offset == offsetof(Config, chmod_spec)) {
mode_t ignored;
if (!chmod_apply(0, config->chmod_spec, &ignored)) {
@@ -1464,7 +1519,9 @@ static bool cli_handle_table_option(CliParseCtx* ctx) {
if (entry->offset == offsetof(Config, per_dir_filter) && config->per_dir_filter_count < INT_MAX)
config->per_dir_filter_count++;
/* A delete-timing flag selects when --delete removes extras, so it
- implies --delete exactly like the rsync options do. */
+ implies --delete exactly like the rsync options do. --delete-commit (the
+ FastSync-only late-commit spelling) is mapped onto delete_after and so is
+ covered here too. */
if (entry->offset == offsetof(Config, delete_before) ||
entry->offset == offsetof(Config, delete_during) ||
entry->offset == offsetof(Config, delete_delay) ||
@@ -1722,6 +1779,7 @@ static bool cli_handle_transfer_flags(CliParseCtx* ctx) {
return true;
}
config->compression_level = (int)level;
+ config->compression_level_set = true;
log_info_message(LOG_INFO_MISC, "Set Compression level to %ld", level);
ctx->i++;
}
@@ -1819,22 +1877,130 @@ static int set_log_file_option(Config* config, const char* log_path) {
return 0;
}
-/* Apply a --bwlimit value (kilobytes per second). Returns 0 on success, -1 on
- * error. */
+/* Faithful port of rsync 3.4.1's `parse_size_arg(bwlimit_arg, 'K', "bwlimit",
+ * 512, -1, True)`: a default KiB suffix, binary (1024) multipliers unless a
+ * `b`/`B` decimal suffix or explicit `iB` is given, an optional decimal
+ * fraction, the P/T/G/M/K suffixes, and the special rules that a value of 0
+ * means "no limit" while any other value below 512 bytes is rejected. The
+ * parsed byte count is then quantized to whole KiB exactly like rsync's
+ * `bwlimit = (size + 512) / 1024`. Returns 0 on success, -1 on a parse error. */
+static int parse_bwlimit_value(const char* value, unsigned long long* bytes_per_sec_out) {
+ if (!value || !bytes_per_sec_out)
+ return -1;
+ const char* arg = value;
+ int reps;
+ long long mult;
+ while (*arg >= '0' && *arg <= '9')
+ arg++;
+ if (*arg != '\0' && (*arg == '.' || *arg == localeconv()->decimal_point[0]))
+ for (arg++; *arg >= '0' && *arg <= '9'; arg++) {
+ }
+
+ char suffix = *arg && *arg != '+' && *arg != '-' ? *arg++ : 'K';
+ switch (suffix) {
+ case 'b':
+ case 'B':
+ reps = 0;
+ break;
+ case 'k':
+ case 'K':
+ reps = 1;
+ break;
+ case 'm':
+ case 'M':
+ reps = 2;
+ break;
+ case 'g':
+ case 'G':
+ reps = 3;
+ break;
+ case 't':
+ case 'T':
+ reps = 4;
+ break;
+ case 'p':
+ case 'P':
+ reps = 5;
+ break;
+ default:
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is invalid", value);
+ return -1;
+ }
+ if (*arg == 'b' || *arg == 'B') {
+ mult = 1000;
+ arg++;
+ } else if (*arg == '\0' || *arg == '+' || *arg == '-') {
+ mult = 1024;
+ } else if ((arg[0] == 'i' || arg[0] == 'I') && (arg[1] == 'b' || arg[1] == 'B')) {
+ mult = 1024;
+ arg += 2;
+ } else {
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is invalid", value);
+ return -1;
+ }
+
+ long long base = 1;
+ for (int i = 0; i < reps; i++) {
+ if (base > LLONG_MAX / mult) {
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is too large", value);
+ return -1;
+ }
+ base *= mult;
+ }
+ /* rsync multiplies the numeric prefix (atof) by mult^reps in a signed
+ * ssize_t, which is undefined on overflow. Scale in double and range-check
+ * before converting, so a huge value is rejected as "too large" (where
+ * rsync's overflow happens to land on a negative result) without invoking
+ * signed-overflow UB. */
+ double scaled = (double)base * strtod(value, NULL);
+ /* (double)LLONG_MAX rounds up to 2^63, which is itself out of range for the
+ * cast, so reject at >= that bound; LLONG_MIN == -2^63 is exactly
+ * representable and thus castable, so the lower bound stays strict. */
+ if (!isfinite(scaled) || scaled >= (double)LLONG_MAX || scaled < (double)LLONG_MIN) {
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is too large", value);
+ return -1;
+ }
+ long long size = (long long)scaled;
+ if ((*arg == '+' || *arg == '-') && arg[1] == '1' && arg != value) {
+ /* The only form accepted here is "+1"/"-1" (a longer number leaves a
+ trailing byte and is rejected below), so apply the delta directly and
+ guard the one overflow direction. */
+ if (*arg == '+') {
+ if (size == LLONG_MAX) {
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is too large", value);
+ return -1;
+ }
+ size += 1;
+ } else {
+ size -= 1;
+ }
+ arg += 2;
+ }
+ if (*arg != '\0' || size < 0) {
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is %s", value, size < 0 ? "too large" : "invalid");
+ return -1;
+ }
+ if (size != 0 && size < 512) {
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is too small (min: 512 or 0 for unlimited)", value);
+ return -1;
+ }
+ long long kib = size == 0 ? 0 : (size + 512) / 1024;
+ if (kib > (long long)(ULLONG_MAX / 1024)) {
+ log_message(LOG_LEVEL_ERROR, "--bwlimit=%s is too large", value);
+ return -1;
+ }
+ *bytes_per_sec_out = (unsigned long long)kib * 1024;
+ return 0;
+}
+
+/* Apply a --bwlimit value using rsync 3.4.1's units/semantics. Returns 0 on
+ * success, -1 on error. */
static int set_bwlimit_option(const char* value) {
- unsigned long long kbps;
- if (parse_ull_arg(value, &kbps, "--bwlimit") != 0)
+ unsigned long long bytes_per_sec;
+ if (parse_bwlimit_value(value, &bytes_per_sec) != 0)
return -1;
- if (kbps == 0) {
- log_message(LOG_LEVEL_ERROR, "--bwlimit must be a positive integer");
- return -1;
- }
- if (kbps > ULLONG_MAX / 1024) {
- log_message(LOG_LEVEL_ERROR, "--bwlimit value too large");
- return -1;
- }
- io_set_bwlimit(kbps * 1024);
- log_info_message(LOG_INFO_MISC, "Set bandwidth limit to %llu KB/s", kbps);
+ io_set_bwlimit(bytes_per_sec);
+ log_info_message(LOG_INFO_MISC, "Set bandwidth limit to %llu KB/s", bytes_per_sec / 1024);
return 0;
}
@@ -2409,6 +2575,18 @@ static bool cli_handle_outbuf_option(CliParseCtx* ctx) {
* -1 on error. */
static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool no_incremental) {
set_log_level(config->quiet ? LOG_LEVEL_ERROR : (verbose ? LOG_LEVEL_DEBUG : LOG_LEVEL_WARNING));
+ /* rsync's plain --delete defaults to delete-during (--del): each directory's
+ extras are removed as that directory is processed, so space is freed
+ progressively and a tight destination never has to hold the whole old+new
+ tree at once. The late whole-tree commit FastSync historically used is
+ still selected explicitly by --delete-after or by the FastSync-only long
+ spelling --delete-commit (an exact alias for --delete-after). Resolve the
+ default on the client, before validation and before the config crosses the
+ wire, so exactly one timing flag is ever set; an explicit timing (including
+ --delete-commit) always wins. */
+ if (config->use_delete && !config->delete_before && !config->delete_during &&
+ !config->delete_delay && !config->delete_after)
+ config->delete_during = true;
if (config->compress_choice) {
int algo = compression_algo_from_name(config->compress_choice);
if (algo >= 0) {
@@ -2416,8 +2594,43 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool
config->use_compression = (algo != (int)COMPRESSION_ALGO_NONE);
}
}
- if (config->use_compression && config->compression_algo == (int)COMPRESSION_ALGO_NONE)
- config->compression_algo = (int)compression_negotiate_default();
+ /* A bare -z (no --compress-choice) resolves like rsync's "auto": the
+ * RSYNC_COMPRESS_LIST preference list first, then the compiled-in order. A
+ * list that names no supported codec is rsync's failed negotiation (exit 4). */
+ if (config->use_compression && !config->compress_choice) {
+ int resolved = compression_choice_resolve();
+ if (resolved < 0) {
+ log_message(LOG_LEVEL_ERROR, "RSYNC_COMPRESS_LIST names no supported compression algorithm");
+ config->cli_exit_code = 4;
+ return -1;
+ }
+ config->compression_algo = resolved;
+ if (resolved == (int)COMPRESSION_ALGO_NONE)
+ config->use_compression = false;
+ }
+ /* Apply rsync's per-codec compression level: an explicit --compress-level is
+ * clamped to the codec's range, otherwise the codec's own default is used. */
+ if (config->use_compression) {
+ CompressionAlgo algo = (CompressionAlgo)config->compression_algo;
+ config->compression_level = config->compression_level_set
+ ? compression_clamp_level(algo, config->compression_level)
+ : compression_default_level(algo);
+ log_debug_message(LOG_DEBUG_UTIL, "Client compression: %s (level %d)",
+ compression_algo_name(algo), config->compression_level);
+ }
+ /* The negotiated checksum is always resolved (rsync negotiates one for the
+ * delta strong sum even without --checksum): RSYNC_CHECKSUM_LIST first, then
+ * the compiled-in order. An explicit --checksum-choice already set it. */
+ if (!config->checksum_choice_set) {
+ int resolved = checksum_choice_resolve();
+ if (resolved < 0) {
+ log_message(LOG_LEVEL_ERROR, "RSYNC_CHECKSUM_LIST names no supported checksum algorithm");
+ config->cli_exit_code = 4;
+ return -1;
+ }
+ config->checksum_algo = resolved;
+ config->checksum_transfer_algo = resolved;
+ }
/* rsync parity: "none" as the pre-transfer checksum cannot be combined with
* --checksum (exit 4). The check runs here because --checksum may appear on
* either side of --checksum-choice. */
@@ -2549,8 +2762,15 @@ static int cli_finalize_config(Config* config, bool verbose, bool no_delta, bool
}
}
}
- config->report_stats = config->stats || config->show_progress || format_needs_wire ||
- (config->dry_run && config->use_delete);
+ /* --info=del on a real --delete run asks the receiver to report the paths it
+ actually removed; the report rides the STATUS_STATS path list, so the wire
+ stats frame must be negotiated too. */
+ config->report_deletes = config->use_delete && !config->dry_run &&
+ ((config->info_level & LOG_INFO_DEL) != 0 || config->itemize_changes ||
+ config->out_format != NULL);
+ config->report_stats = config->stats || config->show_progress ||
+ (config->info_level & LOG_INFO_PROGRESS) || format_needs_wire ||
+ config->report_deletes || (config->dry_run && config->use_delete);
return 0;
}
diff --git a/src/client/client_send.c b/src/client/client_send.c
index 34314d5..767ee21 100644
--- a/src/client/client_send.c
+++ b/src/client/client_send.c
@@ -66,6 +66,16 @@ static void log_server_rejection(const char* context) {
}
}
+/* rsync's --ignore-errors semantics: an I/O error during the transfer normally
+ * suppresses deletion entirely ("IO error encountered -- skipping file
+ * deletion"); --ignore-errors lets the deletion run anyway. FastSync always
+ * continues past an unreadable subdirectory so the readable tree transfers, and
+ * always reports the partial transfer (exit 23); this only decides whether the
+ * deletion phase is skipped. Returns true when deletion may proceed. */
+bool ignore_errors_allows_delete(const Config* config, bool had_io_error) {
+ return !had_io_error || (config && config->ignore_errors);
+}
+
static const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer,
size_t buffer_size) {
if (human_readable && format_human_size_decimal(bytes, buffer, buffer_size))
@@ -83,15 +93,55 @@ static const char* stats_bytes(const Config* config, unsigned long long bytes, c
return buffer;
}
-/* Print the rsync `--stats` block on stdout. Byte totals use the process-wide
- wire counters and the receiver-only counters come from the STATUS_STATS frame;
- the labels, layout and rate/speedup formulas match rsync 3.4.1. Shared by the
- single-threaded and multithreaded send paths. */
-static void report_transfer_stats(const Config* config, int total_files,
- unsigned long long total_bytes, time_t start,
+/* Build rsync's per-type parenthetical: each non-zero category, in
+ reg/dir/link/special order. Empty when every count is zero. */
+static void type_breakdown(unsigned long long reg, unsigned long long dir, unsigned long long link,
+ unsigned long long special, char* out, size_t out_size) {
+ if (reg + dir + link + special == 0) {
+ out[0] = '\0';
+ return;
+ }
+ out[0] = '\0';
+ size_t used = 0;
+ const struct {
+ const char* name;
+ unsigned long long count;
+ } parts[4] = {{"reg", reg}, {"dir", dir}, {"link", link}, {"special", special}};
+ bool first = true;
+ for (size_t i = 0; i < 4; i++) {
+ if (parts[i].count == 0)
+ continue;
+ int written = snprintf(out + used, out_size - used, "%s%s: %llu", first ? "(" : ", ",
+ parts[i].name, parts[i].count);
+ if (written < 0 || (size_t)written >= out_size - used)
+ break;
+ used += (size_t)written;
+ first = false;
+ }
+ if (!first && used + 1 < out_size)
+ out[used++] = ')';
+ out[used] = '\0';
+}
+
+/* Build rsync's `Number of files` parenthetical from the scan's flist counts. */
+static void stats_type_breakdown(const TransferStats* stats, char* out, size_t out_size) {
+ type_breakdown(stats->flist_reg, stats->flist_dir, stats->flist_link, stats->flist_special, out,
+ out_size);
+}
+
+/* Print the rsync `--stats` block on stdout. The source-side flist and
+ transferred counters come from `stats` (filled while scanning/sending), the
+ receiver-only counters from the STATUS_STATS frame, and the wire byte totals
+ from the process-wide protocol counters. The labels, layout and
+ rate/speedup formulas match rsync 3.4.1. Shared by the single-threaded and
+ multithreaded send paths. */
+static void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start,
const ReceiverStats* recv) {
if (!config->stats || config->quiet)
return;
+ TransferStats empty = {0};
+ if (stats == NULL)
+ stats = ∅
ReceiverStats none = {0};
if (recv == NULL)
recv = &none;
@@ -101,11 +151,26 @@ static void report_transfer_stats(const Config* config, int total_files,
double elapsed = difftime(time(NULL), start);
double rate = (double)(sent + received) / (0.5 + elapsed);
char total_buffer[32];
+ char transferred_buffer[32];
+ char literal_buffer[32];
+ char matched_buffer[32];
char sent_buffer[32];
char recv_buffer[32];
char rate_buffer[32] = {0};
char human_rate[32] = {0};
- const char* total = stats_bytes(config, total_bytes, total_buffer, sizeof(total_buffer));
+ const char* total =
+ stats_bytes(config, stats->total_file_size, total_buffer, sizeof(total_buffer));
+ const char* transferred = stats_bytes(config, stats->transferred_file_size, transferred_buffer,
+ sizeof(transferred_buffer));
+ /* Protocol 2.28.0: the receiver reports the bytes it literally stored, which
+ is exact for a delta transfer (the sender's own literal_data counts each
+ stored file's whole source size and is only an upper bound). Fall back to
+ the sender total when the receiver reported no delta/literal accounting
+ (e.g. a local no-server path). */
+ unsigned long long literal_bytes = (recv->literal_bytes != 0 || recv->matched_data != 0)
+ ? recv->literal_bytes
+ : stats->literal_data;
+ const char* literal = stats_bytes(config, literal_bytes, literal_buffer, sizeof(literal_buffer));
const char* sent_s = stats_bytes(config, sent, sent_buffer, sizeof(sent_buffer));
const char* recv_s = stats_bytes(config, received, recv_buffer, sizeof(recv_buffer));
const char* rate_str = rate_buffer;
@@ -116,16 +181,36 @@ static void report_transfer_stats(const Config* config, int total_files,
} else {
snprintf(rate_buffer, sizeof(rate_buffer), "%.2f", rate);
}
- double speedup = (sent + received) > 0 ? (double)total_bytes / (double)(sent + received) : 0.0;
+ double speedup =
+ (sent + received) > 0 ? (double)stats->total_file_size / (double)(sent + received) : 0.0;
+ char breakdown[128];
+ stats_type_breakdown(stats, breakdown, sizeof(breakdown));
+ unsigned long long flist_total =
+ stats->flist_reg + stats->flist_dir + stats->flist_link + stats->flist_special;
+ char created_breakdown[128];
+ type_breakdown(recv->created_reg, recv->created_dir, recv->created_link, recv->created_special,
+ created_breakdown, sizeof(created_breakdown));
+ unsigned long long created_total =
+ recv->created_reg + recv->created_dir + recv->created_link + recv->created_special;
printf("\n");
- printf("Number of files: %d\n", total_files);
- printf("Number of created files: %d\n", total_files);
+ if (breakdown[0] != '\0')
+ printf("Number of files: %llu %s\n", flist_total, breakdown);
+ else
+ printf("Number of files: %llu\n", flist_total);
+ /* Protocol 2.28.0: the receiver reports which destination entries it newly
+ created, split by type, so this line matches rsync exactly. */
+ if (created_breakdown[0] != '\0')
+ printf("Number of created files: %llu %s\n", created_total, created_breakdown);
+ else
+ printf("Number of created files: %llu\n", created_total);
printf("Number of deleted files: %llu\n", recv->deleted_files);
- printf("Number of regular files transferred: %d\n", total_files);
+ printf("Number of regular files transferred: %llu\n", stats->transferred_regular);
printf("Total file size: %s bytes\n", total);
- printf("Total transferred file size: %s bytes\n", total);
- printf("Literal data: %s bytes\n", total);
- printf("Matched data: %llu bytes\n", recv->matched_data);
+ printf("Total transferred file size: %s bytes\n", transferred);
+ printf("Literal data: %s bytes\n", literal);
+ const char* matched =
+ stats_bytes(config, recv->matched_data, matched_buffer, sizeof(matched_buffer));
+ printf("Matched data: %s bytes\n", matched);
printf("File list size: 0\n");
printf("File list generation time: 0.000 seconds\n");
printf("File list transfer time: 0.000 seconds\n");
@@ -138,6 +223,48 @@ static void report_transfer_stats(const Config* config, int total_files,
fflush(stdout);
}
+/* Classify one scanned source entry into the rsync flist counters. Called for
+ every entry the sender walks, transferred or skipped. Directory entries are
+ counted here only for the explicit -d/--dirs generator; a recursive scan's
+ directories are accounted from the scanner's dir_entries list at report time. */
+static void transfer_stats_note_entry(TransferStats* stats, const File* file) {
+ if (stats == NULL || file == NULL)
+ return;
+ if (file->is_dir) {
+ stats->flist_dir++;
+ return;
+ }
+ if (file->is_symlink) {
+ stats->flist_link++;
+ stats->total_file_size += file->symlink_target ? strlen(file->symlink_target) : 0;
+ return;
+ }
+ if (file->is_special) {
+ stats->flist_special++;
+ return;
+ }
+ stats->flist_reg++;
+ stats->total_file_size += file->data ? file->data->size : 0;
+}
+
+/* Account for a regular file (or a whole-file append) the receiver actually
+ stored: rsync's transferred-file count and transferred/literal byte totals.
+ `literal_data` counts the whole source size, which is exact for a whole-file
+ send but an upper bound for a delta send (the receiver reuses basis blocks
+ the sender never ships); see TransferStats.literal_data in format.h. */
+static void transfer_stats_note_transferred(TransferStats* stats, const File* file) {
+ if (stats == NULL || file == NULL)
+ return;
+ if (file->is_dir || file->is_symlink || file->is_special)
+ return;
+ if (file->link_group != 0 && !file->link_first)
+ return;
+ unsigned long long size = file->data ? file->data->size : 0;
+ stats->transferred_regular++;
+ stats->transferred_file_size += size;
+ stats->literal_data += size;
+}
+
/* ---- rsync-style per-file --progress ------------------------------------
* rsync prints, for each transferred regular file, the file name followed by a
* two-frame progress line: the first at the initial 32 KiB read window (always
@@ -148,10 +275,61 @@ static void report_transfer_stats(const Config* config, int total_files,
static const char* delete_display_path(const Config* config, const char* path);
+/* Paths-only pre-count of the source file list, built once at transfer start
+ * when progress output is requested. rsync's `to-chk` denominator is the whole
+ * file list -- every regular file, directory, symlink and special plus the
+ * transfer root -- while the streaming scan never emits directories. A
+ * metadata-only walk (no file reads, no hashing) supplies that total and the
+ * directory names, so the opt-in pass leaves non-progress runs untouched. */
+typedef struct {
+ unsigned long long total;
+ ArrayList* dir_paths; /* owned char* in transfer-relative display form */
+} ProgressPrecount;
+
static bool g_progress_active;
static unsigned long long g_progress_xferred;
-static unsigned long long g_progress_seen;
+static unsigned long long g_progress_index;
+static unsigned long long g_progress_total;
static struct timespec g_progress_file_start;
+static ProgressPrecount g_progress_precount;
+static PathIndex g_progress_dir_index;
+static bool g_progress_dir_index_valid;
+static StrHashSet g_progress_emitted;
+static bool g_progress_emitted_valid;
+static ArrayList* g_progress_emitted_keys;
+
+static bool progress_requested(const Config* config) {
+ return config != NULL && !config->quiet &&
+ (config->show_progress || (config->info_level & LOG_INFO_PROGRESS) != 0);
+}
+
+static void progress_precount_dispose(ProgressPrecount* p) {
+ if (p->dir_paths != NULL) {
+ array_list_delete(p->dir_paths);
+ p->dir_paths = NULL;
+ }
+ p->total = 0;
+}
+
+static void client_progress_cleanup(void) {
+ if (g_progress_dir_index_valid) {
+ path_index_free(&g_progress_dir_index);
+ g_progress_dir_index_valid = false;
+ }
+ if (g_progress_emitted_valid) {
+ str_hash_set_free(&g_progress_emitted);
+ g_progress_emitted_valid = false;
+ }
+ if (g_progress_emitted_keys != NULL) {
+ array_list_delete(g_progress_emitted_keys);
+ g_progress_emitted_keys = NULL;
+ }
+ progress_precount_dispose(&g_progress_precount);
+ g_progress_active = false;
+ g_progress_total = 0;
+ g_progress_index = 0;
+ g_progress_xferred = 0;
+}
static void progress_first_frame(unsigned long long size, char* out, size_t out_size) {
char ofs_buf[32];
@@ -188,19 +366,139 @@ static void progress_final_frame(unsigned long long size, char* out, size_t out_
unsigned long long remain = (unsigned long long)(diff_ms / 1000);
snprintf(rembuf, sizeof(rembuf), "%4u:%02u:%02u", (unsigned)(remain / 3600),
(unsigned)((remain / 60) % 60), (unsigned)(remain % 60));
- unsigned long long to_chk =
- g_progress_seen > g_progress_xferred ? g_progress_seen - g_progress_xferred : 0;
+ /* rsync's `to-chk` denominator is the whole file list (the pre-count); the
+ numerator falls as each entry is processed, root first. Without a
+ pre-count (the paths-only walk failed) fall back to the transferred-file
+ count so the single-file layout stays intact. */
+ unsigned long long total = g_progress_total > 0 ? g_progress_total : g_progress_xferred + 1;
+ unsigned long long to_chk = total > g_progress_index ? total - g_progress_index - 1 : 0;
snprintf(out, out_size, "\r%15s %3d%% %7.2f%s %s (xfr#%llu, to-chk=%llu/%llu)\n", ofs_buf, 100,
- rate, units, rembuf, g_progress_xferred, to_chk, g_progress_seen);
+ rate, units, rembuf, g_progress_xferred, to_chk, total);
+}
+
+static bool info_flag_enabled(const Config* config, LogInfoFlag flag) {
+ return config != NULL && (config->info_level & flag) != 0;
+}
+
+/* Print rsync's deletion lines for a received list of destination-relative
+ * paths: `*deleting PATH` when itemizing, the --out-format expansion when a
+ * format is set, else `deleting PATH` for --info=del. Used by both the dry-run
+ * would-delete report and the real --info=del report. */
+static void print_delete_reports(const Config* config, const ArrayList* paths) {
+ if (!config || !paths || config->quiet)
+ return;
+ if (!(config->itemize_changes || config->out_format != NULL ||
+ info_flag_enabled(config, LOG_INFO_DEL)))
+ return;
+ for (int i = 0; i < paths->size; i++) {
+ const char* raw = (const char*)paths->items[i];
+ const char* path = delete_display_path(config, raw);
+ if (config->out_format != NULL) {
+ ChangeEvent event;
+ memset(&event, 0, sizeof(event));
+ event.decision = CHANGE_SENT;
+ event.deleted = true;
+ event.name = path;
+ event.path = path;
+ char* line = change_render_format(config->out_format, config, &event);
+ if (line) {
+ char* escaped = output_escape(line, config->eight_bit_output);
+ printf("%s\n", escaped ? escaped : line);
+ free(escaped);
+ free(line);
+ }
+ } else {
+ char* escaped = output_escape(path, config->eight_bit_output);
+ if (config->itemize_changes)
+ printf("*deleting %s\n", escaped ? escaped : path);
+ else
+ printf("deleting %s\n", escaped ? escaped : path);
+ free(escaped);
+ }
+ }
+ fflush(stdout);
+}
+
+static void client_progress_emit_ancestors(const Config* config, const char* rel) {
+ if (!g_progress_dir_index_valid || !g_progress_emitted_valid || g_progress_emitted_keys == NULL ||
+ rel == NULL)
+ return;
+ size_t rel_len = strlen(rel);
+ for (size_t i = 0; i < rel_len; i++) {
+ if (rel[i] != '/')
+ continue;
+ char* prefix = malloc(i + 1);
+ if (prefix == NULL)
+ return;
+ memcpy(prefix, rel, i);
+ prefix[i] = '\0';
+ if (path_index_contains(&g_progress_dir_index, prefix) &&
+ !str_hash_set_lookup(&g_progress_emitted, prefix)) {
+ char* key = str_dup(prefix);
+ if (key != NULL && array_list_add(g_progress_emitted_keys, key)) {
+ str_hash_set_insert_ref(&g_progress_emitted, key);
+ char* escaped = output_escape(prefix, config->eight_bit_output);
+ printf("%s/\n", escaped ? escaped : prefix);
+ free(escaped);
+ g_progress_index++;
+ } else {
+ free(key);
+ }
+ }
+ free(prefix);
+ }
+}
+
+/* rsync's --info=name/progress line for one entry: transfer-relative name (a
+ * trailing slash for directories) plus the ` -> target` symlink suffix. */
+static char* progress_entry_line(const File* file, const char* rel) {
+ const char* arrow = NULL;
+ const char* target = NULL;
+ if (file->is_symlink && file->symlink_target != NULL) {
+ arrow = " -> ";
+ target = file->symlink_target;
+ } else if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) {
+ arrow = " => ";
+ target = file->hardlink_target;
+ }
+ size_t rel_len = strlen(rel);
+ bool dir_slash = file->is_dir && (rel_len == 0 || rel[rel_len - 1] != '/');
+ size_t extra = (dir_slash ? 1u : 0u) + (target != NULL ? 4u + strlen(target) : 0u);
+ char* line = malloc(rel_len + extra + 1);
+ if (line == NULL)
+ return NULL;
+ memcpy(line, rel, rel_len);
+ size_t off = rel_len;
+ if (dir_slash)
+ line[off++] = '/';
+ if (target != NULL) {
+ memcpy(line + off, arrow, 4);
+ off += 4;
+ memcpy(line + off, target, strlen(target));
+ off += strlen(target);
+ }
+ line[off] = '\0';
+ return line;
}
static void client_progress_begin(const Config* config) {
- g_progress_active = config->show_progress && !config->quiet;
+ change_reset_name_root();
+ g_progress_active = progress_requested(config);
g_progress_xferred = 0;
- g_progress_seen = 0;
- if (!g_progress_active)
+ g_progress_index = 1; /* the transfer root is file-list entry #0 */
+ if (!g_progress_active) {
+ /* `--info=flist` prints rsync's file-list header even without progress. */
+ if (!config->quiet && info_flag_enabled(config, LOG_INFO_FLIST)) {
+ printf("sending incremental file list\n");
+ fflush(stdout);
+ }
return;
+ }
printf("sending incremental file list\n");
+ /* rsync prints the transfer-root directory's name before the first file when
+ that directory is created; FastSync mirrors the source root below the
+ receive root and creates it on a fresh destination, so emit it here. */
+ printf("./\n");
fflush(stdout);
}
@@ -209,12 +507,14 @@ static void client_progress_begin(const Config* config) {
static void client_progress_file(const Config* config, const File* file) {
if (!g_progress_active || file == NULL || !file->data)
return;
- g_progress_seen++;
g_progress_xferred++;
unsigned long long size = file->data->size;
if (!config->itemize_changes && config->out_format == NULL) {
- const char* name = delete_display_path(config, file_wire_path(file));
- printf("%s\n", name ? name : "");
+ const char* rel = delete_display_path(config, file_wire_path(file));
+ client_progress_emit_ancestors(config, rel);
+ char* escaped = output_escape(rel, config->eight_bit_output);
+ printf("%s\n", escaped ? escaped : (rel ? rel : ""));
+ free(escaped);
}
clock_gettime(CLOCK_MONOTONIC, &g_progress_file_start);
char frame[160];
@@ -222,9 +522,41 @@ static void client_progress_file(const Config* config, const File* file) {
fputs(frame, stdout);
progress_final_frame(size, frame, sizeof(frame));
fputs(frame, stdout);
+ g_progress_index++;
fflush(stdout);
}
+/* Emit the name line for a transferred non-regular entry (directory, symlink,
+ * special or hard-link sibling): rsync prints these in the file list but has no
+ * progress frame for them. */
+static void client_progress_name(const Config* config, const File* file) {
+ if (!g_progress_active || file == NULL)
+ return;
+ const char* rel = delete_display_path(config, file_wire_path(file));
+ if (!config->itemize_changes && config->out_format == NULL) {
+ client_progress_emit_ancestors(config, rel);
+ char* line = progress_entry_line(file, rel ? rel : "");
+ if (line != NULL) {
+ char* escaped = output_escape(line, config->eight_bit_output);
+ printf("%s\n", escaped ? escaped : line);
+ free(escaped);
+ free(line);
+ fflush(stdout);
+ }
+ }
+ g_progress_index++;
+}
+
+/* An entry the receiver already had prints no name under --progress but still
+ * occupies a file-list slot in the `to-chk` numerator. */
+static void client_progress_uptodate(const Config* config, const File* file) {
+ (void)config;
+ (void)file;
+ if (!g_progress_active)
+ return;
+ g_progress_index++;
+}
+
/* Compiled scanner inputs that are shared read-only across scanner instances
* and, in -m mode, across worker threads. `base_filters` owns the compiled
* command-line + -C rules; the FileListSet allow-set lives in the Config.
@@ -309,6 +641,13 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann
options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2;
options->dirs = config->dirs;
options->relative = config->relative;
+ /* A real recursive transfer recreates empty source directories (rsync
+ parity); low-level scanner users leave this off. */
+ options->emit_empty_dirs = true;
+ /* --no-implied-dirs only has meaning with -R (rsync): without it the option
+ is a documented no-op, so the scanner must not suppress directory
+ metadata. */
+ options->no_implied_dirs = config->no_implied_dirs && config->relative;
/* -R/--relative outside --files-from reconstructs every destination path from
* the source spec (rsync's '/./' cut point). With --files-from the listed
* entry already supplies the bare relative path, so no prefix is built. */
@@ -325,6 +664,9 @@ static bool prepare_scanner(const Config* config, int num_threads, PreparedScann
options->prune_empty_dirs = config->prune_empty_dirs;
options->ignore_io_errors = config->ignore_errors;
options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args;
+ options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet;
+ options->send_directory = config->send_directory;
+ options->eight_bit_output = config->eight_bit_output;
options->excluded_paths = NULL;
options->excluded_mutex = NULL;
options->size_skipped_paths = NULL;
@@ -359,6 +701,126 @@ static void prepared_scanner_destroy(PreparedScanner* prepared) {
prepared->relative_prefix = NULL;
}
+static bool progress_precount_add_dir(ProgressPrecount* p, const char* path) {
+ if (path == NULL || path[0] == '\0')
+ return true;
+ char* dup = str_dup(path);
+ if (dup == NULL)
+ return false;
+ if (array_list_add(p->dir_paths, dup))
+ return true;
+ free(dup);
+ return false;
+}
+
+/* Metadata-only walk collecting the full file-list total and every directory
+ * name. It uses its own scanner (fresh filter compilation and hard-link table)
+ * so the data pass's link-group state is never perturbed. */
+static bool progress_precount_scan(const Config* config, ProgressPrecount* out) {
+ out->dir_paths = array_list_create(free);
+ if (out->dir_paths == NULL)
+ return false;
+ out->total = 0;
+ PreparedScanner prepared;
+ memset(&prepared, 0, sizeof(prepared));
+ if (!prepare_scanner(config, 0, &prepared)) {
+ progress_precount_dispose(out);
+ return false;
+ }
+ ScannerOptions local = prepared.options;
+ local.list_dirs = true;
+ local.note_nonreg = false;
+ local.use_metadata = false;
+ local.preserve_xattrs = false;
+ local.preserve_acls = false;
+ local.checksum = false;
+ local.capture_dir_times = false;
+ local.excluded_paths = NULL;
+ local.size_skipped_paths = NULL;
+ local.synced_dirs = NULL;
+ local.plan_dirs = NULL;
+ local.dir_entries = NULL;
+ local.dir_entries_mutex = NULL;
+ local.hardlinks = NULL;
+ DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local);
+ bool ok = scanner != NULL;
+ if (scanner != NULL) {
+ Chunk* chunk;
+ while (ok && (chunk = directory_scanner_next(scanner)) != NULL) {
+ out->total += (unsigned long long)chunk->element_count;
+ for (int i = 0; i < chunk->element_count && ok; i++) {
+ const File* f = chunk->items[i];
+ if (f != NULL && f->is_dir)
+ ok = progress_precount_add_dir(out, delete_display_path(config, file_wire_path(f)));
+ }
+ chunk_destroy(chunk);
+ }
+ if (ok && directory_scanner_failed(scanner))
+ ok = false;
+ directory_scanner_destroy(scanner);
+ }
+ prepared_scanner_destroy(&prepared);
+ if (!ok) {
+ progress_precount_dispose(out);
+ return false;
+ }
+ out->total += 1; /* the transfer root "." */
+ return true;
+}
+
+/* Reuse the --delete-during/--delete-delay keep-set pre-scan: its traversed
+ * directory list already holds every directory and `non_dir_count` the entries
+ * counted during that same pass, so progress costs no second walk. */
+static bool progress_precount_from_plan_dirs(const Config* config, const ArrayList* plan_dirs,
+ unsigned long long non_dir_count,
+ ProgressPrecount* out) {
+ out->dir_paths = array_list_create(free);
+ if (out->dir_paths == NULL)
+ return false;
+ out->total = non_dir_count + 1;
+ for (int i = 0; i < plan_dirs->size; i++) {
+ const char* path = (const char*)plan_dirs->items[i];
+ const char* rel = config->send_directory != NULL
+ ? utils_strip_transfer_root(path, config->send_directory)
+ : path;
+ if (!progress_precount_add_dir(out, rel)) {
+ progress_precount_dispose(out);
+ return false;
+ }
+ }
+ out->total += (unsigned long long)out->dir_paths->size;
+ return true;
+}
+
+/* Build the optional progress pre-count. A failed pre-count is non-fatal: the
+ * transfer proceeds and the progress denominator falls back to the transferred
+ * file count. */
+static void client_progress_prepare(const Config* config, const ArrayList* plan_dirs,
+ unsigned long long plan_non_dir_count) {
+ client_progress_cleanup();
+ g_progress_active = progress_requested(config);
+ if (!g_progress_active)
+ return;
+ bool ok = plan_dirs != NULL ? progress_precount_from_plan_dirs(
+ config, plan_dirs, plan_non_dir_count, &g_progress_precount)
+ : progress_precount_scan(config, &g_progress_precount);
+ if (!ok) {
+ g_progress_total = 0;
+ return;
+ }
+ g_progress_total = g_progress_precount.total;
+ if (g_progress_precount.dir_paths != NULL && g_progress_precount.dir_paths->size > 0 &&
+ path_index_build(&g_progress_dir_index,
+ (const char* const*)g_progress_precount.dir_paths->items,
+ (size_t)g_progress_precount.dir_paths->size))
+ g_progress_dir_index_valid = true;
+ if (str_hash_set_init(&g_progress_emitted, (size_t)(g_progress_precount.dir_paths != NULL
+ ? g_progress_precount.dir_paths->size + 1
+ : 1)))
+ g_progress_emitted_valid = true;
+ g_progress_emitted_keys = array_list_create(free);
+}
+
/* -R/--relative implied directories: rsync transmits the metadata of the
* parent directories implied by the source path (every prefix component above
* the source root) so the receiver applies their attributes to the created
@@ -469,72 +931,6 @@ static const char* delete_plan_walk_root(const Config* config, const ArrayList*
return marker;
}
-/* True when some --files-from entry is an ancestor-or-equal directory of
- * `rel` (an empty entry -- the whole tree "." -- counts as the root). */
-static bool file_list_ancestor_listed(const FileListSet* set, const char* rel) {
- if (!set)
- return true;
- for (int i = 0; i < set->count; i++) {
- const char* listed = set->entries[i];
- if (listed[0] == '\0')
- return true;
- size_t n = strlen(listed);
- if (strncmp(rel, listed, n) == 0 && (rel[n] == '/' || rel[n] == '\0'))
- return true;
- }
- return false;
-}
-
-/* --no-implied-dirs (meaningful only with -R + --files-from): a listed file
- * may only be placed when its parent directory (or one of its ancestors) is
- * itself an explicitly listed entry. rsync omits a file whose implied parent
- * directory is suppressed, and an explicitly listed file that cannot be placed
- * fails the transfer; FastSync fails the whole run up front with a clear error
- * (it has no per-entry skip channel). Without -R or --files-from the option
- * has no effect. */
-static bool no_implied_dirs_files_from_valid(const Config* config) {
- if (!config->no_implied_dirs || !config->relative)
- return true;
- const FileListSet* set = (const FileListSet*)config->files_from_set;
- if (!set)
- return true;
- for (int i = 0; i < set->count; i++) {
- const char* entry = set->entries[i];
- if (entry[0] == '\0')
- continue;
- char* full = path_cat(config->send_directory, entry);
- if (!full)
- return false;
- struct stat st;
- bool is_file = lstat(full, &st) == 0 && S_ISREG(st.st_mode);
- free(full);
- if (!is_file)
- continue;
- const char* slash = strrchr(entry, '/');
- if (!slash)
- continue; /* top-level file: its parent is the receive root */
- size_t parent_len = (size_t)(slash - entry);
- if (parent_len == 0)
- continue;
- char* parent = malloc(parent_len + 1);
- if (!parent)
- return false;
- memcpy(parent, entry, parent_len);
- parent[parent_len] = '\0';
- bool listed = file_list_ancestor_listed(set, parent);
- if (!listed) {
- log_message(LOG_LEVEL_ERROR,
- "--no-implied-dirs: cannot place file '%s': parent directory '%s' is not "
- "explicitly listed (list the directory or drop --no-implied-dirs)",
- entry, parent);
- }
- free(parent);
- if (!listed)
- return false;
- }
- return true;
-}
-
/* The destination-relative mirror path for a missing --files-from entry: where
a PRESENT entry with the same name would have been written. With -R that is
the entry's bare relative path (the bare wire path the receiver uses);
@@ -648,54 +1044,7 @@ static bool files_from_list_check(const Config* config, ArrayList* missing_dest,
"--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out,
*skipped_out == 1 ? "y" : "ies");
}
- return no_implied_dirs_files_from_valid(config);
-}
-
-/* Basis directories are honored by the receiver's per-file incremental check,
- which (like every whole-file payload path in FastSync) is bounded by
- MAX_RECEIVE_WHOLE_FILE_SIZE. rsync would apply basis dirs to files of any
- size; FastSync cannot, so when basis dirs are requested this preflight scan
- refuses the run up front with a clear diagnostic instead of letting the
- receiver abort the whole transfer mid-stream with no client explanation.
- Returns true when the tree can be transferred. */
-static bool basis_oversize_preflight(const Config* config) {
- PreparedScanner prepared;
- if (!prepare_scanner(config, 0, &prepared))
- return false;
- DirectoryScanner* scanner =
- directory_scanner_create_with_options(config->send_directory, &prepared.options);
- if (!scanner) {
- prepared_scanner_destroy(&prepared);
- return false;
- }
- bool ok = true;
- Chunk* chunk;
- while ((chunk = directory_scanner_next(scanner)) != NULL) {
- for (int i = 0; i < chunk->element_count; i++) {
- File* f = chunk->items[i];
- if (f == NULL || f->is_dir || f->data == NULL || f->data->size <= MAX_RECEIVE_WHOLE_FILE_SIZE)
- continue;
- char* escaped = output_escape(file_wire_path(f), config->eight_bit_output);
- log_message(LOG_LEVEL_ERROR,
- "%s is %llu bytes, larger than the %llu-byte whole-file transfer limit; "
- "--compare-dest/--copy-dest/--link-dest cannot sync files above this limit",
- escaped ? escaped : "", (unsigned long long)f->data->size,
- (unsigned long long)MAX_RECEIVE_WHOLE_FILE_SIZE);
- free(escaped);
- ok = false;
- break;
- }
- chunk_destroy(chunk);
- if (!ok)
- break;
- }
- if (directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner))
- ok = false;
- /* The scanner borrows prepared.options' base_filters/hardlinks pointers, so
- prepared must outlive the scanner. */
- directory_scanner_destroy(scanner);
- prepared_scanner_destroy(&prepared);
- return ok;
+ return true;
}
/* Read the daemon's MOTD frame and, unless --no-motd, display it on stdout.
@@ -870,6 +1219,13 @@ static void remove_transferred_sources(const Config* config, ArrayList* paths) {
log_message(LOG_LEVEL_WARNING, "Could not remove source file %s",
escaped_path ? escaped_path : "");
free(escaped_path);
+ } else if (info_flag_enabled(config, LOG_INFO_REMOVE) && !config->quiet) {
+ /* rsync's --info=remove line: the transfer-relative name. */
+ const char* rel = delete_display_path(config, source->path);
+ char* escaped = output_escape(rel, config->eight_bit_output);
+ printf("sender removed %s\n", escaped ? escaped : rel);
+ free(escaped);
+ fflush(stdout);
}
close(dirfd);
}
@@ -959,20 +1315,7 @@ static bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_
static const char* delete_display_path(const Config* config, const char* path) {
if (!config || !path || !config->send_directory)
return path;
- const char* root = config->send_directory;
- while (*root == '/')
- root++;
- const char* rel = path;
- while (*rel == '/')
- rel++;
- size_t root_len = strlen(root);
- while (root_len > 0 && root[root_len - 1] == '/')
- root_len--;
- if (root_len == 0)
- return rel;
- if (strncmp(rel, root, root_len) == 0 && (rel[root_len] == '/' || rel[root_len] == '\0'))
- return rel + root_len + (rel[root_len] == '/' ? 1 : 0);
- return rel;
+ return utils_strip_transfer_root(path, config->send_directory);
}
/* Send the final STATUS_FINISHED frame and await the receiver's verdict.
@@ -993,8 +1336,19 @@ static bool finalize_transfer(Client* client, const Config* config, ArrayList* r
return false;
if (status == STATUS_STATS) {
ReceiverStats scratch;
- if (!receive_stats_record(client->file_descriptor, stats_out ? stats_out : &scratch, NULL))
+ /* A real --info=del run carries the actually-removed paths in the stats
+ frame's path list; collect and print them in rsync's format. */
+ ArrayList* deleted = config->report_deletes ? array_list_create(free) : NULL;
+ if (config->report_deletes && !deleted)
return false;
+ if (!receive_stats_record(client->file_descriptor, stats_out ? stats_out : &scratch, deleted)) {
+ array_list_delete(deleted);
+ return false;
+ }
+ if (deleted) {
+ print_delete_reports(config, deleted);
+ array_list_delete(deleted);
+ }
if (!receive_status(client->file_descriptor, &status))
return false;
}
@@ -1434,16 +1788,29 @@ static bool send_delete_manifest_early(Client* client, ArrayList* manifest,
directory and *io_error_out reports it (the caller still performs the
deletion but reports the run as errored). */
static bool scan_paths_only(const Config* config, const ScannerOptions* options,
- ArrayList* manifest, DeletePlanSender* plans, bool* io_error_out) {
+ ArrayList* manifest, DeletePlanSender* plans, bool* io_error_out,
+ unsigned long long* non_dir_count_out) {
if (io_error_out)
*io_error_out = false;
- DirectoryScanner* scanner =
- directory_scanner_create_with_options(config->send_directory, options);
+ if (non_dir_count_out)
+ *non_dir_count_out = 0;
+ ScannerOptions local = *options;
+ /* The pre-scan is a paths-only pass with no client output; it must not emit
+ --info=nonreg lines (the data pass does that once). */
+ local.note_nonreg = false;
+ DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local);
if (!scanner)
return false;
bool ok = true;
Chunk* chunk;
while ((chunk = directory_scanner_next(scanner)) != NULL) {
+ if (non_dir_count_out) {
+ for (int i = 0; i < chunk->element_count; i++) {
+ const File* f = chunk->items[i];
+ if (f && !f->is_dir)
+ (*non_dir_count_out)++;
+ }
+ }
if (manifest && !add_chunk_to_manifest(manifest, chunk)) {
ok = false;
chunk_destroy(chunk);
@@ -1525,11 +1892,13 @@ static int incremental_check(Client* client, File* file, const Config* config,
return -1;
if (!send_n_data(client->file_descriptor, &mtime_nsec, sizeof(mtime_nsec)))
return -1;
- /* With alternate basis directories the receiver must be able to verify the
- * content of every candidate basis file, so the sender supplies its whole-file
- * digest (computed with the negotiated --checksum-choice algorithm and
- * --checksum-seed) for every file even when --checksum was not requested. */
- if (config->checksum || config_has_basis(config)) {
+ /* The whole-file digest (negotiated --checksum-choice algorithm and
+ * --checksum-seed) is only needed when it drives a decision: --checksum's
+ * per-file quick check, or a --verify-basis content equality. Under the
+ * default metadata quick-check the receiver never reads it, so the sender
+ * skips the full-file read/hash exactly as rsync does for a plain
+ * --link-dest run. */
+ if (config->checksum || config->verify_basis) {
uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
size_t digest_len = 0;
if (!file_checksum(file, (ChecksumAlgo)config->checksum_algo, config->checksum_seed, digest,
@@ -1540,6 +1909,15 @@ static int incremental_check(Client* client, File* file, const Config* config,
!send_n_data(client->file_descriptor, digest, wire_len))
return -1;
}
+ /* Basis directories: the receiver materializes a hit from the basis without a
+ * data frame, so it would otherwise only have the basis inode's metadata.
+ * Transmit the SOURCE metadata with the check (rsync's copy-then-fix) so a
+ * --copy-dest hit / --link-dest copy fallback applies the source's
+ * attributes. Symmetric with incremental_check_receive_request. */
+ if (config_has_basis(config) && config->use_metadata) {
+ if (!metadata_send(client->file_descriptor, file->metadata))
+ return -1;
+ }
Status s;
if (!receive_status(client->file_descriptor, &s))
return -1;
@@ -1779,12 +2157,6 @@ static int send_dry_run_remote(Config* config) {
}
if (missing_args)
array_list_delete(missing_args);
- /* Alternate basis dirs force the whole-file per-file check on the real
- receiver; refuse an oversize source up front exactly as send_files does so
- dry-run reports the same clear diagnostic instead of aborting mid-stream. */
- if (config_has_basis(config) && !basis_oversize_preflight(config))
- return 1;
-
/* A live session may follow, so arm graceful abort handling. */
client_set_abort_armed(true);
Client* client = connect_transfer_client(config);
@@ -1810,11 +2182,41 @@ static int send_dry_run_remote(Config* config) {
DirectoryScanner* scanner = NULL;
ArrayList* dry_manifest = NULL;
ArrayList* dry_dirs = NULL;
+ ArrayList* dry_excluded = NULL;
+ ArrayList* dry_size_skipped = NULL;
if (!config_send(client->file_descriptor, config))
goto dry_fail;
receive_daemon_motd(client, config);
if (!prepare_scanner(config, 0, &prepared))
goto dry_fail;
+ /* -n --delete: build the same keep-set manifest, protected prefixes, and
+ synchronized-directory scope a real run would send, so the receiver's
+ read-only extras walk enumerates exactly the deletions a real run makes. */
+ if (config->use_delete) {
+ dry_manifest = array_list_create(free);
+ dry_dirs = array_list_create(free);
+ dry_size_skipped = array_list_create(free);
+ if (!dry_manifest || !dry_dirs || !dry_size_skipped)
+ goto dry_fail;
+ if (!config->delete_excluded) {
+ dry_excluded = array_list_create(free);
+ if (!dry_excluded)
+ goto dry_fail;
+ prepared.options.excluded_paths = dry_excluded;
+ }
+ prepared.options.size_skipped_paths = dry_size_skipped;
+ /* A --files-from subset confines the extras walk to the directories the
+ scan synchronized; a full recursive transfer marks the root itself. */
+ if (config->files_from_set == NULL) {
+ char* root_marker = delete_scope_root_marker(config);
+ if (!root_marker || !array_list_add(dry_dirs, root_marker)) {
+ free(root_marker);
+ goto dry_fail;
+ }
+ } else {
+ prepared.options.synced_dirs = dry_dirs;
+ }
+ }
scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options);
if (!scanner)
goto dry_fail;
@@ -1822,26 +2224,6 @@ static int send_dry_run_remote(Config* config) {
int file_count = 0;
unsigned long long total_bytes = 0;
char size_buffer[32];
- /* -n --delete: build the same keep-set manifest a real run would send so the
- receiver can enumerate (read-only) the destination extras. Filter-excluded
- and size-pruned protections are not propagated here, so a filtered dry-run
- may over-report; the no-filter case is exact. */
- dry_manifest = config->use_delete ? array_list_create(free) : NULL;
- if (config->use_delete && !dry_manifest)
- goto dry_fail;
- /* Scope the receiver-side extras walk to the receive root (the "." sentinel),
- exactly as the recursive transfer path does. */
- if (config->use_delete) {
- dry_dirs = array_list_create(free);
- char* root_marker = dry_dirs ? str_dup(".") : NULL;
- if (!dry_dirs || !root_marker || !array_list_add(dry_dirs, root_marker)) {
- free(root_marker);
- if (dry_dirs)
- array_list_delete(dry_dirs);
- dry_dirs = NULL;
- goto dry_fail;
- }
- }
if (!config->quiet)
printf("Dry run: files to be transferred\n");
Chunk* chunk;
@@ -1916,8 +2298,8 @@ static int send_dry_run_remote(Config* config) {
the terminal FINISHED. */
bool early_delete = config->use_delete && config_delete_timing_early(config);
if (dry_manifest) {
- if (send_delete_manifest(client->file_descriptor, dry_manifest, NULL, NULL, NULL, dry_dirs) !=
- 0)
+ if (send_delete_manifest(client->file_descriptor, dry_manifest, dry_excluded, dry_size_skipped,
+ NULL, dry_dirs) != 0)
goto dry_fail;
if (early_delete) {
Status ack;
@@ -1942,37 +2324,7 @@ static int send_dry_run_remote(Config* config) {
array_list_delete(would_delete);
goto dry_fail;
}
- /* rsync prints `*deleting PATH` when itemizing (or `deleting PATH` with
- --out-format / -v); the plain-total output used here has no delete
- counterpart, so only the itemize/out-format cases are rendered. */
- if (!config->quiet && (config->itemize_changes || config->out_format != NULL)) {
- for (int i = 0; i < would_delete->size; i++) {
- const char* raw = (const char*)would_delete->items[i];
- const char* path = delete_display_path(config, raw);
- if (config->out_format != NULL) {
- ChangeEvent event;
- memset(&event, 0, sizeof(event));
- event.decision = CHANGE_SENT;
- event.deleted = true;
- event.name = path;
- event.path = path;
- char* line = change_render_format(config->out_format, config, &event);
- if (line) {
- /* Escape the whole rendered line, exactly like change_emit() does
- for a real transfer, so a control byte in the peer-supplied path
- cannot forge output. */
- char* escaped = output_escape(line, config->eight_bit_output);
- printf("%s\n", escaped ? escaped : line);
- free(escaped);
- free(line);
- }
- } else {
- char* escaped = output_escape(path, config->eight_bit_output);
- printf("*deleting %s\n", escaped ? escaped : path);
- free(escaped);
- }
- }
- }
+ print_delete_reports(config, would_delete);
array_list_delete(would_delete);
if (!receive_status(client->file_descriptor, &status))
goto dry_fail;
@@ -1986,7 +2338,16 @@ static int send_dry_run_remote(Config* config) {
else
printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB);
}
- report_transfer_stats(config, file_count, total_bytes, dry_start, &dry_stats);
+ {
+ TransferStats dry_transfer;
+ memset(&dry_transfer, 0, sizeof(dry_transfer));
+ dry_transfer.flist_reg = (unsigned long long)file_count;
+ dry_transfer.total_file_size = total_bytes;
+ dry_transfer.transferred_regular = (unsigned long long)file_count;
+ dry_transfer.transferred_file_size = total_bytes;
+ dry_transfer.literal_data = total_bytes;
+ report_transfer_stats(config, &dry_transfer, dry_start, &dry_stats);
+ }
ret = io_error ? 1 : 0;
dry_fail:
@@ -1994,6 +2355,10 @@ dry_fail:
array_list_delete(dry_manifest);
if (dry_dirs)
array_list_delete(dry_dirs);
+ if (dry_excluded)
+ array_list_delete(dry_excluded);
+ if (dry_size_skipped)
+ array_list_delete(dry_size_skipped);
if (scanner)
directory_scanner_destroy(scanner);
prepared_scanner_destroy(&prepared);
@@ -2236,7 +2601,7 @@ static bool source_is_regular_file(const File* file) {
}
static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
- ArrayList* remove_sources) {
+ ArrayList* remove_sources, TransferStats* stats) {
if (config->use_chunk_serialization) {
if (remove_sources) {
for (int i = 0; i < chunk->element_count; i++) {
@@ -2263,10 +2628,13 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
for (int i = 0; i < chunk->element_count; i++) {
if (chunk->items[i] == NULL)
continue;
+ transfer_stats_note_entry(stats, chunk->items[i]);
if (chunk->items[i]->is_dir)
change_emit_dir_sent(config, chunk->items[i]);
else
change_emit_file_sent(config, chunk->items[i]);
+ if (!chunk->items[i]->is_dir)
+ transfer_stats_note_transferred(stats, chunk->items[i]);
}
return 0;
}
@@ -2275,6 +2643,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
File* f = chunk->items[i];
if (f == NULL)
continue;
+ transfer_stats_note_entry(stats, f);
if (f->is_dir) {
/* Explicit directory entry (--dirs): a MKDIR frame carrying the
destination path (and metadata when negotiated). Directories have no
@@ -2282,6 +2651,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
if (!send_directory_entry(client, f, config))
return -1;
change_emit_dir_sent(config, f);
+ client_progress_name(config, f);
continue;
}
/* --hard-links/-H sibling: a later member of a hard-link group that has no
@@ -2295,6 +2665,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
!send_wire_str(client->file_descriptor, f->hardlink_target))
return -1;
change_emit_file_sent(config, f);
+ client_progress_name(config, f);
continue;
}
/* Symlink entry (-l / -k keep-as-symlink): only the target rides the wire. */
@@ -2302,6 +2673,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
if (!send_symlink_entry(client, f, config))
return -1;
change_emit_file_sent(config, f);
+ client_progress_name(config, f);
continue;
}
/* --devices/--specials: a device/special node is recreated on the receiver,
@@ -2310,6 +2682,7 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
if (!file_send_special(f, client->file_descriptor, config->use_metadata))
return -1;
change_emit_file_sent(config, f);
+ client_progress_name(config, f);
continue;
}
bool stream = f->data->data == NULL && f->data->size > 0;
@@ -2322,12 +2695,15 @@ static int send_chunk_with_removal(Client* client, Chunk* chunk, Config* config,
int rc = send_single_file(client, f, config, config->use_incremental, use_sendfile);
if (rc == 1) {
source_file_destroy(source);
+ change_emit_file_uptodate(config, f);
+ client_progress_uptodate(config, f);
continue;
}
if (rc < 0) {
source_file_destroy(source);
return -1;
}
+ transfer_stats_note_transferred(stats, f);
change_emit_file_sent_bytes(config, f, protocol_bytes_written() - bytes_before,
protocol_bytes_read() - read_before);
client_progress_file(config, f);
@@ -2436,7 +2812,7 @@ static int send_chunks_multithreaded(void* pipeline_context) {
return thrd_error;
}
if (send_chunk_with_removal(client, current_chunk, context->config,
- context->remove_source_files) != 0) {
+ context->remove_source_files, &context->stats) != 0) {
log_message(LOG_LEVEL_ERROR, "unexpected error while sending chunk");
chunk_destroy(current_chunk);
pipeline_cancel(context);
@@ -2482,7 +2858,8 @@ static int send_chunks_multithreaded(void* pipeline_context) {
"unscanned source mirrors are not deleted");
else
log_message(LOG_LEVEL_WARNING, "transfer stopped early (stop deadline)");
- } else if (context->config->use_delete && !context->early_delete && !context->delete_plans) {
+ } else if (context->config->use_delete && !context->early_delete && !context->delete_plans &&
+ !context->delete_suppressed) {
/* Empty keep-set + scan I/O error must not delete the whole destination
(the source may not be genuinely empty -- see send_files). */
bool empty_io;
@@ -2495,12 +2872,20 @@ static int send_chunks_multithreaded(void* pipeline_context) {
"with an empty keep-set (--delete)");
goto send_fail;
}
- if (send_delete_manifest(client->file_descriptor, context->manifest, context->excluded_paths,
- context->size_skipped_paths, context->missing_args,
- context->synced_dirs) != 0)
+ /* rsync default: an I/O error suppresses deletion unless --ignore-errors.
+ The keep-set manifest is not sent, so the receiver removes nothing. */
+ mtx_lock(&context->mutex_scanner);
+ bool scan_io_now = context->scan_had_io_error;
+ mtx_unlock(&context->mutex_scanner);
+ if (!ignore_errors_allows_delete(context->config, scan_io_now)) {
+ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion");
+ } else if (send_delete_manifest(client->file_descriptor, context->manifest,
+ context->excluded_paths, context->size_skipped_paths,
+ context->missing_args, context->synced_dirs) != 0) {
goto send_fail;
+ }
} else if (context->config->delete_missing_args && !context->early_delete &&
- !context->delete_plans) {
+ !context->delete_suppressed && !context->delete_plans) {
/* --delete-missing-args without --delete: no keep-set is built, but the
exact-delete paths still ride the same manifest frame (commit once the
transfer succeeded). */
@@ -2533,13 +2918,12 @@ static int send_chunks_multithreaded(void* pipeline_context) {
"server reported a deletion failure (--delete); see the server log for the reason");
if (ok)
remove_transferred_sources(context->config, context->remove_source_files);
- mtx_lock(&context->mutex_progress);
- int total_files = context->total_files;
- unsigned long long total_bytes = context->total_bytes;
- mtx_unlock(&context->mutex_progress);
- report_transfer_stats(context->config, total_files, total_bytes, start, &recv_stats);
- log_info_message(LOG_INFO_STATS, "Transfer summary: %d files, %.1f MB", total_files,
- (double)total_bytes / (double)BYTES_PER_MIB);
+ if (context->dir_entries)
+ context->stats.flist_dir += (unsigned long long)context->dir_entries->size;
+ report_transfer_stats(context->config, &context->stats, start, &recv_stats);
+ log_info_message(LOG_INFO_STATS, "Transfer summary: %llu files, %.1f MB",
+ context->stats.transferred_regular,
+ (double)context->stats.transferred_file_size / (double)BYTES_PER_MIB);
disconnect_transfer_client(client);
mark_sender_done(context);
protocol_session_unbind();
@@ -2628,7 +3012,8 @@ static int scan_directory_multithreaded(void* pipeline_context) {
failed = use_dscanner ? directory_scanner_failed(dscanner) : parallel_scanner_failed(scanner);
break;
}
- if (context->config->use_delete && !context->early_delete && !context->delete_plans) {
+ if (context->config->use_delete && !context->early_delete && !context->delete_plans &&
+ !context->delete_suppressed) {
mtx_lock(&context->mutex_scanner);
bool manifest_ok = add_chunk_to_manifest(context->manifest, current_chunk);
mtx_unlock(&context->mutex_scanner);
@@ -2813,11 +3198,6 @@ int send_files(Config* config) {
array_list_delete(missing_args);
return 1;
}
- if (config_has_basis(config) && !basis_oversize_preflight(config)) {
- if (missing_args)
- array_list_delete(missing_args);
- return 1;
- }
/* From here on a server session may be live, so Ctrl-C/SIGTERM should set the
abort flag (and be forwarded as STATUS_ABORT) instead of terminating. */
@@ -2858,6 +3238,7 @@ int send_files(Config* config) {
bool delete_per_dir = config->use_delete && config_delete_timing_per_dir(config) && !config->dirs;
bool send_failed = false;
bool had_scan_io = false;
+ unsigned long long per_dir_non_dir_count = 0;
PreparedScanner prepared;
memset(&prepared, 0, sizeof(prepared));
if (!config_send(client->file_descriptor, config))
@@ -2906,7 +3287,8 @@ int send_files(Config* config) {
prepared.options.synced_dirs = synced_dirs;
}
}
- /* The late-timing modes (plain --delete / --delete-after) build the manifest
+ /* The late-timing modes (--delete-after/--delete-commit and a plain --delete
+ that fell back from per-dir mode because of -d/--dirs) build the manifest
while streaming and send it after the last data frame. --delete-before
sends a whole-tree keep-set up front; --delete-during/--delete-delay build a
per-directory plan set up front (paths only) and stream the plans alongside
@@ -2919,8 +3301,9 @@ int send_files(Config* config) {
if (!early_manifest)
goto send_fail;
bool prescan_ok =
- scan_paths_only(config, &prepared.options, early_manifest, NULL, &had_scan_io);
+ scan_paths_only(config, &prepared.options, early_manifest, NULL, &had_scan_io, NULL);
bool early_ok = false;
+ bool skip_delete = false;
if (prescan_ok) {
/* A scan that hit an I/O error and produced NO keep entries is ambiguous
(the source may not be genuinely empty -- part of it was unreadable),
@@ -2932,6 +3315,11 @@ int send_files(Config* config) {
"source scan hit an I/O error before finding any file; refusing to delete "
"with an empty keep-set (--delete)");
prescan_ok = false;
+ } else if (!ignore_errors_allows_delete(config, had_scan_io)) {
+ /* rsync default: an I/O error suppresses deletion unless
+ --ignore-errors. Skip the manifest; the transfer still proceeds. */
+ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion");
+ skip_delete = true;
} else {
early_ok = send_delete_manifest_early(client, early_manifest, excluded, size_skipped,
missing_args, synced_dirs);
@@ -2943,7 +3331,7 @@ int send_files(Config* config) {
prepared.options.excluded_paths = NULL;
prepared.options.size_skipped_paths = NULL;
prepared.options.synced_dirs = NULL;
- if (!prescan_ok || !early_ok)
+ if (!prescan_ok || (!early_ok && !skip_delete))
goto send_fail;
} else if (delete_per_dir) {
/* --delete-during/--delete-delay: build one plan per source directory from a
@@ -2955,8 +3343,10 @@ int send_files(Config* config) {
if (!plan_sender || !plan_dirs)
goto send_fail;
prepared.options.plan_dirs = plan_dirs;
- bool prescan_ok = scan_paths_only(config, &prepared.options, NULL, plan_sender, &had_scan_io);
+ bool prescan_ok = scan_paths_only(config, &prepared.options, NULL, plan_sender, &had_scan_io,
+ &per_dir_non_dir_count);
bool plans_ok = false;
+ bool skip_delete = false;
if (prescan_ok) {
const char* walk_root = delete_plan_walk_root(config, synced_dirs);
const ArrayList* scope =
@@ -2968,6 +3358,15 @@ int send_files(Config* config) {
"source scan hit an I/O error before finding any file; refusing to delete "
"with an empty keep-set (--delete)");
prescan_ok = false;
+ } else if (!ignore_errors_allows_delete(config, had_scan_io)) {
+ /* rsync default: an I/O error suppresses deletion unless
+ --ignore-errors. Drop the plans; the transfer still proceeds. */
+ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion");
+ delete_plan_sender_destroy(plan_sender);
+ plan_sender = NULL;
+ array_list_delete(plan_dirs);
+ plan_dirs = NULL;
+ skip_delete = true;
} else {
plans_ok = delete_plan_send_root(client->file_descriptor, plan_sender) == 0;
}
@@ -2976,13 +3375,18 @@ int send_files(Config* config) {
prepared.options.size_skipped_paths = NULL;
prepared.options.synced_dirs = NULL;
prepared.options.plan_dirs = NULL;
- if (!prescan_ok || !plans_ok)
+ if (!prescan_ok || (!plans_ok && !skip_delete))
goto send_fail;
} else if (config->use_delete) {
manifest = array_list_create(free);
if (!manifest)
goto send_fail;
}
+ /* --progress/--info=progress: pre-count the file list for rsync's to-chk
+ denominator. When a --delete-during/--delete-delay pre-scan already ran,
+ reuse its traversed directory list instead of walking the tree again. */
+ if (progress_requested(config))
+ client_progress_prepare(config, plan_dirs, per_dir_non_dir_count);
/* Phase 6: compute the client-only stop deadline once at transfer start. The
early-delete pre-scan above deliberately ignores it so the keep-set (and
its committed deletion) is always complete and correct. */
@@ -3003,8 +3407,8 @@ int send_files(Config* config) {
goto send_fail;
Chunk* current_chunk;
- unsigned long long total_bytes = 0;
- int total_files = 0;
+ TransferStats transfer_stats;
+ memset(&transfer_stats, 0, sizeof(transfer_stats));
time_t start = time(NULL);
client_progress_begin(config);
/* True when the stop deadline cut the scan short so the keep-set manifest is
@@ -3030,11 +3434,6 @@ int send_files(Config* config) {
scan_stopped_early = true;
break;
}
- unsigned long long chunk_bytes = 0;
- for (int i = 0; i < current_chunk->element_count; i++) {
- chunk_bytes += current_chunk->items[i]->data->size;
- total_files++;
- }
if (manifest && !add_chunk_to_manifest(manifest, current_chunk)) {
chunk_destroy(current_chunk);
goto send_fail;
@@ -3061,13 +3460,13 @@ int send_files(Config* config) {
send_failed = true;
break;
}
- if (send_chunk_with_removal(client, current_chunk, config, remove_sources) != 0) {
+ if (send_chunk_with_removal(client, current_chunk, config, remove_sources, &transfer_stats) !=
+ 0) {
log_message(LOG_LEVEL_ERROR, "Failed to send chunk");
chunk_destroy(current_chunk);
send_failed = true;
break;
}
- total_bytes += chunk_bytes;
chunk_destroy(current_chunk);
}
if (send_failed) {
@@ -3111,7 +3510,18 @@ int send_files(Config* config) {
"an empty keep-set (--delete)");
goto send_fail;
}
- if ((manifest || config->delete_missing_args) && !delete_early && !delete_per_dir) {
+ /* rsync default: a scan I/O error suppresses deletion unless
+ --ignore-errors, even in the late (commit) modes. Drop the keep-set so
+ the receiver removes nothing; the readable tree still transferred. */
+ bool late_delete =
+ (manifest || config->delete_missing_args) && !delete_early && !delete_per_dir;
+ if (late_delete && !ignore_errors_allows_delete(config, had_scan_io)) {
+ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion");
+ if (manifest) {
+ array_list_delete(manifest);
+ manifest = NULL;
+ }
+ } else if (late_delete) {
/* Late (commit) ordering: all file data is out; transmit the manifest so
the receiver commits the extras walk (--delete) and/or the
--delete-missing-args exact-path deletions only after the transfer
@@ -3152,9 +3562,16 @@ int send_files(Config* config) {
"server reported a deletion failure (--delete); see the server log for the reason");
if (ok)
remove_transferred_sources(config, remove_sources);
- report_transfer_stats(config, total_files, total_bytes, start, &recv_stats);
- log_info_message(LOG_INFO_STATS, "Transfer summary: %d files, %.1f MB", total_files,
- (double)total_bytes / (double)BYTES_PER_MIB);
+ /* A recursive -a scan has no directory entries in its chunks; account them
+ from the scanner's captured directory list (present whenever a directory
+ attribute is preserved, e.g. -a/-t/-p). The -d generator counts its
+ explicit directory entries inline instead. */
+ if (dir_entries)
+ transfer_stats.flist_dir += (unsigned long long)dir_entries->size;
+ report_transfer_stats(config, &transfer_stats, start, &recv_stats);
+ log_info_message(LOG_INFO_STATS, "Transfer summary: %llu files, %.1f MB",
+ transfer_stats.transferred_regular,
+ (double)transfer_stats.transferred_file_size / (double)BYTES_PER_MIB);
/* A skipped source entry (--ignore-errors past an unreadable directory, or a
dereferenced symlink with no referent) makes rsync report a partial
transfer (exit 23) even though the rest of the run succeeded. A
@@ -3191,6 +3608,7 @@ send_fail:
if (scanner)
directory_scanner_destroy(scanner);
prepared_scanner_destroy(&prepared);
+ client_progress_cleanup();
disconnect_transfer_client(client);
protocol_session_unbind();
client_set_abort_armed(false);
@@ -3218,11 +3636,6 @@ int send_files_multithreaded(Config** config_ptr) {
array_list_delete(missing_args);
return 1;
}
- if (config_has_basis(config) && !basis_oversize_preflight(config)) {
- if (missing_args)
- array_list_delete(missing_args);
- return 1;
- }
/* Armed only once a session may go live (see send_files). */
client_set_abort_armed(true);
@@ -3274,6 +3687,7 @@ int send_files_multithreaded(Config** config_ptr) {
stop_condition_make(config->stop_after_mins > 0, config->stop_after_mins, config->stop_at_set,
config->stop_at, now_mono);
bool collect_excluded = config->use_delete && !config->delete_excluded;
+ unsigned long long pre_scan_non_dir = 0;
if (config->use_delete) {
if (collect_excluded) {
context->excluded_paths = array_list_create(free);
@@ -3332,8 +3746,9 @@ int send_files_multithreaded(Config** config_ptr) {
prepared_ok = prepared_ok && context->manifest != NULL;
}
bool prebuilt =
- prepared_ok && scan_paths_only(config, &prepared.options, context->manifest,
- context->delete_plans, &context->scan_had_io_error);
+ prepared_ok &&
+ scan_paths_only(config, &prepared.options, context->manifest, context->delete_plans,
+ &context->scan_had_io_error, &pre_scan_non_dir);
prepared_scanner_destroy(&prepared);
if (per_dir && prebuilt) {
const char* walk_root = delete_plan_walk_root(config, context->synced_dirs);
@@ -3358,8 +3773,28 @@ int send_files_multithreaded(Config** config_ptr) {
pipeline_context_sender_destroy(context);
return 1;
}
- if (!per_dir)
+ if (context->scan_had_io_error && !ignore_errors_allows_delete(config, true)) {
+ /* rsync default: an I/O error suppresses deletion unless
+ --ignore-errors. Drop the prebuilt keep-set so nothing is sent; the
+ data pass still transfers the readable tree and exits 23. */
+ log_message(LOG_LEVEL_WARNING, "IO error encountered -- skipping file deletion");
+ if (context->manifest) {
+ array_list_delete(context->manifest);
+ context->manifest = NULL;
+ }
+ if (context->delete_plans) {
+ delete_plan_sender_destroy(context->delete_plans);
+ context->delete_plans = NULL;
+ }
+ if (context->plan_dirs) {
+ array_list_delete(context->plan_dirs);
+ context->plan_dirs = NULL;
+ }
+ /* A later --delete pass must not try to rebuild/send a keep-set. */
+ context->delete_suppressed = true;
+ } else if (!per_dir) {
context->early_delete = true;
+ }
} else {
context->manifest = array_list_create(free);
if (!context->manifest) {
@@ -3370,11 +3805,17 @@ int send_files_multithreaded(Config** config_ptr) {
}
if (config->remove_source_files)
context->remove_source_files = array_list_create(source_file_destroy);
- if ((config->use_delete && !context->manifest && !context->delete_plans) ||
+ if ((config->use_delete && !context->manifest && !context->delete_plans &&
+ !context->delete_suppressed) ||
(config->remove_source_files && !context->remove_source_files)) {
pipeline_context_sender_destroy(context);
return 1;
}
+ /* --progress/--info=progress: pre-count the file list for rsync's to-chk
+ denominator, reusing a --delete-during/--delete-delay pre-scan when one
+ already ran. */
+ if (progress_requested(config))
+ client_progress_prepare(config, context->plan_dirs, pre_scan_non_dir);
thrd_t scanner, loader, sender;
bool scanner_created = false;
@@ -3400,6 +3841,7 @@ int send_files_multithreaded(Config** config_ptr) {
if (scanner_created)
thrd_join(scanner, NULL);
pipeline_context_sender_destroy(context);
+ client_progress_cleanup();
return 1;
}
@@ -3419,6 +3861,7 @@ int send_files_multithreaded(Config** config_ptr) {
transfer (exit 23). A --max-delete-capped commit is a successful transfer
that rsync reports with exit code 25. */
pipeline_context_sender_destroy(context);
+ client_progress_cleanup();
client_set_abort_armed(false);
if (!sender_ok)
return 1;
diff --git a/src/client/client_send.h b/src/client/client_send.h
index 2605899..f8f3706 100644
--- a/src/client/client_send.h
+++ b/src/client/client_send.h
@@ -23,6 +23,12 @@ void client_set_abort_armed(bool armed);
* config_delete() once the call returns). */
int send_files(Config* config);
int send_files_multithreaded(Config** config);
+/* rsync's --ignore-errors deletion gate: with no I/O error during the scan the
+ * deletion phase always proceeds; with one it is suppressed unless
+ * `--ignore-errors` was given. Exposed so the decision can be unit-tested
+ * without a privileged (mode-000) source directory. See client_send.c. */
+bool ignore_errors_allows_delete(const Config* config, bool had_io_error);
+
/* Phase 6 residual-batch (client-only). See client_send.c. */
int write_batch_from_source(const Config* config, const char* batch_path);
int apply_batch_to_dest(const Config* config, const char* batch_path, const char* dest_root);
diff --git a/src/client/scanner.c b/src/client/scanner.c
index 61e7dd9..6d83c8e 100644
--- a/src/client/scanner.c
+++ b/src/client/scanner.c
@@ -432,6 +432,19 @@ static void scanner_record_protected(DirectoryScanner* scanner, const char* fs_p
scanner->failed = true;
}
+/* rsync's `--info=nonreg` line for a non-regular entry that is not being
+ * preserved: `skipping non-regular file "NAME"`. The name is the path relative
+ * to the transfer root, so it matches rsync's displayed name. */
+static void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path) {
+ if (!options || !options->note_nonreg || !fs_path)
+ return;
+ const char* rel = utils_strip_transfer_root(fs_path, options->send_directory);
+ char* escaped = output_escape(rel, options->eight_bit_output);
+ printf("skipping non-regular file \"%s\"\n", escaped ? escaped : rel);
+ free(escaped);
+ fflush(stdout);
+}
+
/* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */
static void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) {
scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths);
@@ -830,7 +843,8 @@ static bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const
const char* fs_path, bool relative_mode,
const char* relative_prefix, bool preserve_atimes,
bool preserve_crtimes, bool preserve_xattrs,
- bool preserve_acls) {
+ bool preserve_acls, bool no_implied_dirs,
+ const FileListSet* file_list) {
if (!dir_entries || !root_path || !fs_path)
return true;
struct stat st;
@@ -839,6 +853,13 @@ static bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const
char* rel = scanner_path_relative(root_path, fs_path);
if (!rel)
return true;
+ /* --no-implied-dirs: an implied parent directory (not listed, and not under
+ a listed directory) keeps the destination's own/default attributes, so its
+ source metadata is not transmitted. */
+ if (no_implied_dirs && file_list && !file_list_dir_in_scope(file_list, rel)) {
+ free(rel);
+ return true;
+ }
if (relative_mode && rel[0] == '\0') {
/* -R + --files-from: the transfer root itself has no bare relative wire
path (matches the -R scan, which never emits the root). */
@@ -902,6 +923,43 @@ static bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const
return true;
}
+/* Recursive scan: emit a payload-less directory entry for the directory that
+ * just finished scanning. rsync creates every source directory at the
+ * destination; FastSync otherwise creates one only implicitly through a
+ * transferred child, so a directory emptied on the transfer side (physically
+ * empty, or all of its entries filtered out) would never appear. The transfer
+ * root is skipped (it maps to the receive root, which already exists), as are
+ * --files-from (only listed items and their implied parents transfer),
+ * --list-only (directory lines are emitted by the caller) and
+ * -m/--prune-empty-dirs. Returns false on allocation failure. */
+static bool scanner_emit_empty_dir(DirectoryScanner* scanner, ArrayList* chunk_data) {
+ if (!scanner->current_path || !scanner->current_rel || scanner->current_rel[0] == '\0')
+ return true;
+ struct stat st;
+ if (lstat(scanner->current_path, &st) != 0 || !S_ISDIR(st.st_mode))
+ return true;
+ File* dir = scanner_build_dir_file(scanner->current_path, &st, &scanner->options);
+ if (!dir)
+ return false;
+ if (scanner->relative_mode) {
+ dir->send_path = str_dup(scanner->current_rel);
+ } else if (scanner->options.relative_prefix) {
+ dir->send_path =
+ scanner_prefix_send_path(scanner->options.relative_prefix, scanner->current_rel);
+ }
+ if ((scanner->relative_mode || scanner->options.relative_prefix) && !dir->send_path) {
+ file_destroy(dir);
+ return false;
+ }
+ if (scanner->options.preserve_xattrs || scanner->options.preserve_acls)
+ dir->xattrs = xattr_capture_path(scanner->current_path, scanner->options.preserve_acls);
+ if (!array_list_add(chunk_data, dir)) {
+ file_destroy(dir);
+ return false;
+ }
+ return true;
+}
+
/* Open the next queued directory and set up its filter context. Returns 1 when
a directory is open, 0 when the queue is exhausted, and -1 on a fatal error.
A directory that cannot be opened is an I/O error: it is recorded on the
@@ -925,6 +983,7 @@ static int open_next_directory(DirectoryScanner* scanner) {
* the directory that enqueued them. */
const FilterNode* inherited = scanner->at_seed_dir ? scanner->seed_node : de->context;
scanner->at_seed_dir = false;
+ scanner->current_dir_produced = false;
free(de);
free(scanner->current_rel);
@@ -954,11 +1013,18 @@ static int open_next_directory(DirectoryScanner* scanner) {
scanner->current_rel = NULL;
free(scanner->current_path);
scanner->current_path = NULL;
- if (!scanner->options.ignore_io_errors || is_root_seed) {
+ if (is_root_seed) {
+ /* The transfer ROOT being unreadable is always fatal: an empty keep-set
+ would delete the whole destination. Mark the scan as errored so the
+ client can report the partial-transfer exit code (rsync's 23). */
+ scanner->root_io_error = true;
scanner->failed = true;
return -1;
}
- /* --ignore-errors: record the I/O error and keep scanning the rest. */
+ /* A subdirectory that cannot be opened is always skipped (rsync continues
+ with a partial transfer), whether or not --ignore-errors is set. The
+ error is recorded so the client exits 23; --ignore-errors only changes
+ what the deletion phase does with the recorded error. */
continue;
}
if (open_directory_filter_context(scanner, inherited) != 0) {
@@ -985,7 +1051,8 @@ static int open_next_directory(DirectoryScanner* scanner) {
scanner->options.dir_entries, scanner->options.dir_entries_mutex, scanner->root_path,
scanner->current_path, scanner->relative_mode, scanner->options.relative_prefix,
scanner->options.preserve_atimes, scanner->options.preserve_crtimes,
- scanner->options.preserve_xattrs, scanner->options.preserve_acls)) {
+ scanner->options.preserve_xattrs, scanner->options.preserve_acls,
+ scanner->options.no_implied_dirs, scanner->options.file_list)) {
closedir(scanner->current_dir);
scanner->current_dir = NULL;
free(scanner->current_path);
@@ -1350,10 +1417,22 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) {
const struct dirent* entry = readdir(scanner->current_dir);
if (entry == NULL) {
+ /* The directory is exhausted: if nothing was transferred or descended
+ from it, recreate it at the destination as an explicit entry. */
+ if (scanner->options.emit_empty_dirs && !scanner->current_dir_produced &&
+ !scanner->options.prune_empty_dirs && !scanner->options.list_dirs &&
+ scanner->options.file_list == NULL) {
+ if (!scanner_emit_empty_dir(scanner, chunk_data))
+ scanner->failed = true;
+ }
closedir(scanner->current_dir);
scanner->current_dir = NULL;
free(scanner->current_path);
scanner->current_path = NULL;
+ if (scanner->failed) {
+ array_list_delete(chunk_data);
+ return NULL;
+ }
continue;
}
@@ -1488,6 +1567,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) {
scanner->failed = true;
break;
}
+ scanner->current_dir_produced = true;
free(cur_path);
continue;
}
@@ -1502,6 +1582,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) {
break;
}
}
+ scanner->current_dir_produced = true;
int next_depth = scanner->current_depth + 1;
if (scanner->options.max_depth <= 0 || next_depth < scanner->options.max_depth) {
DirEntry* de = dir_entry_create(cur_path, next_depth, scanner->current_node);
@@ -1554,6 +1635,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) {
scanner->options.preserve_specials,
scanner->options.copy_devices, file, &stats);
if (special == SCANNER_SPECIAL_SKIP) {
+ scanner_note_nonreg(&scanner->options, file->path);
free(rel_copy);
file_destroy(file);
continue;
@@ -1577,6 +1659,7 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner) {
scanner->failed = true;
break;
}
+ scanner->current_dir_produced = true;
chunk_data_size += file->data->size;
if (chunk_data_size > scanner->options.chunk_size) {
free(rel_copy);
@@ -1604,7 +1687,7 @@ bool directory_scanner_failed(const DirectoryScanner* scanner) {
}
bool directory_scanner_had_io_error(const DirectoryScanner* scanner) {
- return scanner != NULL && scanner->io_error;
+ return scanner != NULL && (scanner->io_error || scanner->root_io_error);
}
typedef struct {
@@ -1974,6 +2057,7 @@ static void scan_root_entry(const ScannerOptions* options, const FilterNode* roo
ScannerSpecial special = scanner_prepare_special(
options->preserve_devices, options->preserve_specials, options->copy_devices, file, &st);
if (special == SCANNER_SPECIAL_SKIP) {
+ scanner_note_nonreg(ps->options, file->path);
free(rel);
file_destroy(file);
return;
@@ -2134,6 +2218,7 @@ ParallelScanner* parallel_scanner_create_with_options(const char* root_directory
return NULL;
}
ps->allocation_session = allocation_session;
+ ps->options = options;
ArrayList* root_files = array_list_create(file_destroy);
ArrayList* subdirs = array_list_create(free);
@@ -2202,11 +2287,11 @@ ParallelScanner* parallel_scanner_create_with_options(const char* root_directory
transfer root itself (it hands the root's immediate subdirectories to
workers), so capture the root's directory time here. */
if (options->capture_dir_times &&
- !scanner_capture_dir_time(options->dir_entries, options->dir_entries_mutex, root_directory,
- root_directory, options->relative && options->file_list != NULL,
- options->relative_prefix, options->preserve_atimes,
- options->preserve_crtimes, options->preserve_xattrs,
- options->preserve_acls)) {
+ !scanner_capture_dir_time(
+ options->dir_entries, options->dir_entries_mutex, root_directory, root_directory,
+ options->relative && options->file_list != NULL, options->relative_prefix,
+ options->preserve_atimes, options->preserve_crtimes, options->preserve_xattrs,
+ options->preserve_acls, options->no_implied_dirs, options->file_list)) {
array_list_delete(root_files);
array_list_delete(subdirs);
parallel_scanner_destroy(ps);
diff --git a/src/client/scanner.h b/src/client/scanner.h
index 3aeab8c..18f61a6 100644
--- a/src/client/scanner.h
+++ b/src/client/scanner.h
@@ -121,9 +121,18 @@ typedef struct {
* it) and to emit its plan after the data stream, when no file frame would
* otherwise trigger it. Guarded by `excluded_mutex`. */
ArrayList* plan_dirs;
- /* --ignore-errors: an unreadable directory during the scan is recorded as an
- * I/O error and skipped instead of aborting the scan. Client-only. */
+ /* --ignore-errors: an unreadable subdirectory no longer aborts the scan (it
+ * is always skipped so the rest of the tree transfers); this flag is kept so
+ * the client can distinguish the option state when deciding deletion policy.
+ * Client-only. */
bool ignore_io_errors;
+ /* --info=nonreg: print rsync's `skipping non-regular file "NAME"` line for a
+ * non-regular entry that is not being preserved. Client-only. */
+ bool note_nonreg;
+ /* Source root and 8-bit-output policy used to render a `--info=nonreg` name
+ * relative to the transfer root. Borrowed read-only. */
+ const char* send_directory;
+ bool eight_bit_output;
/* --ignore-missing-args (implied by --delete-missing-args): an explicitly
* --files-from-listed entry that does not exist under the source is skipped
* instead of failing (the --dirs generator is the only scanner path that
@@ -151,6 +160,17 @@ typedef struct {
bool capture_dir_times;
ArrayList* dir_entries;
mtx_t* dir_entries_mutex;
+ /* Recreate empty source directories on a recursive transfer: emit a
+ * payload-less directory entry for every traversed directory that produced
+ * no transferred/descended child. Off by default so low-level scanner users
+ * (unit helpers, --list-only) see only the historical file list; the real
+ * sender sets it in prepare_scanner. */
+ bool emit_empty_dirs;
+ /* --no-implied-dirs with -R + --files-from: a directory that is only an
+ * implied parent of a listed entry (not itself listed, nor below a listed
+ * directory) must not carry source metadata; it is created with default
+ * attributes at the destination, matching rsync. */
+ bool no_implied_dirs;
} ScannerOptions;
/* Internal per-scanner filter state. FilterNode chains represent the ordered
@@ -168,6 +188,11 @@ typedef struct {
int current_depth;
dev_t root_dev;
bool failed;
+ /* Recursive scan: whether the open directory yielded any transferred or
+ descended entry. When it did not, closing it emits a directory entry so
+ the empty source directory is recreated at the destination (rsync
+ parity). */
+ bool current_dir_produced;
/* Phase 2 (files-from / filter layer). */
char* root_path; /* transfer root (fs path) for rel computation */
char* current_rel; /* rel path of the open directory ("" == root) */
@@ -186,6 +211,10 @@ typedef struct {
--ignore-errors the scan continues past it and the caller decides what to
do; `failed` is reserved for fatal errors that always abort the scan. */
bool io_error;
+ /* The transfer ROOT could not be opened. It is always fatal, even under
+ --ignore-errors, but the client still maps it to rsync's partial-transfer
+ exit (23) rather than a generic failure. */
+ bool root_io_error;
} DirectoryScanner;
typedef struct {
@@ -205,7 +234,8 @@ typedef struct {
int completed;
Chunk* initial_chunk;
ProtocolSession* allocation_session;
- FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
+ FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
+ const ScannerOptions* options; /* borrowed scan options (--info=nonreg output) */
} ParallelScanner;
DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata,
diff --git a/src/client/usage.c b/src/client/usage.c
index c393766..c6f6e96 100644
--- a/src/client/usage.c
+++ b/src/client/usage.c
@@ -62,8 +62,7 @@ void print_usage(void) {
printf(" NOTE: the FastSync batch format is NOT interoperable with rsync's batch\n");
printf(" files (different container format); do not mix the two tools.\n");
printf(" --delete Delete files on receiver not in source\n");
- printf(" (default timing: delete only after the whole\n");
- printf(" transfer has succeeded)\n");
+ printf(" (default timing: delete-during, like rsync --del)\n");
printf(" --delete-before Delete extras before the transfer starts\n");
printf(" (implies --delete)\n");
printf(" --delete-during Delete a directory's extras as that directory is\n");
@@ -72,7 +71,9 @@ void print_usage(void) {
printf(" --delete-delay Record the extras during the scan but remove them\n");
printf(" only after a successful transfer (implies --delete)\n");
printf(" --delete-after Delete only after the whole transfer succeeded\n");
- printf(" (the default --delete timing; implies --delete)\n");
+ printf(" (implies --delete)\n");
+ printf(" --delete-commit FastSync-only: restore the late whole-tree commit\n");
+ printf(" (identical to --delete-after; implies --delete)\n");
printf(" --delete-excluded Also delete destination files that were excluded on\n");
printf(" the source (default protects them, matching rsync)\n");
printf(" --max-delete=NUM Delete at most NUM destination entries per run; if the\n");
@@ -90,10 +91,11 @@ void print_usage(void) {
printf(" entry's destination mirror receiver-side. Independent of\n");
printf(" --delete (it does not imply --delete; a non-empty directory\n");
printf(" mirror is removed only with --force or --delete)\n");
- printf(" -m, --prune-empty-dirs Do not transfer empty directory entries (--dirs mode);\n");
- printf(" recursive transfers never send empty dirs\n");
+ printf(" -m, --prune-empty-dirs Do not create empty directories (a recursive transfer\n");
+ printf(" otherwise recreates them, like rsync)\n");
printf(" Note: each timing flag implies --delete. Combining a timing flag with\n");
- printf(" --no-delete (in either order) is rejected as a config error.\n");
+ printf(" --no-delete (in either order) is rejected as a config error, as is more\n");
+ printf(" than one timing flag.\n");
printf(" --ignore-existing Skip files that already exist on receiver\n");
printf(" --delay-updates Put updated files into place only at the end of transfer\n");
printf(" --dirs, -d, --old-dirs, --old-d Transfer the named directory entries without\n");
@@ -103,8 +105,9 @@ void print_usage(void) {
printf(" -R, --relative With --files-from, preserve each listed entry's relative path\n");
printf(" below the destination root instead of mirroring the full\n");
printf(" source path (no effect without --files-from)\n");
- printf(" --no-implied-dirs With -R --files-from, refuse to place a listed file whose\n");
- printf(" parent directory is not itself listed\n");
+ printf(" --no-implied-dirs With -R, do not apply the source metadata of a listed file's\n");
+ printf(" implied parent directories (they are still created with\n");
+ printf(" default attributes)\n");
printf(" --mkpath Create the destination root directory on the server when it\n");
printf(" does not exist yet\n");
printf(" --exclude , --exclude= Exclude files matching pattern\n");
@@ -137,6 +140,9 @@ void print_usage(void) {
printf(" into the destination instead of transferring its data\n");
printf(" --link-dest Like --copy-dest, but hard-links the unchanged file from DIR\n");
printf(" into the destination (repeatable; earlier DIRs win)\n");
+ printf(" --verify-basis FastSync-only: require a basis hit's content to match the\n");
+ printf(" source by whole-file digest instead of trusting rsync's\n");
+ printf(" size+mtime (or --size-only) quick-check\n");
printf(" --checksum-choice, --cc Whole-file checksum algorithm for --incremental/\n");
printf(" --checksum compares. Accepted: xxh128 (default), xxh3, xxh64\n");
printf(" (aka xxhash), md5, md4, sha1, or none. A two-name\n");
@@ -243,7 +249,9 @@ void print_usage(void) {
printf(" reusable digest is sent (keep the file mode 0600)\n");
printf(" --no-motd Suppress display of the daemon's MOTD (the server\n");
printf(" still sends it; the client just does not show it)\n");
- printf(" --bwlimit Bandwidth limit in kilobytes per second\n");
+ printf(" --bwlimit=RATE Limit socket I/O bandwidth (default unit KiB/s,\n");
+ printf(" rsync-style: 0 = no limit; K/M/G/T/P suffixes are\n");
+ printf(" binary, KB/MB decimal, KiB/MiB binary; decimals allowed)\n");
printf(" --tls Enable TLS encryption\n");
printf(" --cert TLS certificate file (PEM)\n");
printf(" --key TLS private key file (PEM)\n");
diff --git a/src/server/receiver.c b/src/server/receiver.c
index d90e934..178ca34 100644
--- a/src/server/receiver.c
+++ b/src/server/receiver.c
@@ -61,13 +61,18 @@ bool receiver_send_final_success(int fd, const Config* config, const ReceiverOut
}
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
- const struct ArrayList* would_delete) {
+ const struct ArrayList* would_delete,
+ const struct ArrayList* deleted_paths) {
if (!config->report_stats)
return true;
ReceiverStats local;
memset(&local, 0, sizeof(local));
const ReceiverStats* out = stats ? stats : &local;
- size_t count = would_delete ? (size_t)would_delete->size : 0;
+ /* The path list carries the dry-run would-delete set for a -n run and the
+ actually-removed set for a real --info=del run. */
+ const struct ArrayList* paths =
+ config->dry_run ? would_delete : (config->report_deletes ? deleted_paths : NULL);
+ size_t count = paths ? (size_t)paths->size : 0;
if (count > (size_t)MAX_MANIFEST_ENTRIES)
count = MAX_MANIFEST_ENTRIES;
ReceiverStats record = *out;
@@ -76,7 +81,7 @@ bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats
!send_int(fd, (int)count))
return false;
for (size_t i = 0; i < count; i++) {
- const char* path = (const char*)would_delete->items[i];
+ const char* path = (const char*)paths->items[i];
if (!send_wire_str(fd, path ? path : ""))
return false;
}
@@ -90,6 +95,25 @@ static void receiver_tally_deleted(const ReceiverSink* sink, size_t deleted) {
sink->stats->deleted_files += deleted;
}
+/* Observer for --info=del: record each truly-removed destination-relative path
+ in the ArrayList passed as the observer context, so the terminal STATUS_STATS
+ frame can list it. A failed append is best-effort (the deletion already
+ happened; output is cosmetic). Shared by the single-threaded receiver and
+ the -m pipeline's deferred commit. */
+void receiver_record_deleted_path(void* context, const char* rel_path) {
+ ArrayList* paths = context;
+ if (!paths || !rel_path)
+ return;
+ /* Bound the retained list like the keep-set manifest: only MAX_MANIFEST_ENTRIES
+ paths are ever transmitted in the terminal STATUS_STATS frame, so recording
+ more only grows memory. A hostile/huge deletion set is therefore capped. */
+ if ((size_t)paths->size >= (size_t)MAX_MANIFEST_ENTRIES)
+ return;
+ char* copy = str_dup(rel_path);
+ if (copy && !array_list_add(paths, copy))
+ free(copy);
+}
+
static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) {
if (!chunk || !sink || !sink->store_file)
return false;
@@ -281,14 +305,16 @@ int receiver_process(Config* config, int file_descriptor, const ReceiverSink* si
/* Runs the whole receive loop. The delete manifest may legitimately arrive
either FIRST (--delete-before / --delete-during: the sender transmits the
- validated keep-set before any file data) or LAST (plain --delete /
- --delete-after / --delete-delay: the manifest closes the data stream). In
+ validated keep-set before any file data) or LAST (--delete-after /
+ --delete-commit / --delete-delay: the manifest closes the data stream). In
the early modes the receiver deletes as soon as the manifest has been read
and acknowledges with STATUS_OK so the sender only starts streaming once the
deletion has committed (or failed); in the late modes the manifest is held
and the deletion is committed only after the terminal STATUS_FINISHED proves
- the whole transfer succeeded. See receiver_process_pending() for how the -m
- receiver defers that commit until its disk writer has drained. */
+ the whole transfer succeeded. A plain --delete defaults to the per-directory
+ delete-during plan mode (no manifest at all). See
+ receiver_process_pending() for how the -m receiver defers that commit until
+ its disk writer has drained. */
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans) {
Status status;
@@ -398,9 +424,13 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
--max-delete-capped commit still succeeds and the transfer proceeds;
the terminal success frame reports the cap. */
size_t deleted = 0;
- DeleteCommitResult deletion = (config->use_delete || config->delete_missing_args)
- ? manifest_delete_all_counted(config, manifest, &deleted)
- : DELETE_COMMIT_OK;
+ DeletePathObserver observer =
+ (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL;
+ DeleteCommitResult deletion =
+ (config->use_delete || config->delete_missing_args)
+ ? manifest_delete_all_observed(config, manifest, &deleted, observer,
+ (void*)sink->deleted_paths)
+ : DELETE_COMMIT_OK;
receiver_tally_deleted(sink, deleted);
delete_manifest_free(manifest);
if (deletion == DELETE_COMMIT_ERROR) {
@@ -435,8 +465,12 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
send_status(file_descriptor, STATUS_ERROR);
goto fail;
}
- if (!plan_session)
+ if (!plan_session) {
plan_session = delete_plan_session_create(config);
+ if (plan_session && config->report_deletes && sink->deleted_paths)
+ delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path,
+ (void*)sink->deleted_paths);
+ }
if (!plan_session || delete_plan_session_receive(plan_session, config, file_descriptor) != 0)
goto fail;
if (delete_plan_session_limit_reached(plan_session) && !delete_limit_noted &&
@@ -479,8 +513,10 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
deferred_manifest = NULL;
} else {
size_t deleted = 0;
- DeleteCommitResult deletion =
- manifest_delete_all_counted(config, deferred_manifest, &deleted);
+ DeletePathObserver observer =
+ (config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL;
+ DeleteCommitResult deletion = manifest_delete_all_observed(
+ config, deferred_manifest, &deleted, observer, (void*)sink->deleted_paths);
receiver_tally_deleted(sink, deleted);
delete_manifest_free(deferred_manifest);
deferred_manifest = NULL;
@@ -498,6 +534,9 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
hands the session to its caller instead, which commits after the disk
writer drained. */
if (plan_session) {
+ if (config->report_deletes && sink->deleted_paths)
+ delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path,
+ (void*)sink->deleted_paths);
if (pending_plans) {
*pending_plans = plan_session;
plan_session = NULL;
@@ -569,11 +608,15 @@ typedef struct {
--delete would-delete path list collected while processing the manifest. */
ReceiverStats stats;
ArrayList* would_delete;
+ /* --info=del: actually-removed paths collected during the delete commit. */
+ ArrayList* deleted_paths;
} ReceiverSaveContext;
static bool receiver_save_file(File* file, void* context_pointer) {
ReceiverSaveContext* context = context_pointer;
FileSaveResult result = FILE_SAVE_ERROR;
+ bool created = false;
+ unsigned created_dirs = 0;
if (context->config->dry_run) {
/* Defense in depth: a dry-run receiver mutates nothing even if a data
frame reaches the sink (the sender is not supposed to send one). */
@@ -583,12 +626,17 @@ static bool receiver_save_file(File* file, void* context_pointer) {
--remove-source-files sender keeps its source. */
result = FILE_SAVE_SKIPPED;
} else {
- result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
+ result = file_save_to_disk_full_ex(context->config->receive_root_directory, file,
+ context->config, &created, &created_dirs);
}
/* Wire-stats tally: bytes reconstructed from the basis file (delta matches)
count as matched data in the end-of-transfer report. */
if (result != FILE_SAVE_ERROR && file->matched_bytes > 0)
context->stats.matched_data += file->matched_bytes;
+ /* Protocol 2.28.0: receiver-observed literal bytes and the created-entry
+ breakdown (regular/dir/link/special) for the `--stats` report. */
+ if (result == FILE_SAVE_WRITTEN)
+ receiver_stats_note_saved(&context->stats, file, created, created_dirs);
/* A directory's metadata is deferred, never applied inline: collect it now
and apply it at the end. -O/--omit-dir-times and --preserve_perms/-times
are honored by dir_metadata_list_apply's caller (see
@@ -621,7 +669,8 @@ static void receiver_note_delete_limit(void* context_pointer) {
static bool receiver_send_success_frame(int fd, void* context_pointer) {
ReceiverSaveContext* context = context_pointer;
Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK;
- if (!receiver_send_stats_frame(fd, context->config, &context->stats, context->would_delete))
+ if (!receiver_send_stats_frame(fd, context->config, &context->stats, context->would_delete,
+ context->deleted_paths))
return false;
/* Server-contacting --dry-run: nothing was staged or written, so there is
nothing to publish and no directory times to stamp. */
@@ -651,8 +700,15 @@ int receiver_receive_files(Config* config, int file_descriptor) {
ReceiverSaveContext context = {.config = config, .outcomes = {0}};
dir_time_list_init(&context.dir_times);
context.would_delete = array_list_create(free);
- if (!context.would_delete)
+ /* report_deletes (--info=del / -i / --out-format under --delete) is the only
+ reason to retain the actually-removed paths; a plain --delete must not
+ str_dup every removal. NULL is handled by every consumer. */
+ context.deleted_paths = config->report_deletes ? array_list_create(free) : NULL;
+ if (!context.would_delete || (config->report_deletes && !context.deleted_paths)) {
+ array_list_delete(context.would_delete);
+ array_list_delete(context.deleted_paths);
return -1;
+ }
ReceiverSink sink = {receiver_save_file,
&context,
true,
@@ -660,12 +716,14 @@ int receiver_receive_files(Config* config, int file_descriptor) {
receiver_send_success_frame,
receiver_note_delete_limit,
&context.stats,
- context.would_delete};
+ context.would_delete,
+ context.deleted_paths};
int ret = receiver_process(config, file_descriptor, &sink);
if (ret != 0 && config->delay_updates && config->delay_context)
delay_updates_cleanup(config->delay_context);
receiver_outcomes_destroy(&context.outcomes);
dir_time_list_free(&context.dir_times);
array_list_delete(context.would_delete);
+ array_list_delete(context.deleted_paths);
return ret;
}
diff --git a/src/server/receiver.h b/src/server/receiver.h
index 0dc2277..3b2dcff 100644
--- a/src/server/receiver.h
+++ b/src/server/receiver.h
@@ -46,11 +46,20 @@ typedef struct {
carries the -n/--dry-run --delete path list. */
ReceiverStats* stats;
struct ArrayList* would_delete;
+ /* When --info=del requested it, receiver-owned strings for every path the
+ deletion commit ACTUALLY removed, sent in the terminal STATUS_STATS frame's
+ path list so the sender can print rsync's `deleting PATH` lines. */
+ struct ArrayList* deleted_paths;
} ReceiverSink;
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code);
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes);
+/* DeletePathObserver implementation for --info=del: `context` is an ArrayList*
+ that receives owned copies of every truly-removed destination-relative path.
+ Shared by the single-threaded receiver and the -m pipeline's deferred commit. */
+void receiver_record_deleted_path(void* context, const char* rel_path);
+
/* Send the terminal success frame. `final_status` is usually STATUS_OK, or
STATUS_DELETE_LIMIT when a --max-delete commit was capped. */
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
@@ -60,7 +69,8 @@ bool receiver_send_final_success(int fd, const Config* config, const ReceiverOut
non-NULL, a count and that many wire strings) when the wire config requested
report_stats. A no-op otherwise. */
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
- const struct ArrayList* would_delete);
+ const struct ArrayList* would_delete,
+ const struct ArrayList* deleted_paths);
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink);
/* receiver_process with an escape hatch for the commit-style (late) deletion:
diff --git a/src/server/receiver_pipeline.c b/src/server/receiver_pipeline.c
index 40784ed..5d1066e 100644
--- a/src/server/receiver_pipeline.c
+++ b/src/server/receiver_pipeline.c
@@ -31,6 +31,7 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue*
context->delete_limit_reached = false;
memset(&context->stats, 0, sizeof(context->stats));
context->would_delete = NULL;
+ context->deleted_paths = NULL;
atomic_init(&context->cancelled, false);
int init = 0;
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
@@ -46,6 +47,15 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue*
context->would_delete = array_list_create(free);
if (!context->would_delete)
goto fail;
+ /* The actually-removed path list is only needed to render rsync's
+ `deleting PATH` lines, which the client requests via report_deletes
+ (--info=del / -i / --out-format under --delete). A plain --delete run must
+ not allocate it or observe every removal. */
+ if (config->report_deletes) {
+ context->deleted_paths = array_list_create(free);
+ if (!context->deleted_paths)
+ goto fail;
+ }
return context;
fail:
@@ -56,6 +66,12 @@ fail:
cnd_destroy(&context->condition_not_full);
if (init >= 1)
mtx_destroy(&context->mutex);
+ /* Free every list that was already created before the failing allocation:
+ `context` itself is freed below, so they would otherwise leak. */
+ if (context->would_delete)
+ array_list_delete(context->would_delete);
+ if (context->deleted_paths)
+ array_list_delete(context->deleted_paths);
free(context);
return NULL;
}
@@ -71,6 +87,8 @@ void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
dir_time_list_free(&context->dir_times);
if (context->would_delete)
array_list_delete(context->would_delete);
+ if (context->deleted_paths)
+ array_list_delete(context->deleted_paths);
mtx_destroy(&context->mutex);
cnd_destroy(&context->condition_not_full);
cnd_destroy(&context->condition_not_empty);
@@ -184,7 +202,8 @@ int receive_thread(void* pipeline_context) {
NULL,
receiver_pipeline_note_delete_limit,
&context->stats,
- context->would_delete};
+ context->would_delete,
+ context->deleted_paths};
if (receiver_process_pending((Config*)config, file_descriptor, &sink, &context->deferred_manifest,
&context->deferred_plans) != 0) {
receiver_thread_fail(context);
@@ -228,12 +247,23 @@ int write_thread(void* pipeline_context) {
}
size_t file_bytes = file->data ? file->data->size : 0;
FileSaveResult result = FILE_SAVE_SKIPPED;
+ bool created = false;
+ unsigned created_dirs = 0;
/* Server-contacting --dry-run: never write. The receiver thread does not
enqueue anything on the dry-run path, but this keeps the writer thread
provably mutation-free if a data frame ever reached it. */
bool dry_run = context->config->dry_run;
if (save_to_disk && !dry_run) {
- result = file_save_to_disk_full(root_directory, file, context->config);
+ result =
+ file_save_to_disk_full_ex(root_directory, file, context->config, &created, &created_dirs);
+ if (result == FILE_SAVE_WRITTEN) {
+ /* Protocol 2.28.0: fold the receiver-observed literal bytes and the
+ created-entry type into the shared stats block under its mutex (the
+ receive thread also writes stats.matched_data). */
+ mtx_lock(&context->mutex);
+ receiver_stats_note_saved(&context->stats, file, created, created_dirs);
+ mtx_unlock(&context->mutex);
+ }
if (result == FILE_SAVE_ERROR) {
file_destroy(file);
pipeline_context_receiver_note_bytes_released(context, file_bytes);
diff --git a/src/server/receiver_pipeline.h b/src/server/receiver_pipeline.h
index 9ad6110..099bd9c 100644
--- a/src/server/receiver_pipeline.h
+++ b/src/server/receiver_pipeline.h
@@ -61,6 +61,9 @@ typedef struct PipelineContextReceiver {
/* -n/--dry-run --delete would-delete path list, collected by receive_thread
and reported in the STATUS_STATS frame. */
struct ArrayList* would_delete;
+ /* --info=del actually-removed path list, collected by the deferred delete
+ commit in server.c and reported in the STATUS_STATS frame. */
+ struct ArrayList* deleted_paths;
} PipelineContextReceiver;
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
diff --git a/src/server/server.c b/src/server/server.c
index 6172f5e..aebf384 100644
--- a/src/server/server.c
+++ b/src/server/server.c
@@ -956,8 +956,9 @@ void handler(int file_descriptor) {
server-contacting --dry-run deletes nothing (no manifest is sent). */
if (context->deferred_manifest) {
size_t deleted = 0;
- DeleteCommitResult deletion =
- manifest_delete_all_counted(config, context->deferred_manifest, &deleted);
+ DeletePathObserver observer = config->report_deletes ? receiver_record_deleted_path : NULL;
+ DeleteCommitResult deletion = manifest_delete_all_observed(
+ config, context->deferred_manifest, &deleted, observer, (void*)context->deleted_paths);
context->stats.deleted_files += deleted;
if (deletion == DELETE_COMMIT_ERROR) {
transfer_ok = false;
@@ -975,6 +976,9 @@ void handler(int file_descriptor) {
if (context->deferred_plans) {
/* Defence in depth (the enclosing block already excludes dry-run): a
-n run never commits a deletion. */
+ if (config->report_deletes)
+ delete_plan_session_set_delete_observer(
+ context->deferred_plans, receiver_record_deleted_path, (void*)context->deleted_paths);
DeleteCommitResult deletion =
config->dry_run ? DELETE_COMMIT_OK
: delete_plan_session_commit(context->deferred_plans, config);
@@ -1010,7 +1014,7 @@ void handler(int file_descriptor) {
/* Emit the optional wire-stats record first (protocol 2.25.0), then the
success/outcome frame, exactly like the single-threaded receiver. */
if (!receiver_send_stats_frame(file_descriptor, config, &context->stats,
- context->would_delete) ||
+ context->would_delete, context->deleted_paths) ||
!receiver_send_final_success(file_descriptor, config, &context->outcomes, final_status))
transfer_ok = false;
} else {
diff --git a/src/shared/charset.c b/src/shared/charset.c
index 75a2ac5..547ef4c 100644
--- a/src/shared/charset.c
+++ b/src/shared/charset.c
@@ -168,9 +168,14 @@ bool charset_spec_valid_direction(const char* from_charset, const char* to_chars
return direction_probe_valid(from_charset, to_charset);
}
-/* The receiver's real conversion is wire(client REMOTE) -> server-local (the
- * server's own --iconv LOCAL half, or the client's LOCAL half when the server
- * has no --iconv). A dedicated pre-ack check so an impossible direction is
+/* The receiver's conversion is wire charset -> destination charset. rsync's
+ * CONVERT_SPEC is LOCAL,REMOTE and "stays the same whether you're pushing or
+ * pulling", so for a PUSH (FastSync's only direction) the destination end's
+ * charset is the spec's REMOTE half: the client converts LOCAL -> REMOTE on the
+ * sender and the receiver writes the wire bytes verbatim. Only a server that
+ * declares its OWN --iconv (the daemon "charset" analog) has a different local
+ * charset, and then it is that spec's LOCAL half and the receiver converts
+ * wire -> server-local. A dedicated pre-ack check so an impossible direction is
* rejected before the connection instead of refusing mid-transfer. */
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec) {
if (!spec)
@@ -180,7 +185,7 @@ bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec)
if (charset_spec_parse(spec, &local, &remote) != 0)
return false;
const char* wire = remote;
- const char* target_local = local;
+ const char* target_local = remote;
char* server_local = NULL;
char* server_remote = NULL;
if (server_spec) {
@@ -302,13 +307,13 @@ bool charset_wire_init_receiver(const char* spec, const char* server_spec) {
char* remote;
if (charset_spec_parse(spec, &local, &remote) != 0)
return false;
- /* The wire charset is the client spec's REMOTE half; the local charset is
- * the client spec's LOCAL half unless the server was itself started with
- * --iconv naming a different local charset (the server halves above never
- * travel, so the server's own flag is the only way its local charset can
- * differ from what the client assumed). */
+ /* The wire charset is the client spec's REMOTE half (rsync's LOCAL,REMOTE
+ * spec stays the same push or pull, so on a push the destination end's
+ * charset is REMOTE and the receiver writes the wire bytes verbatim). Only a
+ * server started with its own --iconv declares a different local charset (the
+ * server halves above never travel), and then it is that spec's LOCAL half. */
const char* wire = remote;
- const char* target_local = local;
+ const char* target_local = remote;
char* server_local = NULL;
char* server_remote = NULL;
if (server_spec) {
diff --git a/src/shared/charset.h b/src/shared/charset.h
index 49cd0f3..874c901 100644
--- a/src/shared/charset.h
+++ b/src/shared/charset.h
@@ -57,8 +57,9 @@ void charset_conversion_close(void* conversion);
/* Process-wide wire conversion. charset_wire_init_sender (client side) opens
* LOCAL->REMOTE; charset_wire_init_receiver (server side) opens
* wire(REMOTE)->server-local. server_spec is the server's own --iconv, whose
- * LOCAL half may override the local charset the client assumed; NULL reuses
- * the client spec's LOCAL half. Both return false on an unsupported spec.
+ * LOCAL half overrides the destination charset; NULL means the destination
+ * charset is the client spec's REMOTE half (rsync's push semantics: the wire
+ * bytes are written verbatim). Both return false on an unsupported spec.
* The state is freed with charset_wire_free. */
bool charset_wire_init_sender(const char* spec);
bool charset_wire_init_receiver(const char* spec, const char* server_spec);
@@ -66,9 +67,9 @@ void charset_wire_free(void);
bool charset_wire_active(void);
/* Pre-ack receiver-direction sanity (see charset_wire_init_receiver): true
- * when the exact wire->server-local conversion the receiver will use (client
- * spec's REMOTE half into the server's own LOCAL half, or the client's LOCAL
- * half when the server has no --iconv) opens and produces NUL-free output. */
+ * when the exact wire->destination conversion the receiver will use (client
+ * spec's REMOTE half into the server's own LOCAL half, or REMOTE->REMOTE when
+ * the server has no --iconv) opens and produces NUL-free output. */
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec);
/* Convert a path across the wire in the process direction. Returns a malloc'd
diff --git a/src/shared/checksum.c b/src/shared/checksum.c
index d12e339..6546495 100644
--- a/src/shared/checksum.c
+++ b/src/shared/checksum.c
@@ -1,4 +1,5 @@
#include "checksum.h"
+#include "utils.h"
#include
#include
#include
@@ -223,17 +224,33 @@ bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, ui
if (fd < 0)
return false;
+ bool ok = checksum_digest_fd(algo, seed, fd, out, out_capacity, out_len);
+ close(fd);
+ return ok;
+}
+
+bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
+ size_t* out_len) {
+ if (fd < 0 || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
+ return false;
+
+ if (algo == CHECKSUM_ALGO_NONE) {
+ /* No checksum requested: nothing to read; an empty digest succeeds. */
+ *out_len = 0;
+ return true;
+ }
+
uint8_t buffer[64 * 1024];
bool ok = false;
+ lseek(fd, 0, SEEK_SET);
- if (algo == CHECKSUM_ALGO_MD5) {
+ if (algo == CHECKSUM_ALGO_MD5 || algo == CHECKSUM_ALGO_SHA1) {
+ const EVP_MD* md = algo == CHECKSUM_ALGO_MD5 ? EVP_md5() : EVP_sha1();
EVP_MD_CTX* ctx = EVP_MD_CTX_new();
- if (!ctx) {
- close(fd);
+ if (!ctx)
return false;
- }
unsigned int digest_len = 0;
- if (EVP_DigestInit_ex(ctx, EVP_md5(), NULL) == 1) {
+ if (EVP_DigestInit_ex(ctx, md, NULL) == 1) {
ok = true;
ssize_t got;
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
@@ -250,7 +267,22 @@ bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, ui
ok = false;
}
EVP_MD_CTX_free(ctx);
- close(fd);
+ return ok;
+ }
+
+ if (algo == CHECKSUM_ALGO_MD4) {
+ Md4Ctx ctx;
+ md4_init(&ctx);
+ ok = true;
+ ssize_t got;
+ while ((got = read(fd, buffer, sizeof(buffer))) > 0)
+ md4_update(&ctx, buffer, (size_t)got);
+ if (got < 0)
+ ok = false;
+ if (ok) {
+ md4_final(&ctx, out);
+ *out_len = 16;
+ }
return ok;
}
@@ -260,16 +292,13 @@ bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, ui
XXH64_reset(&xxh64, seed);
} else if (algo == CHECKSUM_ALGO_XXH3 || algo == CHECKSUM_ALGO_XXH128) {
xxh3 = XXH3_createState();
- if (!xxh3) {
- close(fd);
+ if (!xxh3)
return false;
- }
if (algo == CHECKSUM_ALGO_XXH3)
XXH3_64bits_reset_withSeed(xxh3, seed);
else
XXH3_128bits_reset_withSeed(xxh3, seed);
} else {
- close(fd);
return false;
}
@@ -303,7 +332,6 @@ bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, ui
}
if (xxh3)
XXH3_freeState(xxh3);
- close(fd);
return ok;
}
@@ -371,7 +399,7 @@ uint8_t checksum_digest_len(ChecksumAlgo algo) {
return 0;
}
-ChecksumAlgo checksum_negotiate_default(void) {
+static ChecksumAlgo compiled_checksum_preference_first(void) {
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
* resolves to xxh128. */
static const ChecksumAlgo preference[] = {
@@ -384,3 +412,16 @@ ChecksumAlgo checksum_negotiate_default(void) {
}
return CHECKSUM_ALGO_XXH64;
}
+
+int checksum_choice_resolve(void) {
+ bool specified = false;
+ int env = env_choice_first("RSYNC_CHECKSUM_LIST", checksum_algo_from_name, &specified);
+ if (specified)
+ return env; /* -1 = the list named no supported checksum */
+ return (int)compiled_checksum_preference_first();
+}
+
+ChecksumAlgo checksum_negotiate_default(void) {
+ int resolved = checksum_choice_resolve();
+ return resolved >= 0 ? (ChecksumAlgo)resolved : compiled_checksum_preference_first();
+}
diff --git a/src/shared/checksum.h b/src/shared/checksum.h
index 36deebc..3614ce1 100644
--- a/src/shared/checksum.h
+++ b/src/shared/checksum.h
@@ -51,6 +51,13 @@ bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
size_t out_capacity, size_t* out_len);
+/* Descriptor form of the streaming digest: rewinds `fd` to the start and hashes
+ * to EOF without closing it. Used by the --verify-basis path to hash an
+ * already-open, root-confined basis descriptor. Same contract as
+ * checksum_digest_file. */
+bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
+ size_t* out_len);
+
/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id.
* Accepts "xxh64"/"xxhash", "xxh3", "xxh128", "md5", "md4", "sha1", "none".
* "auto" is not an algorithm here; the caller resolves it to the negotiated
@@ -72,4 +79,11 @@ uint8_t checksum_digest_len(ChecksumAlgo algo);
* xxh128 xxh3 xxh64 md5 md4 sha1 none). Used to resolve "auto". */
ChecksumAlgo checksum_negotiate_default(void);
+/* Resolve "auto" the way rsync does: the first supported name in
+ * RSYNC_CHECKSUM_LIST (whitespace-separated, client half ends at '&'), then the
+ * compiled-in preference order when the variable is unset/blank. Returns -1
+ * when the variable is set but names no supported checksum (rsync's failed
+ * negotiation), otherwise a valid ChecksumAlgo id. */
+int checksum_choice_resolve(void);
+
#endif /* CHECKSUM_H */
diff --git a/src/shared/compression.c b/src/shared/compression.c
index 428b67b..c1fb5c9 100644
--- a/src/shared/compression.c
+++ b/src/shared/compression.c
@@ -2,6 +2,7 @@
#include "data.h"
#include "log.h"
#include "protocol.h"
+#include "utils.h"
#include
#include
#include
@@ -119,7 +120,7 @@ bool compression_algo_enabled(CompressionAlgo algo) {
return algo != COMPRESSION_ALGO_NONE;
}
-CompressionAlgo compression_negotiate_default(void) {
+static CompressionAlgo compiled_preference_first(void) {
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
* resolves to zstd. */
static const CompressionAlgo preference[] = {
@@ -133,6 +134,57 @@ CompressionAlgo compression_negotiate_default(void) {
return COMPRESSION_ALGO_ZSTD;
}
+int compression_choice_resolve(void) {
+ bool specified = false;
+ int env = env_choice_first("RSYNC_COMPRESS_LIST", compression_algo_from_name, &specified);
+ if (specified)
+ return env; /* -1 = the list named no supported codec */
+ return (int)compiled_preference_first();
+}
+
+CompressionAlgo compression_negotiate_default(void) {
+ int resolved = compression_choice_resolve();
+ return resolved >= 0 ? (CompressionAlgo)resolved : compiled_preference_first();
+}
+
+int compression_default_level(CompressionAlgo algo) {
+ switch (algo) {
+ case COMPRESSION_ALGO_ZSTD:
+ return ZSTD_CLEVEL_DEFAULT;
+ case COMPRESSION_ALGO_ZLIB:
+ case COMPRESSION_ALGO_ZLIBX:
+ return 6; /* rsync resolves zlib's Z_DEFAULT_COMPRESSION (-1) to 6 */
+ case COMPRESSION_ALGO_LZ4:
+ return 1; /* rsync lz4 level is 0/ignored; positive keeps the gate on */
+ case COMPRESSION_ALGO_NONE:
+ return 0;
+ }
+ return 0;
+}
+
+int compression_clamp_level(CompressionAlgo algo, int level) {
+ switch (algo) {
+ case COMPRESSION_ALGO_ZSTD:
+ if (level < 1)
+ return 1;
+ if (level > 22)
+ return 22;
+ return level;
+ case COMPRESSION_ALGO_ZLIB:
+ case COMPRESSION_ALGO_ZLIBX:
+ if (level < 1)
+ return 1;
+ if (level > 9)
+ return 9;
+ return level;
+ case COMPRESSION_ALGO_LZ4:
+ return 1; /* ignored by lz4_compress; keeps the "compress" gate on */
+ case COMPRESSION_ALGO_NONE:
+ return 0;
+ }
+ return level;
+}
+
void compression_set_algo(CompressionAlgo algo) {
if (compression_algo_valid((int)algo))
atomic_store(&g_compression_algo, (int)algo);
diff --git a/src/shared/compression.h b/src/shared/compression.h
index 3e5d928..ca874b8 100644
--- a/src/shared/compression.h
+++ b/src/shared/compression.h
@@ -35,6 +35,26 @@ bool compression_algo_valid(int algo);
* "auto". */
CompressionAlgo compression_negotiate_default(void);
+/* Resolve "auto" the way rsync does: the first supported name in
+ * RSYNC_COMPRESS_LIST (whitespace-separated, client half ends at '&'), then the
+ * compiled-in preference order when the variable is unset/blank. Returns -1
+ * when the variable is set but names no supported codec (rsync's failed
+ * negotiation), otherwise a valid CompressionAlgo id. */
+int compression_choice_resolve(void);
+
+/* rsync 3.4.1's per-codec default level, applied when the user did not pass
+ * --compress-level/--zl. zstd uses ZSTD_CLEVEL_DEFAULT (3) and zlib/zlibx the
+ * resolved Z_DEFAULT_COMPRESSION (6). lz4 has no tunable level in rsync
+ * (always the default acceleration); FastSync returns a positive placeholder so
+ * its "level > 0" compression gate stays engaged, and lz4_compress ignores the
+ * value, so the output is identical to rsync's. none is 0. */
+int compression_default_level(CompressionAlgo algo);
+
+/* Clamp an explicit --compress-level to the codec's accepted range the way
+ * rsync's init_compression_level() does: zstd 1..22, zlib/zlibx 1..9, lz4
+ * ignored (fixed positive placeholder), none 0. */
+int compression_clamp_level(CompressionAlgo algo, int level);
+
/* True when the algorithm actually compresses (i.e. is not NONE). */
bool compression_algo_enabled(CompressionAlgo algo);
diff --git a/src/shared/config.c b/src/shared/config.c
index 03dcebd..d0e25a8 100644
--- a/src/shared/config.c
+++ b/src/shared/config.c
@@ -70,6 +70,8 @@ static void config_set_defaults(Config* config) {
config->ignore_missing_args = false;
config->checksum_transfer_algo = CHECKSUM_ALGO_DEFAULT;
config->cli_exit_code = 0;
+ config->compression_level_set = false;
+ config->checksum_choice_set = false;
config->filters = NULL;
config->files_from = NULL;
config->files_from_set = NULL;
@@ -206,7 +208,8 @@ static bool validate_received_config(const Config* config) {
valid_wire_bool(config->preserve_perms) && valid_wire_bool(config->preserve_times) &&
valid_wire_bool(config->preserve_owner) && valid_wire_bool(config->preserve_group) &&
valid_wire_bool(config->munge_links) && valid_wire_bool(config->keep_dirlinks) &&
- valid_wire_bool(config->fake_super) &&
+ valid_wire_bool(config->fake_super) && valid_wire_bool(config->report_dest_info) &&
+ valid_wire_bool(config->report_stats) && valid_wire_bool(config->report_deletes) &&
(!config->copy_as_set || (config->copy_as_uid >= 0 && config->copy_as_gid >= 0)) &&
(!config->use_compression ||
(config->compression_level >= 1 && config->compression_level <= 22)) &&
@@ -788,6 +791,8 @@ void config_delete(Config* config) {
if (config->filters) {
array_list_delete(config->filters);
}
+ filter_rule_list_free(config->protect_rules);
+ config->protect_rules = NULL;
/* A --delay-updates staging tree is transient receiver state: remove any
leftovers on every exit path (success already emptied it). */
if (config->delay_context)
@@ -1023,6 +1028,124 @@ static bool receive_basis_entries(int fd, Config* c, ConfigStringBudget* budget)
return true;
}
+/* Receiver-side delete-protection rules (protocol 2.28.0). The sender compiles
+ * its command-line selection rules exactly as the scanner does and streams the
+ * result as one bounded, self-describing block (count + per-rule records); the
+ * receiver reconstructs a FilterRuleList for the --delete extras walk. owner
+ * and pattern are charged through the shared ConfigStringBudget and the block
+ * additionally enforces MAX_FILTER_RULES / MAX_FILTER_BYTES. */
+static bool send_protect_entries(int fd, const Config* c) {
+ int count = c->filters ? c->filters->size : 0;
+ const char** texts = NULL;
+ if (count > 0) {
+ texts = malloc((size_t)count * sizeof(char*));
+ if (!texts)
+ return false;
+ for (int i = 0; i < count; i++)
+ texts[i] = (const char*)c->filters->items[i];
+ }
+ char err[160];
+ FilterRuleList* rules =
+ filter_base_build(texts, count, c->cvs_exclude, c->delete_excluded, err, sizeof(err));
+ free(texts);
+ if (!rules) {
+ log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err);
+ return false;
+ }
+ bool ok = send_int(fd, rules->count);
+ for (int i = 0; ok && i < rules->count; i++) {
+ const FilterRule* r = rules->items[i];
+ /* Mirror the receiver's limit so the peer never receives a rule it will
+ reject as a protocol error. */
+ if (r->pattern && strlen(r->pattern) > MAX_PROTECT_PATTERN_LEN) {
+ log_message(LOG_LEVEL_ERROR, "filter pattern exceeds %d bytes", MAX_PROTECT_PATTERN_LEN);
+ filter_rule_list_free(rules);
+ return false;
+ }
+ ok = send_int(fd, (int)r->action) && send_int(fd, (int)r->sides) &&
+ send_int(fd, r->anchored ? 1 : 0) && send_int(fd, r->dir_only ? 1 : 0) &&
+ send_int(fd, r->negate ? 1 : 0) && send_str(fd, r->owner ? r->owner : "") &&
+ send_str(fd, r->pattern ? r->pattern : "");
+ }
+ filter_rule_list_free(rules);
+ return ok;
+}
+
+static bool receive_protect_entries(int fd, Config* c, ConfigStringBudget* budget) {
+ int count;
+ if (!receive_int(fd, &count))
+ return false;
+ if (count < 0 || count > MAX_FILTER_RULES)
+ return false;
+ if (count == 0)
+ return true;
+ FilterRuleList* list = filter_rule_list_create();
+ if (!list)
+ return false;
+ size_t pattern_bytes = 0;
+ for (int i = 0; i < count; i++) {
+ int action;
+ int sides;
+ bool anchored;
+ bool dir_only;
+ bool negate;
+ if (!receive_int(fd, &action) ||
+ (action != FILTER_ACTION_EXCLUDE && action != FILTER_ACTION_INCLUDE) ||
+ !receive_int(fd, &sides) || sides < (int)FILTER_SIDE_SENDER ||
+ sides > (int)(FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER) ||
+ !receive_wire_bool(fd, &anchored) || !receive_wire_bool(fd, &dir_only) ||
+ !receive_wire_bool(fd, &negate))
+ goto fail;
+ char* owner = config_receive_str(fd, budget);
+ if (!owner)
+ goto fail;
+ char* pattern = config_receive_str(fd, budget);
+ if (!pattern || pattern[0] == '\0') {
+ free(owner);
+ free(pattern);
+ goto fail;
+ }
+ /* A pattern too long to be evaluated by glob_match against a PATH_MAX path
+ would silently fail to match and leave a protect rule inert (fail-open:
+ the entry is then deleted). Reject it up front as a protocol error
+ rather than accept a rule that can never shield anything. */
+ if (strlen(pattern) > MAX_PROTECT_PATTERN_LEN) {
+ free(owner);
+ free(pattern);
+ goto fail;
+ }
+ size_t bytes = strlen(owner) + strlen(pattern);
+ if (bytes > MAX_FILTER_BYTES - pattern_bytes) {
+ free(owner);
+ free(pattern);
+ goto fail;
+ }
+ pattern_bytes += bytes;
+ FilterRule* rule = calloc(1, sizeof(FilterRule));
+ if (!rule) {
+ free(owner);
+ free(pattern);
+ goto fail;
+ }
+ rule->action = (FilterAction)action;
+ rule->sides = (unsigned)sides;
+ rule->anchored = anchored;
+ rule->dir_only = dir_only;
+ rule->negate = negate;
+ rule->owner = owner;
+ rule->pattern = pattern;
+ if (!filter_rule_list_add(list, rule)) {
+ filter_rule_free(rule);
+ goto fail;
+ }
+ }
+ c->protect_rules = list;
+ return true;
+fail:
+ filter_rule_list_free(list);
+ return false;
+}
+
static bool send_identity_entries(int fd, const IdentityMap* map, int count) {
for (int i = 0; i < count; i++) {
if (!send_int(fd, map[i].from) || !send_int(fd, map[i].from_hi) || !send_int(fd, map[i].to) ||
@@ -1150,6 +1273,9 @@ fail:
#define CONFIG_RECV_BLOCK_IDMAP(name) \
receive_identity_entries(fd, budget, c->name##_count, &c->name)
+#define CONFIG_SEND_BLOCK_PROTECT_RULES(name) send_protect_entries(fd, c)
+#define CONFIG_RECV_BLOCK_PROTECT_RULES(name) receive_protect_entries(fd, c, budget)
+
/* One table entry, applied in sequence. XSEND/XRECV are statement macros so
* consecutive entries read as a plain sequence of assignments. */
#define XSEND(name, ctype, def, kind) ok = ok && (CONFIG_SEND_##kind(name));
@@ -1187,6 +1313,7 @@ CONFIG_DEFINE_SEND(send_privilege_options, CONFIG_WIRE_PRIVILEGE_FIELDS)
CONFIG_DEFINE_SEND(send_copy_as_options, CONFIG_WIRE_COPY_AS_FIELDS)
CONFIG_DEFINE_SEND(send_output_options, CONFIG_WIRE_OUTPUT_FIELDS)
CONFIG_DEFINE_SEND(send_codec_options, CONFIG_WIRE_CODEC_FIELDS)
+CONFIG_DEFINE_SEND(send_protect_options, CONFIG_WIRE_PROTECT_FIELDS)
CONFIG_DEFINE_RECV(receive_core_fields, CONFIG_WIRE_CORE_FIELDS)
CONFIG_DEFINE_RECV(receive_delta_fields, CONFIG_WIRE_DELTA_FIELDS)
@@ -1207,6 +1334,7 @@ CONFIG_DEFINE_RECV(receive_privilege_options, CONFIG_WIRE_PRIVILEGE_FIELDS)
CONFIG_DEFINE_RECV(receive_copy_as_options, CONFIG_WIRE_COPY_AS_FIELDS)
CONFIG_DEFINE_RECV(receive_output_options, CONFIG_WIRE_OUTPUT_FIELDS)
CONFIG_DEFINE_RECV(receive_codec_options, CONFIG_WIRE_CODEC_FIELDS)
+CONFIG_DEFINE_RECV(receive_protect_options, CONFIG_WIRE_PROTECT_FIELDS)
#undef XSEND
#undef XRECV
@@ -1325,7 +1453,8 @@ bool config_send_wire_block(int file_descriptor, const Config* config) {
send_privilege_options(file_descriptor, config) &&
send_copy_as_options(file_descriptor, config) &&
send_output_options(file_descriptor, config) &&
- send_codec_options(file_descriptor, config);
+ send_codec_options(file_descriptor, config) &&
+ send_protect_options(file_descriptor, config);
}
bool config_send(int file_descriptor, const Config* config) {
@@ -1397,7 +1526,8 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val
!receive_privilege_options(file_descriptor, config, &budget) ||
!receive_copy_as_options(file_descriptor, config, &budget) ||
!receive_output_options(file_descriptor, config, &budget) ||
- !receive_codec_options(file_descriptor, config, &budget))
+ !receive_codec_options(file_descriptor, config, &budget) ||
+ !receive_protect_options(file_descriptor, config, &budget))
goto error;
/* Validate/normalize the negotiated codec. compress_choice is the human
* spelling (NULL or "" when -z was not given); compression_algo is the
diff --git a/src/shared/config.h b/src/shared/config.h
index e9fe715..2eb6d3b 100644
--- a/src/shared/config.h
+++ b/src/shared/config.h
@@ -4,6 +4,7 @@
#include "array_list.h"
#include "checksum.h"
#include "compression.h"
+#include "filter.h"
#include
#include
#include
@@ -82,7 +83,7 @@ typedef struct {
typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF = 2 } SuperMode;
/* ===========================================================================
- * Config wire-field table (single source of truth for protocol 2.26.0).
+ * Config wire-field table (single source of truth for protocol 2.28.0).
*
* Every field below crosses the wire. The table is the ONLY place a
* serialized field is named: config.h expands CONFIG_WIRE_FIELDS() to declare
@@ -197,9 +198,18 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
X(skip_compress_count, int, 0, INT_SKIPCOUNT) \
X(skip_compress_suffixes, char**, NULL, BLOCK_SKIP_SUFFIXES)
+/* FastSync-only --verify-basis (protocol 2.28.0, no version bump by project
+ * decision): restores the stricter content equality on a basis hit. By
+ * default a basis hit is accepted on rsync's metadata quick-check alone (equal
+ * size plus equal mtime, or size alone under --size-only); with this flag the
+ * receiver ALSO requires the basis bytes' whole-file digest (the negotiated
+ * --checksum-choice algorithm) to equal the sender's, exactly FastSync's
+ * historical behavior. It is a receiver policy and crosses the wire so the
+ * receiver knows whether to read and hash the basis content. */
#define CONFIG_WIRE_BASIS_FIELDS(X) \
X(basis_count, int, 0, INT_BASISCOUNT) \
- X(basis_dirs, BasisDest*, NULL, BLOCK_BASIS)
+ X(basis_dirs, BasisDest*, NULL, BLOCK_BASIS) \
+ X(verify_basis, bool, false, BOOL)
#define CONFIG_WIRE_FUZZY_FIELDS(X) X(fuzzy, bool, false, BOOL)
@@ -259,9 +269,18 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
* for -n/--dry-run --delete, the destination-relative paths it WOULD have
* deleted. It is set by the client only when --stats, --progress/-P, an
* --out-format token needs a wire counter (%b/%c), or a dry-run carries
- * --delete; the transfer decision itself is unchanged. */
+ * --delete; the transfer decision itself is unchanged.
+ *
+ * --info wave (protocol 2.27.0). report_deletes tells the receiver to include
+ * the destination-relative paths it ACTUALLY removed in its terminal
+ * STATUS_STATS record (the same path-list field the dry-run would-delete report
+ * uses), so the sender can print rsync's `deleting PATH`/`*deleting` lines for a
+ * real (non-dry-run) deletion. It is set when --delete is active and any of
+ * --info=del, -i/--itemize-changes or --out-format requests per-file change
+ * output; the transfer decision itself is unchanged. */
#define CONFIG_WIRE_OUTPUT_FIELDS(X) \
- X(report_dest_info, bool, false, BOOL) X(report_stats, bool, false, BOOL)
+ X(report_dest_info, bool, false, BOOL) \
+ X(report_stats, bool, false, BOOL) X(report_deletes, bool, false, BOOL)
/* Codec-negotiation wave (protocol 2.26.0). compression_algo is the concrete
* codec the client selected for this transfer (a CompressionAlgo id) and is the
@@ -284,6 +303,20 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
#define CONFIG_WIRE_CODEC_FIELDS(X) \
X(compression_algo, int, COMPRESSION_ALGO_ZSTD, INT_COMPRESSION_ALGO)
+/* Receiver-side delete-protection filter rules (protocol 2.28.0). The sender
+ * compiles its root-level selection rules exactly as the scanner does
+ * (filter_base_build over --filter/-f/--exclude/--include/-C) and streams them
+ * as one self-describing, bounded block (count followed by per-rule records).
+ * The receiver reconstructs `protect_rules` and evaluates them against
+ * DESTINATION-ONLY entries during the --delete extras walk, so a
+ * `protect`/`P` rule protects an extra that never appeared on the sender
+ * (rsync re-derives deletion protection from the filter list; FastSync
+ * historically derived it only from the source scan). `protect_rules` is NULL
+ * on the sender and is owned/freed by the receiver Config. Bounded by
+ * MAX_FILTER_RULES and MAX_FILTER_BYTES; an unknown action/sides is a protocol
+ * error. */
+#define CONFIG_WIRE_PROTECT_FIELDS(X) X(protect_rules, FilterRuleList*, NULL, BLOCK_PROTECT_RULES)
+
/* All serialized fields, in exact wire order. Concatenating the per-segment
* lists here is what keeps the declaration order = the wire order. */
#define CONFIG_WIRE_FIELDS(X) \
@@ -306,7 +339,8 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
CONFIG_WIRE_PRIVILEGE_FIELDS(X) \
CONFIG_WIRE_COPY_AS_FIELDS(X) \
CONFIG_WIRE_OUTPUT_FIELDS(X) \
- CONFIG_WIRE_CODEC_FIELDS(X)
+ CONFIG_WIRE_CODEC_FIELDS(X) \
+ CONFIG_WIRE_PROTECT_FIELDS(X)
typedef struct Config {
/* -j/--threads=N: number of parallel scanner worker threads for the -m
@@ -410,6 +444,12 @@ typedef struct Config {
* for an unsupported checksum/compress algorithm) so main() can mirror it. */
int checksum_transfer_algo;
int cli_exit_code;
+ /* Client-only "the user explicitly chose" bits. They let the per-codec
+ * default level / checksum list be applied only when the corresponding
+ * rsync option was omitted (an explicit --compress-level / --checksum-choice
+ * always wins). Never serialized. */
+ bool compression_level_set;
+ bool checksum_choice_set;
// Issue #129: Advanced file selection. These fields are CLIENT-ONLY: they are
// never serialized to the wire (the receiver must not learn them).
@@ -424,8 +464,10 @@ typedef struct Config {
* repeated -F adds --filter='- .rsync-filter' so they are excluded too. */
int per_dir_filter_count;
bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */
- /* --no-implied-dirs: client-only. With -R + --files-from, refuse to place a
- * listed file whose ancestor directory is not itself explicitly listed. */
+ /* --no-implied-dirs: client-only. With -R, do not transfer the source
+ * metadata of the parent directories implied by a listed path; an unlisted
+ * implied parent is still created (with default attributes) so the listed
+ * file can be placed, matching rsync. */
bool no_implied_dirs;
/* -d/--dirs: client-only. Transfer the directory entries named by the
* source argument / --files-from list without recursing into contents. */
@@ -600,10 +642,14 @@ typedef struct Config {
source directory is streamed in directory order, and the receiver removes
each directory's extras when its plan arrives (during) or snapshots them
and removes them only after a successful transfer (delay). delete_after
- (and plain --delete) keep the whole-tree commit mode: extras are removed
- from a fresh end-of-transfer destination scan only after the whole transfer
- succeeded. See config_delete_timing_early()/config_delete_timing_per_dir()
- below. */
+ keeps the whole-tree commit mode: extras are removed from a fresh
+ end-of-transfer destination scan only after the whole transfer succeeded.
+ A plain --delete with no explicit timing flag defaults to delete_during on
+ the client (cli_finalize_config), matching rsync's --del default; the old
+ late-commit behavior is selected explicitly by --delete-after or the
+ FastSync-only long spelling --delete-commit (an exact alias for
+ --delete-after, mapped onto the same wire field). See
+ config_delete_timing_early()/config_delete_timing_per_dir() below. */
/* partial_dir */
// PR #174: Partial transfer resumption
/* suffix */
@@ -988,11 +1034,51 @@ typedef struct Config {
* boundary, and the strict same-version handshake (config_receive rejects a
* mismatched version before parsing anything else) keeps mixed deployments from
* ever reaching that state. */
-#define PROTOCOL_VERSION "2.26.0"
+/* (7) --info=del report (protocol 2.27.0): the config frame gains one trailing
+ * bool, report_deletes, appended after report_stats. When set, the receiver
+ * lists the paths it actually removed in the terminal STATUS_STATS path list
+ * (the same count-delimited list the -n/--dry-run would-delete report uses), so
+ * the sender can print rsync's `deleting PATH` lines for a real deletion. No
+ * change to the fixed STATUS_STATS record itself; only a new trailing config
+ * bool, which still requires the version bump for the strict lockstep. */
+/* (8) --stats receiver-observed counters (protocol 2.28.0): the fixed
+ * STATUS_STATS record grows from three counters to eight. The receiver now
+ * reports the bytes it literally stored (`literal_data`) and the count of
+ * destination entries it newly CREATED, split by type
+ * (reg/dir/link/special), so the sender can print rsync's exact
+ * `Number of created files: N (reg: X, dir: Y, link: Z, special: W)` line and
+ * an exact `Literal data` total even for delta transfers. The config-frame
+ * LAYOUT is unchanged (no new config field), but the STATUS_STATS body grows,
+ * so a 2.27 peer that does not consume the five new fixed-width counters would
+ * desynchronize on the trailing would-delete path list; the strict
+ * same-version handshake (config_receive rejects a mismatched version before
+ * parsing anything else) keeps mixed deployments from ever reaching that
+ * state. */
+/* (9) Receiver-side delete protection (still protocol 2.28.0): the config frame
+ * gains one trailing self-describing block carrying the sender's compiled base
+ * filter rules so the receiver can protect DESTINATION-ONLY entries from
+ * --delete with `protect`/`risk` rules (rsync parity). The block appends after
+ * compression_algo; see CONFIG_WIRE_PROTECT_FIELDS. */
+#define PROTOCOL_VERSION "2.28.0"
#define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024)
/* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */
#define MAX_BASIS_DIRS 64
+/* Bounds on the received receiver-side delete-protection rule block. The rule
+ * count and the aggregate pattern+owner bytes are each capped so a hostile
+ * peer cannot pin unbounded pre-auth memory; both are validated strictly on
+ * receive (alongside the per-string ConfigStringBudget). */
+/* A peer may supply protect rules; cap the list so a crafted config cannot make
+ * the receiver's delete walk evaluate an unbounded number of glob patterns per
+ * destination entry (glob_match is O(pattern x path)). 1024 is far above any
+ * legitimate selection. */
+#define MAX_FILTER_RULES 1024
+#define MAX_FILTER_BYTES (256 * 1024)
+/* glob_match's DP is capped at 64 Mi work units; a pattern longer than this
+ * could exceed the cap against a PATH_MAX path and silently stop matching,
+ * leaving a protect rule inert. Reject such a rule at receive time. */
+#define MAX_PROTECT_PATTERN_LEN 8192
+
/* Upper bound on the number of --skip-compress suffixes accepted from the wire.
* Each suffix is an independent wire string (up to MAX_STRING_SIZE = 64 KiB), so
* without this a hostile pre-auth client could otherwise retain
@@ -1094,8 +1180,11 @@ bool config_delete_timing_early(const Config* config);
* commits them only after a fully-successful transfer (delay). */
bool config_delete_timing_per_dir(const Config* config);
/* Delete-timing sanity: with deletion enabled at most one timing flag may be
- * set (none = the default delete-after commit timing); without deletion no
- * timing flag may be set (each timing flag implies --delete). */
+ * set; without deletion no timing flag may be set (each timing flag implies
+ * --delete). A plain --delete is normalized to delete_during by
+ * cli_finalize_config on the client, so a transmitted use_delete config always
+ * carries exactly one timing; the zero-timing case remains valid only for a
+ * config that has not been through the CLI. */
bool config_has_valid_delete_timing(const Config* config);
/* Single source of truth for the cross-field ("combination") invariants a
diff --git a/src/shared/delete_plan.c b/src/shared/delete_plan.c
index f5aea91..8aa96f7 100644
--- a/src/shared/delete_plan.c
+++ b/src/shared/delete_plan.c
@@ -331,6 +331,8 @@ static int send_plan_node(int fd, DeletePlanSender* sender, PlanNode* node) {
return -1;
sender->config_sent = true;
}
+ if (!send_int(fd, 1)) /* apply = true */
+ return -1;
if (!send_wire_str(fd, node->dir))
return -1;
if (send_str_section(fd, node->dirs) != 0 || send_str_section(fd, node->files) != 0)
@@ -339,6 +341,29 @@ static int send_plan_node(int fd, DeletePlanSender* sender, PlanNode* node) {
return 0;
}
+/* Transmit the one-shot per-run config block (protected prefixes, size-pruned
+ * mirrors, --delete-missing-args exact paths) on its own carrier frame, with
+ * apply=false so the receiver consumes the config but walks nothing. This is
+ * how the config still reaches the receiver when the scope allows no directory
+ * plan at all (a --files-from list of bare files synchronizes no directory):
+ * without it, the missing-args exact deletions would be lost. Idempotent. */
+static int send_config_only(int fd, DeletePlanSender* sender) {
+ if (!sender || sender->config_sent)
+ return 0;
+ if (!send_status(fd, STATUS_DELETE_PLAN) || !send_int(fd, 1))
+ return -1;
+ if (send_str_section(fd, sender->protected_prefixes) != 0 ||
+ send_str_section(fd, sender->size_skipped) != 0 ||
+ send_str_section(fd, sender->missing_args) != 0)
+ return -1;
+ sender->config_sent = true;
+ if (!send_int(fd, 0)) /* apply = false */
+ return -1;
+ if (!send_wire_str(fd, ".") || !send_int(fd, 0) || !send_int(fd, 0))
+ return -1;
+ return 0;
+}
+
static int send_prefix_plan(int fd, DeletePlanSender* sender, const char* dir) {
PlanNode* node = plan_find(sender, dir);
if (!node || node->sent)
@@ -354,6 +379,10 @@ int delete_plan_send_root(int fd, DeletePlanSender* sender) {
const char* root = sender->walk_root ? sender->walk_root : ".";
if (!plan_ensure(sender, root))
return -1;
+ /* Put the config block on the wire first, on its own carrier frame, so the
+ receiver always sees it even when the scope permits no directory plan. */
+ if (send_config_only(fd, sender) != 0)
+ return -1;
return send_prefix_plan(fd, sender, root);
}
@@ -412,6 +441,17 @@ struct DeletePlanSession {
bool dry_run;
size_t max_delete;
size_t deleted;
+ /* Removals charged against --max-delete. The budget is charged on ACTUAL
+ removals (an unlink/rmdir that succeeded), matching rsync: a snapshotted
+ entry that fails removal consumes nothing, so a later extra is still
+ deleted. `planned` and `deleted` advance together for the inline paths and
+ `apply_missing`; `deleted` is the reported count. */
+ size_t planned;
+ /* Hard bound on the deferred snapshot list. Because the budget is no longer
+ charged at snapshot time, this independent cap keeps a huge destination
+ from growing the list without limit (it matches the receiver's overall
+ deletion bound). */
+ size_t defer_cap;
size_t skipped;
bool limit_hit;
bool limit_logged;
@@ -421,8 +461,16 @@ struct DeletePlanSession {
ArrayList* size_skipped;
ArrayList* missing;
ArrayList* deferred;
+ DeletePathObserver observer;
+ void* observer_context;
};
+/* Report one path the session truly removed (no-op without an observer). */
+static void notify_deleted(DeletePlanSession* session, const char* rel) {
+ if (session && session->observer && rel)
+ session->observer(session->observer_context, rel);
+}
+
DeletePlanSession* delete_plan_session_create(const Config* config) {
if (!config)
return NULL;
@@ -435,6 +483,7 @@ DeletePlanSession* delete_plan_session_create(const Config* config) {
config->max_delete >= 0 && (size_t)config->max_delete < DELETE_PLAN_SERVER_LIMIT;
session->max_delete =
user_limited ? (size_t)config->max_delete : (size_t)DELETE_PLAN_SERVER_LIMIT;
+ session->defer_cap = DELETE_PLAN_SERVER_LIMIT;
session->protected_prefixes = array_list_create(free);
session->size_skipped = array_list_create(free);
session->missing = array_list_create(free);
@@ -526,12 +575,18 @@ static int open_plan_dir(const Config* config, const char* dir) {
typedef struct PlanSkips {
DeleteSkipEntry* entries;
int count;
+ /* Receiver-side delete-protection rules received on the config frame (NULL
+ when the sender sent none). Evaluated per extra so a protect/risk rule is
+ honored under --delete-during/--delete-delay exactly like the whole-tree
+ commit walker. */
+ const FilterRuleList* protect_rules;
} PlanSkips;
static bool build_plan_skips(const Config* config, const DeletePlanSession* session,
PlanSkips* out) {
out->entries = NULL;
out->count = 0;
+ out->protect_rules = config->protect_rules;
int count = (config->delay_updates ? 1 : 0) + config->basis_count +
session->protected_prefixes->size + session->size_skipped->size;
if (count == 0)
@@ -565,7 +620,7 @@ static bool build_plan_skips(const Config* config, const DeletePlanSession* sess
}
static bool budget_available(const DeletePlanSession* session) {
- return session->deleted < session->max_delete;
+ return session->planned < session->max_delete;
}
static void note_skipped(DeletePlanSession* session) {
@@ -579,8 +634,15 @@ static void log_deleted(const char* rel) {
free(escaped);
}
-/* Append a snapshot path for --delete-delay. */
+/* Append a snapshot path for --delete-delay. The budget is NOT charged here:
+ * the remover charges --max-delete only when a path is actually unlinked (see
+ * apply_deferred_path), so a snapshotted entry that survives ENOTEMPTY cannot
+ * deny budget to a later extra. The independent `defer_cap` bounds the list. */
static bool defer_add(DeletePlanSession* session, const char* rel) {
+ if ((size_t)session->deferred->size >= session->defer_cap) {
+ note_skipped(session);
+ return true;
+ }
char* copy = str_dup(rel);
if (!copy)
return false;
@@ -588,7 +650,6 @@ static bool defer_add(DeletePlanSession* session, const char* rel) {
free(copy);
return false;
}
- session->deleted++;
return true;
}
@@ -621,19 +682,21 @@ static bool process_extra_dir(int dirfd, const char* name, const char* child_rel
return false;
if (survives)
return true;
- if (!budget_available(session)) {
- note_skipped(session);
- return true;
- }
if (session->defer && !force_now) {
if (!defer_add(session, child_rel))
return false;
*removed = true;
return true;
}
+ if (!budget_available(session)) {
+ note_skipped(session);
+ return true;
+ }
if (unlinkat(dirfd, name, AT_REMOVEDIR) == 0) {
session->deleted++;
+ session->planned++;
log_deleted(child_rel);
+ notify_deleted(session, child_rel);
*removed = true;
return true;
}
@@ -648,16 +711,18 @@ static bool process_extra_dir(int dirfd, const char* name, const char* child_rel
static bool process_extra_file(int dirfd, const char* name, const char* child_rel, bool force_now,
DeletePlanSession* session) {
+ if (session->defer && !force_now) {
+ return defer_add(session, child_rel);
+ }
if (!budget_available(session)) {
note_skipped(session);
return true;
}
- if (session->defer && !force_now) {
- return defer_add(session, child_rel);
- }
if (unlinkat(dirfd, name, 0) == 0) {
session->deleted++;
+ session->planned++;
log_deleted(child_rel);
+ notify_deleted(session, child_rel);
} else if (errno != ENOENT) {
return false;
}
@@ -703,6 +768,10 @@ static bool process_children(int dirfd, const char* dir_rel, const ArrayList* ke
bool is_dir = S_ISDIR(st.st_mode);
bool in_keep_dirs = is_dir && list_contains_str(keep_dirs, entry->d_name);
bool in_keep_files = !is_dir && list_contains_str(keep_files, entry->d_name);
+ bool rule_protected =
+ skips->protect_rules &&
+ filter_rules_apply_side(skips->protect_rules, child_rel, entry->d_name, is_dir,
+ FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT;
if (in_keep_dirs) {
local_survives = true;
} else if (keep_dirs && !is_dir && list_contains_str(keep_dirs, entry->d_name)) {
@@ -720,11 +789,18 @@ static bool process_children(int dirfd, const char* dir_rel, const ArrayList* ke
else if (!removed)
local_survives = true;
} else if (is_dir) {
- bool removed = false;
- if (!process_extra_dir(dirfd, entry->d_name, child_rel, force_now, skips, session, &removed))
- operation_ok = false;
- else if (!removed)
+ if (rule_protected) {
local_survives = true;
+ } else {
+ bool removed = false;
+ if (!process_extra_dir(dirfd, entry->d_name, child_rel, force_now, skips, session,
+ &removed))
+ operation_ok = false;
+ else if (!removed)
+ local_survives = true;
+ }
+ } else if (rule_protected) {
+ local_survives = true;
} else {
if (!process_extra_file(dirfd, entry->d_name, child_rel, force_now, session))
operation_ok = false;
@@ -768,13 +844,15 @@ static bool apply_missing(DeletePlanSession* session, const Config* config) {
return true;
DeleteManifest manifest = {
.keeps = NULL, .protected = NULL, .missing = session->missing, .dirs = NULL};
- size_t remaining = budget_available(session) ? session->max_delete - session->deleted : 0;
+ size_t remaining = budget_available(session) ? session->max_delete - session->planned : 0;
size_t deleted = 0;
size_t skipped = 0;
bool limit = false;
- bool ok = manifest_delete_missing_args_limited(config, &manifest, remaining, &deleted, &skipped,
- &limit);
+ bool ok = manifest_delete_missing_args_limited_observed(config, &manifest, remaining, &deleted,
+ &skipped, &limit, session->observer,
+ session->observer_context);
session->deleted += deleted;
+ session->planned += deleted;
session->skipped += skipped;
if (limit)
session->limit_hit = true;
@@ -801,6 +879,14 @@ int delete_plan_session_receive(DeletePlanSession* session, const Config* config
}
session->config_seen = true;
}
+ /* apply=false is the config-only carrier frame: the receiver consumes the
+ config (and the missing-args exact deletions) but must not walk any
+ directory. Every real plan carries apply=true. */
+ int apply;
+ if (!receive_int(fd, &apply) || (apply != 0 && apply != 1)) {
+ send_status(fd, STATUS_ERROR);
+ return -1;
+ }
char* dir = receive_wire_str(fd);
ArrayList* dirs = array_list_create(free);
ArrayList* files = array_list_create(free);
@@ -818,7 +904,7 @@ int delete_plan_session_receive(DeletePlanSession* session, const Config* config
if (!session->dry_run && enabled) {
if (!session->defer && !apply_missing(session, config))
ok = false;
- if (ok && !apply_plan_dir(session, config, dir, dirs, files))
+ if (ok && apply && !apply_plan_dir(session, config, dir, dirs, files))
ok = false;
}
free(dir);
@@ -837,9 +923,10 @@ int delete_plan_session_receive(DeletePlanSession* session, const Config* config
}
/* Apply one snapshotted --delete-delay path (post-order: children precede their
- * parent directory). */
+ * parent directory). A directory that is still present is re-scanned so content
+ * created after the plan is removed too; every actual removal charges
+ * --max-delete. */
static bool apply_deferred_path(DeletePlanSession* session, const Config* config, const char* rel) {
- (void)session;
char* full = path_cat(config->receive_root_directory, rel);
if (!full)
return false;
@@ -852,22 +939,86 @@ static bool apply_deferred_path(DeletePlanSession* session, const Config* config
}
struct stat st;
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) {
- bool absent = errno == ENOENT;
+ bool absent = errno == ENOENT || errno == ENOTDIR;
close(parent_fd);
free(leaf);
return absent;
}
- int rc;
- if (S_ISDIR(st.st_mode))
- rc = unlinkat(parent_fd, leaf, AT_REMOVEDIR);
- else
- rc = unlinkat(parent_fd, leaf, 0);
- bool ok = rc == 0 || errno == ENOENT || errno == ENOTEMPTY || errno == EEXIST;
- if (rc == 0)
+ if (S_ISDIR(st.st_mode)) {
+ if (!budget_available(session)) {
+ note_skipped(session);
+ close(parent_fd);
+ free(leaf);
+ return true;
+ }
+ int dirfd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
+ if (dirfd < 0) {
+ bool absent = errno == ENOENT || errno == ENOTDIR;
+ close(parent_fd);
+ free(leaf);
+ return absent;
+ }
+ PlanSkips skips;
+ if (!build_plan_skips(config, session, &skips)) {
+ close(dirfd);
+ close(parent_fd);
+ free(leaf);
+ return false;
+ }
+ bool survives = false;
+ bool ok = process_children(dirfd, rel, NULL, NULL, false, true, &skips, session, &survives);
+ free(skips.entries);
+ close(dirfd);
+ if (!ok) {
+ close(parent_fd);
+ free(leaf);
+ return false;
+ }
+ if (!survives) {
+ if (!budget_available(session)) {
+ note_skipped(session);
+ } else if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) {
+ session->deleted++;
+ session->planned++;
+ log_deleted(rel);
+ notify_deleted(session, rel);
+ } else if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST) {
+ close(parent_fd);
+ free(leaf);
+ return false;
+ }
+ }
+ close(parent_fd);
+ free(leaf);
+ return true;
+ }
+ if (!budget_available(session)) {
+ note_skipped(session);
+ close(parent_fd);
+ free(leaf);
+ return true;
+ }
+ if (unlinkat(parent_fd, leaf, 0) == 0) {
+ session->deleted++;
+ session->planned++;
log_deleted(rel);
+ notify_deleted(session, rel);
+ } else if (errno != ENOENT) {
+ close(parent_fd);
+ free(leaf);
+ return false;
+ }
close(parent_fd);
free(leaf);
- return ok;
+ return true;
+}
+
+void delete_plan_session_set_delete_observer(DeletePlanSession* session,
+ DeletePathObserver observer, void* context) {
+ if (!session)
+ return;
+ session->observer = observer;
+ session->observer_context = context;
}
DeleteCommitResult delete_plan_session_commit(DeletePlanSession* session, const Config* config) {
diff --git a/src/shared/delete_plan.h b/src/shared/delete_plan.h
index 0f4644b..7e36eb4 100644
--- a/src/shared/delete_plan.h
+++ b/src/shared/delete_plan.h
@@ -5,6 +5,7 @@
#include "config.h"
#include "file_receive.h"
#include "protocol.h"
+#include "utils.h"
#include
/* Per-directory delete plans (protocol 2.24.0).
@@ -47,11 +48,14 @@ void delete_plan_sender_finalize(DeletePlanSender* sender, const ArrayList* sync
Directory keep entries do not count, so an I/O error that hid every file
still refuses to delete. */
bool delete_plan_sender_empty(const DeletePlanSender* sender);
-/* Attach the global config sections advertised on the first plan frame. */
+/* Attach the global config sections advertised on the first plan frame. The
+ * block is always transmitted by delete_plan_send_root(), on a config-only
+ * carrier frame when the scope allows no directory plan. */
void delete_plan_sender_set_config(DeletePlanSender* sender, const ArrayList* protected_prefixes,
const ArrayList* size_skipped, const ArrayList* missing_args);
/* Send the root plan (even before any data, so root extras are handled like
- * rsync's first generator directory). Returns -1 on I/O error. */
+ * rsync's first generator directory), after transmitting the per-run config
+ * block on its own carrier frame. Returns -1 on I/O error. */
int delete_plan_send_root(int fd, DeletePlanSender* sender);
/* Send the plans for every ancestor of `path` (root-first) and, when is_dir,
* for `path` itself; already-sent plans are skipped. */
@@ -76,8 +80,16 @@ int delete_plan_session_receive(DeletePlanSession* session, const Config* config
DeleteCommitResult delete_plan_session_commit(DeletePlanSession* session, const Config* config);
/* True once the shared --max-delete budget stopped part of a deletion. */
bool delete_plan_session_limit_reached(const DeletePlanSession* session);
-/* Number of destination entries the session's plans removed (or, for
- --delete-delay, snapshotted for removal), for the end-of-transfer stats. */
+/* Number of destination entries the session actually removed, for the
+ end-of-transfer stats. For --delete-delay this excludes a snapshotted entry
+ that survived (e.g. a refilled directory that failed ENOTEMPTY), even though
+ that entry already consumed --max-delete budget at snapshot time. */
size_t delete_plan_session_deleted(const DeletePlanSession* session);
+/* Install an observer invoked for every destination-relative path the session
+ truly removes (including the deferred --delete-delay commit), so the receiver
+ can report rsync's `deleting PATH` lines through the terminal STATUS_STATS
+ record. Pass NULL/0 to clear. */
+void delete_plan_session_set_delete_observer(DeletePlanSession* session,
+ DeletePathObserver observer, void* context);
#endif
diff --git a/src/shared/file.c b/src/shared/file.c
index 740763f..7e44023 100644
--- a/src/shared/file.c
+++ b/src/shared/file.c
@@ -15,6 +15,7 @@
#include
#include "data.h"
+#include "checksum.h"
#include "delta.h"
#include "file.h"
#include "file_store.h"
@@ -24,6 +25,13 @@
#include "utils.h"
#include "protocol.h"
#include "xattr.h"
+#include
+#include
+
+/* Files larger than this are not loaded whole for transfer (the sender streams
+ * them); a whole-file digest is computed from the path instead. Kept in sync
+ * with the sender's streaming threshold. */
+#define STREAM_THRESHOLD (64ULL * 1024 * 1024)
static bool write_all(int fd, const void* data, unsigned long long size) {
const unsigned char* p = data;
@@ -39,6 +47,31 @@ static bool write_all(int fd, const void* data, unsigned long long size) {
return true;
}
+/* Streaming copy of an open source descriptor into the just-created destination
+ `fd` (already at offset 0). Used by the --copy-dest basis install so a basis
+ larger than any in-memory whole-file bound still materializes without
+ buffering the entire file. `expected_size` is the caller-verified basis
+ size; the copy must produce exactly that many bytes (a short source is a hard
+ error, never a silently truncated destination). The final ftruncate drops
+ any residual tail a raced-in longer source might have left. */
+static bool copy_fd_all(int dst_fd, int src_fd, unsigned long long expected_size) {
+ unsigned char buf[1 << 20];
+ unsigned long long done = 0;
+ while (done < expected_size) {
+ unsigned long long remaining = expected_size - done;
+ size_t want = remaining < sizeof(buf) ? (size_t)remaining : sizeof(buf);
+ ssize_t n = read(src_fd, buf, want);
+ if (n < 0 && errno == EINTR)
+ continue;
+ if (n <= 0)
+ return false;
+ if (!write_all(dst_fd, buf, (unsigned long long)n))
+ return false;
+ done += (unsigned long long)n;
+ }
+ return ftruncate(dst_fd, (off_t)expected_size) == 0;
+}
+
/* Preallocate `size` bytes on `fd` before any data is written (--preallocate).
* fallocate(2) reserves real disk blocks, so an out-of-space condition
* (ENOSPC/EDQUOT) surfaces up front instead of partway through a transfer;
@@ -140,6 +173,12 @@ bool file_checksum(File* file, ChecksumAlgo algo, uint64_t seed, uint8_t* out, s
if (file->data->size == 0) {
return checksum_digest(algo, seed, "", 0, out, out_capacity, out_len);
}
+ /* A streamed source (data not loaded) may exceed any in-memory whole-file
+ bound; hash it from the file path in bounded buffers instead of forcing a
+ full load. This is the same digest the receiver recomputes on the basis. */
+ if (!file->data->data && file->path && file->data->size > STREAM_THRESHOLD &&
+ checksum_digest_file(algo, seed, file->path, out, out_capacity, out_len))
+ return true;
if (!file->data->data && !file_load_data(file))
return false;
return checksum_digest(algo, seed, file->data->data, file->data->size, out, out_capacity,
@@ -176,6 +215,7 @@ File* file_create(const char* path) {
file->is_dir = false;
file->dir_time_only = false;
file->basis_link = NULL;
+ file->basis_copy = NULL;
file->link_group = 0;
file->link_first = false;
file->hardlink_target = NULL;
@@ -187,6 +227,7 @@ File* file_create(const char* path) {
file->xattrs = NULL;
file->dest_state = (OutputDestState){0};
file->matched_bytes = 0;
+ file->literal_bytes = 0;
return file;
}
@@ -204,6 +245,8 @@ void file_destroy(void* item) {
file->send_path = NULL;
free(file->basis_link);
file->basis_link = NULL;
+ free(file->basis_copy);
+ file->basis_copy = NULL;
free(file->hardlink_target);
file->hardlink_target = NULL;
free(file->symlink_target);
@@ -614,6 +657,103 @@ static int open_dir_beneath_root(const char* resolved, const char* root) {
}
int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs) {
+ return file_open_secure_parent_counted(path, leaf_out, create_dirs, NULL, NULL);
+}
+
+/* The logical transfer root expressed in the same coordinate as the secure
+ * parent walk's `rel_buf` (relative to the authorized root, with a leading
+ * '/'), used as the floor at or below which a created directory is a real
+ * file-list entry. The on-disk transfer root is the receive root joined to the
+ * wire path; the mirror scaffolding above it (the absolute source path below
+ * the destination root) is not an rsync entry. Returns an allocated string or
+ * NULL (count every created component). */
+static char* transfer_root_floor(const Config* config) {
+ if (!config || !config->send_directory || config->send_directory[0] == '\0')
+ return NULL;
+ const char* spec = config->send_directory;
+ const char* after = spec;
+ if (spec[0] == '.' && spec[1] == '/') {
+ after = spec + 2;
+ } else {
+ const char* cut = strstr(spec, "/./");
+ if (cut)
+ after = cut + 3;
+ }
+ while (*after == '/')
+ after++;
+ char* wire_root = str_dup(after);
+ if (!wire_root)
+ return NULL;
+ size_t wlen = strlen(wire_root);
+ while (wlen > 0 && wire_root[wlen - 1] == '/')
+ wire_root[--wlen] = '\0';
+ if (wlen == 0) {
+ free(wire_root);
+ return NULL;
+ }
+ char* disk_root = config->receive_root_directory
+ ? path_cat(config->receive_root_directory, wire_root)
+ : str_dup(wire_root);
+ free(wire_root);
+ if (!disk_root)
+ return NULL;
+ const char* root_path = utils_get_authorized_root_path();
+ const char* floor = disk_root;
+ if (root_path && root_path[0] == '/') {
+ size_t rl = strlen(root_path);
+ while (rl > 0 && root_path[rl - 1] == '/')
+ rl--;
+ if (strncmp(disk_root, root_path, rl) == 0 && (disk_root[rl] == '/' || disk_root[rl] == '\0'))
+ floor = disk_root + rl;
+ }
+ while (*floor == '/')
+ floor++;
+ char* out = str_dup(floor);
+ free(disk_root);
+ if (!out)
+ return NULL;
+ if (out[0] == '\0') {
+ free(out);
+ return NULL;
+ }
+ return out;
+}
+
+/* A created parent component counts toward `Number of created files` only when
+ * its receive-root-relative path is at or below the logical transfer root
+ * (`count_floor`). The transfer root itself corresponds to rsync's `.` entry
+ * (created on a fresh destination, pre-existing otherwise); the mirror
+ * scaffolding above it is FastSync's absolute-path layout, not an rsync entry. */
+static bool created_dir_counts(const char* count_floor, const char* rel_buf,
+ const char* component) {
+ if (!count_floor)
+ return true;
+ char candidate[PATH_MAX];
+ int n = snprintf(candidate, sizeof(candidate), "%s/%s", rel_buf, component);
+ if (n < 0 || (size_t)n >= sizeof(candidate))
+ return false;
+ const char* cand = candidate;
+ while (*cand == '/')
+ cand++;
+ size_t fl = strlen(count_floor);
+ if (strncmp(cand, count_floor, fl) != 0)
+ return false;
+ return cand[fl] == '\0' || cand[fl] == '/';
+}
+
+/* Public wrapper for the receiver's created-directory accounting: the logical
+ * transfer root expressed receive-root-relative, or NULL when the wire paths
+ * carry no mirror scaffolding above it (--relative and --files-from, whose
+ * paths are already relative to the transfer root). The caller frees a
+ * non-NULL result. */
+char* file_transfer_root_floor(const Config* config) {
+ if (!config || config->relative || config->files_from_set != NULL)
+ return NULL;
+ return transfer_root_floor(config);
+}
+
+int file_open_secure_parent_counted(const char* path, char** leaf_out, bool create_dirs,
+ unsigned* dirs_created, const char* count_floor) {
char* copy = str_dup(path);
if (!copy)
return -1;
@@ -674,6 +814,14 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs)
if (next < 0 && create_dirs && errno == ENOENT) {
bool created = mkdirat(fd, component, (mode_t)(0777 & ~(mode_t)file_process_umask())) == 0;
if (created || errno == EEXIST) {
+ /* Protocol 2.28.0: only directories the logical file list would
+ create count toward `Number of created files`; the mirror
+ scaffolding above the transfer root (e.g. the absolute source path
+ under the destination root) is not an rsync entry. `count_floor`
+ is a receive-root-relative prefix that must be reached before a
+ created component is counted. */
+ if (created && dirs_created && created_dir_counts(count_floor, rel_buf, component))
+ (*dirs_created)++;
/* P7 Wave E: --copy-as owns EVERY entry, including the intermediate
directories this walk creates implicitly. Its target ids are a
global policy, so they are available here without per-entry source
@@ -807,6 +955,25 @@ bool file_ensure_directory_secure(const char* path) {
} else if (errno == EEXIST) {
dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
}
+ } else if (dir_fd < 0 && errno == ENOTDIR) {
+ /* rsync replaces a destination non-directory (regular file) with an
+ incoming directory. Confined to the already-opened secure parent fd:
+ the leaf is unlinked by name (never followed) and only a non-directory
+ is ever removed, so this cannot escape the authorized root or remove a
+ pre-existing directory tree. A symlink is left alone (openat with
+ O_NOFOLLOW reports ELOOP, which takes no branch here), since replacing
+ it is not required for FastSync's transferred directories and keeps
+ --keep-dirlinks semantics untouched. */
+ struct stat leaf_st;
+ if (fstatat(parent_fd, leaf, &leaf_st, AT_SYMLINK_NOFOLLOW) == 0 && !S_ISDIR(leaf_st.st_mode) &&
+ !S_ISLNK(leaf_st.st_mode)) {
+ if (unlinkat(parent_fd, leaf, 0) == 0) {
+ if (mkdirat(parent_fd, leaf, (mode_t)(0777 & ~(mode_t)file_process_umask())) == 0)
+ created = true;
+ /* On failure dir_fd stays < 0 below, so the caller still sees it. */
+ dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
+ }
+ }
}
bool ok = dir_fd >= 0;
/* --copy-as owns a directory this call just created (the final component;
@@ -1007,15 +1174,14 @@ static void restore_extra_fd(int fd, const FileMetadata* metadata, const FileXat
}
}
-static bool file_to_disk_secure_impl(const char* path, const void* data,
- unsigned long long data_size, bool inplace, bool sparse,
- bool preallocate, const FileMetadata* metadata,
- FileAttrPolicy policy, bool update, bool no_replace,
- bool use_fsync, const char* temp_dir,
- const FileXattrList* xattrs, bool fake_super,
- bool keep_partial) {
+static bool
+file_to_disk_secure_impl(const char* path, const void* data, unsigned long long data_size,
+ bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
+ FileAttrPolicy policy, bool update, bool no_replace, bool use_fsync,
+ const char* temp_dir, const FileXattrList* xattrs, bool fake_super,
+ bool keep_partial, unsigned* dirs_created, const char* count_floor) {
char* leaf = NULL;
- int dirfd = file_open_secure_parent(path, &leaf, true);
+ int dirfd = file_open_secure_parent_counted(path, &leaf, true, dirs_created, count_floor);
if (dirfd < 0)
return false;
int fd = -1;
@@ -1306,7 +1472,7 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
"non-atomic copy into the destination directory");
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
policy, update, no_replace, use_fsync, NULL, xattrs, fake_super,
- keep_partial);
+ keep_partial, dirs_created, count_floor);
}
return ok;
}
@@ -1315,7 +1481,8 @@ bool file_to_disk_secure(const char* path, const void* data, unsigned long long
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
FileAttrPolicy policy, const char* temp_dir) {
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
- policy, false, false, false, temp_dir, NULL, false, false);
+ policy, false, false, false, temp_dir, NULL, false, false, NULL,
+ NULL);
}
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
@@ -1323,7 +1490,8 @@ bool file_to_disk_secure_update(const char* path, const void* data, unsigned lon
const FileMetadata* metadata, FileAttrPolicy policy,
const char* temp_dir) {
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
- policy, true, false, false, temp_dir, NULL, false, false);
+ policy, true, false, false, temp_dir, NULL, false, false, NULL,
+ NULL);
}
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
@@ -1331,7 +1499,8 @@ bool file_to_disk_secure_with_fsync(const char* path, const void* data,
bool preallocate, const FileMetadata* metadata,
FileAttrPolicy policy, bool use_fsync, const char* temp_dir) {
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
- policy, false, false, use_fsync, temp_dir, NULL, false, false);
+ policy, false, false, use_fsync, temp_dir, NULL, false, false,
+ NULL, NULL);
}
bool file_to_disk_secure_no_replace(const char* path, const void* data,
@@ -1339,7 +1508,8 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
const FileMetadata* metadata, FileAttrPolicy policy,
const char* temp_dir) {
return file_to_disk_secure_impl(path, data, data_size, false, sparse, preallocate, metadata,
- policy, false, true, false, temp_dir, NULL, false, false);
+ policy, false, true, false, temp_dir, NULL, false, false, NULL,
+ NULL);
}
/* Receiver write-path variant that also applies the per-file xattrs (-X/-A)
@@ -1352,9 +1522,21 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
bool fake_super, bool keep_partial, const char* temp_dir) {
+ return file_to_disk_secure_attrs_counted(path, data, data_size, inplace, sparse, preallocate,
+ metadata, policy, update, no_replace, use_fsync, xattrs,
+ fake_super, keep_partial, temp_dir, NULL, NULL);
+}
+
+bool file_to_disk_secure_attrs_counted(const char* path, const void* data,
+ unsigned long long data_size, bool inplace, bool sparse,
+ bool preallocate, const FileMetadata* metadata,
+ FileAttrPolicy policy, bool update, bool no_replace,
+ bool use_fsync, const FileXattrList* xattrs, bool fake_super,
+ bool keep_partial, const char* temp_dir,
+ unsigned* dirs_created, const char* count_floor) {
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
policy, update, no_replace, use_fsync, temp_dir, xattrs,
- fake_super, keep_partial);
+ fake_super, keep_partial, dirs_created, count_floor);
}
/* Atomic --link-dest install. The destination is replaced (via a temporary
@@ -1372,16 +1554,176 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
* basis). Likewise `xattrs`/`fake_super` are applied only on the copy
* fallback, so a fallback copy preserves the per-file attributes instead of
* silently dropping them. */
+/* Streaming --copy-dest basis install: atomically materialize `path` from the
+ * bytes of `basis_path` without holding the file in memory, so a basis larger
+ * than any whole-file bound still works. Mirrors the ordinary secure store
+ * path (confined parent walk, temp + rename, --update/--ignore-existing/
+ * --preallocate/--temp-dir) but sources the data from the basis descriptor
+ * rather than a caller buffer, and applies the SOURCE metadata (rsync copies
+ * then fixes attributes). A hard-link install that falls back to a byte copy
+ * also routes through here when the caller supplies the basis path. */
+static bool file_copy_basis_stream_impl(const char* path, const char* basis_path,
+ unsigned long long expected_size, bool preallocate,
+ const FileMetadata* metadata, FileAttrPolicy policy,
+ bool update, bool no_replace, bool use_fsync,
+ const FileXattrList* xattrs, bool fake_super,
+ const char* temp_dir, unsigned* dirs_created,
+ const char* count_floor) {
+ if (!path || !basis_path)
+ return false;
+ char* leaf = NULL;
+ int dirfd = file_open_secure_parent_counted(path, &leaf, true, dirs_created, count_floor);
+ if (dirfd < 0)
+ return false;
+
+ char* basis_leaf = NULL;
+ int basis_dirfd = file_open_secure_parent(basis_path, &basis_leaf, false);
+ int src_fd = -1;
+ if (basis_dirfd >= 0 && basis_leaf != NULL) {
+ /* O_NONBLOCK rejects a raced-in FIFO without blocking; the S_ISREG gate
+ below is the real type check. */
+ src_fd = openat(basis_dirfd, basis_leaf, O_RDONLY | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK);
+ struct stat src_st;
+ if (src_fd >= 0 && (fstat(src_fd, &src_st) != 0 || !S_ISREG(src_st.st_mode))) {
+ close(src_fd);
+ src_fd = -1;
+ }
+ }
+ if (basis_dirfd >= 0)
+ close(basis_dirfd);
+ free(basis_leaf);
+ if (src_fd < 0) {
+ close(dirfd);
+ free(leaf);
+ return false;
+ }
+
+ struct stat destination_stat;
+ bool destination_is_regular = fstatat(dirfd, leaf, &destination_stat, AT_SYMLINK_NOFOLLOW) == 0 &&
+ S_ISREG(destination_stat.st_mode);
+ if (update && metadata && destination_is_regular && stat_is_newer(&destination_stat, metadata)) {
+ close(src_fd);
+ close(dirfd);
+ free(leaf);
+ return true;
+ }
+ if (no_replace && file_path_exists_secure(path)) {
+ close(src_fd);
+ close(dirfd);
+ free(leaf);
+ return true;
+ }
+
+ int scratch_dirfd = -1;
+ if (temp_dir) {
+ scratch_dirfd = file_open_temp_dir(temp_dir);
+ if (scratch_dirfd < 0) {
+ int saved_errno = errno;
+ log_message(LOG_LEVEL_ERROR,
+ "--temp-dir '%s' could not be opened (rsync requires it to already exist): %s",
+ temp_dir, strerror(saved_errno));
+ close(src_fd);
+ close(dirfd);
+ free(leaf);
+ return false;
+ }
+ }
+
+ int target_dirfd = scratch_dirfd >= 0 ? scratch_dirfd : dirfd;
+ int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%llu", leaf, (long)getpid(), ~0ULL);
+ char* tmp = NULL;
+ bool ok = false;
+ if (tmp_size >= 0)
+ tmp = malloc((size_t)tmp_size + 1);
+ if (tmp) {
+ for (unsigned int i = 0; i < 100 && !ok; ++i) {
+ if (scratch_dirfd >= 0)
+ snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%llu", leaf, (long)getpid(),
+ next_temp_sequence());
+ else
+ snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i);
+ int fd =
+ openat(target_dirfd, tmp, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | O_NOFOLLOW, 0600);
+ if (fd < 0) {
+ if (errno != EEXIST)
+ break;
+ continue;
+ }
+ bool wrote = true;
+ if (preallocate && expected_size > 0 && preallocate_fd(fd, expected_size) != 0)
+ wrote = false;
+ if (wrote)
+ wrote = copy_fd_all(fd, src_fd, expected_size);
+ if (wrote && metadata) {
+ if (!policy.perms &&
+ fchmod(fd, file_mode_base(metadata, destination_is_regular,
+ destination_is_regular ? destination_stat.st_mode & 0777
+ : 0)) != 0)
+ wrote = false;
+ if (wrote)
+ wrote = file_restore_metadata_fd(fd, metadata, policy);
+ } else if (wrote && fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0) {
+ wrote = false;
+ }
+ if (wrote)
+ restore_extra_fd(fd, metadata, xattrs, fake_super, policy);
+ if (wrote && use_fsync)
+ wrote = fsync(fd) == 0;
+ if (close(fd) != 0)
+ wrote = false;
+ if (wrote && renameat(target_dirfd, tmp, dirfd, leaf) != 0)
+ wrote = false;
+ if (!wrote)
+ unlinkat(target_dirfd, tmp, 0);
+ ok = wrote;
+ }
+ free(tmp);
+ }
+ if (!ok && scratch_dirfd >= 0) {
+ /* Retry once with no scratch dir (rsync's EXDEV fallback). */
+ close(scratch_dirfd);
+ close(src_fd);
+ close(dirfd);
+ free(leaf);
+ return file_copy_basis_stream_impl(path, basis_path, expected_size, preallocate, metadata,
+ policy, update, no_replace, use_fsync, xattrs, fake_super,
+ NULL, dirs_created, count_floor);
+ }
+ if (scratch_dirfd >= 0)
+ close(scratch_dirfd);
+ close(src_fd);
+ close(dirfd);
+ free(leaf);
+ return ok;
+}
+
+/* --copy-dest basis install (streaming). Applies the source metadata and the
+ per-file xattrs / --fake-super record. */
+bool file_copy_basis_stream_attrs(const char* path, const char* basis_path,
+ unsigned long long expected_size, bool preallocate,
+ const FileMetadata* metadata, FileAttrPolicy policy, bool update,
+ bool use_fsync, const FileXattrList* xattrs, bool fake_super,
+ const char* temp_dir) {
+ return file_copy_basis_stream_impl(path, basis_path, expected_size, preallocate, metadata, policy,
+ update, false, use_fsync, xattrs, fake_super, temp_dir, NULL,
+ NULL);
+}
+
static bool file_to_disk_secure_link_impl(const char* path, const char* basis_path,
const void* data, unsigned long long data_size,
bool preallocate, const FileMetadata* metadata,
FileAttrPolicy policy, bool use_fsync,
const FileXattrList* xattrs, bool fake_super,
- const char* temp_dir) {
+ const char* temp_dir, unsigned* dirs_created,
+ const char* count_floor) {
if (!path || !basis_path)
return false;
+ /* The caller-supplied buffer is no longer used: the copy fallback streams
+ from the basis path (which may hold an over-limit file). Kept in the
+ signature for the existing API. */
+ (void)data;
char* leaf = NULL;
- int dirfd = file_open_secure_parent(path, &leaf, true);
+ int dirfd = file_open_secure_parent_counted(path, &leaf, true, dirs_created, count_floor);
if (dirfd < 0)
return false;
@@ -1463,10 +1805,17 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
close(dirfd);
free(leaf);
/* The basis file could not be linked in (missing, cross-device, refused
- by the filesystem). Write a byte-identical local copy instead. */
- return file_to_disk_secure_attrs(path, data, data_size, false, false, preallocate, metadata,
- policy, false, false, use_fsync, xattrs, fake_super, false,
- temp_dir);
+ by the filesystem). Stream a byte-identical local copy from the basis
+ itself (never the possibly-absent caller buffer) so an over-limit basis
+ still materializes. When the basis path is not a readable regular file
+ (e.g. a directory raced in), fall back to the caller-supplied bytes. */
+ if (file_copy_basis_stream_impl(path, basis_path, data_size, preallocate, metadata, policy,
+ false, false, use_fsync, xattrs, fake_super, temp_dir,
+ dirs_created, count_floor))
+ return true;
+ return file_to_disk_secure_attrs_counted(
+ path, data, data_size, false, false, preallocate, metadata, policy, false, false, use_fsync,
+ xattrs, fake_super, false, temp_dir, dirs_created, count_floor);
}
if (scratch_dirfd >= 0)
@@ -1481,7 +1830,7 @@ bool file_to_disk_secure_link(const char* path, const char* basis_path, const vo
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
const char* temp_dir) {
return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata,
- policy, use_fsync, NULL, false, temp_dir);
+ policy, use_fsync, NULL, false, temp_dir, NULL, NULL);
}
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
@@ -1489,8 +1838,21 @@ bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, co
const FileMetadata* metadata, FileAttrPolicy policy,
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
const char* temp_dir) {
+ return file_to_disk_secure_link_attrs_counted(path, basis_path, data, data_size, preallocate,
+ metadata, policy, use_fsync, xattrs, fake_super,
+ temp_dir, NULL, NULL);
+}
+
+bool file_to_disk_secure_link_attrs_counted(const char* path, const char* basis_path,
+ const void* data, unsigned long long data_size,
+ bool preallocate, const FileMetadata* metadata,
+ FileAttrPolicy policy, bool use_fsync,
+ const FileXattrList* xattrs, bool fake_super,
+ const char* temp_dir, unsigned* dirs_created,
+ const char* count_floor) {
return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata,
- policy, use_fsync, xattrs, fake_super, temp_dir);
+ policy, use_fsync, xattrs, fake_super, temp_dir,
+ dirs_created, count_floor);
}
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
diff --git a/src/shared/file.h b/src/shared/file.h
index 939d86d..612c50b 100644
--- a/src/shared/file.h
+++ b/src/shared/file.h
@@ -87,6 +87,11 @@ bool file_path_exists_secure(const char* path);
bool file_stat_secure(const char* path, struct stat* st);
bool file_destination_is_newer_secure(const char* path, const FileMetadata* metadata);
int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs);
+/* Protocol 2.28.0 variant: also increments *dirs_created for every missing
+ * parent directory this walk creates that lies strictly below `count_floor`
+ * (a receive-root-relative path, or NULL to count all of them). */
+int file_open_secure_parent_counted(const char* path, char** leaf_out, bool create_dirs,
+ unsigned* dirs_created, const char* count_floor);
bool file_ensure_directory_secure(const char* path);
bool file_directory_exists_secure(const char* path);
bool file_rename_secure(const char* old_path, const char* new_path);
@@ -158,5 +163,37 @@ bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, co
const FileMetadata* metadata, FileAttrPolicy policy,
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
const char* temp_dir);
+/* Streaming --copy-dest install: atomically materialize `path` by copying the
+ * bytes of `basis_path` through a bounded buffer (no whole-file buffering, so
+ * an arbitrarily large basis works), applying the SOURCE metadata and the
+ * per-file xattrs / --fake-super record. `update` honors a newer destination;
+ * a --temp-dir scratch location falls back to a direct write on EXDEV. */
+bool file_copy_basis_stream_attrs(const char* path, const char* basis_path,
+ unsigned long long expected_size, bool preallocate,
+ const FileMetadata* metadata, FileAttrPolicy policy, bool update,
+ bool use_fsync, const FileXattrList* xattrs, bool fake_super,
+ const char* temp_dir);
+/* Protocol 2.28.0 receiver-stat variants: like the two above but additionally
+ * report through `dirs_created` (when non-NULL) how many parent directories the
+ * confined secure walk had to create that lie strictly below `count_floor` (a
+ * receive-root-relative prefix, or NULL for all). Used to reproduce rsync's
+ * `Number of created files` directory count on a fresh destination. */
+bool file_to_disk_secure_attrs_counted(const char* path, const void* data,
+ unsigned long long data_size, bool inplace, bool sparse,
+ bool preallocate, const FileMetadata* metadata,
+ FileAttrPolicy policy, bool update, bool no_replace,
+ bool use_fsync, const FileXattrList* xattrs, bool fake_super,
+ bool keep_partial, const char* temp_dir,
+ unsigned* dirs_created, const char* count_floor);
+bool file_to_disk_secure_link_attrs_counted(const char* path, const char* basis_path,
+ const void* data, unsigned long long data_size,
+ bool preallocate, const FileMetadata* metadata,
+ FileAttrPolicy policy, bool use_fsync,
+ const FileXattrList* xattrs, bool fake_super,
+ const char* temp_dir, unsigned* dirs_created,
+ const char* count_floor);
+/* The logical transfer root expressed receive-root-relative, or NULL when the
+ * wire paths carry no mirror scaffolding above it. Caller frees non-NULL. */
+char* file_transfer_root_floor(const Config* config);
#endif
diff --git a/src/shared/file_receive.c b/src/shared/file_receive.c
index c9f9c18..b724c4f 100644
--- a/src/shared/file_receive.c
+++ b/src/shared/file_receive.c
@@ -37,7 +37,12 @@
#define MANIFEST_ENTRY_OVERHEAD (sizeof(char*) + 16)
bool file_save_to_disk(const char* root_directory, const File* file, const Config* config) {
- return file_save_to_disk_full(root_directory, file, config) != FILE_SAVE_ERROR;
+ return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL) != FILE_SAVE_ERROR;
+}
+
+FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file,
+ const Config* config) {
+ return file_save_to_disk_full_ex(root_directory, file, config, NULL, NULL);
}
/* --delay-updates receiver path: write the file into a private staging tree
@@ -94,6 +99,12 @@ static FileSaveResult file_stage_delayed_update(const char* root_directory,
if (file->basis_link) {
ok = file_to_disk_secure_link(staged_path, file->basis_link, file->data->data, file->data->size,
config->preallocate, metadata, policy, config->use_fsync, NULL);
+ } else if (file->basis_copy) {
+ /* --copy-dest basis hit: stream the basis into the staging tree (bounded
+ buffers, so an over-limit basis still stages). */
+ ok = file_copy_basis_stream_attrs(staged_path, file->basis_copy, file->data->size,
+ config->preallocate, metadata, policy, config->update,
+ config->use_fsync, file->xattrs, config->fake_super, NULL);
} else {
ok =
file_to_disk_secure_attrs(staged_path, file->data->data, file->data->size, false, sparse,
@@ -208,13 +219,14 @@ static FileSaveResult hardlink_sibling_absent_first(const char* destination_path
--existing/--ignore-existing/--update policies are decided against the final
destination like every normal write. */
static FileSaveResult file_save_hardlink_sibling(const char* root_directory, const File* file,
- const Config* config) {
+ const Config* config, bool* created) {
Config* cfg = (Config*)config;
if (!root_directory || !file || !file->path || !file->hardlink_target)
return FILE_SAVE_ERROR;
char* destination_path = path_cat(root_directory, file->path);
if (!destination_path)
return FILE_SAVE_ERROR;
+ bool existed = file_path_exists_secure(destination_path);
if (cfg->existing && !file_path_exists_secure(destination_path)) {
free(destination_path);
@@ -278,6 +290,8 @@ static FileSaveResult file_save_hardlink_sibling(const char* root_directory, con
free(staged_first);
free(staged_sibling);
free(destination_path);
+ if (ok && created && !existed)
+ *created = true;
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
@@ -324,6 +338,8 @@ static FileSaveResult file_save_hardlink_sibling(const char* root_directory, con
free(content);
free(first_disk);
free(destination_path);
+ if (ok && created && !existed)
+ *created = true;
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
@@ -360,7 +376,7 @@ bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode) {
* non-device entry must carry an empty rdev.
*/
static FileSaveResult file_save_special_to_disk(const char* root_directory, const File* file,
- const Config* config) {
+ const Config* config, bool* created) {
/* The empty-path and structural checks stay unconditional; the redundant
".." list-path re-check is skipped under --trust-sender exactly like the
receive layer (confinement is deferred to the secure parent walk below,
@@ -412,6 +428,7 @@ static FileSaveResult file_save_special_to_disk(const char* root_directory, cons
char* destination = path_cat(root_directory, file->path);
if (!destination)
return FILE_SAVE_ERROR;
+ bool existed = file_path_exists_secure(destination);
char* leaf = NULL;
int parent_fd = file_open_secure_parent(destination, &leaf, true);
if (parent_fd < 0) {
@@ -532,6 +549,8 @@ static FileSaveResult file_save_special_to_disk(const char* root_directory, cons
free(destination);
/* A failed required --copy-as ownership marks the node as failed; every other
* identity policy stays best-effort. */
+ if (owner_ok && created && !existed)
+ *created = true;
return owner_ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
@@ -610,8 +629,13 @@ static FileSaveResult file_save_write_device(const char* root_directory, const F
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_SKIPPED;
}
-FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file,
- const Config* config) {
+FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file,
+ const Config* config, bool* created,
+ unsigned* created_dirs) {
+ if (created)
+ *created = false;
+ if (created_dirs)
+ *created_dirs = 0;
/* Central no-mutation guard: a server-contacting --dry-run (or a local batch
apply that somehow carries dry_run) must never touch the destination, no
matter which caller reached this primitive. The per-caller guards remain,
@@ -634,7 +658,8 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
char* destination_path = NULL;
char *backup_path = NULL, *parent_copy = NULL;
- if (!file || !file->path || !file->data || (file->data->size != 0 && !file->data->data) ||
+ if (!file || !file->path || !file->data ||
+ (file->data->size != 0 && !file->data->data && !file->basis_link && !file->basis_copy) ||
(!file_get_trust_sender() && has_path_traversal(file->path)) ||
(backup_enabled &&
(!backup_suffix || backup_suffix[0] == '\0' || strchr(backup_suffix, '/') != NULL ||
@@ -658,7 +683,7 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
/* Device/special node (--devices/--specials): recreate the node instead of
writing content (privilege-gated, confined, rdev-validated). */
if (file->is_special)
- return file_save_special_to_disk(root_directory, file, config);
+ return file_save_special_to_disk(root_directory, file, config, created);
/* --write-devices: write straight into an existing device node. Writing
into a device is a super-user activity, so --no-super must suppress it just
like device-node creation; the default AUTO/--super attempt it (the wide
@@ -689,6 +714,7 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
char* dir_path = path_cat(root_directory, file->path);
if (!dir_path)
return FILE_SAVE_ERROR;
+ bool dir_existed = file_path_exists_secure(dir_path);
bool ok = file_ensure_directory_secure(dir_path);
/* P7 Wave E: apply the negotiated ownership to the directory ITSELF (not
just the files inside it). --copy-as and every explicit identity policy
@@ -712,6 +738,8 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
free(leaf);
}
free(dir_path);
+ if (ok && created && !dir_existed)
+ *created = true;
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
@@ -728,6 +756,7 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
char* link_path = path_cat(root_directory, file->path);
if (!link_path)
return FILE_SAVE_ERROR;
+ bool link_existed = file_path_exists_secure(link_path);
/* The link value is stored verbatim (rsync -l parity: absolute and
".."-bearing targets are preserved; the scanner's --safe-links /
--copy-unsafe-links decide which links are sent at all). --munge-links
@@ -768,6 +797,8 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
ok = file_restore_symlink_metadata(link_path, file->metadata, link_policy,
config->omit_link_times);
}
+ if (ok && created && !link_existed)
+ *created = true;
free(link_path);
return ok ? FILE_SAVE_WRITTEN : FILE_SAVE_ERROR;
}
@@ -777,7 +808,7 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
byte-identical copy of) the group's first member. Handled entirely here,
before the normal data-write paths (which would create an empty file). */
if (file->link_group != 0 && !file->link_first && file->hardlink_target != NULL) {
- return file_save_hardlink_sibling(root_directory, file, config);
+ return file_save_hardlink_sibling(root_directory, file, config, created);
}
/* These options arrive from the client. --backup-dir, --partial-dir and
@@ -808,12 +839,18 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
free(disk_path);
return FILE_SAVE_ERROR;
}
+ /* Snapshot the final destination's existence BEFORE any backup/force/partial
+ step can move or remove it, so the receiver can report rsync's
+ `Number of created files` (protocol 2.28.0). */
+ bool dest_existed = file_path_exists_secure(destination_path);
/* --delay-updates diverts the whole write into the staging tree; the rest of
this function is the immediate-install path. */
if (config && config->delay_updates) {
FileSaveResult result =
file_stage_delayed_update(root_directory, destination_path, file, (Config*)config);
+ if (result == FILE_SAVE_WRITTEN && created && !dest_existed)
+ *created = true;
free(confined_backup);
free(confined_partial);
free(destination_path);
@@ -939,19 +976,30 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
existing/ignore-existing/update/backup preamble above has already made the
policy decision. */
bool ok;
+ char* count_floor = file_transfer_root_floor(config);
if (config && file->basis_link) {
- ok = file_to_disk_secure_link_attrs(
+ ok = file_to_disk_secure_link_attrs_counted(
disk_path, file->basis_link, file->data->data, file->data->size, config->preallocate,
- metadata, policy, config->use_fsync, file->xattrs, config->fake_super, confined_temp);
+ metadata, policy, config->use_fsync, file->xattrs, config->fake_super, confined_temp,
+ created_dirs, count_floor);
+ } else if (config && file->basis_copy) {
+ /* --copy-dest: stream the basis bytes through a bounded buffer so a basis
+ larger than any whole-file bound still materializes. The source
+ metadata was transmitted with the check frame. */
+ ok = file_copy_basis_stream_attrs(
+ disk_path, file->basis_copy, file->data->size, config->preallocate, metadata, policy,
+ config->update, config->use_fsync, file->xattrs, config->fake_super, confined_temp);
} else {
/* The plain no-replace / update / with-fsync engines, plus per-file xattr
(-X/-A) and --fake-super application on the written fd. */
- ok = file_to_disk_secure_attrs(
+ ok = file_to_disk_secure_attrs_counted(
disk_path, file->data->data, file->data->size, inplace, sparse,
config && config->preallocate, metadata, policy, config && config->update,
config && config->ignore_existing, config && config->use_fsync, file->xattrs,
- config ? config->fake_super : false, config ? config->partial : false, confined_temp);
+ config ? config->fake_super : false, config ? config->partial : false, confined_temp,
+ created_dirs, count_floor);
}
+ free(count_floor);
free(confined_temp);
confined_temp = NULL;
if (!ok)
@@ -972,6 +1020,8 @@ FileSaveResult file_save_to_disk_full(const char* root_directory, const File* fi
free(confined_partial);
free(destination_path);
free(disk_path);
+ if (created && !dest_existed)
+ *created = true;
return FILE_SAVE_WRITTEN;
fail:
@@ -984,6 +1034,37 @@ fail:
return FILE_SAVE_ERROR;
}
+void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created,
+ unsigned created_dirs) {
+ if (!stats || !file)
+ return;
+ /* A basis-dir hit (--link-dest/--copy-dest) materializes bytes the sender
+ * never transferred. rsync reports no literal data and no created entry for
+ * such a file, and does not count the parent directories it creates only to
+ * hold it, so exclude the whole entry from the receiver tallies. */
+ bool basis_sourced = file->basis_link != NULL || file->basis_copy != NULL;
+ if (basis_sourced)
+ return;
+ bool is_sibling = file->link_group != 0 && !file->link_first;
+ if (!file->is_dir && !file->is_symlink && !file->is_special && !is_sibling) {
+ unsigned long long literal = file->literal_bytes;
+ if (literal == 0 && file->matched_bytes == 0)
+ literal = file->data ? file->data->size : 0;
+ stats->literal_bytes += literal;
+ }
+ stats->created_dir += created_dirs;
+ if (!created)
+ return;
+ if (file->is_dir)
+ stats->created_dir++;
+ else if (file->is_symlink)
+ stats->created_link++;
+ else if (file->is_special)
+ stats->created_special++;
+ else
+ stats->created_reg++;
+}
+
/* Receive a file's xattr block (when the config enables xattr transport) and
* attach it to `file`. Returns false on a malformed/oversized frame. */
static bool receive_file_xattrs(File* file, int fd, const Config* config) {
@@ -1094,11 +1175,15 @@ static File* receive_delta_file(int fd, const Config* config, const char* check_
return NULL;
}
/* Wire-stats tally: bytes taken straight from the basis file (matched
- delta blocks). Computed before the delta is destroyed. */
+ delta blocks) and bytes shipped literally (protocol 2.28.0). Computed
+ before the delta is destroyed. */
unsigned long long matched = 0;
+ unsigned long long literal = 0;
for (uint32_t k = 0; k < delta->instruction_count; k++) {
if (delta->instructions[k].type == DELTA_INSTR_BLOCK_MATCH)
matched += delta->instructions[k].match.length;
+ else if (delta->instructions[k].type == DELTA_INSTR_LITERAL)
+ literal += delta->instructions[k].literal.length;
}
void* new_data = delta_apply(old_data, old_size, delta, config->delta_block_size);
delta_destroy(delta);
@@ -1119,6 +1204,7 @@ static File* receive_delta_file(int fd, const Config* config, const char* check_
return NULL;
}
file->matched_bytes = matched;
+ file->literal_bytes = literal;
if (config->use_metadata) {
int meta_ok = 1;
@@ -1233,16 +1319,17 @@ static File* receive_delta_file(int fd, const Config* config, const char* check_
/* ---- Alternate basis directories (--compare-dest / --copy-dest / --link-dest) ----
* The receiver consults the ordered basis-dir list only when the destination
- * entry is NOT already up to date. An "exact match" requires an equal size,
- * an equal mtime (unless --size-only), and an equal content xxHash64, so a
- * hard link / local copy is only ever made from byte-identical content. */
+ * entry is NOT already up to date. By default an "exact match" is rsync's
+ * metadata quick-check: an equal size and an equal mtime (unless --size-only).
+ * The FastSync-only --verify-basis additionally requires an equal whole-file
+ * content digest, so a hard link / local copy is only then made from
+ * byte-verified content. */
typedef struct BasisMatch {
bool hit;
BasisDestType type;
char* basis_path; /* owned absolute path of the matched basis file */
struct stat st; /* fstat() of the matched basis file */
- Data* content; /* owned basis bytes (or empty Data), NULL when not loaded */
} BasisMatch;
static void basis_match_free(BasisMatch* match) {
@@ -1250,8 +1337,6 @@ static void basis_match_free(BasisMatch* match) {
return;
free(match->basis_path);
match->basis_path = NULL;
- data_destroy(match->content);
- match->content = NULL;
match->hit = false;
match->type = BASIS_DEST_NONE;
}
@@ -1283,32 +1368,10 @@ static bool basis_open_regular(const char* path, unsigned long long expected_siz
return true;
}
-/* Read the whole remaining content of an open descriptor. A zero-length file
- yields an empty Data (data pointer NULL). */
-static Data* basis_read_content(int fd, unsigned long long size) {
- if (size == 0)
- return data_create_reserve(0);
- if (size > MAX_RECEIVE_WHOLE_FILE_SIZE || size > SIZE_MAX)
- return NULL;
- void* buf = protocol_alloc((size_t)size);
- if (!buf)
- return NULL;
- size_t got = 0;
- while (got < (size_t)size) {
- ssize_t n = read(fd, (char*)buf + got, (size_t)size - got);
- if (n <= 0) {
- free(buf);
- return NULL;
- }
- got += (size_t)n;
- }
- return data_create(buf, (size_t)size);
-}
-
/* --ignore-times forces every file to be updated, so no basis hit is ever
declared (matching rsync, where -I prevents link-dest from linking). */
-static bool basis_quick_matches(const Config* config, const struct stat* st, time_t check_mtime,
- long check_mtime_nsec) {
+bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime,
+ long check_mtime_nsec) {
if (config->size_only)
return true;
long mtime_nsec = 0;
@@ -1319,28 +1382,39 @@ static bool basis_quick_matches(const Config* config, const struct stat* st, tim
config->modify_window);
}
-/* Search the basis-dir list in command-line order and return the first exact
- match. When load_content is true the matched bytes are kept in out->content
- so the caller can materialize the file without re-reading it.
+/* True when a basis hit must be confirmed by a whole-file content digest
+ (--verify-basis). False is the rsync-parity default: the metadata
+ quick-check alone decides a hit. */
+bool file_basis_content_required(const Config* config) {
+ return config != NULL && config->verify_basis;
+}
- An exact match ALSO requires the basis bytes' digest to equal the source's,
- so `hash_content` gates the content read/hash itself. A server-contacting
- --dry-run passes hash_content=false: no basis file may be read or hashed
- (that would be a 1-bit content oracle against a client-supplied digest), so a
- metadata-only pass can never confirm a hit and declines it. The real path
- always passes hash_content=true, keeping its behavior byte-for-byte. */
+/* Search the basis-dir list in command-line order and return the first match.
+ By default (no --verify-basis) rsync's metadata quick-check is sufficient:
+ basis_open_regular has already required an equal size, and
+ file_basis_quick_match applies rsync's mtime (or --size-only) rule.
+ --verify-basis additionally requires the basis bytes' whole-file digest to
+ equal the sender's, restoring FastSync's historical content equality; that
+ digest is computed by streaming the open basis descriptor, so an arbitrarily
+ large basis is verified without buffering it. A copy/link install re-reads
+ the basis from its path in bounded buffers, so no content buffer is kept.
+
+ `hash_content` gates content READS under --verify-basis: a server-contacting
+ --dry-run passes false because hashing a basis against a client-supplied
+ digest would be a 1-bit content oracle. Without --verify-basis a dry-run can
+ still confirm the metadata-only hit without reading any basis bytes, matching
+ rsync's read-only quick-check. */
static bool basis_match_find(const Config* config, const char* check_path,
unsigned long long check_size, time_t check_mtime,
long check_mtime_nsec, const uint8_t* check_digest,
- size_t check_digest_len, bool load_content, bool hash_content,
- BasisMatch* out) {
+ size_t check_digest_len, bool hash_content, BasisMatch* out) {
memset(out, 0, sizeof(*out));
if (!config || !config_has_basis(config) || config->ignore_times)
return false;
- /* Dry-run: never read/hash basis content. A hit cannot be decided from
- metadata alone, so report no match (the caller treats it as would-transfer)
- without touching the file's contents. */
- if (!hash_content)
+ /* --verify-basis needs the basis content; a content-blind (dry-run) pass can
+ never confirm it and must not read the file, so decline without touching
+ the basis bytes. */
+ if (file_basis_content_required(config) && !hash_content)
return false;
for (int i = 0; i < config->basis_count; i++) {
const BasisDest* entry = &config->basis_dirs[i];
@@ -1359,29 +1433,26 @@ static bool basis_match_find(const Config* config, const char* check_path,
int fd;
struct stat st;
if (basis_open_regular(candidate, check_size, &fd, &st)) {
- if (basis_quick_matches(config, &st, check_mtime, check_mtime_nsec)) {
- Data* content = basis_read_content(fd, check_size);
- if (content) {
+ if (file_basis_quick_match(config, &st, check_mtime, check_mtime_nsec)) {
+ bool hit = true;
+ if (file_basis_content_required(config)) {
uint8_t basis_digest[CHECKSUM_MAX_DIGEST_LEN];
size_t basis_len = 0;
- bool hashed = checksum_digest((ChecksumAlgo)config->checksum_algo, config->checksum_seed,
- content->data, content->size, basis_digest,
- sizeof(basis_digest), &basis_len);
- if (hashed && basis_len == check_digest_len && check_digest_len > 0 &&
- memcmp(basis_digest, check_digest, check_digest_len) == 0) {
- out->hit = true;
- out->type = entry->type;
- out->basis_path = candidate;
- candidate = NULL; /* ownership transferred to out */
- out->st = st;
- out->content = load_content ? content : NULL;
- if (!load_content)
- data_destroy(content);
- close(fd);
- return true;
- }
+ bool hashed =
+ checksum_digest_fd((ChecksumAlgo)config->checksum_algo, config->checksum_seed, fd,
+ basis_digest, sizeof(basis_digest), &basis_len);
+ hit = hashed && basis_len == check_digest_len && check_digest_len > 0 &&
+ memcmp(basis_digest, check_digest, check_digest_len) == 0;
+ }
+ if (hit) {
+ out->hit = true;
+ out->type = entry->type;
+ out->basis_path = candidate;
+ candidate = NULL; /* ownership transferred to out */
+ out->st = st;
+ close(fd);
+ return true;
}
- data_destroy(content);
}
close(fd);
}
@@ -1806,6 +1877,12 @@ typedef struct {
long long check_mtime_nsec;
uint8_t check_digest[CHECKSUM_MAX_DIGEST_LEN];
size_t check_digest_len;
+ /* Source metadata carried alongside the check frame whenever a basis dir is
+ configured (rsync keeps the whole file list; FastSync's sender-driven
+ incremental path otherwise never transmits metadata for a SKIPPED file).
+ A basis materialization applies these SOURCE attributes instead of the
+ basis inode's, matching rsync's "copy then fix attributes". */
+ FileMetadata* source_metadata;
bool dest_exists; /* any destination entry exists (lstat succeeded) */
bool has_old_file;
int old_fd;
@@ -1838,6 +1915,8 @@ static void incremental_check_state_cleanup(IncrementalCheckState* state) {
if (state->old_fd >= 0)
close(state->old_fd);
state->old_fd = -1;
+ file_metadata_destroy(state->source_metadata);
+ state->source_metadata = NULL;
free(state->full_path);
state->full_path = NULL;
free(state->check_path);
@@ -1862,7 +1941,7 @@ static IncrementalCheckOutcome incremental_check_receive_request(IncrementalChec
send_error_detail(fd, "invalid check mtime nanoseconds");
return INCREMENTAL_ERROR;
}
- if ((config->checksum || config_has_basis(config))) {
+ if ((config->checksum || config->verify_basis)) {
uint8_t wire_len;
if (!receive_n_data(fd, &wire_len, sizeof(wire_len)) || wire_len == 0 ||
wire_len > CHECKSUM_MAX_DIGEST_LEN ||
@@ -1874,8 +1953,24 @@ static IncrementalCheckOutcome incremental_check_receive_request(IncrementalChec
if (!receive_n_data(fd, state->check_digest, state->check_digest_len))
return INCREMENTAL_ERROR;
}
+ /* The sender transmits the source metadata with every basis-configured check
+ so a basis hit can be materialized with the SOURCE's attributes (rsync
+ copies/copies-then-fixes; the receiver would otherwise only have the basis
+ inode's stat). The block is symmetric and consumed unconditionally here,
+ whether or not this file ends up as a basis hit. */
+ if (config_has_basis(config) && config->use_metadata) {
+ int meta_ok = 1;
+ state->source_metadata = metadata_receive(fd, &meta_ok);
+ if (!meta_ok)
+ return INCREMENTAL_ERROR;
+ }
- if (state->check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) {
+ /* A basis-configured run may materialize a file larger than the whole-file
+ payload bound: a basis hit is streamed from the basis path (bounded
+ buffers), so the check size is not itself an allocation. Every other
+ path (delta/append/full) still applies MAX_RECEIVE_WHOLE_FILE_SIZE, and a
+ miss simply falls through to the normal transfer with its own bound. */
+ if (!config_has_basis(config) && state->check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) {
send_error_detail(fd, "check size exceeds receiver limit");
return INCREMENTAL_ERROR;
}
@@ -1965,6 +2060,20 @@ incremental_check_ignore_existing(const IncrementalCheckState* state) {
return INCREMENTAL_SKIP;
}
+/* Metadata for a materialized basis hit: prefer the SOURCE metadata the sender
+ transmitted with the check frame (rsync copies then fixes the destination to
+ the source's attributes); fall back to the basis inode's own stat when
+ metadata was not negotiated. Consumes state->source_metadata on success. */
+static FileMetadata* basis_take_metadata(IncrementalCheckState* state,
+ const struct stat* basis_st) {
+ if (state->source_metadata) {
+ FileMetadata* meta = state->source_metadata;
+ state->source_metadata = NULL;
+ return meta;
+ }
+ return file_metadata_create(NULL, basis_st, false, false);
+}
+
/* --link-dest relink of an already up-to-date destination. rsync hard-links a
destination entry to a matching basis even when the entry is already correct,
so a run over an existing tree still maximizes sharing with the basis. Only a
@@ -1982,7 +2091,7 @@ static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalChe
BasisMatch basis;
basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime,
(long)state->check_mtime_nsec, state->check_digest, state->check_digest_len,
- true, true, &basis);
+ true, &basis);
/* Only a link-dest hit relinks; a copy-dest/compare-dest hit (or a miss) lets
the up-to-date check below keep the existing destination. */
if (!basis.hit || basis.type != BASIS_DEST_LINK) {
@@ -1995,11 +2104,16 @@ static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalChe
return INCREMENTAL_CONTINUE;
}
File* materialized = file_create(state->check_path);
- if (materialized && basis.content) {
+ if (materialized) {
data_destroy(materialized->data);
- materialized->data = basis.content;
- basis.content = NULL;
- materialized->metadata = file_metadata_create(NULL, &basis.st, false, false);
+ materialized->data = data_create_reserve((size_t)state->check_size);
+ if (!materialized->data) {
+ file_destroy(materialized);
+ materialized = NULL;
+ }
+ }
+ if (materialized) {
+ materialized->metadata = basis_take_metadata(state, &basis.st);
materialized->skip = true;
materialized->basis_link = basis.basis_path;
basis.basis_path = NULL;
@@ -2007,9 +2121,6 @@ static IncrementalCheckOutcome incremental_check_link_dest_relink(IncrementalChe
file_destroy(materialized);
materialized = NULL;
}
- } else {
- file_destroy(materialized);
- materialized = NULL;
}
if (materialized) {
if (!send_status(state->fd, STATUS_OK)) {
@@ -2110,12 +2221,13 @@ static IncrementalCheckOutcome incremental_check_quick_skip(IncrementalCheckStat
materialize nothing (no basis link/copy, no append/delta/full transfer) and
the sender must send no data, so answer STATUS_DRY_RUN_TRANSFER and stop.
- The basis lookup is deliberately content-blind: a real run would only accept
- a --compare-dest exact hit after hashing the basis file and comparing it with
- the client-supplied digest, which in a dry-run is a 1-bit content oracle.
- Under dry_run no basis bytes may be read, so an otherwise-matching entry is
- treated as would-transfer instead of a skip. Everything read here (the
- destination file's metadata, basis candidates' metadata) is read-only. */
+ The basis lookup is content-blind: under the default metadata quick-check a
+ hit needs no basis bytes and is honored here just as in a real run; under
+ --verify-basis a real run hashes the basis against the client-supplied digest,
+ which in a dry-run is a 1-bit content oracle, so no basis bytes may be read
+ and an otherwise-matching entry is reported as would-transfer. Everything
+ read here (the destination file's metadata, basis candidates' metadata) is
+ read-only. */
static IncrementalCheckOutcome incremental_check_dry_run_shortcut(IncrementalCheckState* state,
bool* skipped,
bool* would_transfer) {
@@ -2126,12 +2238,14 @@ static IncrementalCheckOutcome incremental_check_dry_run_shortcut(IncrementalChe
bool skip_via_compare = false;
if (config_has_basis(config) && !config->ignore_times) {
BasisMatch basis;
- /* hash_content=false: a dry-run must not read or hash the basis file. No
- content comparison is possible, so no compare-dest hit can be confirmed
- and an otherwise-matching file is reported as would-transfer. */
+ /* hash_content=false: a dry-run must not read or hash the basis file, so
+ under --verify-basis no compare-dest hit can be confirmed and an
+ otherwise-matching file is reported as would-transfer. Without
+ --verify-basis the metadata quick-check confirms it without touching any
+ basis bytes. */
basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime,
(long)state->check_mtime_nsec, state->check_digest, state->check_digest_len,
- false, false, &basis);
+ false, &basis);
if (basis.hit && basis.type == BASIS_DEST_COMPARE && !state->has_old_file)
skip_via_compare = true;
basis_match_free(&basis);
@@ -2159,7 +2273,7 @@ static IncrementalCheckOutcome incremental_check_try_basis(IncrementalCheckState
BasisMatch basis;
basis_match_find(config, state->check_path, state->check_size, (time_t)state->check_mtime,
(long)state->check_mtime_nsec, state->check_digest, state->check_digest_len,
- true, true, &basis);
+ true, &basis);
if (basis.hit) {
if (basis.type == BASIS_DEST_COMPARE) {
basis_match_free(&basis);
@@ -2169,24 +2283,30 @@ static IncrementalCheckOutcome incremental_check_try_basis(IncrementalCheckState
return INCREMENTAL_SKIP;
}
} else {
+ /* Copy/link installs source their bytes from the basis PATH at install
+ time (bounded buffers), so no whole-file content buffer is needed here
+ even for an over-limit basis. */
File* materialized = file_create(state->check_path);
- if (materialized && basis.content) {
+ if (materialized) {
data_destroy(materialized->data);
- materialized->data = basis.content;
- basis.content = NULL;
- materialized->metadata = file_metadata_create(NULL, &basis.st, false, false);
- materialized->skip = true; /* receiver must not ack this as a data file */
- if (basis.type == BASIS_DEST_LINK) {
- materialized->basis_link = basis.basis_path;
- basis.basis_path = NULL;
+ materialized->data = data_create_reserve((size_t)state->check_size);
+ if (!materialized->data) {
+ file_destroy(materialized);
+ materialized = NULL;
}
+ }
+ if (materialized) {
+ materialized->metadata = basis_take_metadata(state, &basis.st);
+ materialized->skip = true; /* receiver must not ack this as a data file */
+ if (basis.type == BASIS_DEST_LINK)
+ materialized->basis_link = basis.basis_path;
+ else
+ materialized->basis_copy = basis.basis_path;
+ basis.basis_path = NULL;
if (!materialized->metadata) {
file_destroy(materialized);
materialized = NULL;
}
- } else {
- file_destroy(materialized);
- materialized = NULL;
}
if (materialized) {
if (!send_status(fd, STATUS_OK)) {
@@ -3195,8 +3315,9 @@ char* file_receive_basis_delete_relative(const Config* config, const char* path)
alternate basis directories are never destination content and are skipped at
any depth. Returns true unless a traversal/unlink error aborted the walk;
the budget's limit_hit/skipped fields report a cap-stopped run. */
-static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest,
- DeleteBudgetState* budget) {
+static bool delete_extras_budgeted_observed(const Config* config, DeleteManifest* manifest,
+ DeleteBudgetState* budget, DeletePathObserver observer,
+ void* observer_context) {
if (!config || !manifest || !manifest->keeps)
return false;
fprintf(stderr, "Deleting files not in manifest...\n");
@@ -3260,9 +3381,9 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes
remaining = budget->max_delete - budget->deleted;
size_t deleted = 0;
size_t skipped = 0;
- DeleteWalkResult result =
- delete_extras_limited(config->receive_root_directory, manifest->keeps, manifest->dirs,
- remaining, skips, used, &deleted, &skipped);
+ DeleteWalkResult result = delete_extras_limited_observed(
+ config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips, used,
+ config->protect_rules, &deleted, &skipped, observer, observer_context);
if (owned_prefixes) {
for (int i = 0; i < config->basis_count; i++)
free(owned_prefixes[i]);
@@ -3282,6 +3403,31 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes
return true;
}
+static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifest,
+ DeleteBudgetState* budget) {
+ return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL);
+}
+
+/* Prefixes every observed path with a fixed subtree root, so a nested walk
+ (a recursively removed missing-arg directory) reports receive-root-relative
+ names like the rest of the delete output. */
+typedef struct {
+ DeletePathObserver inner;
+ void* inner_context;
+ const char* prefix;
+} PrefixedDeleteObserver;
+
+static void prefixed_delete_observer(void* context, const char* rel) {
+ PrefixedDeleteObserver* prefixed = context;
+ if (!prefixed->inner || !rel)
+ return;
+ char* joined = path_cat((char*)prefixed->prefix, rel);
+ if (joined) {
+ prefixed->inner(prefixed->inner_context, joined);
+ free(joined);
+ }
+}
+
/* --delete-missing-args exact-path deletions: each destination mirror in
manifest->missing is an explicit user request, so it is removed even when the
ordinary extras walk (with its protected prefixes) would leave it alone. The
@@ -3295,8 +3441,10 @@ static bool delete_extras_budgeted(const Config* config, DeleteManifest* manifes
--max-delete budget: once it is exhausted the remaining requests are skipped
and counted. Returns false only on a genuine error (a confinement failure on
a validated path or an I/O error), which fails the run. */
-static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* manifest,
- DeleteBudgetState* budget) {
+static bool delete_missing_args_budgeted_observed(const Config* config, DeleteManifest* manifest,
+ DeleteBudgetState* budget,
+ DeletePathObserver observer,
+ void* observer_context) {
if (!config || !manifest)
return false;
if (!manifest->missing || manifest->missing->size == 0)
@@ -3413,9 +3561,12 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m
budget->deleted >= budget->max_delete ? 0 : budget->max_delete - budget->deleted;
size_t contents_deleted = 0;
size_t contents_skipped = 0;
+ PrefixedDeleteObserver nested = {observer, observer_context, rel};
DeleteWalkResult walk =
- no_keeps ? delete_extras_limited(full, no_keeps, NULL, remaining, NULL, 0,
- &contents_deleted, &contents_skipped)
+ no_keeps ? delete_extras_limited_observed(full, no_keeps, NULL, remaining, NULL, 0,
+ NULL, &contents_deleted, &contents_skipped,
+ observer ? prefixed_delete_observer : NULL,
+ observer ? &nested : NULL)
: DELETE_WALK_ERROR;
if (no_keeps)
array_list_delete(no_keeps);
@@ -3455,6 +3606,8 @@ static bool delete_missing_args_budgeted(const Config* config, DeleteManifest* m
}
if (removed) {
budget->deleted++;
+ if (observer)
+ observer(observer_context, rel);
char* escaped = output_escape(rel, log_get_8_bit_output());
fprintf(stderr, " Deleted: %s\n", escaped ? escaped : "");
free(escaped);
@@ -3522,7 +3675,7 @@ bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest,
used = idx;
}
bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs,
- skips, used, out, count_out);
+ skips, used, config->protect_rules, out, count_out);
if (owned_prefixes) {
for (int i = 0; i < config->basis_count; i++)
free(owned_prefixes[i]);
@@ -3541,15 +3694,25 @@ bool manifest_delete_extras(const Config* config, DeleteManifest* manifest) {
bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest) {
DeleteBudgetState budget = {
.max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false};
- return delete_missing_args_budgeted(config, manifest, &budget);
+ return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL);
}
bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest,
size_t max_delete, size_t* deleted, size_t* skipped,
bool* limit_hit) {
+ return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted,
+ skipped, limit_hit, NULL, NULL);
+}
+
+bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest,
+ size_t max_delete, size_t* deleted,
+ size_t* skipped, bool* limit_hit,
+ DeletePathObserver observer,
+ void* observer_context) {
DeleteBudgetState budget = {
.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
- bool ok = delete_missing_args_budgeted(config, manifest, &budget);
+ bool ok =
+ delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context);
if (deleted)
*deleted = budget.deleted;
if (skipped)
@@ -3572,6 +3735,12 @@ DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* man
DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest,
size_t* deleted) {
+ return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL);
+}
+
+DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest,
+ size_t* deleted, DeletePathObserver observer,
+ void* observer_context) {
if (deleted)
*deleted = 0;
if (!config || !manifest)
@@ -3590,9 +3759,11 @@ DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManif
.deleted = 0,
.skipped = 0,
.limit_hit = false};
- if (config->delete_missing_args && !delete_missing_args_budgeted(config, manifest, &budget))
+ if (config->delete_missing_args &&
+ !delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context))
return DELETE_COMMIT_ERROR;
- if (config->use_delete && !delete_extras_budgeted(config, manifest, &budget))
+ if (config->use_delete &&
+ !delete_extras_budgeted_observed(config, manifest, &budget, observer, observer_context))
return DELETE_COMMIT_ERROR;
if (deleted)
*deleted = budget.deleted;
diff --git a/src/shared/file_receive.h b/src/shared/file_receive.h
index 2bf0c4e..e316000 100644
--- a/src/shared/file_receive.h
+++ b/src/shared/file_receive.h
@@ -3,6 +3,7 @@
#include "config.h"
#include "file_types.h"
+#include "utils.h"
#include
/* Server-side file receive/save path. */
@@ -23,6 +24,16 @@ File* file_receive_hardlink(int file_descriptor);
File* file_receive_symlink(int file_descriptor, const Config* config);
File* file_receive_special(int file_descriptor);
bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode);
+/* Testable basis quick-check / verification policy. file_basis_quick_match is
+ * rsync's metadata quick-check for a basis candidate (equal size is required
+ * separately by the caller; this adds the --size-only / mtime / --modify-window
+ * leg). file_basis_content_required reports whether a hit must ALSO be
+ * confirmed by a whole-file content digest (--verify-basis; false is the
+ * default rsync-parity behavior). */
+bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime,
+ long check_mtime_nsec);
+bool file_basis_content_required(const Config* config);
+
File* receive_incremental_check(int fd, const Config* config, bool* skipped);
/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set
* true only on the server-contacting --dry-run path when the file is not up to
@@ -126,6 +137,13 @@ bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest
bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest,
size_t max_delete, size_t* deleted, size_t* skipped,
bool* limit_hit);
+/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may
+ be NULL) is invoked for every destination-relative path truly removed. */
+bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest,
+ size_t max_delete, size_t* deleted,
+ size_t* skipped, bool* limit_hit,
+ DeletePathObserver observer,
+ void* observer_context);
/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's
partial --max-delete result: the budget allowed some deletions and the rest
were skipped (the run still stores all file data but the client exits 25). */
@@ -146,6 +164,11 @@ DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* man
removed (for the end-of-transfer wire stats). `deleted` may be NULL. */
DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest,
size_t* deleted);
+/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL)
+ is invoked for every destination-relative path truly removed. */
+DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest,
+ size_t* deleted, DeletePathObserver observer,
+ void* observer_context);
/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as
the delete pass would and append (strdup'd) destination-relative paths that
@@ -166,6 +189,22 @@ typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2
FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file,
const Config* config);
+/* Protocol 2.28.0 variant: also reports through `created` (when non-NULL)
+ * whether the destination entry did not exist before this save, and through
+ * `created_dirs` how many parent directories the confined walk created, so the
+ * receiver can build rsync's `Number of created files` breakdown. The plain
+ * file_save_to_disk_full() is this with both out-params NULL. */
+FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file,
+ const Config* config, bool* created,
+ unsigned* created_dirs);
bool file_save_to_disk(const char* root_directory, const File* file, const Config* config);
+/* Protocol 2.28.0 receiver counter accumulator: fold one successfully saved
+ * entry into `stats`, adding its receiver-observed literal bytes and, when
+ * `created`, the matching created-by-type counter (regular file / symlink /
+ * special) plus `created_dirs` implicitly-created parent directories.
+ * Non-first hardlink siblings contribute no literal bytes. */
+void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created,
+ unsigned created_dirs);
+
#endif
diff --git a/src/shared/file_types.h b/src/shared/file_types.h
index c2c127d..1a3bd82 100644
--- a/src/shared/file_types.h
+++ b/src/shared/file_types.h
@@ -57,6 +57,11 @@ typedef struct {
* equals the incoming file, and `data` is kept as the cross-filesystem
* fallback (a local copy) if the hard link cannot be created. */
char* basis_link;
+ /* Receiver-only, --copy-dest: when set (and basis_link is NULL), stream the
+ * basis file's bytes into the destination instead of `data`/`data->size`.
+ * This lets a basis larger than any whole-file bound materialize without
+ * buffering it; the source metadata on `metadata` is applied afterwards. */
+ char* basis_copy;
/* --hard-links (-H), sender + receiver wire state. link_group is a run-local
* id shared by every member of one source inode (0 = not part of a group).
* The FIRST member (link_first == true) carries its data on the wire and is
@@ -97,6 +102,11 @@ typedef struct {
* basis file (matched delta blocks) for this entry. 0 when the file was sent
* whole. Accumulated into ReceiverStats.matched_data by the receiver sink. */
unsigned long long matched_bytes;
+ /* Receiver-only (protocol 2.28.0) wire-stats tally: the literal delta fragment
+ * bytes this entry carried (DELTA_INSTR_LITERAL). 0 when the file was sent
+ * whole; the sink then falls back to the whole payload size. Accumulated
+ * into ReceiverStats.literal_bytes. */
+ unsigned long long literal_bytes;
} File;
/* The path that should be sent on the wire and used for the receiver-side
diff --git a/src/shared/filter.h b/src/shared/filter.h
index 97a373f..691860b 100644
--- a/src/shared/filter.h
+++ b/src/shared/filter.h
@@ -53,7 +53,7 @@ typedef struct {
char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */
} FilterRule;
-typedef struct {
+typedef struct FilterRuleList {
FilterRule** items; /* owned array of rule pointers */
int count;
int capacity;
diff --git a/src/shared/format.c b/src/shared/format.c
index 147d002..df2c463 100644
--- a/src/shared/format.c
+++ b/src/shared/format.c
@@ -105,25 +105,27 @@ bool format_dest_state_receive(int fd, OutputDestState* state) {
bool format_stats_send(int fd, const ReceiverStats* stats) {
if (!stats)
return false;
- unsigned long long matched = stats->matched_data;
- unsigned long long deleted = stats->deleted_files;
- unsigned long long would = stats->would_delete_count;
- return send_n_data(fd, &matched, sizeof(matched)) && send_n_data(fd, &deleted, sizeof(deleted)) &&
- send_n_data(fd, &would, sizeof(would));
+ unsigned long long fields[8] = {
+ stats->matched_data, stats->deleted_files, stats->would_delete_count, stats->literal_bytes,
+ stats->created_reg, stats->created_dir, stats->created_link, stats->created_special,
+ };
+ return send_n_data(fd, fields, sizeof(fields));
}
bool format_stats_receive(int fd, ReceiverStats* stats) {
if (!stats)
return false;
- unsigned long long matched = 0;
- unsigned long long deleted = 0;
- unsigned long long would = 0;
- if (!receive_n_data(fd, &matched, sizeof(matched)) ||
- !receive_n_data(fd, &deleted, sizeof(deleted)) || !receive_n_data(fd, &would, sizeof(would)))
+ unsigned long long fields[8] = {0};
+ if (!receive_n_data(fd, fields, sizeof(fields)))
return false;
memset(stats, 0, sizeof(*stats));
- stats->matched_data = matched;
- stats->deleted_files = deleted;
- stats->would_delete_count = would;
+ stats->matched_data = fields[0];
+ stats->deleted_files = fields[1];
+ stats->would_delete_count = fields[2];
+ stats->literal_bytes = fields[3];
+ stats->created_reg = fields[4];
+ stats->created_dir = fields[5];
+ stats->created_link = fields[6];
+ stats->created_special = fields[7];
return true;
}
diff --git a/src/shared/format.h b/src/shared/format.h
index fc904d4..9da9a93 100644
--- a/src/shared/format.h
+++ b/src/shared/format.h
@@ -57,14 +57,26 @@ bool format_dest_state_send(int fd, const OutputDestState* state);
bool format_dest_state_receive(int fd, OutputDestState* state);
/* End-of-transfer receiver counters reported through STATUS_STATS (protocol
- * 2.25.0) when the wire config carries report_stats. `would_delete_count` is
- * the number of destination-relative paths the receiver would have deleted in a
- * -n/--dry-run --delete run; that many wire strings immediately follow the
- * fixed record (sent/read by the caller). */
+ * 2.25.0, extended in 2.28.0) when the wire config carries report_stats.
+ * `would_delete_count` is the number of destination-relative paths the receiver
+ * would have deleted in a -n/--dry-run --delete run; that many wire strings
+ * immediately follow the fixed record (sent/read by the caller).
+ *
+ * Protocol 2.28.0 adds the receiver-observed counters the sender cannot see:
+ * `literal_bytes` is the file data the receiver actually stored literally
+ * (whole files plus the literal fragments of a delta) and the four `created_*`
+ * counters split the destination entries the receiver newly created by type,
+ * reproducing rsync's `Number of created files` breakdown and an exact
+ * `Literal data` for a delta run. */
typedef struct {
unsigned long long matched_data;
unsigned long long deleted_files;
unsigned long long would_delete_count;
+ unsigned long long literal_bytes;
+ unsigned long long created_reg;
+ unsigned long long created_dir;
+ unsigned long long created_link;
+ unsigned long long created_special;
} ReceiverStats;
/* Fixed-width STATUS_STATS counter record. The status frame and the optional
@@ -73,4 +85,27 @@ typedef struct {
bool format_stats_send(int fd, const ReceiverStats* stats);
bool format_stats_receive(int fd, ReceiverStats* stats);
+/* Sender-side file-list accounting for rsync's `--stats` block. Filled while
+ * the scan/send loops walk each entry: the flist counters describe every
+ * scanned source entry (transferred or skipped), while the transferred/literal
+ * counters describe only the regular files the receiver actually stored. The
+ * type split lets the client print rsync's `Number of files` breakdown; the
+ * receiver-only counters (matched data, deleted, created) come from
+ * STATUS_STATS. */
+typedef struct {
+ unsigned long long flist_reg;
+ unsigned long long flist_dir;
+ unsigned long long flist_link;
+ unsigned long long flist_special;
+ unsigned long long total_file_size; /* sum of entry sizes (link target len) */
+ unsigned long long transferred_regular; /* regular files actually stored */
+ unsigned long long transferred_file_size; /* source size of those files */
+ /* Whole-file accuracy: the `--stats` "Literal data" row. The sender counts
+ * the source size of every stored file, so a whole-file transfer matches
+ * rsync. A delta run actually ships only the literal fragments of the diff
+ * (the rest is matched/copied), so here the value is an upper bound, not
+ * rsync's literal-byte total; see RSYNC_COMPAT.md's `--stats` row. */
+ unsigned long long literal_data;
+} TransferStats;
+
#endif
diff --git a/src/shared/log.h b/src/shared/log.h
index 0e5a2ff..52cf38f 100644
--- a/src/shared/log.h
+++ b/src/shared/log.h
@@ -21,7 +21,26 @@ typedef enum {
LOG_INFO_MISC = 1u << 1,
LOG_INFO_SKIP = 1u << 2,
LOG_INFO_STATS = 1u << 3,
- LOG_INFO_ALL = LOG_INFO_COPY | LOG_INFO_MISC | LOG_INFO_SKIP | LOG_INFO_STATS,
+ /* rsync categories that map to a FastSync event (emitted in rsync's line
+ * format): del (deletions), remove (sender-side source removal), name
+ * (transferred entry names), flist (file-list header), nonreg (skipped
+ * non-regular files), progress (per-file progress). rsync's `backup`
+ * category is accepted for CLI parity but stays silent: the receiver does the
+ * backing-up and FastSync has no backup event to report from the sender. */
+ LOG_INFO_DEL = 1u << 4,
+ LOG_INFO_REMOVE = 1u << 5,
+ LOG_INFO_NAME = 1u << 6,
+ LOG_INFO_FLIST = 1u << 7,
+ LOG_INFO_NONREG = 1u << 8,
+ LOG_INFO_PROGRESS = 1u << 9,
+ /* Marker for `--info=name2` and higher: also print rsync's
+ "NAME is uptodate" line for entries the receiver already has. It rides in
+ the info_level bitset (there is no separate Config field) and is never set
+ by --info=all (which selects level 1). */
+ LOG_INFO_NAME_UPTODATE = 1u << 10,
+ LOG_INFO_ALL = LOG_INFO_COPY | LOG_INFO_MISC | LOG_INFO_SKIP | LOG_INFO_STATS | LOG_INFO_DEL |
+ LOG_INFO_REMOVE | LOG_INFO_NAME | LOG_INFO_FLIST | LOG_INFO_NONREG |
+ LOG_INFO_PROGRESS,
} LogInfoFlag;
void log_message(LogLevel log_level, const char* message, ...);
diff --git a/src/shared/multiprocessing.c b/src/shared/multiprocessing.c
index a4f56bb..c45109a 100644
--- a/src/shared/multiprocessing.c
+++ b/src/shared/multiprocessing.c
@@ -38,10 +38,12 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que
context->remove_source_files = NULL;
context->early_delete = false;
context->delete_plans = NULL;
+ context->delete_suppressed = false;
context->scan_stopped_early = false;
context->total_files = 0;
context->progress_bytes = 0;
context->total_bytes = 0;
+ memset(&context->stats, 0, sizeof(context->stats));
context->sender_done = false;
atomic_init(&context->cancelled, false);
protocol_session_init(&context->allocation_session, -1, -1);
diff --git a/src/shared/multiprocessing.h b/src/shared/multiprocessing.h
index d2e5284..a6fda8c 100644
--- a/src/shared/multiprocessing.h
+++ b/src/shared/multiprocessing.h
@@ -9,6 +9,7 @@
#include "config.h"
#include "delete_plan.h"
#include "file.h"
+#include "format.h"
#include "protocol.h"
#include "queue.h"
#include "stop_condition.h"
@@ -82,10 +83,22 @@ typedef struct {
thread transmits the root plan before any data and the remaining plans
alongside the chunks. Set once before the worker threads start. */
DeletePlanSender* delete_plans;
+ /* A scan I/O error without --ignore-errors suppressed deletion: the prebuilt
+ keep-set/plans were dropped, and the streaming scanner must not build a
+ fresh manifest or re-send the per-directory plans. Set once before the
+ worker threads start. */
+ bool delete_suppressed;
mtx_t mutex_progress;
int total_files;
unsigned long long progress_bytes;
unsigned long long total_bytes;
+ /* Per-type flist / transferred accounting for the rsync --stats breakdown and
+ the progress `to-chk` denominator. Owned by the sender thread: it is the
+ only writer (the entry/transfer notes in send_chunks_multithreaded) and it
+ reads the totals in its completion tail, so no lock is needed. This is NOT
+ guarded by mutex_progress (which covers total_files/progress_bytes/
+ total_bytes/sender_done). */
+ TransferStats stats;
bool sender_done;
atomic_bool cancelled;
ProtocolSession allocation_session;
diff --git a/src/shared/protocol.c b/src/shared/protocol.c
index 70c0d8a..2a3f844 100644
--- a/src/shared/protocol.c
+++ b/src/shared/protocol.c
@@ -183,12 +183,28 @@ void io_set_bwlimit(unsigned long long bytes_per_sec) {
mtx_unlock(&bw_mutex);
}
+unsigned long long io_get_bwlimit(void) {
+ return global_bwlimit();
+}
+
+/* rsync's throttle (io.c sleep_for_bwlimit) sleeps once its unslept debt
+ * reaches ~100 ms of bandwidth, so its effective initial burst is about 0.1 s
+ * worth of bytes, not a full second. FastSync models the same with a token
+ * bucket whose capacity is bwlimit/10, so a throttled run paces like rsync
+ * instead of sending a full second's worth up front. */
+static long long bw_burst_capacity(unsigned long long bwlimit) {
+ if (bwlimit == 0)
+ return 0;
+ long long burst = (long long)(bwlimit / 10);
+ return burst > 0 ? burst : 1;
+}
+
void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec) {
if (!session)
return;
session->bwlimit =
bytes_per_sec > (unsigned long long)LLONG_MAX ? (unsigned long long)LLONG_MAX : bytes_per_sec;
- session->bw_tokens = (long long)session->bwlimit;
+ session->bw_tokens = bw_burst_capacity(session->bwlimit);
struct timespec now;
clock_gettime(CLOCK_MONOTONIC, &now);
session->bw_last_refill_sec = now.tv_sec;
@@ -222,8 +238,9 @@ static void bw_throttle_session(ProtocolSession* session, size_t bytes_written)
long long tokens_to_add = (long long)((double)session->bwlimit * elapsed_ns / 1000000000.0);
session->bw_tokens += tokens_to_add;
- if (session->bw_tokens > (long long)session->bwlimit)
- session->bw_tokens = (long long)session->bwlimit;
+ long long burst = bw_burst_capacity(session->bwlimit);
+ if (session->bw_tokens > burst)
+ session->bw_tokens = burst;
session->bw_tokens -= bytes_written;
@@ -234,7 +251,11 @@ static void bw_throttle_session(ProtocolSession* session, size_t bytes_written)
poll(NULL, 0, (int)(deficit_us / 1000));
else
usleep((useconds_t)deficit_us);
+ /* Reset the bucket AFTER the sleep: crediting the sleep duration as elapsed
+ refill time would cancel half the throttle (the next call would see the
+ whole sleep as refill and immediately grant a fresh burst). */
session->bw_tokens = 0;
+ clock_gettime(CLOCK_MONOTONIC, &now);
session->bw_last_refill_sec = now.tv_sec;
session->bw_last_refill_nsec = now.tv_nsec;
}
diff --git a/src/shared/protocol.h b/src/shared/protocol.h
index 30bc5af..4348f09 100644
--- a/src/shared/protocol.h
+++ b/src/shared/protocol.h
@@ -191,7 +191,9 @@ enum NET_STATUS {
* (--delete-delay). Payload: an int32 has_config flag (1 on the first plan
* of the run, 0 afterwards); when set, the three global config sections
* (protected-prefix count+paths, size-skipped count+paths, missing-args
- * count+paths); then the destination-relative directory path wire string
+ * count+paths); then an int32 apply flag (1 for a real plan, 0 for a
+ * config-only carrier frame that must not walk a directory); then the
+ * destination-relative directory path wire string
* ("." for the receive root); then the child-directory count + names and the
* child-file count + names that must be kept. Appended after
* STATUS_DEST_INFO so no existing status is renumbered. */
@@ -207,6 +209,7 @@ enum NET_STATUS {
void io_set_fds(int read_fd, int write_fd);
void io_set_bwlimit(unsigned long long bytes_per_sec);
+unsigned long long io_get_bwlimit(void);
void io_set_ssl(SSL* ssl);
SSL* io_get_ssl(void);
diff --git a/src/shared/utils.c b/src/shared/utils.c
index 14bdd7d..52bb5b2 100644
--- a/src/shared/utils.c
+++ b/src/shared/utils.c
@@ -2,6 +2,7 @@
#include "array_list.h"
#include "log.h"
#include
+#include
#include
#include
#include
@@ -52,6 +53,31 @@ bool path_is_within_root(const char* root, const char* path) {
return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/');
}
+/* Borrowed transfer-relative view of `path`: strip any leading '/' and then a
+ * `root` prefix (its own leading/trailing slashes tolerated), returning a
+ * pointer into `path`. Non-allocating, so it is safe on the hot scan/print
+ * paths. A NULL/empty root, or a path not under `root`, leaves only the
+ * leading-slash strip. `path` must be NUL-terminated and live in the caller. */
+const char* utils_strip_transfer_root(const char* path, const char* root) {
+ if (path == NULL)
+ return NULL;
+ const char* rel = path;
+ while (*rel == '/')
+ rel++;
+ if (root == NULL)
+ return rel;
+ while (*root == '/')
+ root++;
+ size_t root_len = strlen(root);
+ while (root_len > 0 && root[root_len - 1] == '/')
+ root_len--;
+ if (root_len == 0)
+ return rel;
+ if (strncmp(rel, root, root_len) == 0 && (rel[root_len] == '/' || rel[root_len] == '\0'))
+ return rel + root_len + (rel[root_len] == '/' ? 1 : 0);
+ return rel;
+}
+
/* Open the destination root directory itself, confined to the authorized root.
* NOTE (do not merge with file_open_secure_parent): this walk opens dest_root
* (a directory that must already exist) and returns its fd, whereas
@@ -117,6 +143,47 @@ char* str_dup(const char* string) {
return new_string;
}
+int env_choice_first(const char* env_name, int (*resolve)(const char*), bool* specified) {
+ if (specified)
+ *specified = false;
+ if (!env_name || !resolve)
+ return -1;
+ const char* env = getenv(env_name);
+ if (!env)
+ return -1;
+
+ bool saw_nonblank = false;
+ const char* p = env;
+ while (*p) {
+ if (*p == '&')
+ break;
+ if (isspace((unsigned char)*p)) {
+ p++;
+ continue;
+ }
+ saw_nonblank = true;
+ char token[64];
+ size_t len = 0;
+ while (*p && *p != '&' && !isspace((unsigned char)*p)) {
+ if (len < sizeof(token) - 1)
+ token[len++] = *p;
+ p++;
+ }
+ token[len] = '\0';
+ if (len > 0) {
+ int id = resolve(token);
+ if (id >= 0) {
+ if (specified)
+ *specified = true;
+ return id;
+ }
+ }
+ }
+ if (specified)
+ *specified = saw_nonblank;
+ return -1;
+}
+
#define STR_HASH_SET_MIN_CAPACITY 16
static size_t str_hash_set_hash(const char* key, size_t len) {
@@ -615,8 +682,10 @@ static bool is_synced_dir(const PathIndex* dirs, const char* rel) {
like any other non-directory extra (never followed). */
static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep,
const PathIndex* dirs, DeleteBudget* budget,
- const DeleteSkipEntry* skips, int skip_count, bool parent_deletable,
- bool* all_removed) {
+ const DeleteSkipEntry* skips, int skip_count,
+ const FilterRuleList* protect_rules, bool parent_deletable,
+ bool* all_removed, DeletePathObserver observer,
+ void* observer_context) {
/* openat(dirfd, ".") opens an independent file description: a dup() would
share dirfd's file offset and a prior pass could leave the stream drained. */
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
@@ -661,12 +730,23 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k
free(child_rel);
continue;
}
- if (S_ISDIR(st.st_mode)) {
+ bool is_dir = S_ISDIR(st.st_mode);
+ if (protect_rules && filter_rules_apply_side(protect_rules, child_rel, entry->d_name, is_dir,
+ FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) {
+ /* A first-match protect rule shields the extra; for a directory the whole
+ subtree is shielded (rsync prunes an excluded directory), so do not
+ descend. */
+ local_survives = true;
+ free(child_rel);
+ continue;
+ }
+ if (is_dir) {
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
bool child_all_removed = false;
if (childfd >= 0) {
- if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, deletable,
- &child_all_removed))
+ if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count,
+ protect_rules, deletable, &child_all_removed, observer,
+ observer_context))
operation_ok = false;
close(childfd);
} else if (errno != ENOENT) {
@@ -693,6 +773,8 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k
local_survives = true;
} else {
budget->deleted++;
+ if (observer)
+ observer(observer_context, child_rel);
}
} else {
local_survives = true;
@@ -713,6 +795,8 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k
local_survives = true;
} else {
budget->deleted++;
+ if (observer)
+ observer(observer_context, child_rel);
char* escaped_path = output_escape(child_rel, log_get_8_bit_output());
fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : "");
free(escaped_path);
@@ -730,7 +814,8 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k
reportable children (depth-first), matching the delete pass's ordering. */
static bool list_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep,
const PathIndex* dirs, ArrayList* out, size_t* recorded,
- const DeleteSkipEntry* skips, int skip_count, bool parent_deletable,
+ const DeleteSkipEntry* skips, int skip_count,
+ const FilterRuleList* protect_rules, bool parent_deletable,
bool* all_removed) {
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
if (scanfd < 0)
@@ -764,12 +849,21 @@ static bool list_extras_fd(int dirfd, const char* rel_path, const PathIndex* kee
free(child_rel);
continue;
}
- if (S_ISDIR(st.st_mode)) {
+ bool is_dir = S_ISDIR(st.st_mode);
+ if (protect_rules && filter_rules_apply_side(protect_rules, child_rel, entry->d_name, is_dir,
+ FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) {
+ /* Mirror the delete walk: a protected entry is never reported as a
+ would-delete and a protected directory's subtree is not enumerated. */
+ local_survives = true;
+ free(child_rel);
+ continue;
+ }
+ if (is_dir) {
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
bool child_all_removed = false;
if (childfd >= 0) {
if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count,
- deletable, &child_all_removed))
+ protect_rules, deletable, &child_all_removed))
operation_ok = false;
close(childfd);
} else if (errno != ENOENT) {
@@ -820,7 +914,7 @@ static bool list_extras_fd(int dirfd, const char* rel_path, const PathIndex* kee
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
- ArrayList* out, size_t* count_out) {
+ const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out) {
if (count_out)
*count_out = 0;
if (!manifest || !out)
@@ -856,7 +950,7 @@ bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
bool all_removed = false;
size_t recorded = 0;
bool ok = list_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, out, &recorded, skips,
- skip_count, false, &all_removed);
+ skip_count, protect_rules, false, &all_removed);
if (close(rootfd) != 0)
ok = false;
path_index_free(&keep);
@@ -867,10 +961,13 @@ bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
return ok;
}
-DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
- const ArrayList* synced_dirs, size_t max_delete,
- const DeleteSkipEntry* skips, int skip_count,
- size_t* deleted_out, size_t* skipped_out) {
+DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
+ const ArrayList* synced_dirs, size_t max_delete,
+ const DeleteSkipEntry* skips, int skip_count,
+ const FilterRuleList* protect_rules,
+ size_t* deleted_out, size_t* skipped_out,
+ DeletePathObserver observer,
+ void* observer_context) {
if (deleted_out)
*deleted_out = 0;
if (skipped_out)
@@ -910,8 +1007,9 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* m
}
DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
bool all_removed = false;
- bool ok = delete_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &budget, skips,
- skip_count, false, &all_removed);
+ bool ok =
+ delete_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &budget, skips, skip_count,
+ protect_rules, false, &all_removed, observer, observer_context);
if (close(rootfd) != 0)
ok = false;
path_index_free(&keep);
@@ -926,8 +1024,18 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* m
return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK;
}
+DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
+ const ArrayList* synced_dirs, size_t max_delete,
+ const DeleteSkipEntry* skips, int skip_count,
+ const FilterRuleList* protect_rules, size_t* deleted_out,
+ size_t* skipped_out) {
+ return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips,
+ skip_count, protect_rules, deleted_out, skipped_out, NULL,
+ NULL);
+}
+
bool delete_extras(const char* dest_root, const ArrayList* manifest) {
- return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL) ==
+ return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) ==
DELETE_WALK_OK;
}
diff --git a/src/shared/utils.h b/src/shared/utils.h
index 578706f..5d746f6 100644
--- a/src/shared/utils.h
+++ b/src/shared/utils.h
@@ -2,6 +2,7 @@
#define UTILS_H
#include "array_list.h"
+#include "filter.h"
#include
#include
#include
@@ -81,6 +82,17 @@ bool path_index_has_descendant(const PathIndex* index, const char* path);
char* str_dup(const char* string);
char* output_escape(const char* string, bool eight_bit_output);
+
+/* Resolve the first supported name from a rsync algorithm-preference
+ * environment variable (RSYNC_COMPRESS_LIST / RSYNC_CHECKSUM_LIST). `resolve`
+ * maps a case-insensitive name to an algorithm id (>= 0) or -1 for an unknown
+ * name. rsync's syntax is a whitespace-separated list (comma/colon are NOT
+ * separators); the client-side half ends at '&'. Unknown entries are skipped
+ * and the first resolvable one wins. *specified is set true when the variable
+ * holds at least one non-blank character. Returns the first resolvable id, or
+ * -1 when the variable is unset/blank or names no supported algorithm. */
+int env_choice_first(const char* env_name, int (*resolve)(const char*), bool* specified);
+
/* Upper bound on one line/token read from a local list file (--files-from,
* --exclude-from/--include-from, .rsync-filter). Mirrors MAX_STRING_SIZE and
* stops a hostile multi-gigabyte line from forcing unbounded allocation. */
@@ -137,7 +149,27 @@ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSki
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
const ArrayList* synced_dirs, size_t max_delete,
const DeleteSkipEntry* skips, int skip_count,
- size_t* deleted_out, size_t* skipped_out);
+ const FilterRuleList* protect_rules, size_t* deleted_out,
+ size_t* skipped_out);
+
+/* Optional per-deletion observer: called for each destination-relative path
+ actually removed (a file, symlink, or directory), in removal order, so the
+ receiver can stream rsync's `--info=del`/`--info=remove` lines. */
+typedef void (*DeletePathObserver)(void* context, const char* rel_path);
+
+/* `delete_extras_limited_observed` is delete_extras_limited with an optional
+ * observer; the observer is invoked only for entries truly removed. When
+ * `protect_rules` is non-NULL its receiver-side verdict is evaluated for every
+ * candidate extra: a first-match PROTECT leaves the entry (and, for a
+ * directory, its whole subtree) in place, while RISK/NONE fall through to the
+ * ordinary skip-prefix/keep-set logic. */
+DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
+ const ArrayList* synced_dirs, size_t max_delete,
+ const DeleteSkipEntry* skips, int skip_count,
+ const FilterRuleList* protect_rules,
+ size_t* deleted_out, size_t* skipped_out,
+ DeletePathObserver observer,
+ void* observer_context);
/* Read-only companion to delete_extras_limited: walk the destination exactly as
the delete pass would and APPEND (strdup'd) destination-relative paths that
WOULD be removed, without touching disk. Used for -n/--dry-run --delete
@@ -145,7 +177,7 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* m
strings appended to `out` and receives their count in *count_out. */
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
- ArrayList* out, size_t* count_out);
+ const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out);
bool delete_extras(const char* dest_root, const ArrayList* manifest);
/* Open the existing destination directory at `dest_root`, confined to the
authorized root with an O_NOFOLLOW component walk (the same confinement the
@@ -177,6 +209,11 @@ const char* utils_get_authorized_root_path(void);
* callers guarantee this); this is containment by string, not by resolved
* symlinks. Shared by the utils and file secure-walk root confinement. */
bool path_is_within_root(const char* root, const char* path);
+/* Non-allocating transfer-relative view of `path`: strip any leading '/' and
+ * then a `root` prefix (leading/trailing slashes tolerated), returning a
+ * borrowed pointer into `path`. A NULL/empty root, or a path not under
+ * `root`, yields just the leading-slash strip. `path`/`root` must stay alive. */
+const char* utils_strip_transfer_root(const char* path, const char* root);
/* True when `path` contains a ".." component. This is a purely lexical
* dot-dot check: an absolute path is NOT rejected here, because default
* (non-relative) transfers legitimately put the sender's absolute source path
diff --git a/test_data-manual/dst/f.bin b/test_data-manual/dst/f.bin
new file mode 100644
index 0000000..870ff8b
--- /dev/null
+++ b/test_data-manual/dst/f.bin
@@ -0,0 +1 @@
+OLDDEST
diff --git a/test_data-manual/dst/test_data-manual/src/f.bin b/test_data-manual/dst/test_data-manual/src/f.bin
new file mode 100644
index 0000000..924c75c
--- /dev/null
+++ b/test_data-manual/dst/test_data-manual/src/f.bin
@@ -0,0 +1 @@
+NEWCONTENT
diff --git a/test_data-manual/src/f.bin b/test_data-manual/src/f.bin
new file mode 100644
index 0000000..924c75c
--- /dev/null
+++ b/test_data-manual/src/f.bin
@@ -0,0 +1 @@
+NEWCONTENT
diff --git a/tests/fuzz/fuzz_config_receive.c b/tests/fuzz/fuzz_config_receive.c
index 95a01d3..f4e43a6 100644
--- a/tests/fuzz/fuzz_config_receive.c
+++ b/tests/fuzz/fuzz_config_receive.c
@@ -119,6 +119,13 @@ static void build_canonical_frame(void) {
cfg->usermap[0].to = MAP_TO;
cfg->usermap[0].to_name = NULL;
}
+ /* Force a non-empty receiver delete-protection block so the fuzzer mutates
+ * its rule count, action/sides codes and pattern strings. */
+ cfg->filters = array_list_create(free);
+ if (cfg->filters) {
+ array_list_add(cfg->filters, str_dup("P *.log"));
+ array_list_add(cfg->filters, str_dup("+r keep/**"));
+ }
if (!cfg->send_directory || !cfg->receive_root_directory || !cfg->usermap) {
config_delete(cfg);
return;
diff --git a/tests/integration/README.md b/tests/integration/README.md
new file mode 100644
index 0000000..bd250cc
--- /dev/null
+++ b/tests/integration/README.md
@@ -0,0 +1,80 @@
+# Integration tests
+
+The integration suite drives the built `build/server` and `build/client`
+against local corpora. Unit tests live in `tests/` (the custom C framework);
+the Python suite here covers the full transfer pipeline, transports, features,
+and rsync parity.
+
+## Running
+
+```bash
+# Full suite (excludes privilege-dependent tests on CI runners)
+python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
+
+# Fast PR subset only
+python3 -m pytest tests/integration/ -n 4 --dist=load -m ci
+```
+
+The tests expect `build/server` and `build/client` (configure/build with CMake
+first); `common.py` derives `BUILD_DIR` from the repository root.
+
+## Differential rsync-parity gate
+
+`test_differential_parity.py` runs the **same** transfer with real
+`rsync 3.4.1` and with FastSync over separate destinations, then compares:
+
+- the destination trees — relative paths, file content hashes, symlink
+ targets, modes (where the case is about perms), and hard-link grouping;
+- the normalized stdout for output-oriented flags (`-i`,
+ `--out-format=...`, `--stats`), after stripping volatile fields
+ (timings, rates, wire byte counts) and directory-only itemize lines that
+ FastSync's recursive scanner documents as absent.
+
+FastSync mirrors the absolute source path under its receive root (see
+`get_dest_received_dir`); the harness normalizes that layout (and the
+`-R`/`--files-from` layouts) before comparing.
+
+```bash
+# Fast subset that guards the ✅ surface on pull requests
+python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci
+
+# Full differential case table (`_CASES`): every row is marked `parity`, and a
+# case with an allowlisted residual in `parity_caveats.py` is included too.
+python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity
+```
+
+`-m parity` selects only the `_CASES` table in this module. Differential
+coverage for options outside that table (`--temp-dir`, `--delay-updates`,
+`--dry-run`, `--fuzzy`, the basis-dir options, `-M` over daemon/TCP, and
+receiver filter-protect) lives in dedicated modules (`test_option_parity.py`,
+`test_parity_blockers.py`, `test_parity_quickwins.py`, ...) and is not part of
+this gate. The suite skips cleanly when `rsync` is not installed.
+
+## Allowlist (`parity_caveats.py`)
+
+`parity_caveats.py` is the single data-driven allowlist of known differences.
+Each entry maps a case id to the aspects that may differ (`tree`, `stdout`,
+`extra`, `rc`) and cites the governing row in `RSYNC_COMPAT.md`:
+
+```python
+CAVEATS = {
+ # no known residuals at present -- the burn-down reached zero
+ # "some_case_id": {"tree": "documented residual ... ref: RSYNC_COMPAT.md ..."},
+}
+```
+
+A differential mismatch in an aspect that is **not** listed fails the gate with
+a readable tree/stdout diff.
+
+If a case is allowlisted but now matches rsync, the gate emits a loud warning
+naming the stale entry — that is the parity burn-down signal. Run with
+`FASTSYNC_PARITY_STRICT=1` to make stale entries fail instead (the full CI
+parity job sets this). To add a residual:
+
+1. Reproduce it with `-m parity` and read the failure's tree/stdout diff.
+2. Confirm it is a documented `⚠️`/`❌` residual (or get the `✅` row
+ reclassified) and cite the row.
+3. Add the case id and aspect(s) to `CAVEATS`, keeping the reason concise.
+
+Do not allowlist an undocumented divergence from a `✅` row — fix it or get the
+row reclassified first.
diff --git a/tests/integration/parity_caveats.py b/tests/integration/parity_caveats.py
new file mode 100644
index 0000000..c80ee6b
--- /dev/null
+++ b/tests/integration/parity_caveats.py
@@ -0,0 +1,34 @@
+"""Data-driven allowlist for the differential rsync-parity gate.
+
+Every entry maps a case id (see ``test_differential_parity.py``) to the aspects
+that are *known* to differ from ``rsync 3.4.1`` and the documented reason. A
+differential mismatch in an aspect that is **not** listed here fails the gate.
+
+Aspect keys
+-----------
+``tree`` destination tree differs (paths, file hashes, symlink targets,
+ modes, hardlink grouping)
+``stdout`` normalized output for ``-i`` / ``--stats`` / ``--out-format``
+``extra`` a case-specific assertion differs (basis/inode checks, ...)
+``rc`` exit status differs
+
+Burn-down
+---------
+If a case is listed here but now matches rsync, the gate emits a loud
+``pytest`` warning naming the stale entry: delete the entry (and, when the
+underlying row in ``RSYNC_COMPAT.md`` is now parity, update that row). Set
+``FASTSYNC_PARITY_STRICT=1`` to turn stale entries into failures in CI.
+
+Keep the values concise but cite the governing row so the entry can be
+re-triaged when the row moves.
+"""
+
+# case id -> {aspect: "reason (ref: RSYNC_COMPAT.md ...)"}
+CAVEATS = {}
+
+# Accepted aspect names (guards against typos in this file).
+ASPECTS = ("tree", "stdout", "extra", "rc")
+
+
+def caveat_for(case_id: str) -> dict:
+ return CAVEATS.get(case_id, {})
diff --git a/tests/integration/parity_harness.py b/tests/integration/parity_harness.py
new file mode 100644
index 0000000..23c2ea7
--- /dev/null
+++ b/tests/integration/parity_harness.py
@@ -0,0 +1,584 @@
+"""Differential rsync-parity harness.
+
+Runs the SAME transfer with real ``rsync`` and with FastSync over separate
+destinations and compares the resulting trees and (optionally) normalized
+stdout. ``test_differential_parity.py`` drives this module with a table of
+cases; ``parity_caveats.py`` is the data-driven allowlist of documented
+residuals.
+
+Design notes
+------------
+FastSync mirrors the *absolute* source path below its receive root, while
+rsync copies the source contents directly into the destination. ``Case.layout``
+tells the harness which pair of directory roots to compare:
+
+* ``MIRROR`` -- rsync ``DEST/`` vs FastSync ``DEST//`` (the common
+ case; matches ``common.get_dest_received_dir``).
+* ``MIRROR_ABS`` -- ``rsync -R`` without a cut lays the full absolute path
+ under the destination, so rsync ``DEST//`` is compared against the
+ same FastSync mirror path.
+* ``RELATIVE`` -- ``rsync -R --files-from`` lays bare relative paths under the
+ destination and FastSync does the same, so both destination roots compare
+ directly.
+
+Only ``tests/integration/common.py`` is used to reach the build products and the
+server manager; the harness never duplicates that plumbing.
+"""
+import difflib
+import hashlib
+import os
+import re
+import shutil
+import subprocess
+import sys
+from dataclasses import dataclass
+from typing import Callable, Dict, List, Optional, Tuple
+
+sys.path.insert(0, os.path.dirname(__file__))
+from common import ( # noqa: E402 (path bootstrap above)
+ TEST_DATA_DIR,
+ clean_dir,
+ get_dest_received_dir,
+ run_client,
+)
+
+RSYNC = shutil.which("rsync")
+
+# Comparison layouts (see module docstring).
+MIRROR = "mirror"
+MIRROR_ABS = "mirror_abs"
+RELATIVE = "relative"
+
+# stdout comparators.
+STDOUT_NONE = None
+STDOUT_ITEMIZE = "itemize"
+STDOUT_OUTFMT = "outfmt"
+STDOUT_STATS = "stats"
+STDOUT_PROGRESS = "progress"
+
+# rsync --stats lines that are protocol-independent and must match exactly.
+# `Number of files` and `Number of created files` carry rsync's per-type
+# breakdown; protocol 2.28.0 reports the receiver-created split over
+# STATUS_STATS. Deliberately excluded: Total bytes sent/received (protocol
+# framing differs, see the `--stats` row in RSYNC_COMPAT.md).
+STATS_KEYS = (
+ "Number of files",
+ "Number of created files",
+ "Number of deleted files",
+ "Number of regular files transferred",
+ "Total file size",
+ "Total transferred file size",
+ "Literal data",
+ "Matched data",
+ "File list size",
+)
+
+_ITEMIZE_RE = re.compile(r"^(<|>|c|h|\.|\*)[fdLDS][.+\-][.+\-][.+\-][.+\-]")
+_PROGRESS_TOTAL_RE = re.compile(r"to-chk=\d+/(\d+)")
+_PROGRESS_XFR_RE = re.compile(r"xfr#(\d+)")
+
+
+@dataclass
+class Case:
+ """One differential scenario: a corpus, a flag set, and how to compare."""
+
+ id: str
+ corpus: str
+ flags: List[str]
+ fastsync_flags: Optional[List[str]] = None
+ layout: str = MIRROR
+ server_args: Tuple[str, ...] = ("--allow-super",)
+ seed: Optional[Callable] = None
+ stdout: Optional[str] = STDOUT_NONE
+ compare_modes: bool = False
+ compare_hardlinks: bool = False
+ ignore_paths: Tuple[str, ...] = ()
+ extra_check: Optional[Callable] = None
+ files_from: Optional[Tuple[str, ...]] = None
+ # rsync receives ``src + "/"``; FastSync mirrors the path it is given, so a
+ # trailing-slash-sensitive case must hand FastSync the same form.
+ fs_src_suffix: str = ""
+ # Some cases have an unspecified result (e.g. which extras survive a
+ # partial --max-delete abort): assert the case-specific invariants via
+ # extra_check and skip the exact-tree comparison.
+ compare_tree: bool = True
+ ci: bool = False
+ ref: str = ""
+
+ def fs_flags(self) -> List[str]:
+ return list(self.flags if self.fastsync_flags is None else self.fastsync_flags)
+
+
+# ---------------------------------------------------------------------------
+# Corpora
+# ---------------------------------------------------------------------------
+
+# Deterministic mtimes so quick-check decisions are reproducible.
+_SRC_MTIME = 1_600_000_000
+
+
+def _write(path: str, data: bytes, mode: Optional[int] = None) -> None:
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ with open(path, "wb") as fh:
+ fh.write(data)
+ os.utime(path, (_SRC_MTIME, _SRC_MTIME))
+ if mode is not None:
+ os.chmod(path, mode)
+
+
+def _set_mode(path: str, mode: int) -> None:
+ os.chmod(path, mode)
+
+
+def corpus_basic(root: str) -> None:
+ """Regular files + nested dirs (dirs are implied by their files)."""
+ clean_dir(root)
+ _write(os.path.join(root, "a.txt"), b"hello world\n")
+ _write(os.path.join(root, "sub", "b.bin"),
+ bytes((i * 7) & 0xFF for i in range(5000)))
+ _write(os.path.join(root, "sub", "deep", "c.txt"), "w\u00f6rld\n".encode())
+
+
+def corpus_unicode(root: str) -> None:
+ clean_dir(root)
+ _write(os.path.join(root, "uni \u00f1\u6587.txt"), b"unicode\n")
+ _write(os.path.join(root, "sub", "sp ace \u00e9.dat"), b"spaced\n")
+ _set_mode(os.path.join(root, "sub"), 0o750)
+
+
+def corpus_links(root: str) -> None:
+ corpus_basic(root)
+ os.symlink("a.txt", os.path.join(root, "rel_link"))
+ os.symlink("/etc/hostname", os.path.join(root, "abs_link"))
+ os.symlink("nowhere/target", os.path.join(root, "broken_link"))
+
+
+def corpus_hardlinks(root: str) -> None:
+ clean_dir(root)
+ _write(os.path.join(root, "h1.txt"), b"hardlinked payload\n")
+ os.link(os.path.join(root, "h1.txt"), os.path.join(root, "h2.txt"))
+ _write(os.path.join(root, "other.txt"), b"other\n")
+
+
+def corpus_sparse(root: str) -> None:
+ clean_dir(root)
+ _write(os.path.join(root, "small.txt"), b"small\n")
+ sparse = os.path.join(root, "sparse.bin")
+ with open(sparse, "wb") as fh:
+ fh.seek(1024 * 1024 - 1)
+ fh.write(b"\0")
+ os.utime(sparse, (_SRC_MTIME, _SRC_MTIME))
+
+
+def corpus_filters(root: str) -> None:
+ clean_dir(root)
+ _write(os.path.join(root, "keep.txt"), b"keep\n")
+ _write(os.path.join(root, "drop.log"), b"log\n")
+ _write(os.path.join(root, "sub", "keep2.txt"), b"keep2\n")
+ _write(os.path.join(root, "sub", "drop2.log"), b"log2\n")
+ _write(os.path.join(root, "sub", "data.bin"), b"bin\n")
+
+
+def corpus_empty_dir(root: str) -> None:
+ clean_dir(root)
+ _write(os.path.join(root, "keep.txt"), b"keep\n")
+ os.makedirs(os.path.join(root, "emptydir"), exist_ok=True)
+ os.utime(os.path.join(root, "emptydir"), (_SRC_MTIME, _SRC_MTIME))
+ _write(os.path.join(root, "nonempty", "f.txt"), b"f\n")
+
+
+def corpus_multidir(root: str) -> None:
+ """Multi-directory tree for the --progress file-list naming/denominator.
+
+ Nested files, a directory-only branch, an empty directory and a symlink
+ exercise every file-list entry type rsync counts in `to-chk` but FastSync's
+ streaming scanner never emits as a transfer entry.
+ """
+ clean_dir(root)
+ _write(os.path.join(root, "a.txt"), b"alpha\n")
+ _write(os.path.join(root, "b.txt"), b"bravo\n")
+ _write(os.path.join(root, "sub1", "c.txt"), b"charlie\n")
+ _write(os.path.join(root, "sub1", "deep", "d.txt"), b"delta\n")
+ _write(os.path.join(root, "sub2", "e.txt"), b"echo\n")
+ os.symlink("a.txt", os.path.join(root, "link1"))
+ os.makedirs(os.path.join(root, "emptydir"), exist_ok=True)
+ os.utime(os.path.join(root, "emptydir"), (_SRC_MTIME, _SRC_MTIME))
+
+
+def corpus_relative(root: str) -> None:
+ """Tree for the -R/--files-from cases."""
+ clean_dir(root)
+ _write(os.path.join(root, "a.txt"), b"a\n")
+ _write(os.path.join(root, "b.txt"), b"b\n")
+ _write(os.path.join(root, "sub", "x.txt"), b"x\n")
+ _write(os.path.join(root, "sub", "y.txt"), b"y\n")
+ os.makedirs(os.path.join(root, "dir1"), exist_ok=True)
+ os.utime(os.path.join(root, "dir1"), (_SRC_MTIME, _SRC_MTIME))
+ _write(os.path.join(root, "dir1", "keep.txt"), b"keep\n")
+
+
+def corpus_iconv(root: str) -> None:
+ """Latin-1 (ISO-8859-1) encoded filenames, matching the --iconv direction."""
+ clean_dir(root)
+ for rel, data in ((b"caf\xe9.txt", b"caf\xe9\n"),
+ (os.path.join(b"sub", b"\xfcber.txt"), b"\xfcber\n")):
+ full = os.path.join(os.fsencode(root), rel)
+ os.makedirs(os.path.dirname(full), exist_ok=True)
+ with open(full, "wb") as fh:
+ fh.write(data)
+ os.utime(full, (_SRC_MTIME, _SRC_MTIME))
+
+
+# Payload for the --fuzzy basis corpus: large enough for the delta engine's
+# 16 KiB minimum and with repeated content so a coinciding basis yields a
+# non-zero (and identical) Matched data count in both tools.
+FUZZY_PAYLOAD = (b"the quick brown fox jumps over the lazy dog\n" * 2000)[:65536]
+
+
+def corpus_fuzzy(root: str) -> None:
+ """A named regular file; the `fuzzy` seed adds the similar-suffix sibling."""
+ clean_dir(root)
+ _write(os.path.join(root, "report_v2.txt"), FUZZY_PAYLOAD)
+
+
+CORPORA: Dict[str, Callable[[str], None]] = {
+ "basic": corpus_basic,
+ "unicode": corpus_unicode,
+ "links": corpus_links,
+ "hardlinks": corpus_hardlinks,
+ "sparse": corpus_sparse,
+ "filters": corpus_filters,
+ "empty_dir": corpus_empty_dir,
+ "multidir": corpus_multidir,
+ "relative": corpus_relative,
+ "iconv": corpus_iconv,
+ "fuzzy": corpus_fuzzy,
+}
+
+
+# ---------------------------------------------------------------------------
+# Tree snapshotting / comparison
+# ---------------------------------------------------------------------------
+
+def snapshot(root: str, compare_modes: bool = False) -> Dict[str, tuple]:
+ """Map relative path -> descriptor for every entry below ``root``.
+
+ Files hash their contents with SHA-256 (structural comparison, so differing
+ quick-check metadata cannot mask a payload difference). Symlinks record
+ their target. Empty directories are included (as ``("dir", ...)``) so the
+ recursive-empty-directory residual is observable.
+ """
+ out: Dict[str, tuple] = {}
+ if not os.path.isdir(root):
+ return out
+
+ def describe(path: str) -> Optional[tuple]:
+ st = os.lstat(path)
+ if os.path.islink(path):
+ return ("link", os.readlink(path))
+ if os.path.isdir(path):
+ mode = oct(st.st_mode & 0o7777) if compare_modes else None
+ return ("dir", mode)
+ h = hashlib.sha256()
+ with open(path, "rb") as fh:
+ for chunk in iter(lambda: fh.read(65536), b""):
+ h.update(chunk)
+ mode = oct(st.st_mode & 0o7777) if compare_modes else None
+ return ("file", h.hexdigest()[:16], mode)
+
+ # The comparison root itself is not part of the tree diff: a no-transfer
+ # result legitimately leaves FastSync's mirror directory absent while rsync
+ # leaves an existing (empty) destination root.
+ for dirpath, dirnames, filenames in os.walk(root, followlinks=False):
+ dirnames.sort()
+ for name in sorted(dirnames):
+ p = os.path.join(dirpath, name)
+ rel = os.path.relpath(p, root)
+ if os.path.islink(p):
+ out[rel] = ("link", os.readlink(p))
+ dirnames.remove(name)
+ else:
+ out[rel] = describe(p)
+ for name in sorted(filenames):
+ p = os.path.join(dirpath, name)
+ out[os.path.relpath(p, root)] = describe(p)
+ return out
+
+
+def _hardlink_groups(root: str) -> Dict[str, str]:
+ """Assign a stable group letter to each inode shared by >1 regular file."""
+ inodes: Dict[tuple, List[str]] = {}
+ for dirpath, _dirs, filenames in os.walk(root, followlinks=False):
+ for name in filenames:
+ p = os.path.join(dirpath, name)
+ if os.path.islink(p):
+ continue
+ st = os.lstat(p)
+ if st.st_nlink > 1:
+ inodes.setdefault((st.st_dev, st.st_ino), []).append(
+ os.path.relpath(p, root))
+ groups: Dict[str, str] = {}
+ for i, (_key, members) in enumerate(sorted(inodes.items())):
+ for rel in members:
+ groups[rel] = chr(ord("A") + i)
+ return groups
+
+
+def _drop_ignored(tree: Dict[str, tuple], ignore_paths) -> Dict[str, tuple]:
+ if not ignore_paths:
+ return tree
+ out = {}
+ for rel, desc in tree.items():
+ if any(rel == ig or rel.startswith(ig.rstrip("/") + "/") for ig in ignore_paths):
+ continue
+ out[rel] = desc
+ return out
+
+
+def tree_diff(rsync_root: str, fs_root: str, case: Case) -> List[str]:
+ """Return a list of human-readable differences (empty when identical)."""
+ rtree = _drop_ignored(snapshot(rsync_root, case.compare_modes), case.ignore_paths)
+ ftree = _drop_ignored(snapshot(fs_root, case.compare_modes), case.ignore_paths)
+ if case.compare_hardlinks:
+ rgroups = _hardlink_groups(rsync_root)
+ fgroups = _hardlink_groups(fs_root)
+ else:
+ rgroups = fgroups = {}
+ diffs: List[str] = []
+ for rel in sorted(set(rtree) | set(ftree)):
+ r = rtree.get(rel)
+ f = ftree.get(rel)
+ if r == f:
+ continue
+ if r is None:
+ diffs.append(f"+ fastsync-only: {rel!r} {f}")
+ elif f is None:
+ diffs.append(f"- rsync-only: {rel!r} {r}")
+ else:
+ diffs.append(f"~ differs: {rel!r} rsync={r} fastsync={f}")
+ if case.compare_hardlinks:
+ for rel in sorted(set(rgroups) | set(fgroups)):
+ if rgroups.get(rel) != fgroups.get(rel):
+ diffs.append(
+ f"~ hardlink group: {rel!r} rsync={rgroups.get(rel)} "
+ f"fastsync={fgroups.get(rel)}")
+ return diffs
+
+
+# ---------------------------------------------------------------------------
+# stdout normalization
+# ---------------------------------------------------------------------------
+
+def _parse_bytes(text: str) -> str:
+ m = re.match(r"([\d,]+)", text.strip())
+ return m.group(1).replace(",", "") if m else text.strip()
+
+
+def normalize_stdout(text: str, mode: Optional[str]) -> object:
+ if mode == STDOUT_ITEMIZE:
+ lines = []
+ for line in (text or "").splitlines():
+ line = line.rstrip()
+ if not line:
+ continue
+ if line.startswith("*deleting"):
+ lines.append(line)
+ continue
+ if not _ITEMIZE_RE.match(line):
+ continue
+ # Directories are not transfer entries in FastSync's recursive
+ # scanner, so rsync's `cd+++++++++ name/` lines have no counterpart
+ # (documented recursive-empty-dir residual). Compare file/link
+ # itemization only.
+ if line.rsplit(" ", 1)[-1].endswith("/"):
+ continue
+ lines.append(line)
+ return sorted(lines)
+ if mode == STDOUT_OUTFMT:
+ lines = []
+ for line in (text or "").splitlines():
+ line = line.rstrip()
+ if not line:
+ continue
+ # Directory entries are emitted by rsync but not by FastSync's
+ # recursive scanner (documented residual). Tokens are either
+ # `%n %l` (path first) or `%i %n` (path last); drop a line when
+ # either end-token is a directory path.
+ first = line.split(" ", 1)[0]
+ last = line.rsplit(" ", 1)[-1]
+ if first.endswith("/") or last.endswith("/"):
+ continue
+ lines.append(line)
+ return sorted(lines)
+ if mode == STDOUT_STATS:
+ found = {}
+ for line in (text or "").splitlines():
+ for key in STATS_KEYS:
+ if line.startswith(key + ":"):
+ found[key] = _parse_bytes(line.split(":", 1)[1])
+ return found
+ if mode == STDOUT_PROGRESS:
+ # rsync prints the file-list entries in sorted depth-first order while
+ # FastSync's streaming scan emits them in readdir/BFS order; only the
+ # entry set and deterministic fields are compared. The transfer-root
+ # `./` line's trigger condition is a separate documented residual, and
+ # the per-frame rate/elapsed/xfr#/to-chk numerator are wall-clock- or
+ # order-dependent, so only the `to-chk` denominator and the name set are
+ # asserted.
+ names = []
+ totals = set()
+ max_xfr = 0
+ for line in (text or "").splitlines():
+ line = line.rstrip()
+ if not line:
+ continue
+ if "%" in line:
+ m = _PROGRESS_TOTAL_RE.search(line)
+ if m:
+ totals.add(int(m.group(1)))
+ mx = _PROGRESS_XFR_RE.search(line)
+ if mx:
+ max_xfr = max(max_xfr, int(mx.group(1)))
+ continue
+ if line == "sending incremental file list":
+ continue
+ if line.startswith("created directory "):
+ continue
+ if line == "./":
+ continue
+ names.append(line)
+ return {"names": sorted(names), "total": sorted(totals), "xfr": max_xfr}
+ # raw
+ return sorted(l.rstrip() for l in (text or "").splitlines() if l.strip())
+
+
+def stdout_diff(rsync_out: str, fs_out: str, mode: Optional[str]) -> List[str]:
+ r = normalize_stdout(rsync_out, mode)
+ f = normalize_stdout(fs_out, mode)
+ if r == f:
+ return []
+ if mode == STDOUT_STATS:
+ return [f"stats rsync={r}", f"stats fastsync={f}"]
+ if mode == STDOUT_PROGRESS:
+ return [f"progress rsync={r}", f"progress fastsync={f}"]
+ return list(difflib.unified_diff(
+ [str(x) for x in r], [str(x) for x in f],
+ fromfile="rsync", tofile="fastsync", lineterm=""))
+
+
+# ---------------------------------------------------------------------------
+# Running one case
+# ---------------------------------------------------------------------------
+
+def run_rsync(src: str, rdst: str, flags: List[str]) -> subprocess.CompletedProcess:
+ args = [RSYNC] + list(flags) + [src + "/", rdst + "/"]
+ return subprocess.run(
+ args, capture_output=True, text=True,
+ env=dict(os.environ, LC_ALL="C"), timeout=180)
+
+
+def run_fastsync(src: str, fdst: str, flags: List[str], port: int):
+ return run_client(src, fdst, flags=list(flags), port=port)
+
+
+def run_differential( # noqa: PLR0913 (explicit scenario parameters)
+ src: str,
+ rdst: str,
+ fdst: str,
+ rs_flags: List[str],
+ fs_flags: List[str],
+ server,
+ layout: str = MIRROR,
+ seed: Optional[Callable] = None,
+ stdout: Optional[str] = STDOUT_NONE,
+ compare_modes: bool = False,
+ compare_hardlinks: bool = False,
+ ignore_paths: Tuple[str, ...] = (),
+ extra_check: Optional[Callable] = None,
+ files_from: Optional[Tuple[str, ...]] = None,
+ fs_src_suffix: str = "",
+ compare_tree: bool = True,
+) -> Dict[str, object]:
+ """Run one rsync/FastSync pair and return the diff aspects.
+
+ Returned dict keys: ``rsync_rc``, ``fastsync_rc``, ``rsync_stderr``,
+ ``fastsync_stderr``, ``tree``, ``stdout``, ``extra``.
+ """
+ clean_dir(rdst)
+ clean_dir(fdst)
+ abs_src = os.path.abspath(src)
+ rel = abs_src.lstrip(os.sep)
+ if layout == RELATIVE:
+ rroot, froot = rdst, fdst
+ elif layout == MIRROR_ABS:
+ rroot, froot = os.path.join(rdst, rel), get_dest_received_dir(fdst, src)
+ else:
+ rroot, froot = rdst, get_dest_received_dir(fdst, src)
+ # rsync's destination root always exists (clean_dir created it). FastSync's
+ # logical transfer root is the mirror path below the destination argument,
+ # so pre-create it too: `Number of created files` counts the root only when
+ # it is genuinely absent, and the two tools must start from the same state.
+ os.makedirs(froot, exist_ok=True)
+ if seed:
+ seed(src, rroot, froot)
+
+ rs_flags = list(rs_flags)
+ fs_flags = list(fs_flags)
+ if files_from is not None:
+ list_path = os.path.join(TEST_DATA_DIR, "parity_" +
+ os.path.basename(src) + ".list")
+ write_list(list_path, files_from)
+ rs_flags.append(f"--files-from={list_path}")
+ fs_flags.append(f"--files-from={list_path}")
+
+ rs = run_rsync(src, rdst, rs_flags)
+ fs_result, _ = run_fastsync(src + fs_src_suffix, fdst, fs_flags, server.port)
+
+ class _View:
+ """Adapter so tree_diff/extra_check keep the Case-shaped interface."""
+
+ def __init__(self) -> None:
+ self.compare_modes = compare_modes
+ self.compare_hardlinks = compare_hardlinks
+ self.ignore_paths = ignore_paths
+
+ result = {
+ "rsync_rc": rs.returncode,
+ "fastsync_rc": fs_result.returncode,
+ "rsync_stderr": rs.stderr,
+ "fastsync_stderr": fs_result.stderr or fs_result.stdout,
+ "tree": tree_diff(rroot, froot, _View()) if compare_tree else [],
+ "stdout": [],
+ "extra": [],
+ }
+ if stdout is not None:
+ result["stdout"] = stdout_diff(rs.stdout, fs_result.stdout, stdout)
+ if extra_check:
+ result["extra"] = list(extra_check(src, rroot, froot, rs, fs_result) or [])
+ return result
+
+
+def execute_case(case: Case, server) -> Dict[str, object]:
+ """Run a table-driven case and return the diff aspects."""
+ tag = case.id
+ src = os.path.join(TEST_DATA_DIR, f"parity_{tag}_src")
+ rdst = os.path.join(TEST_DATA_DIR, f"parity_{tag}_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, f"parity_{tag}_fdst")
+ CORPORA[case.corpus](src)
+ return run_differential(
+ src, rdst, fdst,
+ case.flags, case.fs_flags(), server,
+ layout=case.layout, seed=case.seed, stdout=case.stdout,
+ compare_modes=case.compare_modes, compare_hardlinks=case.compare_hardlinks,
+ ignore_paths=case.ignore_paths, extra_check=case.extra_check,
+ files_from=case.files_from, fs_src_suffix=case.fs_src_suffix,
+ compare_tree=case.compare_tree,
+ )
+
+
+def write_list(path: str, entries) -> str:
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ with open(path, "w", encoding="utf-8") as fh:
+ for e in entries:
+ fh.write(e + "\n")
+ return path
diff --git a/tests/integration/test_client_cli.py b/tests/integration/test_client_cli.py
new file mode 100644
index 0000000..91d582f
--- /dev/null
+++ b/tests/integration/test_client_cli.py
@@ -0,0 +1,213 @@
+"""Differential tests for the client CLI's codec defaults and env lists.
+
+Track 3a of the rsync-parity plan pins two rsync 3.4.1 behaviors that are
+resolved entirely on the client:
+
+* the per-codec default ``--compress-level`` (zstd 3, zlib/zlibx 6, lz4
+ ignored) applied when the user omits ``--compress-level``/``--zl``, with an
+ explicit level clamped to the codec's range; and
+* the ``RSYNC_COMPRESS_LIST`` / ``RSYNC_CHECKSUM_LIST`` preference lists that
+ rsync's ``auto`` consults before its compiled-in order (whitespace-separated,
+ unknown names skipped, first supported wins, all-unknown is exit 4).
+
+The rsync side is observed through ``--debug=NSTR1``; FastSync publishes its
+resolved codec/level through ``--debug=util``. The checksum side is confirmed
+byte-for-byte through ``--out-format %C``. The rsync-based tests skip cleanly
+when rsync is not installed.
+"""
+import os
+import re
+import shutil
+import subprocess
+import sys
+
+import pytest
+
+sys.path.insert(0, os.path.dirname(__file__))
+from common import (
+ TEST_DATA_DIR,
+ run_client,
+ clean_dir,
+ get_dest_received_dir,
+)
+
+RSYNC = shutil.which("rsync")
+requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
+
+CODEC_ROOT = os.path.join(TEST_DATA_DIR, "cli_differential")
+
+_COMPRESS_RE = re.compile(r"compress(?:ion)?: (\w+) \(level (-?\d+)\)")
+
+
+def _rsync(args):
+ env = dict(os.environ, LC_ALL="C")
+ return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
+
+
+def _scratch(tag):
+ path = os.path.join(CODEC_ROOT, tag)
+ clean_dir(path)
+ os.makedirs(path, exist_ok=True)
+ return path
+
+
+def _make_corpus(root):
+ clean_dir(root)
+ os.makedirs(root, exist_ok=True)
+ with open(os.path.join(root, "big.bin"), "wb") as fh:
+ fh.write(b"FastSync codec payload " * 4096)
+ with open(os.path.join(root, "small.txt"), "wb") as fh:
+ fh.write(b"hello codec world\n" * 32)
+ return root
+
+
+def _rsync_compress_level(choice, level):
+ src = _make_corpus(_scratch(f"lvl_src_{choice}_{level}"))
+ dst = _scratch(f"lvl_rsync_{choice}_{level}")
+ args = ["-a", "-z", f"--zc={choice}"]
+ if level is not None:
+ args.append(f"--zl={level}")
+ args += ["--debug=NSTR1", src + "/", dst + "/"]
+ result = _rsync(args)
+ assert result.returncode == 0, result.stderr
+ match = _COMPRESS_RE.search(result.stdout + result.stderr)
+ assert match, (result.stdout, result.stderr)
+ return match.group(1), int(match.group(2))
+
+
+def _fastsync_compress_level(choice, level, shared_server):
+ src = _make_corpus(_scratch(f"lvl_src_fs_{choice}_{level}"))
+ dst = _scratch(f"lvl_fs_{choice}_{level}")
+ args = ["-a", "-z", f"--zc={choice}"]
+ if level is not None:
+ args.append(f"--zl={level}")
+ args += ["-v", "--debug=util"]
+ result, _ = run_client(src, dst, flags=args, port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ match = _COMPRESS_RE.search(result.stdout)
+ assert match, result.stdout[:500]
+ return match.group(1), int(match.group(2))
+
+
+class TestPerCodecCompressionLevelDefaults:
+ """``--compress-level`` defaults and clamping match rsync per codec."""
+
+ # FastSync uses a positive lz4 placeholder because its "level > 0" gate
+ # enables compression; lz4_compress ignores the value, so rsync's level 0
+ # and FastSync's level 1 produce the same bytes.
+ CASES = [
+ ("zstd", None, 3),
+ ("zlib", None, 6),
+ ("zlibx", None, 6),
+ ("lz4", None, 1),
+ ("zstd", 10, 10),
+ ("zlib", 15, 9),
+ ("zlib", 3, 3),
+ ("lz4", 15, 1),
+ ]
+
+ @requires_rsync
+ @pytest.mark.ci
+ @pytest.mark.parametrize("choice,level,fs_level", CASES)
+ def test_level_matches_rsync(self, choice, level, fs_level, shared_server):
+ rsync_algo, rsync_level = _rsync_compress_level(choice, level)
+ fs_algo, fs_level_actual = _fastsync_compress_level(choice, level, shared_server)
+ assert rsync_algo == choice
+ assert fs_algo == choice
+ if choice == "lz4":
+ assert rsync_level == 0 and fs_level_actual > 0
+ else:
+ assert rsync_level == fs_level
+ assert fs_level_actual == fs_level
+
+
+class TestEnvPreferenceLists:
+ """``RSYNC_COMPRESS_LIST`` / ``RSYNC_CHECKSUM_LIST`` drive auto like rsync."""
+
+ # (env value, expected codec, rsync level, FastSync level)
+ COMPRESS_CASES = [
+ ("zlib lz4", "zlib", 6, 6),
+ ("lz4 zstd", "lz4", 0, 1),
+ ("bogus zstd zlib", "zstd", 3, 3),
+ (" ", "zstd", 3, 3),
+ ]
+
+ @requires_rsync
+ @pytest.mark.ci
+ @pytest.mark.parametrize("env,algo,rsync_level,fs_level", COMPRESS_CASES)
+ def test_compress_list_matches_rsync(self, env, algo, rsync_level, fs_level, shared_server,
+ monkeypatch):
+ monkeypatch.setenv("RSYNC_COMPRESS_LIST", env)
+ src = _make_corpus(_scratch(f"envc_src_{algo}"))
+ rdst = _scratch(f"envc_rsync_{algo}")
+ rs = _rsync(["-a", "-z", "--debug=NSTR1", src + "/", rdst + "/"])
+ assert rs.returncode == 0, rs.stderr
+ rm = _COMPRESS_RE.search(rs.stdout + rs.stderr)
+ assert rm, (rs.stdout, rs.stderr)
+ assert rm.group(1) == algo
+ assert int(rm.group(2)) == rsync_level
+
+ fdst = _scratch(f"envc_fs_{algo}")
+ result, _ = run_client(src, fdst, flags=["-a", "-z", "-v", "--debug=util"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ fm = _COMPRESS_RE.search(result.stdout)
+ assert fm, result.stdout[:500]
+ assert fm.group(1) == algo
+ assert int(fm.group(2)) == fs_level
+ received = get_dest_received_dir(fdst, src)
+ assert _tree_bytes(received) == _tree_bytes(src)
+
+ @requires_rsync
+ @pytest.mark.ci
+ @pytest.mark.parametrize("env,algo", [("md5", "md5"), ("sha1", "sha1"), ("xxh3 md5", "xxh3")])
+ def test_checksum_list_matches_rsync(self, env, algo, shared_server, monkeypatch):
+ monkeypatch.setenv("RSYNC_CHECKSUM_LIST", env)
+ src = _make_corpus(_scratch(f"envcc_src_{algo}"))
+ rdst = _scratch(f"envcc_rsync_{algo}")
+ rs = _rsync(["-a", "--checksum", "--out-format=%C %n", src + "/", rdst + "/"])
+ assert rs.returncode == 0, rs.stderr
+ rs_digests = _digests(rs.stdout)
+
+ fdst = _scratch(f"envcc_fs_{algo}")
+ result, _ = run_client(src, fdst, flags=["-a", "--checksum", "--out-format=%C %n"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ assert _digests(result.stdout) == rs_digests
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_all_unknown_lists_fail_like_rsync(self, shared_server, monkeypatch):
+ src = _make_corpus(_scratch("envbad_src"))
+ monkeypatch.setenv("RSYNC_COMPRESS_LIST", "bogus")
+ rs = _rsync(["-a", "-z", src + "/", _scratch("envbad_rsync_c") + "/"])
+ assert rs.returncode == 4, rs.stderr
+ result, _ = run_client(src, _scratch("envbad_fs_c"), flags=["-a", "-z"],
+ port=shared_server.port)
+ assert result.returncode == 4, (result.stderr or result.stdout)[:200]
+
+ monkeypatch.setenv("RSYNC_CHECKSUM_LIST", "bogus")
+ rs = _rsync(["-a", src + "/", _scratch("envbad_rsync_s") + "/"])
+ assert rs.returncode == 4, rs.stderr
+ result, _ = run_client(src, _scratch("envbad_fs_s"), flags=["-a"],
+ port=shared_server.port)
+ assert result.returncode == 4, (result.stderr or result.stdout)[:200]
+
+
+def _tree_bytes(root):
+ out = {}
+ for dirpath, _dirs, files in os.walk(root):
+ for name in files:
+ path = os.path.join(dirpath, name)
+ with open(path, "rb") as fh:
+ out[os.path.relpath(path, root)] = fh.read()
+ return out
+
+
+def _digests(output):
+ out = {}
+ for line in output.splitlines():
+ parts = line.split()
+ if len(parts) == 2 and parts[0]:
+ out[parts[1]] = parts[0]
+ return out
diff --git a/tests/integration/test_codecs.py b/tests/integration/test_codecs.py
index 0e0dfbe..8c17f0d 100644
--- a/tests/integration/test_codecs.py
+++ b/tests/integration/test_codecs.py
@@ -9,6 +9,7 @@ directory when the shared test server is launched), so every scratch tree lives
under ``TEST_DATA_DIR`` rather than pytest's ``tmp_path``.
"""
import os
+import random
import shutil
import subprocess
import sys
@@ -68,6 +69,84 @@ def _tree_bytes(root):
return out
+_CC_DELTA_T0 = 1_600_000_000
+_CC_DELTA_T1 = 1_600_000_100
+
+_DELTA_STATS_KEYS = (
+ "Number of created files",
+ "Number of regular files transferred",
+ "Total transferred file size",
+ "Literal data",
+ "Matched data",
+)
+
+
+def _pin_tree(root, mtime):
+ for dirpath, dirnames, filenames in os.walk(root):
+ for name in dirnames + filenames:
+ path = os.path.join(dirpath, name)
+ if not os.path.islink(path):
+ os.utime(path, (mtime, mtime))
+ os.utime(root, (mtime, mtime))
+
+
+def _make_delta_basis(src, size=512 * 1024):
+ """Build a source and a matching pre-modification basis tree.
+
+ The source's ``big.bin`` is then modified in a few disjoint places and given
+ a newer mtime so both tools take the delta path. Returns the basis dir.
+ """
+ clean_dir(src)
+ original = random.Random(20240101).randbytes(size)
+ with open(os.path.join(src, "big.bin"), "wb") as fh:
+ fh.write(original)
+ with open(os.path.join(src, "small.txt"), "wb") as fh:
+ fh.write(b"hello world\n")
+ _pin_tree(src, _CC_DELTA_T0)
+
+ basis = src.rstrip("/") + "_basis"
+ clean_dir(basis)
+ shutil.copy2(os.path.join(src, "big.bin"), os.path.join(basis, "big.bin"))
+ shutil.copy2(os.path.join(src, "small.txt"), os.path.join(basis, "small.txt"))
+ _pin_tree(basis, _CC_DELTA_T0)
+
+ modified = bytearray(original)
+ for off in (0, size // 3, 2 * size // 3, size - 64):
+ for i in range(32):
+ modified[off + i] ^= 0x5A
+ with open(os.path.join(src, "big.bin"), "wb") as fh:
+ fh.write(bytes(modified))
+ os.utime(os.path.join(src, "big.bin"), (_CC_DELTA_T1, _CC_DELTA_T1))
+ return basis
+
+
+def _seed_from_basis(basis, target):
+ clean_dir(target)
+ for name in os.listdir(basis):
+ shutil.copy2(os.path.join(basis, name), os.path.join(target, name))
+
+
+def _delta_stats(text):
+ found = {}
+ for line in text.splitlines():
+ for key in _DELTA_STATS_KEYS:
+ if line.startswith(key + ":"):
+ found[key] = line.split(":", 1)[1].strip()
+ return found
+
+
+def _big_bin_outfmt(text):
+ """The ``(c, C)`` pair from the ``big.bin`` out-format line (`%c|%C %n`)."""
+ for line in text.splitlines():
+ stripped = line.strip()
+ if "|" not in stripped or not stripped.endswith("big.bin"):
+ continue
+ c_field, rest = stripped.split("|", 1)
+ fields = rest.split()
+ return c_field.strip(), (fields[0] if fields else "")
+ return None, None
+
+
class TestCodecChoiceMatrix:
"""The CLI accept/reject set and exit codes must match rsync 3.4.1."""
@@ -191,6 +270,76 @@ class TestCodecTransferDifferential:
assert _tree_bytes(received) == _tree_bytes(rsync_dst)
+class TestChecksumChoiceDeltaSurface:
+ """--checksum-choice does not move the delta-transfer parity surface.
+
+ FastSync's delta BLOCK strong checksum is a fixed xxHash32, so the
+ negotiated algorithm only selects the whole-file comparison digest (and the
+ ``%C`` transfer digest). A pre-seeded delta transfer must therefore land
+ byte-identical bytes and report the same counters for every choice, while
+ ``%C`` -- the one token that tracks the choice -- stays byte-identical to
+ rsync. This pins the Track-3b reclassification in RSYNC_COMPAT.md.
+ """
+
+ CHOICES = ["xxh64", "xxh128", "xxh3", "md5", "md4", "sha1",
+ "xxh64,sha1", "sha1,xxh64"]
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_delta_surface_invariant_to_checksum_choice(self, shared_server):
+ src = _scratch("ccdelta_src")
+ basis = _make_delta_basis(src)
+ source_bytes = _tree_bytes(src)
+
+ rsync_c, fastsync_c, digests = {}, {}, {}
+ rsync_stats, fastsync_stats = {}, {}
+ for choice in self.CHOICES:
+ tag = choice.replace(",", "_")
+ rsync_dst = _scratch(f"ccdelta_rs_{tag}")
+ fs_dst = _scratch(f"ccdelta_fs_{tag}")
+ fs_root = get_dest_received_dir(fs_dst, src)
+ _seed_from_basis(basis, rsync_dst)
+ _seed_from_basis(basis, fs_root)
+
+ # Pin the block size on both ends so the literal/matched split is
+ # comparable (rsync's adaptive default would otherwise differ from
+ # FastSync's 8192-byte default).
+ rsync_result = _rsync(["-a", "--no-whole-file", "-B8192", "--stats",
+ "--out-format=%c|%C %n", f"--cc={choice}",
+ src + "/", rsync_dst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(
+ src, fs_dst,
+ flags=["-a", "--incremental", "--delta", "-B8192", "--stats",
+ "--out-format=%c|%C %n", f"--cc={choice}"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+
+ assert _tree_bytes(rsync_dst) == source_bytes, choice
+ assert _tree_bytes(fs_root) == source_bytes, choice
+ assert _delta_stats(rsync_result.stdout) == _delta_stats(result.stdout), choice
+
+ rs_c, rs_C = _big_bin_outfmt(rsync_result.stdout)
+ fs_c, fs_C = _big_bin_outfmt(result.stdout)
+ assert rs_C == fs_C, f"{choice}: %C rsync={rs_C!r} fastsync={fs_C!r}"
+ rsync_c[choice] = rs_c
+ fastsync_c[choice] = fs_c
+ digests[choice] = fs_C
+ rsync_stats[choice] = _delta_stats(rsync_result.stdout)
+ fastsync_stats[choice] = _delta_stats(result.stdout)
+
+ # The choice is only observable in %C, and it is effective (the digests
+ # are not all the same algorithm's output).
+ assert len(set(digests.values())) > 1, digests
+ # The compared --stats counters are invariant across choices in each tool
+ # (and were asserted equal cross-tool inside the loop).
+ assert len({tuple(sorted(s.items())) for s in rsync_stats.values()}) == 1, rsync_stats
+ assert len({tuple(sorted(s.items())) for s in fastsync_stats.values()}) == 1, fastsync_stats
+ # The block-checksum token (%c) is invariant across choices in each tool.
+ assert len(set(rsync_c.values())) == 1, rsync_c
+ assert len(set(fastsync_c.values())) == 1, fastsync_c
+
+
class TestCodecNegotiationFallback:
"""FastSync's auto negotiation and deterministic fallback order."""
diff --git a/tests/integration/test_delete_delay_budget_parity.py b/tests/integration/test_delete_delay_budget_parity.py
new file mode 100644
index 0000000..c8c05a8
--- /dev/null
+++ b/tests/integration/test_delete_delay_budget_parity.py
@@ -0,0 +1,185 @@
+"""Differential coverage for ``--delete-delay`` + ``--max-delete`` with a
+refilled deferred directory.
+
+FastSync snapshots a directory's extras at plan time (``defer_add``) but charges
+``--max-delete`` only when a path is actually removed, and its deferred commit
+re-scans a queued directory and removes content created after the plan -- the
+same rules as rsync. These tests run both tools on the same fixture and assert
+both sides remove the late content (recursively) and bound the deletion with
+``--max-delete`` identically.
+
+They are not part of the fast PR gate because the rsync side needs a wide
+real-time injection window (a throttled transfer), while the FastSync side uses
+the existing byte-deterministic slicing proxy.
+
+The refilled directory sits at the transfer ROOT, whose delete plan is always
+processed before any subdirectory's, so the budget is deterministically charged
+to the refilled entry; the second extra lives under ``b`` and is skipped.
+"""
+import os
+import shutil
+import subprocess
+import sys
+import threading
+import time
+
+import pytest
+
+sys.path.insert(0, os.path.dirname(__file__))
+from common import ( # noqa: E402
+ TEST_DATA_DIR,
+ ServerManager,
+ clean_dir,
+ get_dest_received_dir,
+ run_client,
+)
+from test_delete_timing_parity import _SlicingProxy # noqa: E402
+
+RSYNC = shutil.which("rsync")
+requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
+
+BIG_BYTES = 8 * 1024 * 1024
+MID_TRANSFER_BYTES = 256 * 1024
+PROXY_THROTTLE = 0.001
+# rsync is driven locally, so the refill is injected on a wall-clock delay while
+# a throttled ~8 s transfer is in flight. 1.5 s is safely after rsync's plan
+# scan (t=0) and well before the deferred commit at the end.
+RSYNC_BWLIMIT = 1024 # 1 MiB/s
+RSYNC_INJECT_DELAY = 1.5
+
+
+def _write(path, content):
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ with open(path, "wb") as fh:
+ fh.write(content)
+
+
+def _seed_source(tag):
+ source = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_src")
+ clean_dir(source)
+ _write(os.path.join(source, "a", "keep.bin"), b"B" * BIG_BYTES)
+ _write(os.path.join(source, "b", "keep.txt"), b"keep\n")
+ return source
+
+
+def _seed_fastsync(tag):
+ """FastSync mirrors the absolute source path under its receive root, so the
+ extras live below ``received``."""
+ source = _seed_source(tag)
+ dest = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_dst")
+ clean_dir(dest)
+ received = get_dest_received_dir(dest, source)
+ os.makedirs(os.path.join(received, "xdir"), exist_ok=True)
+ os.makedirs(os.path.join(received, "b", "ydir"), exist_ok=True)
+ return source, dest, received
+
+
+def _seed_rsync(tag):
+ """rsync mirrors the source contents directly into the destination, so the
+ extras are flat under ``rsync_dst``."""
+ source = _seed_source(tag)
+ rsync_dst = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_dst")
+ clean_dir(rsync_dst)
+ os.makedirs(os.path.join(rsync_dst, "xdir"), exist_ok=True)
+ os.makedirs(os.path.join(rsync_dst, "b", "ydir"), exist_ok=True)
+ return source, rsync_dst
+
+
+def _deleted_count(text):
+ for line in text.splitlines():
+ if line.startswith("Number of deleted files:"):
+ return int(line.split(":", 1)[1].split()[0])
+ return None
+
+
+def _rsync(args, timeout=120):
+ env = dict(os.environ, LC_ALL="C")
+ return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=timeout)
+
+
+class TestDeleteDelayRefilledDirVsRsync:
+ """Both tools charge --max-delete on actual removals and recurse."""
+
+ def _fastsync_refilled(self, tag, max_delete=None):
+ """Run FastSync with the refill injected deterministically by the proxy
+ hook (fired once the receiver has processed the plan frames)."""
+ source, dest, received = _seed_fastsync(tag)
+ late = os.path.join(received, "xdir", "new.txt")
+
+ def hook():
+ _write(late, b"created mid-transfer\n")
+
+ flags = ["--delete-delay", "--incremental", "--ignore-times", "--stats"]
+ if max_delete is not None:
+ flags.append(f"--max-delete={max_delete}")
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ proxy = _SlicingProxy(server.port, hook=hook, hook_after=MID_TRANSFER_BYTES,
+ throttle=PROXY_THROTTLE, wait_for_reply=True)
+ result, _ = run_client(source, dest, flags=flags, port=proxy.port)
+ proxy.finish()
+ assert proxy.hook_called.is_set(), "refill hook never fired"
+ return result, received, late
+
+ @requires_rsync
+ def test_max_delete_budget_bound_matches(self):
+ # --- FastSync: the one actual removal is the late file; dirs survive ---
+ result, received, late = self._fastsync_refilled("budget_fs", max_delete=1)
+ assert result.returncode == 25, (result.stderr or result.stdout)[:300]
+ assert _deleted_count(result.stdout) == 1, result.stdout
+ assert not os.path.exists(late), "FastSync kept the late content of a queued dir"
+ assert os.path.isdir(os.path.join(received, "xdir"))
+ assert os.path.isdir(os.path.join(received, "b", "ydir")), (
+ "FastSync did not bound the deletion with --max-delete=1"
+ )
+
+ # --- rsync: same budget rule and recursive removal ---
+ source, rsync_dst = _seed_rsync("budget_rsync")
+
+ def inject():
+ time.sleep(RSYNC_INJECT_DELAY)
+ _write(os.path.join(rsync_dst, "xdir", "new.txt"), b"created mid-transfer\n")
+
+ t = threading.Thread(target=inject)
+ t.start()
+ rsync_result = _rsync(
+ ["-a", "--delete-delay", "--max-delete=1", "--stats",
+ f"--bwlimit={RSYNC_BWLIMIT}", source + "/", rsync_dst + "/"]
+ )
+ t.join()
+ assert rsync_result.returncode == 25, rsync_result.stderr
+ assert _deleted_count(rsync_result.stdout) == _deleted_count(result.stdout)
+ assert not os.path.exists(os.path.join(rsync_dst, "xdir", "new.txt")), (
+ "rsync kept late content inside a queued directory"
+ )
+ assert os.path.isdir(os.path.join(rsync_dst, "b", "ydir")), (
+ "rsync did not bound the deletion with --max-delete=1"
+ )
+ assert os.path.isdir(os.path.join(received, "b", "ydir"))
+
+ @requires_rsync
+ def test_refilled_extra_dir_recursive_removal_matches(self):
+ """Without --max-delete both tools remove the refilled extra directory
+ (and its late content)."""
+ result, received, late = self._fastsync_refilled("recur_fs")
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ assert not os.path.exists(late), "FastSync kept the refilled directory's late content"
+ assert not os.path.isdir(os.path.join(received, "xdir"))
+
+ source, rsync_dst = _seed_rsync("recur_rsync")
+
+ def inject():
+ time.sleep(RSYNC_INJECT_DELAY)
+ _write(os.path.join(rsync_dst, "xdir", "new.txt"), b"created mid-transfer\n")
+
+ t = threading.Thread(target=inject)
+ t.start()
+ rsync_result = _rsync(
+ ["-a", "--delete-delay", "--stats", f"--bwlimit={RSYNC_BWLIMIT}",
+ source + "/", rsync_dst + "/"]
+ )
+ t.join()
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ assert not os.path.exists(os.path.join(rsync_dst, "xdir")), (
+ "rsync did not recursively remove the refilled extra directory"
+ )
diff --git a/tests/integration/test_delete_timing_parity.py b/tests/integration/test_delete_timing_parity.py
index b12355c..4f4c78d 100644
--- a/tests/integration/test_delete_timing_parity.py
+++ b/tests/integration/test_delete_timing_parity.py
@@ -217,7 +217,21 @@ class _SlicingProxy:
class TestDeleteTimingFinalStateParity:
- """On a successful transfer the per-directory timings match rsync's result."""
+ """On a successful transfer the per-directory timings match rsync's result.
+
+ Plain ``--delete`` has no rsync-incompatible spelling: it defaults to
+ delete-during on both tools, so it is compared against rsync's own default.
+ ``--delete-commit`` is FastSync-only and selects the late whole-tree commit,
+ which is rsync's ``--delete-after`` timing.
+ """
+
+ # (fastsync flag, rsync flag)
+ PAIRS = [
+ ("--delete", "--delete"),
+ ("--delete-during", "--delete-during"),
+ ("--delete-delay", "--delete-delay"),
+ ("--delete-commit", "--delete-after"),
+ ]
def _run_fastsync(self, tag, timing):
source, dest, received = _seed_pair(tag)
@@ -226,28 +240,32 @@ class TestDeleteTimingFinalStateParity:
result, _ = run_client(source, dest, flags=[timing], port=server.port)
return result, received
- @pytest.mark.parametrize("timing", ["--delete-during", "--delete-delay"])
+ @pytest.mark.parametrize("fs_timing,rs_timing", PAIRS)
@requires_rsync
- def test_success_final_state_matches_rsync(self, timing):
+ def test_success_final_state_matches_rsync(self, fs_timing, rs_timing):
+ # Worker-safe names: xdist may run the parametrizations concurrently, so
+ # the flags are part of every fixture path.
+ label = f"{fs_timing.lstrip('-')}_vs_{rs_timing.lstrip('-')}"
# Build the rsync fixture from the same seed so both sides start equal.
- source, dest, received = _seed_pair("parity_rsync")
+ source, dest, received = _seed_pair(f"parity_rsync_{label}")
source2 = source
- rsync_dst = os.path.join(TEST_DATA_DIR, "dtp_parity_rsync_dst")
+ rsync_dst = os.path.join(TEST_DATA_DIR, f"dtp_rsync_{label}_dst")
clean_dir(rsync_dst)
# rsync mirrors src/ into dst/; seed the same extra.
_write(os.path.join(rsync_dst, "d", "old_extra"), b"stale extra\n")
- rsync_result = _rsync(["-a", timing, source2 + "/", rsync_dst + "/"])
+ rsync_result = _rsync(["-a", rs_timing, source2 + "/", rsync_dst + "/"])
assert rsync_result.returncode == 0, rsync_result.stderr
rsync_tree = _tree(rsync_dst)
with ServerManager() as server:
server.start(extra_args=["--allow-delete"])
- result, _ = run_client(source, dest, flags=[timing], port=server.port)
+ result, _ = run_client(source, dest, flags=[fs_timing], port=server.port)
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
fastsync_tree = _tree(received)
assert fastsync_tree == rsync_tree, (
- f"{timing}: fastsync tree {fastsync_tree} != rsync tree {rsync_tree}"
+ f"{fs_timing} vs rsync {rs_timing}: fastsync tree {fastsync_tree} != "
+ f"rsync tree {rsync_tree}"
)
@@ -258,7 +276,8 @@ class TestDeleteTimingTypeConflictParity:
@pytest.mark.parametrize("timing", ["--delete-during", "--delete-delay"])
@requires_rsync
def test_type_conflicts_match_rsync(self, timing):
- source = os.path.join(TEST_DATA_DIR, "dtc_src")
+ label = timing.lstrip("-")
+ source = os.path.join(TEST_DATA_DIR, f"dtc_{label}_src")
clean_dir(source)
_write(os.path.join(source, "foo"), b"now a file\n")
_write(os.path.join(source, "bar", "inner.txt"), b"now a dir\n")
@@ -268,13 +287,13 @@ class TestDeleteTimingTypeConflictParity:
_write(os.path.join(root, "foo", "inner.txt"), b"was a dir\n")
_write(os.path.join(root, "bar"), b"was a file\n")
- rsync_dst = os.path.join(TEST_DATA_DIR, "dtc_rsync_dst")
+ rsync_dst = os.path.join(TEST_DATA_DIR, f"dtc_{label}_rsync_dst")
seed_dest(rsync_dst)
rsync_result = _rsync(["-a", timing, source + "/", rsync_dst + "/"])
assert rsync_result.returncode == 0, rsync_result.stderr
rsync_tree = _tree(rsync_dst)
- dest = os.path.join(TEST_DATA_DIR, "dtc_dst")
+ dest = os.path.join(TEST_DATA_DIR, f"dtc_{label}_dst")
clean_dir(dest)
received = get_dest_received_dir(dest, source)
seed_dest(received)
@@ -288,17 +307,28 @@ class TestDeleteTimingTypeConflictParity:
class TestDeleteTimingFailure:
- """A mid-transfer failure distinguishes during from delay."""
+ """A mid-transfer failure distinguishes the during timings from the late
+ commit timings.
+
+ Plain ``--delete`` must behave like ``--delete-during`` (the rsync default),
+ removing the extras of the directories already reached; ``--delete-commit``
+ must behave like ``--delete-after`` and remove nothing until the transfer
+ has fully succeeded.
+ """
@pytest.mark.parametrize("mt", [False, True])
def test_during_removes_delay_preserves_on_failure(self, mt):
- source, dest, received = _seed_pair("failure", big=True)
+ source, dest, received = _seed_pair(f"failure_mt{int(mt)}", big=True)
extra = os.path.join(received, "d", "old_extra")
assert os.path.exists(extra)
with ServerManager() as server:
server.start(extra_args=["--allow-delete"])
- for timing, expect_removed in (("--delete-during", True),
- ("--delete-delay", False)):
+ for timing, expect_removed in (
+ ("--delete-during", True),
+ ("--delete", True),
+ ("--delete-delay", False),
+ ("--delete-commit", False),
+ ("--delete-after", False)):
# Re-seed the extra before each run.
_write(extra, b"stale extra\n")
proxy = _SlicingProxy(server.port, forward_limit=MID_TRANSFER_BYTES, throttle=PROXY_THROTTLE)
@@ -313,13 +343,99 @@ class TestDeleteTimingFailure:
)
+class TestDeleteDelayDeletedCount:
+ """The reported deleted count must reflect entries actually removed."""
+
+ def test_refilled_deferred_dir_is_recursively_removed_and_counted(self):
+ """A directory snapshotted into a --delete-delay plan that is refilled
+ before the commit is re-scanned and removed recursively (rsync parity):
+ the late file and the directory are both counted as deleted."""
+ source = os.path.join(TEST_DATA_DIR, "ddc_src")
+ dest = os.path.join(TEST_DATA_DIR, "ddc_dst")
+ clean_dir(source)
+ clean_dir(dest)
+ _write(os.path.join(source, "d", "keep.txt"), b"kept payload\n")
+ _write(os.path.join(source, "d", "big.bin"), b"B" * BIG_BYTES)
+ received = get_dest_received_dir(dest, source)
+ extra_dir = os.path.join(received, "d", "extradir")
+ os.makedirs(extra_dir, exist_ok=True)
+
+ def hook():
+ # Runs while big.bin is in flight, after d's delete plan was processed.
+ _write(os.path.join(extra_dir, "new.txt"), b"created mid-transfer\n")
+
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ proxy = _SlicingProxy(server.port, hook=hook, hook_after=MID_TRANSFER_BYTES,
+ throttle=PROXY_THROTTLE, wait_for_reply=True)
+ flags = ["--delete-delay", "--incremental", "--ignore-times", "--stats"]
+ result, _ = run_client(source, dest, flags=flags, port=proxy.port)
+ proxy.finish()
+ assert result.returncode == 0, (result.stderr or result.stdout)[:400]
+ assert proxy.hook_called.is_set(), "hook never fired"
+ assert not os.path.exists(os.path.join(extra_dir, "new.txt")), "late file survived"
+ assert not os.path.isdir(extra_dir), "refilled extra dir survived"
+ deleted = None
+ for line in result.stdout.splitlines():
+ if line.startswith("Number of deleted files:"):
+ deleted = int(line.split(":", 1)[1].split()[0])
+ assert deleted == 2, (deleted, result.stdout)
+
+
+class TestDeleteDelayMaxDeleteParity:
+ """--max-delete with --delete-delay: a partial deletion still reports the
+ number of entries actually removed, matching rsync (the exact surviving set
+ can differ; only the count is compared)."""
+
+ @requires_rsync
+ def test_max_delete_count_matches_rsync(self):
+ source = os.path.join(TEST_DATA_DIR, "ddm_src")
+ rsync_dst = os.path.join(TEST_DATA_DIR, "ddm_rsync_dst")
+ clean_dir(source)
+ clean_dir(rsync_dst)
+ _write(os.path.join(source, "d", "keep.txt"), b"keep\n")
+ for i in range(1, 6):
+ _write(os.path.join(rsync_dst, "d", f"e{i}.txt"), f"extra{i}\n".encode())
+
+ rsync_result = _rsync(["-a", "--delete-delay", "--max-delete=2", "--stats",
+ source + "/", rsync_dst + "/"])
+ # rsync exits 25 ("the --max-delete limit stopped deletions").
+ assert rsync_result.returncode == 25, rsync_result.stderr
+ rsync_count = _deleted_count(rsync_result.stdout)
+ assert rsync_count == 2, rsync_result.stdout
+
+ dest = os.path.join(TEST_DATA_DIR, "ddm_dst")
+ clean_dir(dest)
+ received = get_dest_received_dir(dest, source)
+ for i in range(1, 6):
+ _write(os.path.join(received, "d", f"e{i}.txt"), f"extra{i}\n".encode())
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(
+ source, dest,
+ flags=["--delete-delay", "--max-delete=2", "--stats"],
+ port=server.port,
+ )
+ # A capped --max-delete commit is a successful transfer that both tools
+ # report with exit 25.
+ assert result.returncode == 25, (result.stderr or result.stdout)[:300]
+ assert _deleted_count(result.stdout) == rsync_count, result.stdout
+
+
+def _deleted_count(text):
+ for line in text.splitlines():
+ if line.startswith("Number of deleted files:"):
+ return int(line.split(":", 1)[1].split()[0])
+ return None
+
+
class TestDeleteDelayVsAfterSnapshot:
"""A destination entry created after its directory's scan survives under
--delete-delay but is removed by --delete-after's fresh end scan."""
@pytest.mark.parametrize("mt", [False, True])
def test_late_created_extra_survives_delay_not_after(self, mt):
- source, dest, received = _seed_pair("latecreate", big=True)
+ source, dest, received = _seed_pair(f"latecreate_mt{int(mt)}", big=True)
old_extra = os.path.join(received, "d", "old_extra")
new_extra = os.path.join(received, "d", "new_extra")
with ServerManager() as server:
@@ -356,3 +472,68 @@ class TestDeleteDelayVsAfterSnapshot:
f"{timing} (mt={mt}): new_extra present="
f"{os.path.exists(new_extra)}, expected survives={new_survives}"
)
+
+
+class TestDeleteAfterThreadsKeepSet:
+ """Regression: -j/--threads must still transmit the delete keep-set in every
+ timing. PipelineContextSender.delete_suppressed was left uninitialized, so a
+ garbage true silently skipped the late keep-set manifest under --threads.
+ Plain --delete now uses the per-directory plans, while --delete-commit /
+ --delete-after keep exercising the late whole-tree manifest."""
+
+ @pytest.mark.parametrize("delete_flag", ["--delete", "--delete-commit", "--delete-after"])
+ def test_threads_delete_after_sends_keep_set(self, delete_flag):
+ source, dest, received = _seed_pair("mtkeep")
+ extra = os.path.join(received, "d", "old_extra")
+ assert os.path.exists(extra)
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(source, dest, flags=["--threads", delete_flag],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ assert not os.path.exists(extra), (
+ f"{delete_flag} --threads did not remove an extra: delete keep-set was suppressed"
+ )
+
+class TestDeleteDelayMaxDeleteRefilledDir:
+ """--delete-delay charges the --max-delete budget on ACTUAL removals: the
+ refilled directory's late content is removed first (consuming the one slot),
+ so the directory itself and a later extra are skipped, matching rsync.
+
+ The refilled directory is at the destination ROOT (its plan is always sent
+ first) and the skipped extra is under a separate source directory, so the
+ ordering that decides the budget charge is deterministic -- not readdir
+ order. The refill is injected through the byte-barrier proxy so it is
+ causally after the plan frame."""
+
+ def test_budget_charged_on_actual_removal(self):
+ source = os.path.join(TEST_DATA_DIR, "ddmb_src")
+ dest = os.path.join(TEST_DATA_DIR, "ddmb_dst")
+ clean_dir(source)
+ clean_dir(dest)
+ _write(os.path.join(source, "a", "keep.bin"), b"B" * BIG_BYTES)
+ _write(os.path.join(source, "b", "keep.txt"), b"keep\n")
+ received = get_dest_received_dir(dest, source)
+ refilled_dir = os.path.join(received, "xdir")
+ os.makedirs(refilled_dir, exist_ok=True)
+ later_dir = os.path.join(received, "b", "ydir")
+ os.makedirs(later_dir, exist_ok=True)
+
+ def hook():
+ _write(os.path.join(refilled_dir, "new.txt"), b"created mid-transfer\n")
+
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ proxy = _SlicingProxy(server.port, hook=hook, hook_after=MID_TRANSFER_BYTES,
+ throttle=PROXY_THROTTLE, wait_for_reply=True)
+ flags = ["--delete-delay", "--max-delete=1", "--incremental", "--ignore-times", "--stats"]
+ result, _ = run_client(source, dest, flags=flags, port=proxy.port)
+ proxy.finish()
+ assert result.returncode == 25, (result.stderr or result.stdout)[:400]
+ assert proxy.hook_called.is_set(), "hook never fired"
+ # The late content consumes the single budget slot; the refilled
+ # directory itself and the later extra are skipped.
+ assert not os.path.exists(os.path.join(refilled_dir, "new.txt")), "late file survived"
+ assert os.path.isdir(later_dir), "later extra was not skipped by the budget"
+ # The one actual removal is reported.
+ assert _deleted_count(result.stdout) == 1, result.stdout
diff --git a/tests/integration/test_differential_parity.py b/tests/integration/test_differential_parity.py
new file mode 100644
index 0000000..a5ee3d2
--- /dev/null
+++ b/tests/integration/test_differential_parity.py
@@ -0,0 +1,719 @@
+"""Differential rsync-parity gate.
+
+Runs real ``rsync 3.4.1`` and FastSync over the same corpora and flags, then
+compares the destination trees and the normalized output of the
+output-oriented flags. This is the executable counterpart of
+``RSYNC_COMPAT.md``: the fast subset (``-m parity_ci``) guards the ✅ surface on
+every pull request, and the full set (``-m parity``) burns the documented
+⚠️/❌ residuals down.
+
+Known, documented differences live in ``parity_caveats.py``; anything else
+fails with a readable tree/stdout diff. A stale allowlist entry is reported
+loudly (and fails when ``FASTSYNC_PARITY_STRICT=1``).
+
+Run locally::
+
+ python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci
+ python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity
+"""
+import os
+import shutil
+import sys
+import warnings
+
+import pytest
+
+sys.path.insert(0, os.path.dirname(__file__))
+from common import ( # noqa: E402
+ ServerManager,
+ TEST_DATA_DIR,
+ clean_dir,
+ get_dest_received_dir,
+)
+from parity_caveats import ASPECTS, caveat_for # noqa: E402
+import parity_harness as H # noqa: E402
+
+RSYNC = shutil.which("rsync")
+requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
+
+# `--allow-super` matches the rest of the integration suite; `--allow-delete`
+# is needed only by the delete cases.
+SUPER = ("--allow-super",)
+DELETE = ("--allow-super", "--allow-delete")
+_OLD_MTIME = 1_500_000_000
+
+parity = pytest.mark.parity
+parity_ci = pytest.mark.parity_ci
+
+
+@pytest.fixture(scope="session")
+def parity_server_factory():
+ """Lazily start one server per distinct extra-argument set, per xdist worker."""
+ servers = {}
+
+ def get(extra):
+ key = tuple(extra)
+ if key not in servers:
+ s = ServerManager()
+ s.start(extra_args=list(extra))
+ servers[key] = s
+ return servers[key]
+
+ yield get
+ for s in servers.values():
+ s.stop()
+
+
+def _pin(path, mtime):
+ os.utime(path, (mtime, mtime))
+
+
+def _mk(path, data, mtime=None):
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ with open(path, "wb") as fh:
+ fh.write(data)
+ if mtime is not None:
+ _pin(path, mtime)
+
+
+# --- destination seeds ------------------------------------------------------
+
+def seed_extras(_src, rroot, froot):
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "extra.txt"), b"extra\n")
+ _mk(os.path.join(root, "extradir", "z.txt"), b"z\n")
+
+
+def seed_update(_src, rroot, froot):
+ for root in (rroot, froot):
+ p = os.path.join(root, "a.txt")
+ _mk(p, b"destination is newer and longer\n", 2_000_000_000)
+
+
+def seed_ignore_existing(_src, rroot, froot):
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "a.txt"), b"destination-kept\n", _OLD_MTIME)
+
+
+def seed_append(_src, rroot, froot):
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "a.txt"), b"hello ", _OLD_MTIME)
+
+
+def seed_backup(_src, rroot, froot):
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "a.txt"), b"OLD-CONTENT\n", _OLD_MTIME)
+
+
+def seed_size_only(_src, rroot, froot):
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "a.txt"), b"XXXXXXXXXXX\n", _OLD_MTIME)
+
+
+def seed_delete_excluded(_src, rroot, froot):
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "drop.log"), b"stale log\n", _OLD_MTIME)
+ _mk(os.path.join(root, "extra.txt"), b"extra\n", _OLD_MTIME)
+ _mk(os.path.join(root, "keep.txt"), b"keep\n", _OLD_MTIME)
+
+
+def seed_filter_protect(_src, rroot, froot):
+ """Destination-only entries, including nested ones, for the receiver-side
+ `protect` rule: the `.log` extras must survive --delete, the rest go."""
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "extra.log"), b"dest-only log\n", _OLD_MTIME)
+ _mk(os.path.join(root, "other.txt"), b"dest-only other\n", _OLD_MTIME)
+ _mk(os.path.join(root, "sub", "extra2.log"), b"nested dest-only log\n", _OLD_MTIME)
+ _mk(os.path.join(root, "sub", "other2.txt"), b"nested dest-only other\n", _OLD_MTIME)
+
+
+def seed_max_delete(_src, rroot, froot):
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "extra1.txt"), b"e1\n", _OLD_MTIME)
+ _mk(os.path.join(root, "extra2.txt"), b"e2\n", _OLD_MTIME)
+
+
+def fuzzy_basis_seed(_src, rroot, froot):
+ """Seed a same-suffix sibling whose name is one edit from the source and
+ whose content matches it, with a DIFFERENT mtime so rsync's exact
+ size+mtime pass cannot fire: both tools must select it via the
+ name-distance pass. Where the two tools' basis choices coincide the
+ block-level results are identical when the block size is pinned."""
+ for root in (rroot, froot):
+ _mk(os.path.join(root, "report_v1.txt"), H.FUZZY_PAYLOAD, _OLD_MTIME)
+
+
+def max_delete_count_check(_src, rroot, froot, _rs, _fs):
+ """The exact survivor set is order-dependent; the count must still match."""
+ r = H.snapshot(rroot)
+ f = H.snapshot(froot)
+ if len(r) != len(f):
+ return [f"survivor count differs: rsync={len(r)} fastsync={len(f)}"]
+ return []
+
+
+# --- case table -------------------------------------------------------------
+
+_CASES = [
+ # --- core archive / recursion -----------------------------------------
+ H.Case("archive", "basic", ["-a"], ci=True, ref="-a/--archive"),
+ H.Case("recursive", "basic", ["-r"], ci=True, ref="-r/--recursive"),
+ H.Case("unicode_names", "unicode", ["-a"], ci=True, ref="-a unicode names"),
+ H.Case("links_archive", "links", ["-a"], ci=True, ref="-l/--links"),
+ H.Case("copy_links", "links", ["-aL"], ref="-L/--copy-links"),
+ H.Case("hardlinks", "hardlinks", ["-a", "-H"], compare_hardlinks=True,
+ ci=True, ref="-H/--hard-links"),
+ H.Case("hardlinks_without_H", "hardlinks", ["-a"], compare_hardlinks=True,
+ ref="hardlinks without -H"),
+ H.Case("sparse", "sparse", ["-a", "-S"], ref="-S/--sparse"),
+
+ # --- compression / checksums ------------------------------------------
+ H.Case("compress_zstd", "basic", ["-a", "-z"], ci=True, ref="-z/--compress"),
+ H.Case("checksum", "basic", ["-a", "-c"], ref="-c/--checksum"),
+ H.Case("checksum_choice_xxh64", "basic",
+ ["-a", "-c", "--checksum-choice=xxh64"], ref="--checksum-choice"),
+
+ # --- selection --------------------------------------------------------
+ H.Case("exclude", "filters", ["-a", "--exclude=*.log"], ci=True,
+ ref="--exclude"),
+ H.Case("include_exclude", "filters",
+ ["-a", "--include=*.txt", "--exclude=*"], ci=True,
+ ref="--include/--exclude ordering"),
+ H.Case("filter_rules", "filters",
+ ["-a", "-f", "- *.log", "-f", "+ *.txt", "-f", "- *"],
+ ref="--filter/-f grammar"),
+ H.Case("max_size", "basic", ["-a", "--max-size=1000"], ref="--max-size"),
+ H.Case("min_size", "basic", ["-a", "--min-size=1000"], ref="--min-size"),
+
+ # --- output-oriented --------------------------------------------------
+ H.Case("stats", "basic", ["-a", "--stats"], stdout=H.STDOUT_STATS,
+ ci=True, ref="--stats"),
+ H.Case("itemize", "links", ["-a", "-i"], stdout=H.STDOUT_ITEMIZE,
+ ci=True, ref="-i/--itemize-changes"),
+ H.Case("out_format_n_l", "basic", ["-a", "--out-format=%n %l"],
+ stdout=H.STDOUT_OUTFMT, ref="--out-format %n %l"),
+ H.Case("out_format_i_n", "basic", ["-a", "--out-format=%i %n"],
+ stdout=H.STDOUT_OUTFMT, ref="--out-format %i %n"),
+ H.Case("progress", "multidir", ["-a", "--progress"], stdout=H.STDOUT_PROGRESS,
+ ci=True, ref="--progress multi-directory file list"),
+ H.Case("progress_threads", "multidir", ["-a", "--progress"],
+ fastsync_flags=["-a", "--progress", "--threads"],
+ stdout=H.STDOUT_PROGRESS, ci=True,
+ ref="--progress multi-directory file list (--threads)"),
+
+ # --- transfer modifications -------------------------------------------
+ H.Case("update", "basic", ["-a", "--update"], seed=seed_update,
+ ref="-u/--update"),
+ H.Case("ignore_existing", "basic", ["-a", "--ignore-existing"],
+ seed=seed_ignore_existing, ci=True, ref="--ignore-existing"),
+ H.Case("size_only", "basic",
+ ["-a", "--size-only"], fastsync_flags=["-a", "--incremental", "--size-only"],
+ seed=seed_size_only, ref="--size-only"),
+ H.Case("append", "basic", ["-a", "--append"], seed=seed_append,
+ ref="--append"),
+ H.Case("append_verify", "basic", ["-a", "--append-verify"], seed=seed_append,
+ ref="--append-verify"),
+ H.Case("backup", "basic", ["-a", "--backup"], seed=seed_backup,
+ ref="--backup"),
+ H.Case("chmod", "basic", ["-a", "--chmod=Fu+rwx"], compare_modes=True,
+ ci=True, ref="--chmod"),
+
+ # --- delta / similar-file basis (--fuzzy) -----------------------------
+ # Basis choices coincide here (same-suffix sibling, name distance one edit,
+ # content identical); with the block size pinned both tools report the same
+ # Matched/Literal/transferred counters. The residual (FastSync's narrower
+ # delta size window) is covered by TestFuzzy in test_parity_quickwins.py.
+ H.Case("fuzzy_basis", "fuzzy",
+ ["-a", "--no-whole-file", "--fuzzy", "--stats", "-B8192"],
+ fastsync_flags=["-a", "--incremental", "--delta", "--fuzzy",
+ "--stats", "--delta-block=8192"],
+ seed=fuzzy_basis_seed, stdout=H.STDOUT_STATS, ci=True,
+ ref="-y/--fuzzy similar-file basis"),
+
+ # --- deletion ---------------------------------------------------------
+ H.Case("delete", "basic", ["-a", "--delete"], seed=seed_extras,
+ server_args=DELETE, ci=True, ref="--delete"),
+ H.Case("delete_before", "basic", ["-a", "--delete-before"], seed=seed_extras,
+ server_args=DELETE, ref="--delete-before"),
+ H.Case("delete_during", "basic", ["-a", "--delete-during"], seed=seed_extras,
+ server_args=DELETE, ref="--delete-during"),
+ H.Case("delete_delay", "basic", ["-a", "--delete-delay"], seed=seed_extras,
+ server_args=DELETE, ref="--delete-delay"),
+ H.Case("delete_after", "basic", ["-a", "--delete-after"], seed=seed_extras,
+ server_args=DELETE, ref="--delete-after"),
+ H.Case("delete_commit", "basic", ["-a", "--delete-after"], seed=seed_extras,
+ fastsync_flags=["-a", "--delete-commit"], server_args=DELETE,
+ ref="FastSync-only --delete-commit == rsync --delete-after"),
+ H.Case("delete_excluded", "filters",
+ ["-a", "--delete", "--delete-excluded", "--exclude=*.log"],
+ seed=seed_delete_excluded, server_args=DELETE, ref="--delete-excluded"),
+ H.Case("exclude_protect_dest_only", "filters",
+ ["-a", "--delete", "--exclude=*.log"],
+ seed=seed_delete_excluded, server_args=DELETE, ci=True,
+ ref="--delete protects a destination-only excluded entry like rsync"),
+ H.Case("max_delete", "basic", ["-a", "--delete", "--max-delete=1"],
+ seed=seed_max_delete, server_args=DELETE,
+ extra_check=max_delete_count_check, compare_tree=False,
+ ref="--max-delete"),
+ H.Case("filter_protect", "filters",
+ ["-a", "--delete", "--filter=P *.log"],
+ seed=seed_filter_protect, server_args=DELETE, ci=True,
+ ref="--filter P/--protect receiver-side delete protection (default during)"),
+ H.Case("filter_protect_during", "filters",
+ ["-a", "--delete-during", "--filter=P *.log"],
+ seed=seed_filter_protect, server_args=DELETE, ci=True,
+ ref="--filter P/--protect under --delete-during"),
+ H.Case("filter_protect_delay", "filters",
+ ["-a", "--delete-delay", "--filter=P *.log"],
+ seed=seed_filter_protect, server_args=DELETE, ci=True,
+ ref="--filter P/--protect under --delete-delay"),
+ H.Case("filter_protect_after", "filters",
+ ["-a", "--delete-after", "--filter=P *.log"],
+ seed=seed_filter_protect, server_args=DELETE, ci=True,
+ ref="--filter P/--protect under the whole-tree --delete-after commit"),
+
+ # --- relative / dirs --------------------------------------------------
+ H.Case("relative_general", "basic", ["-a", "-R"], layout=H.MIRROR_ABS,
+ compare_modes=True, ref="-R/--relative"),
+ H.Case("relative_no_implied_dirs", "basic",
+ ["-a", "-R", "--no-implied-dirs"], layout=H.MIRROR_ABS,
+ compare_modes=True, ref="--no-implied-dirs"),
+ H.Case("files_from", "relative", ["--dirs", "-R"],
+ files_from=("dir1", "sub/x.txt"), layout=H.RELATIVE, ci=True,
+ ref="-d/--dirs + --files-from"),
+ H.Case("dirs_plain", "basic", ["-d"], fs_src_suffix="/",
+ ref="-d/--dirs (plain)"),
+ H.Case("empty_dirs_recursive", "empty_dir", ["-a"],
+ ref="recursive empty-directory residual"),
+ H.Case("empty_dirs_files_from", "empty_dir", ["--dirs", "-R"],
+ files_from=("emptydir",), layout=H.RELATIVE, ci=True,
+ ref="-d/--dirs explicit empty directory"),
+
+ # --- codecs -----------------------------------------------------------
+ H.Case("iconv_identity", "basic", ["-a", "--iconv=UTF-8,UTF-8"],
+ ref="--iconv identity"),
+ H.Case("iconv_convert", "iconv",
+ ["-a", "--iconv=ISO-8859-1,UTF-8"],
+ server_args=("--allow-super", "--iconv=UTF-8"),
+ ref="--iconv conversion (receiver declares its own charset)"),
+ # rsync's spec is LOCAL,REMOTE and the destination end's charset is REMOTE
+ # on a push, so a default server writes the wire (UTF-8) names verbatim.
+ H.Case("iconv_default_server", "iconv",
+ ["-a", "--iconv=ISO-8859-1,UTF-8"],
+ ref="--iconv push direction (default receiver charset = REMOTE)"),
+
+ # --- partial ----------------------------------------------------------
+ H.Case("partial_complete", "basic", ["-a", "--partial"], ref="--partial"),
+]
+
+# Cases that must always be tolerated (documented ⚠️/❌ residuals) get an
+# allowlist entry; the table below stays the exact ✅ surface.
+ALL_CASES = _CASES
+
+
+def _params():
+ out = []
+ for case in ALL_CASES:
+ marks = [parity]
+ if case.ci:
+ marks.append(parity_ci)
+ out.append(pytest.param(case, id=case.id, marks=marks))
+ return out
+
+
+def _aspects_to_check(result):
+ return {
+ "tree": result["tree"],
+ "stdout": result["stdout"],
+ "extra": result["extra"],
+ }
+
+
+def _assert_no_unexpected(case_id, mismatches, caveat, ref=""):
+ unexpected = {a: v for a, v in mismatches.items() if v and a not in caveat}
+ if unexpected:
+ lines = [f"differential parity mismatch for case {case_id!r}:"]
+ lines.append(f" ref: {ref or 'see RSYNC_COMPAT.md'}")
+ for aspect, detail in unexpected.items():
+ lines.append(f" --- {aspect} ---")
+ lines.extend(" " + str(d) for d in detail)
+ lines.append("If this is a documented residual, add it to "
+ "tests/integration/parity_caveats.py with a RSYNC_COMPAT.md "
+ "reference. Do not allowlist an undocumented divergence.")
+ pytest.fail("\n".join(lines))
+
+ stale = [a for a in ASPECTS
+ if a in caveat and a != "rc" and not mismatches.get(a)]
+ if stale:
+ msg = (f"stale parity allowlist entry for case {case_id!r}, aspect(s) "
+ f"{stale}: FastSync now matches rsync. Remove it from "
+ f"parity_caveats.py (and update RSYNC_COMPAT.md if the row moved).")
+ if os.environ.get("FASTSYNC_PARITY_STRICT") == "1":
+ pytest.fail(msg)
+ warnings.warn(msg, stacklevel=2)
+
+
+@requires_rsync
+@pytest.mark.parametrize("case", _params())
+def test_differential_case(case, parity_server_factory):
+ server = parity_server_factory(case.server_args)
+ result = H.execute_case(case, server)
+ caveat = caveat_for(case.id)
+
+ mismatches = _aspects_to_check(result)
+ if result["rsync_rc"] != result["fastsync_rc"]:
+ mismatches["rc"] = [
+ f"rsync rc={result['rsync_rc']} fastsync rc={result['fastsync_rc']} "
+ f"(rsync stderr: {result['rsync_stderr'][:200]!r}, "
+ f"fastsync stderr: {result['fastsync_stderr'][:200]!r})"]
+ _assert_no_unexpected(case.id, mismatches, caveat, ref=case.ref)
+
+
+# ---------------------------------------------------------------------------
+# Multi-run and setup-heavy scenarios (kept as explicit tests)
+# ---------------------------------------------------------------------------
+
+def _result_aspects(result):
+ return _aspects_to_check(result)
+
+
+_STANDALONE_REFS = {
+ "incremental_modified": "-i/--itemize-changes + incremental second run",
+ "compare_dest": "--compare-dest",
+ "copy_dest": "--copy-dest",
+ "link_dest": "--link-dest",
+ "link_dest_stats": "--link-dest + --stats",
+ "verify_basis": "--verify-basis (FastSync-only)",
+ "verify_basis_default": "--verify-basis (default quick-check vs rsync)",
+ "added_and_deleted": "--delete across two runs",
+ "added_and_deleted_seed": "--delete across two runs",
+ "one_file_system": "-x/--one-file-system",
+}
+
+
+def _run_and_check(case_id, result, ref=""):
+ mismatches = _result_aspects(result)
+ if result["rsync_rc"] != result["fastsync_rc"]:
+ mismatches["rc"] = [
+ f"rsync rc={result['rsync_rc']} fastsync rc={result['fastsync_rc']} "
+ f"(rsync stderr: {result['rsync_stderr'][:200]!r}, "
+ f"fastsync stderr: {result['fastsync_stderr'][:200]!r})"]
+ _assert_no_unexpected(case_id, mismatches, caveat_for(case_id),
+ ref=ref or _STANDALONE_REFS.get(case_id, ""))
+
+
+@requires_rsync
+@parity
+def test_incremental_modified_file(parity_server_factory):
+ """A second run sends only the modified file; destinations stay identical."""
+ case_id = "incremental_modified"
+ src = os.path.join(TEST_DATA_DIR, "parity_inc_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_inc_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_inc_fdst")
+ H.CORPORA["basic"](src)
+ server = parity_server_factory(SUPER)
+
+ # Seed both destinations with the initial content.
+ H.run_differential(src, rdst, fdst, ["-a"], ["-a"], server,
+ extra_check=lambda *a: [])
+ with open(os.path.join(src, "a.txt"), "wb") as fh:
+ fh.write(b"hello world, now modified and longer\n")
+ _pin(os.path.join(src, "a.txt"), 1_650_000_000)
+
+ result = H.run_differential(
+ src, rdst, fdst, ["-a", "-i"], ["-a", "-i", "--incremental"], server,
+ stdout=H.STDOUT_ITEMIZE)
+ _run_and_check(case_id, result)
+
+
+def _seed_basis(rel_entries):
+ def seed(src, rroot, froot):
+ for root in (rroot, froot):
+ os.makedirs(root, exist_ok=True)
+ for rel, data in rel_entries.items():
+ _mk(os.path.join(root, rel), data)
+ return seed
+
+
+@requires_rsync
+@parity
+def test_compare_dest_skips_basis(parity_server_factory):
+ """--compare-dest: a file present in the basis is not copied."""
+ case_id = "compare_dest"
+ src = os.path.join(TEST_DATA_DIR, "parity_cmpd_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_cmpd_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_cmpd_fdst")
+ clean_dir(src)
+ _mk(os.path.join(src, "f.txt"), b"basis-content\n")
+ _pin(os.path.join(src, "f.txt"), _OLD_MTIME)
+ server = parity_server_factory(SUPER)
+ rel = os.path.abspath(src).lstrip(os.sep)
+
+ # rsync resolves --compare-dest relative to the destination dir; FastSync
+ # resolves it under the receive root and appends the mirrored source path.
+ # Both rely on rsync's size+mtime quick-check, so the basis mtime is pinned
+ # to the source's to keep the match deterministic across a second boundary.
+ def seed(_src, rroot, froot):
+ _mk(os.path.join(rroot, "basis", "f.txt"), b"basis-content\n", _OLD_MTIME)
+ _mk(os.path.join(fdst, "basis", rel, "f.txt"), b"basis-content\n", _OLD_MTIME)
+
+ def extra(_src, rroot, froot, _rs, _fs):
+ out = []
+ for label, root in (("rsync", rroot), ("fastsync", froot)):
+ if os.path.exists(os.path.join(root, "f.txt")):
+ out.append(f"{label} copied a file that is present in the "
+ f"compare basis")
+ return out
+
+ result = H.run_differential(
+ src, rdst, fdst,
+ ["-a", "--compare-dest=basis"],
+ ["-a", f"--compare-dest={os.path.join(fdst, 'basis')}", "--incremental"],
+ server, seed=seed, ignore_paths=("basis",), extra_check=extra)
+ _run_and_check(case_id, result)
+
+
+@requires_rsync
+@parity
+def test_link_dest_hardlinks_basis(parity_server_factory):
+ """--link-dest: an unchanged file is hard-linked to the basis, not copied."""
+ case_id = "link_dest"
+ src = os.path.join(TEST_DATA_DIR, "parity_linkd_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_linkd_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_linkd_fdst")
+ clean_dir(src)
+ _mk(os.path.join(src, "f.txt"), b"link-basis-content\n")
+ _pin(os.path.join(src, "f.txt"), _OLD_MTIME)
+ server = parity_server_factory(SUPER)
+ rel = os.path.abspath(src).lstrip(os.sep)
+
+ def seed(_src, rroot, froot):
+ _mk(os.path.join(rroot, "basis", "f.txt"), b"link-basis-content\n", _OLD_MTIME)
+ _mk(os.path.join(fdst, "basis", rel, "f.txt"), b"link-basis-content\n", _OLD_MTIME)
+
+ def extra(_src, rroot, froot, _rs, _fs):
+ r_basis = os.stat(os.path.join(rroot, "basis", "f.txt")).st_ino
+ f_basis = os.stat(os.path.join(fdst, "basis", rel, "f.txt")).st_ino
+ out = []
+ for label, root, basis in (("rsync", rroot, r_basis),
+ ("fastsync", froot, f_basis)):
+ target = os.path.join(root, "f.txt")
+ if not os.path.exists(target):
+ out.append(f"{label}: f.txt missing")
+ elif os.stat(target).st_ino != basis:
+ out.append(f"{label}: f.txt is not hard-linked to the basis")
+ return out
+
+ result = H.run_differential(
+ src, rdst, fdst,
+ ["-a", "--link-dest=basis"],
+ ["-a", f"--link-dest={os.path.join(fdst, 'basis')}", "--incremental"],
+ server, seed=seed, ignore_paths=("basis",), extra_check=extra)
+ _run_and_check(case_id, result)
+
+
+@requires_rsync
+@parity
+def test_link_dest_stats_matches_rsync(parity_server_factory):
+ """A basis hit must not be counted as created or literal data: rsync reports
+ zero for both, so FastSync's receiver tallies must too (regression for the
+ basis materialization over-report)."""
+ case_id = "link_dest_stats"
+ src = os.path.join(TEST_DATA_DIR, "parity_linkds_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_linkds_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_linkds_fdst")
+ clean_dir(src)
+ _mk(os.path.join(src, "f.txt"), b"link-basis-content\n")
+ _pin(os.path.join(src, "f.txt"), _OLD_MTIME)
+ server = parity_server_factory(SUPER)
+ rel = os.path.abspath(src).lstrip(os.sep)
+
+ def seed(_src, rroot, froot):
+ _mk(os.path.join(rroot, "basis", "f.txt"), b"link-basis-content\n", _OLD_MTIME)
+ _mk(os.path.join(fdst, "basis", rel, "f.txt"), b"link-basis-content\n", _OLD_MTIME)
+
+ result = H.run_differential(
+ src, rdst, fdst,
+ ["-a", "--link-dest=basis", "--stats"],
+ ["-a", f"--link-dest={os.path.join(fdst, 'basis')}", "--incremental", "--stats"],
+ server, seed=seed, ignore_paths=("basis",), stdout=H.STDOUT_STATS)
+ _run_and_check(case_id, result, ref="--link-dest + --stats")
+
+
+@requires_rsync
+@parity
+def test_copy_dest_copies_basis(parity_server_factory):
+ """--copy-dest: a basis match is materialized as an independent copy with the
+ source's attributes, matching rsync (copy then fix attributes)."""
+ case_id = "copy_dest"
+ src = os.path.join(TEST_DATA_DIR, "parity_copyd_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_copyd_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_copyd_fdst")
+ clean_dir(src)
+ _mk(os.path.join(src, "f.txt"), b"copy-basis-content\n")
+ _pin(os.path.join(src, "f.txt"), 1_600_000_000)
+ os.chmod(os.path.join(src, "f.txt"), 0o755)
+ server = parity_server_factory(SUPER)
+ rel = os.path.abspath(src).lstrip(os.sep)
+
+ def seed(_src, rroot, froot):
+ # Basis content matches the source; give the basis a different mode so a
+ # wrong "keep basis attributes" implementation is visible.
+ _mk(os.path.join(rroot, "basis", "f.txt"), b"copy-basis-content\n",
+ 1_600_000_000)
+ os.chmod(os.path.join(rroot, "basis", "f.txt"), 0o644)
+ _mk(os.path.join(fdst, "basis", rel, "f.txt"), b"copy-basis-content\n",
+ 1_600_000_000)
+ os.chmod(os.path.join(fdst, "basis", rel, "f.txt"), 0o644)
+
+ def extra(_src, rroot, froot, _rs, _fs):
+ out = []
+ bases = {"rsync": os.path.join(rroot, "basis", "f.txt"),
+ "fastsync": os.path.join(fdst, "basis", rel, "f.txt")}
+ for label, root in (("rsync", rroot), ("fastsync", froot)):
+ target = os.path.join(root, "f.txt")
+ if not os.path.exists(target):
+ out.append(f"{label}: f.txt missing")
+ continue
+ if os.stat(target).st_ino == os.stat(bases[label]).st_ino:
+ out.append(f"{label}: f.txt is hard-linked, not copied")
+ if (os.stat(target).st_mode & 0o777) != 0o755:
+ out.append(f"{label}: f.txt mode "
+ f"{oct(os.stat(target).st_mode & 0o777)} != 0o755")
+ return out
+
+ result = H.run_differential(
+ src, rdst, fdst,
+ ["-a", "--copy-dest=basis"],
+ ["-a", f"--copy-dest={os.path.join(fdst, 'basis')}", "--incremental"],
+ server, seed=seed, ignore_paths=("basis",), extra_check=extra,
+ compare_modes=True)
+ _run_and_check(case_id, result)
+
+
+@requires_rsync
+@parity
+def test_verify_basis_restores_strict_content(parity_server_factory):
+ """Default matches rsync's metadata quick-check; FastSync-only
+ `--verify-basis` restores strict content equality and transfers the source
+ when a same-size/different-content basis would otherwise be trusted."""
+ case_id = "verify_basis"
+ src = os.path.join(TEST_DATA_DIR, "parity_vbasis_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_vbasis_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_vbasis_fdst")
+ clean_dir(src)
+ _mk(os.path.join(src, "f.txt"), b"AAAA\n")
+ _pin(os.path.join(src, "f.txt"), _OLD_MTIME)
+ server = parity_server_factory(SUPER)
+ rel = os.path.abspath(src).lstrip(os.sep)
+
+ def seed(_src, rroot, froot):
+ # Same size and mtime as the source, different bytes: a metadata
+ # quick-check trusts it; --verify-basis must not.
+ for root, basis_rel in ((rroot, os.path.join("basis", "f.txt")),
+ (fdst, os.path.join("basis", rel, "f.txt"))):
+ _mk(os.path.join(root, basis_rel), b"BBBB\n", _OLD_MTIME)
+
+ # Default: both tools trust the basis (rsync's quick check), so the
+ # destination carries the basis bytes and the trees match.
+ result = H.run_differential(
+ src, rdst, fdst,
+ ["-a", "--link-dest=basis"],
+ ["-a", f"--link-dest={os.path.join(fdst, 'basis')}", "--incremental"],
+ server, seed=seed, ignore_paths=("basis",))
+ _run_and_check(case_id + "_default", result)
+
+ # --verify-basis (FastSync only): the digest mismatch rejects the basis and
+ # the source is transferred, so the destination is the source bytes. rsync
+ # has no such flag; assert the FastSync outcome directly against the source.
+ fdst2 = os.path.join(TEST_DATA_DIR, "parity_vbasis_fdst2")
+ clean_dir(fdst2)
+ for root, basis_rel in ((fdst2, os.path.join("basis", rel, "f.txt")),):
+ _mk(os.path.join(root, basis_rel), b"BBBB\n", _OLD_MTIME)
+ result, _ = H.run_fastsync(src, fdst2,
+ ["-a", f"--link-dest={os.path.join(fdst2, 'basis')}",
+ "--incremental", "--verify-basis"], server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ target = os.path.join(get_dest_received_dir(fdst2, src), "f.txt")
+ with open(target, "rb") as fh:
+ assert fh.read() == b"AAAA\n", \
+ "--verify-basis must reject the same-size/different-content basis"
+
+
+@requires_rsync
+@parity
+def test_added_and_deleted_between_runs(parity_server_factory):
+ """A source deletion and addition sync correctly under --delete."""
+ case_id = "added_and_deleted"
+ src = os.path.join(TEST_DATA_DIR, "parity_addel_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_addel_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_addel_fdst")
+ server = parity_server_factory(DELETE)
+ H.CORPORA["basic"](src)
+
+ seed = seed_extras
+ result = H.run_differential(
+ src, rdst, fdst, ["-a", "--delete"], ["-a", "--delete"], server,
+ seed=seed)
+ _run_and_check(case_id + "_seed", result)
+
+ os.remove(os.path.join(src, "a.txt"))
+ _mk(os.path.join(src, "added.txt"), b"added between runs\n")
+ result = H.run_differential(
+ src, rdst, fdst, ["-a", "--delete", "-i"],
+ ["-a", "--delete", "-i", "--incremental"], server,
+ stdout=H.STDOUT_ITEMIZE)
+ _run_and_check(case_id, result)
+
+
+@requires_rsync
+@parity
+def test_one_file_system(parity_server_factory):
+ """-x emits the mount-point directory but not its contents."""
+ case_id = "one_file_system"
+ local = os.stat(".")
+ shm = "/dev/shm"
+ if not os.path.isdir(shm):
+ pytest.skip("/dev/shm not available")
+ if os.stat(shm).st_dev == local.st_dev:
+ pytest.skip("no cross-device filesystem available")
+
+ src = os.path.join(TEST_DATA_DIR, "parity_ofs_src")
+ rdst = os.path.join(TEST_DATA_DIR, "parity_ofs_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "parity_ofs_fdst")
+ clean_dir(src)
+ _mk(os.path.join(src, "keep.txt"), b"keep\n")
+ probe = os.path.join(shm, f"fastsync_ofs_{os.getpid()}")
+ shutil.rmtree(probe, ignore_errors=True)
+ os.makedirs(probe)
+ _mk(os.path.join(probe, "inside.txt"), b"cross\n")
+ try:
+ os.symlink(probe, os.path.join(src, "nested_link"))
+ server = parity_server_factory(SUPER)
+ result = H.run_differential(
+ src, rdst, fdst,
+ ["-a", "--copy-links", "-x"],
+ ["-a", "--copy-links", "-x"], server)
+ _run_and_check(case_id, result)
+ finally:
+ shutil.rmtree(probe, ignore_errors=True)
+
+
+@requires_rsync
+@parity
+def test_parity_caveats_reference_known_cases():
+ """Every allowlist entry must name a real case id and aspect."""
+ from parity_caveats import CAVEATS
+ known = {c.id for c in ALL_CASES} | {
+ "incremental_modified", "compare_dest", "link_dest",
+ "added_and_deleted", "added_and_deleted_seed", "one_file_system",
+ }
+ problems = []
+ for case_id, entry in CAVEATS.items():
+ if case_id not in known:
+ problems.append(f"unknown case id in parity_caveats.py: {case_id!r}")
+ for aspect in entry:
+ if aspect not in ASPECTS:
+ problems.append(
+ f"{case_id!r}: unknown aspect {aspect!r} (expected {ASPECTS})")
+ assert not problems, "\n".join(problems)
diff --git a/tests/integration/test_fault_injection.py b/tests/integration/test_fault_injection.py
index 80d6a8d..7b8eb46 100644
--- a/tests/integration/test_fault_injection.py
+++ b/tests/integration/test_fault_injection.py
@@ -36,7 +36,7 @@ from common import ( # noqa: E402
verify_transfer,
)
-PROTOCOL_VERSION = b"2.26.0"
+PROTOCOL_VERSION = b"2.28.0"
STATUS_MANIFEST = 5
STATUS_OK = 0
diff --git a/tests/integration/test_features.py b/tests/integration/test_features.py
index 2f9e686..1a2c05c 100644
--- a/tests/integration/test_features.py
+++ b/tests/integration/test_features.py
@@ -583,6 +583,57 @@ class TestRemoteDryRun:
assert os.path.exists(extra), f"{flags} deleted an extra in dry-run"
assert _snapshot_tree(received) == before, f"{flags} mutated the destination"
+ @pytest.mark.skipif(shutil.which("rsync") is None, reason="rsync not installed")
+ def test_dry_run_delete_lines_match_rsync(self):
+ """`-n --delete` lists exactly the destination extras rsync would remove.
+
+ Covers the three cases that a real run protects: the file being updated
+ (in the keep set), a filter-excluded source entry (protected prefix), and
+ a --max-size-pruned source entry (always-protected prefix). Track 4a
+ adds a fourth: a destination-only entry matching the exclude rule is
+ re-derived on the receiver and also protected, so only the genuine
+ destination-only `extra.txt` appears.
+ """
+ source = os.path.join(TEST_DATA_DIR, "dryrep_src")
+ rdst = os.path.join(TEST_DATA_DIR, "dryrep_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "dryrep_fdst")
+ clean_dir(source)
+ clean_dir(rdst)
+ clean_dir(fdst)
+ for name, data in (("a.txt", b"new content\n"), ("keep.log", b"log\n"),
+ ("big.bin", b"B" * 2000)):
+ with open(os.path.join(source, name), "wb") as fh:
+ fh.write(data)
+ os.utime(os.path.join(source, "a.txt"), (1_700_000_000, 1_700_000_000))
+ received = get_dest_received_dir(fdst, source)
+ os.makedirs(received, exist_ok=True)
+ for root in (rdst, received):
+ for name, data in (("a.txt", b"old\n"), ("keep.log", b"log\n"),
+ ("big.bin", b"B" * 2000), ("extra.txt", b"extra\n"),
+ ("stray.log", b"dest only\n")):
+ with open(os.path.join(root, name), "wb") as fh:
+ fh.write(data)
+ os.utime(os.path.join(root, name), (1_500_000_000, 1_500_000_000))
+
+ flags = ["-a", "-n", "-i", "--delete", "--exclude=*.log", "--max-size=1000"]
+ r = subprocess.run(["rsync", "-an", "-i", "--delete", "--exclude=*.log",
+ "--max-size=1000", source + "/", rdst + "/"],
+ capture_output=True, text=True,
+ env=dict(os.environ, LC_ALL="C"))
+ assert r.returncode == 0, r.stderr
+ rsync_del = sorted(l for l in r.stdout.splitlines() if l.startswith("*deleting"))
+ assert rsync_del == ["*deleting extra.txt"], f"unexpected rsync set: {rsync_del}"
+
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(source, fdst, flags=flags, port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ fs_del = sorted(l for l in (result.stdout or "").splitlines()
+ if l.startswith("*deleting"))
+ assert fs_del == rsync_del, f"rsync={rsync_del}\nfastsync={fs_del}"
+ assert os.path.exists(os.path.join(received, "stray.log")), \
+ "destination-only exclude match must be protected in the dry-run report"
+
@pytest.mark.ci
def test_remote_dry_run_quiet_is_silent(self, shared_server):
source = os.path.join(TEST_DATA_DIR, "remote_dry_quiet_src")
@@ -2376,6 +2427,61 @@ class TestTempDir:
assert not mismatches, f"Mismatch: {mismatches}"
self._assert_clean_scratch(os.path.join(dest, "scratch"))
+ @pytest.mark.skipif(shutil.which("rsync") is None, reason="rsync not installed")
+ def test_relative_temp_dir_matches_rsync_absolute_rejected(self):
+ """Differential: a relative --temp-dir is resolved under the destination
+ by both (rsync 3.4.1 and FastSync), producing identical trees. An
+ absolute --temp-dir is used verbatim by rsync standalone, but the
+ receiver deliberately confines it to the receive root (security
+ invariant), so FastSync rejects it without writing outside the root.
+ """
+ source = self._make_source("tempdir_diff_src")
+ rdst = os.path.join(TEST_DATA_DIR, "tempdir_diff_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "tempdir_diff_fdst")
+ clean_dir(rdst)
+ clean_dir(fdst)
+ os.makedirs(os.path.join(rdst, "scratch"), exist_ok=True)
+ os.makedirs(os.path.join(fdst, "scratch"), exist_ok=True)
+ r = subprocess.run(["rsync", "-a", "--temp-dir=scratch", source + "/", rdst + "/"],
+ capture_output=True, text=True,
+ env=dict(os.environ, LC_ALL="C"))
+ assert r.returncode == 0, r.stderr
+ with ServerManager() as server:
+ server.start()
+ result, _ = run_client(source, fdst, flags=["--temp-dir=scratch"],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ # rsync lays the source contents directly in rdst; FastSync mirrors the
+ # absolute source path below fdst. Compare the mirrored content trees
+ # (the scratch dir lives at each destination root).
+ rtree = sorted(os.path.relpath(os.path.join(dp, n), rdst)
+ for dp, dn, fn in os.walk(rdst)
+ for n in dn + fn if os.path.join(dp, n) != os.path.join(rdst, "scratch"))
+ mirror = get_dest_received_dir(fdst, source)
+ ftree = sorted(os.path.relpath(os.path.join(dp, n), mirror)
+ for dp, dn, fn in os.walk(mirror) for n in dn + fn)
+ assert rtree == ftree, f"relative temp-dir tree mismatch: {rtree} != {ftree}"
+ assert _walk_tmp_files(os.path.join(rdst, "scratch")) == []
+ assert _walk_tmp_files(os.path.join(fdst, "scratch")) == []
+
+ # Absolute temp dir: rsync accepts it; FastSync rejects it safely.
+ abs_scratch = os.path.join(TEST_DATA_DIR, "tempdir_diff_abs")
+ clean_dir(abs_scratch)
+ rdst2 = os.path.join(TEST_DATA_DIR, "tempdir_diff_rdst2")
+ clean_dir(rdst2)
+ r2 = subprocess.run(["rsync", "-a", "--temp-dir=" + abs_scratch, source + "/", rdst2 + "/"],
+ capture_output=True, text=True,
+ env=dict(os.environ, LC_ALL="C"))
+ assert r2.returncode == 0, r2.stderr
+ fdst2 = os.path.join(TEST_DATA_DIR, "tempdir_diff_fdst2")
+ clean_dir(fdst2)
+ with ServerManager() as server:
+ server.start()
+ result2, _ = run_client(source, fdst2, flags=["--temp-dir", abs_scratch],
+ port=server.port)
+ assert result2.returncode != 0, "an absolute --temp-dir must be rejected (confined)"
+ assert os.listdir(abs_scratch) == [], "receiver wrote into an unconfined temp dir"
+
def test_default_behavior_has_no_scratch_dir(self, shared_server):
source = self._make_source("tempdir_default_src")
dest = os.path.join(TEST_DATA_DIR, "tempdir_default_dst")
@@ -2837,6 +2943,41 @@ class TestDelayUpdates:
assert not os.path.isdir(os.path.join(delay_dest, self.STAGING)), \
"staging directory left behind after a successful delayed transfer"
+ @pytest.mark.skipif(shutil.which("rsync") is None, reason="rsync not installed")
+ def test_delay_updates_staging_name_collision_residual(self):
+ """Documented residual (RSYNC_COMPAT.md `--delay-updates` row): FastSync
+ uses a fixed `.fastsync-stage` staging name and wipes a pre-existing tree
+ of that name at the start of a delayed run (crash-leftover cleanup),
+ even without `--delete`; rsync leaves a genuine destination entry of that
+ name untouched. Pins the divergence that keeps the row Divergent."""
+ source = self._make_source("delay_collide_src")
+ rdst = os.path.join(TEST_DATA_DIR, "delay_collide_rdst")
+ fdst = os.path.join(TEST_DATA_DIR, "delay_collide_fdst")
+ clean_dir(rdst)
+ clean_dir(fdst)
+ for root in (rdst, fdst):
+ with open(os.path.join(root, "top.txt"), "wb") as fh:
+ fh.write(b"old\n")
+ stage = os.path.join(root, self.STAGING)
+ os.makedirs(stage, exist_ok=True)
+ with open(os.path.join(stage, "keepme.txt"), "wb") as fh:
+ fh.write(b"genuine user data\n")
+
+ r = subprocess.run(["rsync", "-a", "--delay-updates", source + "/", rdst + "/"],
+ capture_output=True, text=True,
+ env=dict(os.environ, LC_ALL="C"))
+ assert r.returncode == 0, r.stderr
+ assert os.path.exists(os.path.join(rdst, self.STAGING, "keepme.txt")), \
+ "rsync removed an unrelated destination entry named like the staging dir"
+
+ with ServerManager() as server:
+ server.start()
+ result, _ = run_client(source, fdst, flags=["--delay-updates"],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ assert not os.path.exists(os.path.join(fdst, self.STAGING)), \
+ "FastSync did not wipe the reserved staging name (residual changed)"
+
@pytest.mark.parametrize("mt", [False, True])
def test_delay_updates_incremental_rerun_no_leftovers(self, shared_server, mt):
source = self._make_source("delay_rerun_src")
@@ -3523,23 +3664,25 @@ class TestMissingArgs:
class TestNoImpliedDirs:
- """--no-implied-dirs (only meaningful with -R + --files-from) refuses to
- place a listed file whose parent directory is not itself listed."""
+ """--no-implied-dirs (meaningful with -R) omits the source metadata of a
+ listed path's implied parent directories but still creates those parents
+ with default attributes, matching rsync 3.4.1."""
def _make(self):
return _make_relative_source("noimplied_src")
@pytest.mark.parametrize("mt", [False, True])
- def test_implied_dir_only_fails_entry(self, shared_server, mt):
+ def test_implied_dir_created_with_default_attrs(self, shared_server, mt):
source = self._make()
dest = os.path.join(TEST_DATA_DIR, "noimplied_dst")
clean_dir(dest)
lst = _write_rel_list(b"a/b.txt\n") # "a" itself is not listed
flags = ["--files-from", lst, "-R", "--no-implied-dirs"] + (["--threads"] if mt else [])
result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
- assert result.returncode != 0, "implied parent directory was not rejected"
- assert "--no-implied-dirs" in (result.stderr or result.stdout)
- assert not os.path.exists(os.path.join(dest, "a", "b.txt"))
+ assert result.returncode == 0, \
+ f"implied parent directory was not created: {result.stderr[:200]}"
+ assert os.path.isdir(os.path.join(dest, "a")), "implied parent 'a' was not created"
+ assert _read_file(os.path.join(dest, "a", "b.txt")) == b"nested\n"
@pytest.mark.parametrize("mt", [False, True])
def test_listed_dir_allows_file(self, shared_server, mt):
@@ -3666,9 +3809,10 @@ class TestDirs:
files.extend(os.path.relpath(os.path.join(root, n), mirror) for n in names)
assert files == [], f"--dirs descended into contents: {files}"
- def test_dirs_listed_dir_colliding_with_file_fails(self, shared_server):
- """A listed directory that already exists as a regular file at the
- destination fails the transfer cleanly instead of clobbering the file."""
+ def test_dirs_listed_dir_replaces_blocking_file(self, shared_server):
+ """rsync parity: a listed directory replaces a regular file already at
+ its destination path (rsync removes the non-directory and creates the
+ directory)."""
source = self._make()
dest = os.path.join(TEST_DATA_DIR, "dirs_coll_dst")
clean_dir(dest)
@@ -3678,8 +3822,10 @@ class TestDirs:
lst = _write_rel_list(b"dir1\n")
result, _ = run_client(source, dest, flags=["--files-from", lst, "--dirs", "-R"],
port=shared_server.port)
- assert result.returncode != 0, "dir entry over an existing file did not fail"
- assert os.path.isfile(blocker), "blocking regular file was clobbered"
+ assert result.returncode == 0, \
+ f"dir entry over an existing file failed: {(result.stderr or result.stdout)[:300]}"
+ assert os.path.isdir(blocker) and not os.path.islink(blocker), \
+ "blocking regular file was not replaced by the incoming directory"
class TestMkpath:
@@ -3836,15 +3982,16 @@ class TestDeleteTiming:
assert _read_file(os.path.join(received, "sub", "deep.txt")) == b"deeply nested file\n", \
f"{flag}: nested file was not written after the early deletion"
- @pytest.mark.parametrize("flag", ["--delete", "--delete-after"])
+ @pytest.mark.parametrize("flag", ["--delete-commit", "--delete-after"])
@pytest.mark.parametrize("mt", [False, True])
def test_late_flags_commit_only_after_success(self, flag, mt):
- """Plain --delete/--delete-after defer deletion until the whole transfer
+ """--delete-commit/--delete-after defer deletion until the whole transfer
succeeds: a mid-transfer write failure must leave every extra in place
- (commit-style safety). The -m receiver must also keep the extras: the
- deferred keep-set is committed by the server only after the disk-writer
- thread has finished, and a failing writer means the manifest is freed,
- never applied."""
+ (commit-style safety). Plain --delete no longer defers (it defaults to
+ delete-during), so only the explicitly late timings are exercised here.
+ The --threads receiver must also keep the extras: the deferred keep-set is
+ committed by the server only after the disk-writer thread has finished,
+ and a failing writer means the manifest is freed, never applied."""
source = self._seed("late")
dest = os.path.join(TEST_DATA_DIR, "deltiming_late_dst")
clean_dir(dest)
@@ -4463,11 +4610,11 @@ class TestDeletePolicy:
@pytest.mark.parametrize("mt", [False, True])
@pytest.mark.setpriv
def test_ignore_errors_keeps_deletion_active_on_scan_error(self, mt):
- """A source I/O error (unreadable subdirectory) aborts the run so no
- deletion happens by default; --ignore-errors continues, still transfers
- the readable tree and still deletes, single-threaded and under -m. Run
- as an unprivileged user so the mode-000 directory is genuinely
- unreadable."""
+ """rsync's --ignore-errors semantics: a source I/O error (unreadable
+ subdirectory) makes the run continue and transfer the readable tree, but
+ the default suppresses deletion ("IO error encountered -- skipping file
+ deletion"); --ignore-errors lets deletion proceed. Both exit 23. Run as
+ an unprivileged user so the mode-000 directory is genuinely unreadable."""
if os.geteuid() != 0 or shutil.which("setpriv") is None:
pytest.skip("requires root + setpriv to drop privileges for the client")
tag = f"ioerr_{os.getpid()}_{mt}"
@@ -4487,11 +4634,15 @@ class TestDeletePolicy:
try:
os.chmod(os.path.join(source, "locked"), 0)
- # Default: scan error aborts the run; nothing is deleted.
+ # Default: the scan continues past the unreadable dir and the
+ # readable tree transfers, but deletion is skipped (exit 23).
self._write(os.path.join(received, "extra.txt"), b"extra\n")
flags = ["--delete"] + (["--threads"] if mt else [])
result = self._run_client_as_nobody(source, dest, server.port, flags)
- assert result.returncode != 0, "unreadable source dir did not fail the run"
+ assert result.returncode == 23, \
+ f"unreadable source dir should exit 23 (got {result.returncode})"
+ assert os.path.exists(os.path.join(received, "top.txt")), \
+ "readable tree did not transfer past the I/O error"
assert os.path.exists(os.path.join(received, "extra.txt")), \
"default run deleted although the scan hit an I/O error"
@@ -4499,6 +4650,8 @@ class TestDeletePolicy:
self._write(os.path.join(received, "extra.txt"), b"extra\n")
flags = ["--delete", "--ignore-errors"] + (["--threads"] if mt else [])
result = self._run_client_as_nobody(source, dest, server.port, flags)
+ assert result.returncode == 23, \
+ f"--ignore-errors run should still exit 23 (got {result.returncode})"
assert not os.path.exists(os.path.join(received, "extra.txt")), \
f"--ignore-errors did not keep deletion active: {result.stderr[:300]}"
assert not os.path.exists(os.path.join(received, "locked")), \
@@ -4543,11 +4696,11 @@ class TestDeletePolicy:
finally:
os.chmod(source, 0o755)
- def test_delete_excluded_protection_is_sender_derived(self):
+ def test_delete_protection_reapplied_on_receiver(self):
"""Plain --delete protects destination mirrors of files the SOURCE scan
- excluded, but a destination-only file that merely matches an exclude
- rule is still an extra and is removed (protection never re-applies rules
- to the destination)."""
+ excluded, and (track 4a) also protects a destination-only file matching
+ an exclude rule because the compiled rule set is re-applied on the
+ receiver, matching rsync."""
source = os.path.join(TEST_DATA_DIR, "senderderived_src")
clean_dir(source)
self._write(os.path.join(source, "keep.txt"), b"kept\n")
@@ -4567,8 +4720,8 @@ class TestDeletePolicy:
f"delete sync failed: {(result.stderr or result.stdout)[:300]}"
assert os.path.exists(os.path.join(received, "secret.log")), \
"source-excluded mirror was deleted under plain --delete"
- assert not os.path.exists(os.path.join(received, "stray.log")), \
- "destination-only file matching the exclude rule was left (should be deleted)"
+ assert os.path.exists(os.path.join(received, "stray.log")), \
+ "destination-only file matching the exclude rule must be protected like rsync"
def _pin_mtime(path, ts):
@@ -4588,11 +4741,12 @@ class TestBasisDestDirs:
STAGING = ".fastsync-stage"
TS = 1577836800 # 2020-01-01 00:00:00 UTC, used to pin matching mtimes
- # fixture files: source and basis share the mtime pin, so a basis "match"
- # is decided purely by content (xxHash). unchanged.txt is byte-identical;
- # changed.txt is byte-DIFFERENT but has the SAME SIZE as the source (and
- # the same pinned mtime), which is what forces the content-hash gate;
- # added.txt does not exist in the basis at all.
+ # fixture files: source and basis share the mtime pin, so the DEFAULT
+ # (rsync-parity) quick-check is a size+mtime match and trusts the basis even
+ # when the body differs. unchanged.txt is byte-identical; changed.txt is
+ # byte-DIFFERENT but has the SAME SIZE as the source (and the same pinned
+ # mtime), which is what the FastSync-only --verify-basis content gate
+ # rejects; added.txt does not exist in the basis at all.
UNCHANGED = "unchanged.txt"
CHANGED = "changed.txt"
ADDED = "added.txt"
@@ -4632,18 +4786,19 @@ class TestBasisDestDirs:
}
def _basis_tree(self, prefix):
- # unchanged.txt is identical to the source; changed.txt has the SAME
- # byte size and pinned mtime but a different body (equal size forces
- # the xxHash gate); added.txt is missing from the basis.
+ # unchanged.txt is identical to the source; changed.txt has a DIFFERENT
+ # size (and body) so the size leg of the quick-check fails and it is
+ # transferred normally; added.txt is missing from the basis.
return {
self.UNCHANGED: b"stable content v1\n",
- self.CHANGED: b"CHANGED CONTENT NOW\n",
+ self.CHANGED: b"CHANGED CONTENT NOW AND LONGER\n",
}
- def test_same_size_different_content_is_not_a_basis_match(self, shared_server):
- # Core safety property: equal size + pinned mtime but different content
- # must NEVER be hard-linked or copied from the basis -- the xxHash gate
- # rejects it and the sender's data is transferred instead.
+ def test_same_size_different_content_default_trusts_quick_check(self, shared_server):
+ # Default rsync-parity behavior: equal size + pinned mtime is a basis
+ # match, so the basis body is materialized/linked without reading it.
+ # This mirrors rsync 3.4.1's quick check (differential-tested in
+ # test_differential_parity.py::test_verify_basis_restores_strict_content).
for flag, basis_dir in (("--link-dest", "szlb"), ("--copy-dest", "szcp"),
("--compare-dest", "szcmp")):
source = self._make_source("basis_same_size_src",
@@ -4655,7 +4810,36 @@ class TestBasisDestDirs:
result, _ = run_client(source, dest, flags=[f"{flag}={basis_dir}"],
port=shared_server.port)
assert result.returncode == 0, \
- f"{flag} same-size mismatch failed: {result.stderr[:300]}"
+ f"{flag} same-size quick-check failed: {result.stderr[:300]}"
+ received = get_dest_received_dir(dest, source)
+ dest_file = os.path.join(received, self.UNCHANGED)
+ if flag == "--compare-dest":
+ assert not os.path.exists(dest_file), \
+ f"{flag}: compare-dest must leave a matching file sparse"
+ else:
+ assert _read_file(dest_file) == b"SAME LENGTH BODY!", \
+ f"{flag}: default quick-check did not trust the basis body"
+ if flag == "--link-dest":
+ assert os.stat(dest_file).st_ino == os.stat(basis_file).st_ino, \
+ f"{flag}: basis was not hard-linked"
+
+ def test_verify_basis_rejects_same_size_different_content(self, shared_server):
+ # FastSync-only --verify-basis: the whole-file digest gate rejects the
+ # same-size/different-content basis, so the source data is transferred
+ # instead of the wrong basis bytes.
+ for flag, basis_dir in (("--link-dest", "vszlb"), ("--copy-dest", "vszcp"),
+ ("--compare-dest", "vszcmp")):
+ source = self._make_source("basis_verify_src",
+ {self.UNCHANGED: b"same length body\n"})
+ dest = os.path.join(TEST_DATA_DIR, f"basis_verify_dst_{basis_dir}")
+ clean_dir(dest)
+ basis_file = self._seed_basis_file(dest, source, basis_dir, self.UNCHANGED,
+ b"SAME LENGTH BODY!")
+ result, _ = run_client(source, dest,
+ flags=[f"{flag}={basis_dir}", "--verify-basis"],
+ port=shared_server.port)
+ assert result.returncode == 0, \
+ f"{flag} --verify-basis failed: {result.stderr[:300]}"
received = get_dest_received_dir(dest, source)
dest_file = os.path.join(received, self.UNCHANGED)
assert _read_file(dest_file) == b"same length body\n", \
@@ -4687,11 +4871,13 @@ class TestBasisDestDirs:
self._source_tree("c")[self.ADDED], "added file not transferred"
@pytest.mark.ci
- def test_dry_run_compare_dest_does_not_read_basis(self, shared_server):
- # A dry-run --compare-dest must never read/hash the basis file: doing so
- # is a 1-bit content oracle against the client-supplied digest. Even a
- # byte-identical basis with a matching size+mtime is therefore reported
- # as would-transfer, and nothing is created.
+ def test_dry_run_compare_dest_quick_check_does_not_read_basis(self, shared_server):
+ # A dry-run --compare-dest must never read/hash the basis file. Under
+ # the default metadata quick-check a matching basis is reported as a
+ # skip (matching rsync) without reading it; nothing is created. Under
+ # --verify-basis, which would require hashing, the dry-run cannot
+ # confirm the hit (that would be a 1-bit content oracle) and reports
+ # would-transfer instead.
source = self._make_source("basis_dry_src", {self.UNCHANGED: b"stable content v1\n"})
dest = os.path.join(TEST_DATA_DIR, "basis_dry_dst")
clean_dir(dest)
@@ -4702,11 +4888,30 @@ class TestBasisDestDirs:
port=shared_server.port)
assert result.returncode == 0, \
f"dry-run compare-dest failed: {result.stderr[:300]}"
- assert self.UNCHANGED in result.stdout, (
- "dry-run compare-dest silently skipped: receiver read the basis content"
+ assert self.UNCHANGED not in result.stdout, (
+ "dry-run compare-dest did not honor the metadata quick-check "
+ "(reported would-transfer for a matching basis)"
)
assert _snapshot_tree(dest) == before, "dry-run compare-dest mutated the destination"
+ # --verify-basis: the hit needs the basis content, which a dry-run must
+ # not read, so the file is reported as would-transfer.
+ dest2 = os.path.join(TEST_DATA_DIR, "basis_dry_verify_dst")
+ clean_dir(dest2)
+ self._seed_basis(dest2, source, "drybasis", {self.UNCHANGED: b"stable content v1\n"})
+ before2 = _snapshot_tree(dest2)
+ result, _ = run_client(source, dest2,
+ flags=["--compare-dest=drybasis", "--dry-run",
+ "--verify-basis"],
+ port=shared_server.port)
+ assert result.returncode == 0, \
+ f"dry-run --verify-basis compare-dest failed: {result.stderr[:300]}"
+ assert self.UNCHANGED in result.stdout, (
+ "dry-run --verify-basis must not read the basis to confirm a hit"
+ )
+ assert _snapshot_tree(dest2) == before2, \
+ "dry-run --verify-basis compare-dest mutated the destination"
+
def test_compare_dest_content_mismatch_forces_transfer(self, shared_server):
# The basis holds a file with a DIFFERENT body: even though it shares
# the mtime pin, the xxHash check fails and the data must be sent.
@@ -4945,27 +5150,36 @@ class TestBasisDestDirs:
assert os.stat(dest_file).st_ino != os.stat(basis_file).st_ino, \
"--ignore-times must not hard-link to a basis file"
- def test_basis_refuses_file_above_whole_file_limit(self, shared_server):
- # Every whole-file payload path in FastSync (basis dirs included) is
- # bounded by MAX_RECEIVE_WHOLE_FILE_SIZE. rsync supports basis dirs for
- # arbitrary sizes; FastSync refuses such a run up front with a clear
- # diagnostic instead of letting the receiver abort the whole transfer
- # mid-stream with no client-side explanation.
+ def test_basis_handles_file_above_whole_file_limit(self, shared_server):
+ # Track 5a: a basis hit streams the copy (and the --verify-basis digest
+ # streams the basis), so a source larger than the whole-file payload
+ # bound is supported for basis dirs exactly like rsync. A basis MISS
+ # still falls back to the normal transfer, which keeps its own bound.
source = self._make_source("basis_oversize_src", {"small.txt": b"ok\n"})
big = os.path.join(source, "huge.bin")
with open(big, "wb") as fh:
os.ftruncate(fh.fileno(), 256 * 1024 * 1024 + 4096)
dest = os.path.join(TEST_DATA_DIR, "basis_oversize_dst")
clean_dir(dest)
- result, _ = run_client(source, dest, flags=["--link-dest=nope"],
- port=shared_server.port)
- assert result.returncode != 0, \
- "basis run with an over-limit file unexpectedly succeeded"
- assert "larger than" in result.stderr, \
- f"no clear over-limit diagnostic: {result.stderr[:300]}"
received = get_dest_received_dir(dest, source)
- assert not os.path.exists(received), \
- "over-limit basis run transferred files before failing"
+ rel = os.path.relpath(received, dest)
+ basis_big = os.path.join(dest, "ob", rel, "huge.bin")
+ os.makedirs(os.path.dirname(basis_big), exist_ok=True)
+ shutil.copyfile(big, basis_big)
+ os.utime(basis_big, (self.TS, self.TS))
+ os.utime(big, (self.TS, self.TS))
+
+ result, _ = run_client(source, dest,
+ flags=["--link-dest=ob", "--incremental"],
+ port=shared_server.port)
+ assert result.returncode == 0, \
+ f"over-limit basis run failed: {result.stderr[:300]}"
+ dest_big = os.path.join(received, "huge.bin")
+ assert os.path.exists(dest_big), "over-limit basis hit was not materialized"
+ assert os.path.getsize(dest_big) == 256 * 1024 * 1024 + 4096
+ assert os.stat(dest_big).st_ino == os.stat(basis_big).st_ino, \
+ "over-limit --link-dest did not hard-link to the basis"
+ assert _read_file(os.path.join(received, "small.txt")) == b"ok\n"
def _random_payloads(size=2 * 1024 * 1024, changed=64 * 1024, seed=1234):
@@ -6690,11 +6904,10 @@ class TestDirectoryAndSymlinkTimes:
@pytest.mark.ci
@pytest.mark.parametrize("mt", [False, True])
- def test_preserve_does_not_create_empty_source_dir(self, shared_server, mt):
- """P7 Wave D #1: a captured-but-EMPTY source directory is never created
- at the destination. The scanner records its time (it is transmitted via
- STATUS_DIR_TIMES), but the receiver treats that entry as record-only, so
- `-a` keeps the documented "empty dirs are never transferred" behavior."""
+ def test_preserve_creates_empty_source_dir(self, shared_server, mt):
+ """rsync parity: a recursive `-a` transfer recreates an empty source
+ directory at the destination (the scanner emits it as an explicit
+ directory entry)."""
source = os.path.join(TEST_DATA_DIR, f"empty_dir_{'m' if mt else 's'}_src")
dest = os.path.join(TEST_DATA_DIR, f"empty_dir_{'m' if mt else 's'}_dst")
clean_dir(source)
@@ -6705,8 +6918,8 @@ class TestDirectoryAndSymlinkTimes:
flags = ["-a"] + (["--threads"] if mt else [])
received = self._run(source, dest, flags, shared_server)
assert os.path.isfile(os.path.join(received, "keep.txt")), "regular file missing"
- assert not os.path.lexists(os.path.join(received, "empty_sub")), \
- f"-a created an empty source directory at {received}/empty_sub"
+ assert os.path.isdir(os.path.join(received, "empty_sub")), \
+ f"-a did not recreate the empty source directory at {received}/empty_sub"
@pytest.mark.ci
@pytest.mark.parametrize("mt", [False, True])
@@ -6729,10 +6942,10 @@ class TestDirectoryAndSymlinkTimes:
@pytest.mark.ci
@pytest.mark.parametrize("mt", [False, True])
- def test_collision_at_dir_time_path_does_not_abort(self, shared_server, mt):
- """P7 Wave D #1: a pre-existing regular file at a source-empty-dir's
- mirror path must not abort the transfer (the old mkdir failed and failed
- the run) and must not be clobbered."""
+ def test_collision_at_empty_dir_path_replaces_blocker(self, shared_server, mt):
+ """rsync parity: a pre-existing regular file at a source empty-dir's
+ mirror path is replaced by the incoming directory (rsync removes the
+ non-directory and creates the directory); the run succeeds."""
source = os.path.join(TEST_DATA_DIR, f"dirtime_collide_{'m' if mt else 's'}_src")
dest = os.path.join(TEST_DATA_DIR, f"dirtime_collide_{'m' if mt else 's'}_dst")
clean_dir(source)
@@ -6751,10 +6964,8 @@ class TestDirectoryAndSymlinkTimes:
assert result.returncode == 0, \
f"-a aborted on a pre-existing file at an empty-dir path: " \
f"{(result.stderr or result.stdout)[:400]}"
- assert os.path.isfile(blocker) and not os.path.islink(blocker), \
- "the pre-existing blocker was replaced by a directory"
- with open(blocker, "rb") as fh:
- assert fh.read() == b"pre-existing blocker\n", "the blocker file was clobbered"
+ assert os.path.isdir(blocker) and not os.path.islink(blocker), \
+ "the pre-existing blocker was not replaced by the incoming directory"
assert os.path.isfile(os.path.join(received, "keep.txt")), "regular file missing"
diff --git a/tests/integration/test_iconv.py b/tests/integration/test_iconv.py
index 4a69aa1..fe7522f 100644
--- a/tests/integration/test_iconv.py
+++ b/tests/integration/test_iconv.py
@@ -1,10 +1,12 @@
"""--iconv=CONVERT_SPEC file-NAME charset conversion integration tests.
-The client converts every source file name from LOCAL to REMOTE before it goes
-on the wire, and the receiver converts it back from REMOTE to LOCAL, so a
-source tree using one charset can be written into a destination tree using
-another (rsync compatibility; content bytes are never touched).
+rsync's spec is ``--iconv=LOCAL,REMOTE`` (the order is the same push or pull).
+The sender converts each source name from LOCAL to REMOTE for the wire, and on
+a PUSH the receiver's charset is the spec's REMOTE half, so it writes the wire
+bytes verbatim (only a server with its own ``--iconv`` declares a different
+destination charset and re-converts). Content bytes are never touched.
"""
+import codecs
import os
import shutil
@@ -16,6 +18,11 @@ LATIN1_NAME = b"caf\xe9.txt"
UTF8_NAME = "caf\u00e9.txt".encode("utf-8")
+def _to_utf8(name_bytes):
+ """The UTF-8 encoding of a name that is stored as ISO-8859-1 bytes."""
+ return codecs.encode(codecs.decode(name_bytes, "iso-8859-1"), "utf-8")
+
+
def _make(tag):
source = os.path.join(TEST_DATA_DIR, f"iconv_{tag}_src")
dest = os.path.join(TEST_DATA_DIR, f"iconv_{tag}_dst")
@@ -41,10 +48,10 @@ def _dest_file(source, dest, name):
@pytest.mark.ci
-def test_iconv_latin1_roundtrip(shared_server):
- """A source file whose name is ISO-8859-1 bytes is transferred with
- --iconv=iso-8859-1,utf-8 and lands on the destination with the ORIGINAL
- latin1 name (the wire carried it as UTF-8)."""
+def test_iconv_latin1_to_utf8_dest(shared_server):
+ """rsync push parity: --iconv=iso-8859-1,utf-8 converts a latin1 source name
+ to the spec's REMOTE (UTF-8) on the wire and the default receiver writes it
+ verbatim, so the destination name is UTF-8 (not the source's latin1)."""
source, dest = _make("latin1")
_place_bytes(source, LATIN1_NAME)
@@ -53,8 +60,10 @@ def test_iconv_latin1_roundtrip(shared_server):
)
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
- dst = _dest_file(source, dest, LATIN1_NAME)
- assert os.path.exists(dst), f"dest latin1-named file not found under {dest}"
+ dst = _dest_file(source, dest, UTF8_NAME)
+ assert os.path.exists(dst), f"dest UTF-8-named file not found under {dest}"
+ assert not os.path.exists(_dest_file(source, dest, LATIN1_NAME)), \
+ "destination kept the latin1 name instead of the wire (UTF-8) charset"
@pytest.mark.ci
@@ -153,7 +162,7 @@ def test_iconv_expanding_name_growth(shared_server):
)
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
- assert os.path.exists(_dest_file(source, dest, name_bytes))
+ assert os.path.exists(_dest_file(source, dest, _to_utf8(name_bytes)))
def test_iconv_symlink_path_and_target(shared_server):
@@ -170,11 +179,13 @@ def test_iconv_symlink_path_and_target(shared_server):
)
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
- dst_target = _dest_file(source, dest, target)
- dst_link = _dest_file(source, dest, b"link\xe9")
- assert os.path.exists(dst_target), "dest latin1 target file missing"
- assert os.path.islink(dst_link), "dest latin1 symlink missing"
- assert os.readlink(dst_link) == target, "symlink target not preserved/decoded"
+ utf8_target = _to_utf8(target)
+ utf8_link = _to_utf8(b"link\xe9")
+ dst_target = _dest_file(source, dest, utf8_target)
+ dst_link = _dest_file(source, dest, utf8_link)
+ assert os.path.exists(dst_target), "dest UTF-8 target file missing"
+ assert os.path.islink(dst_link), "dest UTF-8 symlink missing"
+ assert os.readlink(dst_link) == utf8_target, "symlink target not wire-converted"
with open(dst_link, "rb") as fh:
assert fh.read() == b"t\n"
@@ -198,8 +209,8 @@ def test_iconv_hardlink_path_and_target(shared_server):
)
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
- dst_a = _dest_file(source, dest, a)
- dst_b = _dest_file(source, dest, b)
+ dst_a = _dest_file(source, dest, _to_utf8(a))
+ dst_b = _dest_file(source, dest, _to_utf8(b))
assert os.path.exists(dst_a) and os.path.exists(dst_b)
assert os.stat(dst_a).st_ino == os.stat(dst_b).st_ino, \
"hard-link relationship not preserved across the transfer"
@@ -221,16 +232,16 @@ def test_iconv_delete_manifest_consistent(shared_server):
flags = ["--iconv=iso-8859-1,utf-8"]
result, _ = run_client(source, dest, flags=flags, port=server.port)
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
- assert os.path.exists(_dest_file(source, dest, keep))
- assert os.path.exists(_dest_file(source, dest, gone))
+ assert os.path.exists(_dest_file(source, dest, _to_utf8(keep)))
+ assert os.path.exists(_dest_file(source, dest, _to_utf8(gone)))
os.remove(os.path.join(os.fsencode(source), gone))
result, _ = run_client(
source, dest, flags=flags + ["--delete"], port=server.port
)
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
- assert os.path.exists(_dest_file(source, dest, keep)), "kept file deleted"
- assert not os.path.exists(_dest_file(source, dest, gone)), \
+ assert os.path.exists(_dest_file(source, dest, _to_utf8(keep))), "kept file deleted"
+ assert not os.path.exists(_dest_file(source, dest, _to_utf8(gone))), \
"missing file was not deleted"
@@ -247,4 +258,4 @@ def test_iconv_chunk_serialization_blob(shared_server):
)
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
- assert os.path.exists(_dest_file(source, dest, name))
\ No newline at end of file
+ assert os.path.exists(_dest_file(source, dest, _to_utf8(name)))
\ No newline at end of file
diff --git a/tests/integration/test_option_parity.py b/tests/integration/test_option_parity.py
new file mode 100644
index 0000000..52a2540
--- /dev/null
+++ b/tests/integration/test_option_parity.py
@@ -0,0 +1,538 @@
+"""Differential parity tests for the option wave (bwlimit, --info=*, -M,
+--ignore-errors, --filter protect).
+
+Every differential here runs the SAME scenario with real ``rsync 3.4.1`` and
+with fastsync and compares the observable result, so the modules are skipped
+when rsync is unavailable. The privilege-dependent --ignore-errors differential
+drops the client to an unprivileged uid so a mode-000 source directory is
+genuinely unreadable; it is marked ``setpriv`` (run as root locally, excluded
+from the root PR gate exactly like the other privilege tests).
+"""
+import os
+import shutil
+import subprocess
+import sys
+import time
+
+import pytest
+
+sys.path.insert(0, os.path.dirname(__file__))
+from common import ( # noqa: E402
+ CLIENT_CMD,
+ TEST_DATA_DIR,
+ ServerManager,
+ clean_dir,
+ get_dest_received_dir,
+ run_client,
+)
+
+RSYNC = shutil.which("rsync")
+requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
+
+
+def _rsync(args, timeout=120, as_nobody=False):
+ env = dict(os.environ, LC_ALL="C")
+ cmd = [RSYNC] + args
+ if as_nobody:
+ cmd = ["setpriv", "--reuid=65534", "--regid=65534", "--clear-groups"] + cmd
+ return subprocess.run(cmd, capture_output=True, text=True, env=env, timeout=timeout)
+
+
+def _write(path, content):
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ with open(path, "wb") as fh:
+ fh.write(content)
+
+
+class TestBwlimitParity:
+ """--bwlimit must accept rsync 3.4.1's spellings and pace like it."""
+
+ ACCEPTED = ["100", "0", "1.5", "100K", "100KB", "100KiB", "1M", "1MB", "1m", "1G", "512"]
+ REJECTED = ["-1", "abc", "1x", "1 000"]
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_parse_acceptance_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "bwp_src")
+ clean_dir(source)
+ _write(os.path.join(source, "f.txt"), b"payload\n")
+
+ for value in self.ACCEPTED + self.REJECTED:
+ rdst = os.path.join(TEST_DATA_DIR, "bwp_rdst")
+ clean_dir(rdst)
+ rsync_result = _rsync(["-a", "--bwlimit=" + value, source + "/", rdst + "/"])
+ dest = os.path.join(TEST_DATA_DIR, "bwp_dst")
+ clean_dir(dest)
+ result, _ = run_client(source, dest, flags=["-a", "--bwlimit=" + value],
+ port=shared_server.port)
+ assert (result.returncode == 0) == (rsync_result.returncode == 0), (
+ f"--bwlimit={value}: fastsync rc={result.returncode} "
+ f"({(result.stderr or result.stdout)[:120]!r}) "
+ f"rsync rc={rsync_result.returncode} ({rsync_result.stderr[:120]!r})"
+ )
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_throttle_rate_matches_rsync(self, shared_server):
+ """A 4 MiB transfer at --bwlimit=2048 (2 MiB/s) must take about the same
+ wall-clock time for both tools (~2 s with rsync's leaky bucket)."""
+ source = os.path.join(TEST_DATA_DIR, "bwt_src")
+ clean_dir(source)
+ _write(os.path.join(source, "big.bin"), os.urandom(4 * 1024 * 1024))
+ dest = os.path.join(TEST_DATA_DIR, "bwt_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "bwt_rdst")
+
+ clean_dir(rdst)
+ start = time.monotonic()
+ rsync_result = _rsync(["-a", "--bwlimit=2048", source + "/", rdst + "/"])
+ rsync_secs = time.monotonic() - start
+ assert rsync_result.returncode == 0, rsync_result.stderr
+
+ clean_dir(dest)
+ result, fast_secs = run_client(source, dest, flags=["-a", "--bwlimit=2048"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ # 4 MiB at 2 MiB/s rendezvous near 2 s. Use a coarse band on each side
+ # (an unthrottled transfer finishes well under 1.5 s) plus a generous
+ # cross-tolerance so a loaded CI runner cannot flake the parity assert.
+ lo, hi = 1.5, 4.5
+ assert lo <= fast_secs <= hi, f"fastsync throttle out of band: {fast_secs:.2f}s"
+ assert lo <= rsync_secs <= hi, f"rsync throttle out of band: {rsync_secs:.2f}s"
+ assert abs(fast_secs - rsync_secs) < 2.0, (
+ f"fastsync {fast_secs:.2f}s vs rsync {rsync_secs:.2f}s"
+ )
+
+
+def _output_tree(root):
+ clean_dir(root)
+ os.makedirs(os.path.join(root, "sub"))
+ _write(os.path.join(root, "a.txt"), b"top\n")
+ _write(os.path.join(root, "sub", "b.txt"), b"nested\n")
+ os.symlink("a.txt", os.path.join(root, "link"))
+
+
+class TestInfoParity:
+ """The --info categories that map to a FastSync event must print rsync's
+ line format."""
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_flist_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "inf_fl_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_fl_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_fl_rdst")
+ _output_tree(source)
+ clean_dir(dest)
+ clean_dir(rdst)
+ rsync_result = _rsync(["-a", "--info=flist", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(source, dest, flags=["-a", "--info=flist"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ assert "sending incremental file list" in result.stdout
+ assert "sending incremental file list" in rsync_result.stdout
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_name_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "inf_nm_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_nm_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_nm_rdst")
+ _output_tree(source)
+ clean_dir(dest)
+ clean_dir(rdst)
+ rsync_result = _rsync(["-a", "--info=name", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(source, dest, flags=["-a", "--info=name"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+
+ def entries(text):
+ # Compare the transferred entries only: rsync also prints the
+ # transfer-root `./` and every directory (FastSync records dirs),
+ # which are a separate documented divergence.
+ out = []
+ for line in text.splitlines():
+ if not line or line.startswith("sending ") or line.startswith("created "):
+ continue
+ if line == "./" or line.endswith("/"):
+ continue
+ out.append(line)
+ return sorted(out)
+
+ assert entries(result.stdout) == entries(rsync_result.stdout), (
+ f"rsync={entries(rsync_result.stdout)} fastsync={entries(result.stdout)}"
+ )
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_name_root_line_matches_rsync(self, shared_server):
+ """A fresh destination: rsync prints `created directory`, then the
+ transfer-root `./` name line before the entries; FastSync must emit the
+ same `./` line."""
+ source = os.path.join(TEST_DATA_DIR, "inf_root_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_root_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_root_rdst")
+ clean_dir(source)
+ _write(os.path.join(source, "f.bin"), b"payload\n")
+ clean_dir(dest)
+ shutil.rmtree(rdst, ignore_errors=True)
+ rsync_result = _rsync(["-a", "--info=name", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(source, dest, flags=["-a", "--info=name"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+
+ def names(text):
+ return [l for l in text.splitlines()
+ if l and not l.startswith("created directory")
+ and not (l.endswith("/") and l != "./")]
+
+ assert names(rsync_result.stdout) == ["./", "f.bin"], names(rsync_result.stdout)
+ assert names(result.stdout) == ["./", "f.bin"], names(result.stdout)
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_name2_uptodate_matches_rsync(self, shared_server):
+ """--info=name2 prints `NAME is uptodate` for entries the receiver
+ already has, matching rsync byte-for-byte."""
+ source = os.path.join(TEST_DATA_DIR, "inf_up_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_up_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_up_rdst")
+ clean_dir(source)
+ os.makedirs(os.path.join(source, "sub"))
+ _write(os.path.join(source, "a.txt"), b"a\n")
+ _write(os.path.join(source, "sub", "b.txt"), b"b\n")
+ clean_dir(rdst)
+ assert _rsync(["-a", source + "/", rdst + "/"]).returncode == 0
+ rsync_result = _rsync(["-a", "--info=name2", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+
+ clean_dir(dest)
+ seed, _ = run_client(source, dest, flags=["-a", "--incremental"],
+ port=shared_server.port)
+ assert seed.returncode == 0, (seed.stderr or seed.stdout)[:200]
+ result, _ = run_client(source, dest, flags=["-a", "--incremental", "--info=name2"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
+ if l.endswith("is uptodate"))
+ fast_lines = sorted(l for l in result.stdout.splitlines()
+ if l.endswith("is uptodate"))
+ assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
+ assert fast_lines == ["a.txt is uptodate", "sub/b.txt is uptodate"], fast_lines
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_nonreg_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "inf_nr_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_nr_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_nr_rdst")
+ clean_dir(source)
+ os.mkfifo(os.path.join(source, "fifo"))
+ _write(os.path.join(source, "a.txt"), b"a\n")
+ clean_dir(dest)
+ clean_dir(rdst)
+ rsync_result = _rsync(["-rlt", "--info=nonreg", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(source, dest, flags=["-rlt", "--info=nonreg"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
+ if l.startswith("skipping non-regular"))
+ fast_lines = sorted(l for l in result.stdout.splitlines()
+ if l.startswith("skipping non-regular"))
+ assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
+ assert fast_lines, "no non-regular skip line emitted"
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_del_real_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "inf_dl_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_dl_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_dl_rdst")
+ clean_dir(source)
+ _write(os.path.join(source, "keep.txt"), b"keep\n")
+ clean_dir(rdst)
+ _write(os.path.join(rdst, "extra.txt"), b"x\n")
+ _write(os.path.join(rdst, "extra2.txt"), b"y\n")
+ rsync_result = _rsync(["-a", "--delete", "--info=del", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
+ if l.startswith("deleting "))
+
+ clean_dir(dest)
+ received = get_dest_received_dir(dest, source)
+ _write(os.path.join(received, "extra.txt"), b"x\n")
+ _write(os.path.join(received, "extra2.txt"), b"y\n")
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(source, dest, flags=["-a", "--delete", "--info=del"],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ fast_lines = sorted(l for l in result.stdout.splitlines()
+ if l.startswith("deleting "))
+ assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
+ assert fast_lines, "no deletion lines emitted"
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_del_itemize_real_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "inf_di_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_di_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_di_rdst")
+ clean_dir(source)
+ _write(os.path.join(source, "keep.txt"), b"keep\n")
+ clean_dir(rdst)
+ _write(os.path.join(rdst, "extra.txt"), b"x\n")
+ rsync_result = _rsync(["-a", "-i", "--delete", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
+ if l.startswith("*deleting"))
+
+ clean_dir(dest)
+ received = get_dest_received_dir(dest, source)
+ _write(os.path.join(received, "extra.txt"), b"x\n")
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(source, dest, flags=["-a", "-i", "--delete"],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ fast_lines = sorted(l for l in result.stdout.splitlines()
+ if l.startswith("*deleting"))
+ assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_del_dry_run_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "inf_dd_src")
+ dest = os.path.join(TEST_DATA_DIR, "inf_dd_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "inf_dd_rdst")
+ clean_dir(source)
+ _write(os.path.join(source, "keep.txt"), b"keep\n")
+ clean_dir(rdst)
+ _write(os.path.join(rdst, "extra.txt"), b"x\n")
+ rsync_result = _rsync(["-a", "-n", "--delete", "--info=del", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
+ if l.startswith("deleting "))
+
+ clean_dir(dest)
+ received = get_dest_received_dir(dest, source)
+ _write(os.path.join(received, "extra.txt"), b"x\n")
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(source, dest, flags=["-a", "-n", "--delete", "--info=del"],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ fast_lines = sorted(l for l in result.stdout.splitlines()
+ if l.startswith("deleting "))
+ assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_info_remove_matches_rsync(self, shared_server):
+ tag = "inf_rm"
+ rsync_src = os.path.join(TEST_DATA_DIR, f"{tag}_rsrc")
+ rsync_dst = os.path.join(TEST_DATA_DIR, f"{tag}_rdst")
+ fast_src = os.path.join(TEST_DATA_DIR, f"{tag}_fsrc")
+ fast_dst = os.path.join(TEST_DATA_DIR, f"{tag}_fdst")
+ for root in (rsync_src, rsync_dst, fast_src, fast_dst):
+ clean_dir(root)
+ _write(os.path.join(rsync_src, "a.txt"), b"a\n")
+ _write(os.path.join(rsync_src, "sub", "b.txt"), b"b\n")
+ _write(os.path.join(fast_src, "a.txt"), b"a\n")
+ _write(os.path.join(fast_src, "sub", "b.txt"), b"b\n")
+
+ rsync_result = _rsync(["-a", "--remove-source-files", "--info=remove",
+ rsync_src + "/", rsync_dst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
+ if l.startswith("sender removed "))
+
+ result, _ = run_client(fast_src, fast_dst,
+ flags=["-a", "--remove-source-files", "--info=remove"],
+ port=shared_server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ fast_lines = sorted(l for l in result.stdout.splitlines()
+ if l.startswith("sender removed "))
+ assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
+ assert fast_lines, "no source-removal lines emitted"
+
+
+class TestIgnoreErrorsParity:
+ """--ignore-errors: a source I/O error skips deletion by default; the flag
+ lets deletion proceed. Both exit 23. Run the client as an unprivileged user
+ so the mode-000 directory is genuinely unreadable."""
+
+ @pytest.mark.setpriv
+ def test_delete_after_io_error_matches_rsync(self):
+ if os.geteuid() != 0 or shutil.which("setpriv") is None:
+ pytest.skip("requires root + setpriv to drop privileges for the client")
+ tag = f"ie_{os.getpid()}"
+ source = os.path.join(TEST_DATA_DIR, f"{tag}_src")
+ rsync_dst = os.path.join(TEST_DATA_DIR, f"{tag}_rdst")
+ dest = os.path.join(TEST_DATA_DIR, f"{tag}_dst")
+ clean_dir(source)
+ clean_dir(rsync_dst)
+ clean_dir(dest)
+ _write(os.path.join(source, "top.txt"), b"top\n")
+ _write(os.path.join(source, "locked", "blocked.txt"), b"blocked\n")
+ os.chmod(os.path.join(source, "locked"), 0)
+ os.chmod(TEST_DATA_DIR, 0o777)
+ os.chmod(source, 0o755)
+ os.chmod(rsync_dst, 0o777)
+ os.chmod(dest, 0o777)
+ try:
+ for ignore in (False, True):
+ flags = ["-a", "--delete-after"] + (["--ignore-errors"] if ignore else [])
+ # rsync side
+ _write(os.path.join(rsync_dst, "extra.txt"), b"x\n")
+ os.chmod(os.path.join(rsync_dst, "extra.txt"), 0o666)
+ rres = _rsync(flags + [source + "/", rsync_dst + "/"], as_nobody=True)
+ rsync_extra = os.path.exists(os.path.join(rsync_dst, "extra.txt"))
+
+ # fastsync side
+ received = get_dest_received_dir(dest, source)
+ _write(os.path.join(received, "extra.txt"), b"x\n")
+ os.chmod(os.path.join(received, "extra.txt"), 0o666)
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ fflags = (["--delete", "--ignore-errors"] if ignore else ["--delete"])
+ cmd = CLIENT_CMD + ["--source-dir", source, "--dest-dir", dest,
+ "--save-to-disk", "--server-port", str(server.port)] + fflags
+ fres = subprocess.run(
+ ["setpriv", "--reuid=65534", "--regid=65534", "--clear-groups"] + cmd,
+ text=True, capture_output=True)
+ fast_extra = os.path.exists(os.path.join(received, "extra.txt"))
+
+ assert rres.returncode == 23, (ignore, rres.returncode, rres.stderr[:200])
+ assert fres.returncode == 23, (ignore, fres.returncode, fres.stderr[:200])
+ assert rsync_extra == fast_extra, (
+ f"ignore_errors={ignore}: rsync extra={rsync_extra} fastsync extra={fast_extra}"
+ )
+ assert fast_extra is (not ignore), (ignore, fast_extra)
+ finally:
+ os.chmod(os.path.join(source, "locked"), 0o755)
+
+
+class TestRemoteOptionDaemon:
+ """rsync forwards -M/--remote-option to its remote process over a daemon
+ connection; FastSync's daemon has no per-connection argv channel and rejects
+ it. This pins the documented divergence with evidence."""
+
+ @requires_rsync
+ def test_rsync_forwards_M_over_daemon_and_fastsync_rejects(self, tmp_path):
+ import socket
+
+ with socket.socket() as probe:
+ probe.bind(("127.0.0.1", 0))
+ port = probe.getsockname()[1]
+
+ module_root = tmp_path / "mod"
+ module_root.mkdir()
+ os.chmod(module_root, 0o777)
+ source = tmp_path / "src"
+ source.mkdir()
+ (source / "a.txt").write_bytes(b"hello\n")
+ conf = tmp_path / "rsyncd.conf"
+ conf.write_text(
+ f"port = {port}\nuse chroot = no\n[m]\npath = {module_root}\nread only = no\n"
+ )
+ daemon = subprocess.Popen(
+ [RSYNC, "--daemon", "--no-detach", "--port", str(port), "--config", str(conf)],
+ stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
+ try:
+ deadline = time.monotonic() + 5
+ while time.monotonic() < deadline:
+ try:
+ with socket.create_connection(("127.0.0.1", port), timeout=0.3):
+ break
+ except OSError:
+ time.sleep(0.05)
+ else:
+ pytest.skip("rsync daemon did not start")
+
+ # A well-formed -M option is forwarded and accepted by the daemon...
+ ok = _rsync(["-a", "-M--safe-links", source.as_posix() + "/",
+ f"rsync://127.0.0.1:{port}/m/"])
+ # ...and a bogus one is rejected ON THE REMOTE with "unknown option",
+ # which proves the option reached the daemon's parser.
+ bogus = _rsync(["-a", "-M--totally-bogus", source.as_posix() + "/",
+ f"rsync://127.0.0.1:{port}/m/"])
+ assert bogus.returncode != 0
+ assert "unknown option" in (bogus.stderr + bogus.stdout), bogus.stderr
+ del ok
+ finally:
+ daemon.terminate()
+ try:
+ daemon.wait(timeout=5)
+ except subprocess.TimeoutExpired:
+ daemon.kill()
+
+ # FastSync rejects -M for a non-SSH transport up front.
+ dest = os.path.join(TEST_DATA_DIR, "ro_dst")
+ clean_dir(dest)
+ result, _ = run_client(source.as_posix(), dest, flags=["-a", "-M--safe-links"])
+ assert result.returncode != 0
+ assert "remote-option" in (result.stderr + result.stdout)
+
+
+class TestFilterProtect:
+ """Receiver-derived delete protection: a `protect`/`P` rule is compiled by
+ the sender and sent on the config frame, so the receiver shields a
+ destination-only entry that never appeared on the sender, matching rsync."""
+
+ @requires_rsync
+ @pytest.mark.ci
+ def test_protect_dest_only_matches_rsync(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "fpd_src")
+ dest = os.path.join(TEST_DATA_DIR, "fpd_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "fpd_rdst")
+ clean_dir(source)
+ _write(os.path.join(source, "keep.txt"), b"keep\n")
+ clean_dir(rdst)
+ _write(os.path.join(rdst, "extra.log"), b"extra\n")
+ _write(os.path.join(rdst, "other.txt"), b"other\n")
+
+ rsync_result = _rsync(["-a", "--delete", "--filter=P *.log", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ assert os.path.exists(os.path.join(rdst, "extra.log")), "rsync did not protect extra.log"
+ assert not os.path.exists(os.path.join(rdst, "other.txt")), "rsync did not delete other.txt"
+
+ clean_dir(dest)
+ received = get_dest_received_dir(dest, source)
+ _write(os.path.join(received, "extra.log"), b"extra\n")
+ _write(os.path.join(received, "other.txt"), b"other\n")
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(source, dest,
+ flags=["-a", "--delete", "--filter=P *.log"],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:200]
+ assert os.path.exists(os.path.join(received, "extra.log")), (
+ "FastSync must protect a destination-only P match like rsync")
+ assert not os.path.exists(os.path.join(received, "other.txt"))
+
+ @pytest.mark.ci
+ def test_protect_dest_only_dry_run_enumeration(self, shared_server):
+ source = os.path.join(TEST_DATA_DIR, "fpd_nd_src")
+ dest = os.path.join(TEST_DATA_DIR, "fpd_nd_dst")
+ clean_dir(source)
+ _write(os.path.join(source, "keep.txt"), b"keep\n")
+ received = get_dest_received_dir(dest, source)
+ clean_dir(received)
+ _write(os.path.join(received, "keep.txt"), b"keep\n")
+ _write(os.path.join(received, "extra.log"), b"extra\n")
+ _write(os.path.join(received, "other.txt"), b"other\n")
+ with ServerManager() as server:
+ server.start(extra_args=["--allow-delete"])
+ result, _ = run_client(source, dest,
+ flags=["-a", "-n", "--delete", "--out-format=%n",
+ "--filter=P *.log"],
+ port=server.port)
+ assert result.returncode == 0, (result.stderr or result.stdout)[:300]
+ assert "other.txt" in result.stdout, result.stdout
+ assert "extra.log" not in result.stdout, result.stdout
+ assert os.path.exists(os.path.join(received, "extra.log"))
+ assert os.path.exists(os.path.join(received, "other.txt"))
diff --git a/tests/integration/test_output_parity.py b/tests/integration/test_output_parity.py
index 81a4efb..8cae816 100644
--- a/tests/integration/test_output_parity.py
+++ b/tests/integration/test_output_parity.py
@@ -289,6 +289,49 @@ def _make_one_file(root, name="f.bin", size=100):
fh.write(bytes((i * 7 + 3) & 0xFF for i in range(size)))
+def _make_multidir_tree(root):
+ """Multi-directory corpus for the --progress file-list tests: nested files,
+ a directory-only branch, an empty directory and a symlink."""
+ clean_dir(root)
+ for rel, data in (("a.txt", b"alpha\n"), ("b.txt", b"bravo\n"),
+ ("sub1/c.txt", b"charlie\n"), ("sub1/deep/d.txt", b"delta\n"),
+ ("sub2/e.txt", b"echo\n")):
+ path = os.path.join(root, rel)
+ os.makedirs(os.path.dirname(path), exist_ok=True)
+ with open(path, "wb") as fh:
+ fh.write(data)
+ os.symlink("a.txt", os.path.join(root, "link1"))
+ os.makedirs(os.path.join(root, "emptydir"), exist_ok=True)
+
+
+def _parse_progress(text):
+ """Name lines and the `to-chk` denominators from a --progress run."""
+ names = []
+ totals = set()
+ for line in text.splitlines():
+ line = line.rstrip()
+ if not line or line == "sending incremental file list":
+ continue
+ if "%" in line:
+ match = re.search(r"to-chk=\d+/(\d+)", line)
+ if match:
+ totals.add(int(match.group(1)))
+ continue
+ if line == "./": # root-line trigger is a separate documented residual
+ continue
+ names.append(line)
+ return sorted(names), totals
+
+
+def _pick_stats(text, keys):
+ out = {}
+ for line in text.splitlines():
+ for key in keys:
+ if line.startswith(key + ":"):
+ out[key] = line
+ return out
+
+
class TestWireStatsParity:
"""Wire-counter output parity: --out-format %b/%c/%C, --progress and
--stats versus real rsync 3.4.1."""
@@ -447,6 +490,99 @@ class TestWireStatsParity:
assert fast_frames[0] == rsync_frames[0], (rsync_frames[0], fast_frames[0])
assert "(xfr#1," in fast_frames[-1], fast_frames[-1]
+ @requires_rsync
+ @pytest.mark.ci
+ def test_progress_leading_root_line_and_to_chk_match_rsync(self, shared_server):
+ """A single-file transfer: rsync emits the transfer-root `./` name line
+ and a `to-chk=0/2` denominator that counts that root entry. Both must
+ match FastSync byte-for-byte for the deterministic frames."""
+ source = os.path.join(TEST_DATA_DIR, "wire_pgroot_src")
+ dest = os.path.join(TEST_DATA_DIR, "wire_pgroot_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "wire_pgroot_rdst")
+ _make_one_file(source, "f.bin", 100)
+ clean_dir(dest)
+ # rsync prints the `./` root line only when the transfer root itself is
+ # created, so make the rsync destination absent. The "created directory"
+ # line it then emits has no FastSync counterpart (different mirror
+ # layout), so only the name/frame lines are compared.
+ shutil.rmtree(rdst, ignore_errors=True)
+ rsync_result = _rsync(["-a", "--progress", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(source, dest, flags=["-a", "--progress"],
+ port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+
+ # subprocess text mode normalizes \r to \n (universal newlines).
+ def lines_of(text):
+ return [ln for ln in text.splitlines() if ln and not ln.startswith("created directory")]
+
+ rsync_lines = lines_of(rsync_result.stdout)
+ fast_lines = lines_of(result.stdout)
+ rsync_names = [ln for ln in rsync_lines if "%" not in ln]
+ fast_names = [ln for ln in fast_lines if "%" not in ln]
+
+ assert rsync_names == ["sending incremental file list", "./", "f.bin"], rsync_names
+ assert fast_names == rsync_names, (rsync_names, fast_names)
+ # The final frame's to-chk denominator must include the source-root entry.
+ assert "to-chk=0/2" in fast_lines[-1], fast_lines[-1]
+ assert fast_lines[-1] == rsync_lines[-1], (rsync_lines[-1], fast_lines[-1])
+
+ @requires_rsync
+ @pytest.mark.ci
+ @pytest.mark.parametrize("mt", [False, True])
+ def test_progress_multidir_file_list_matches_rsync(self, shared_server, mt):
+ """A multi-directory tree: the paths-only pre-count must reproduce
+ rsync's file-list set and `to-chk` denominator. Per-directory name
+ lines are emitted for directories, symlinks and the empty directory; the
+ name set and the denominator (every entry plus the transfer root) match
+ rsync, while the emitted *order* remains a documented residual (rsync
+ sorts depth-first, FastSync streams in readdir/BFS order)."""
+ source = os.path.join(TEST_DATA_DIR, "wire_pgmd_src")
+ dest = os.path.join(TEST_DATA_DIR, "wire_pgmd_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "wire_pgmd_rdst")
+ _make_multidir_tree(source)
+ clean_dir(dest)
+ clean_dir(rdst)
+
+ rsync_result = _rsync(["-a", "--progress", source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ flags = ["-a", "--progress"] + (["--threads"] if mt else [])
+ result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+
+ rsync_names, rsync_totals = _parse_progress(rsync_result.stdout)
+ fast_names, fast_totals = _parse_progress(result.stdout)
+ assert sorted(rsync_names) == [
+ "a.txt", "b.txt", "emptydir/", "link1 -> a.txt", "sub1/",
+ "sub1/c.txt", "sub1/deep/", "sub1/deep/d.txt", "sub2/", "sub2/e.txt",
+ ], rsync_names
+ assert fast_names == rsync_names, (rsync_names, fast_names)
+ # 10 entries + the transfer-root "." counted by rsync's file list.
+ assert rsync_totals == {11}, rsync_totals
+ assert fast_totals == rsync_totals, (rsync_totals, fast_totals)
+
+ @pytest.mark.ci
+ def test_progress_delete_during_reuses_pre_scan(self):
+ """--delete-during + --progress reuses the keep-set pre-scan instead of
+ walking the tree a second time: the file-list total and directory name
+ lines are identical to a plain --progress run."""
+ source = os.path.join(TEST_DATA_DIR, "wire_pgdel_src")
+ dest = os.path.join(TEST_DATA_DIR, "wire_pgdel_dst")
+ _make_multidir_tree(source)
+ clean_dir(dest)
+ server = ServerManager()
+ server.start(extra_args=["--allow-super", "--allow-delete"])
+ try:
+ result, _ = run_client(source, dest, flags=["-a", "--progress", "--delete-during"],
+ port=server.port)
+ finally:
+ server.stop()
+ assert result.returncode == 0, result.stderr[:300]
+ names, totals = _parse_progress(result.stdout)
+ assert totals == {11}, totals
+ assert "sub1/" in names and "sub1/deep/" in names and "emptydir/" in names, names
+ assert "link1 -> a.txt" in names, names
+
@requires_rsync
@pytest.mark.ci
@pytest.mark.parametrize("mt", [False, True])
@@ -459,6 +595,9 @@ class TestWireStatsParity:
_make_one_file(source, "f.bin", 6000)
clean_dir(dest)
clean_dir(rdst)
+ # Start both tools from the same state: rsync's destination root exists,
+ # so pre-create FastSync's mirrored logical root as well.
+ os.makedirs(get_dest_received_dir(dest, source), exist_ok=True)
rsync_result = _rsync(["-a", "--stats", source + "/", rdst + "/"])
assert rsync_result.returncode == 0, rsync_result.stderr
flags = ["-a", "--stats"] + (["--threads"] if mt else [])
@@ -488,24 +627,19 @@ class TestWireStatsParity:
@requires_rsync
@pytest.mark.ci
- def test_stats_file_count_breakdown_residual(self, shared_server):
- """Residual (row #3): rsync prints the `Number of files` and
- `Number of created files` lines with a per-type breakdown
- (`(reg: X, dir: Y, link: Z)`).
-
- FastSync cannot reproduce it from what the sender currently knows: the
- scanner does not put directory entries in the transfer list (directories
- are created implicitly), and without a per-entry destination-probe the
- sender cannot tell which entries the receiver newly created. So FastSync
- prints the bare transferred-entry count. This test pins the divergence
- explicitly -- the row must not be marked ✅.
- """
+ def test_stats_file_count_breakdown_matches_rsync(self, shared_server):
+ """`Number of files` and `Number of created files` both carry rsync's
+ per-type breakdown (protocol 2.28.0 reports the receiver-created
+ reg/dir/link/special split over STATUS_STATS)."""
source = os.path.join(TEST_DATA_DIR, "wire_stc_src")
dest = os.path.join(TEST_DATA_DIR, "wire_stc_dst")
rdst = os.path.join(TEST_DATA_DIR, "wire_stc_rdst")
_make_one_file(source, "f.bin", 6000)
clean_dir(dest)
clean_dir(rdst)
+ # Start both tools from the same state: rsync's destination root exists,
+ # so pre-create FastSync's mirrored logical root as well.
+ os.makedirs(get_dest_received_dir(dest, source), exist_ok=True)
rsync_result = _rsync(["-a", "--stats", source + "/", rdst + "/"])
assert rsync_result.returncode == 0, rsync_result.stderr
result, _ = run_client(source, dest, flags=["-a", "--stats"],
@@ -523,14 +657,104 @@ class TestWireStatsParity:
f_files = stats_line(result.stdout, "Number of files")
f_created = stats_line(result.stdout, "Number of created files")
- # rsync always carries the type breakdown (the source root counts as a
- # directory; the single regular file as reg).
assert re.match(r"Number of files: 2 \(reg: 1, dir: 1\)$", r_files), r_files
+ assert r_files == f_files, (r_files, f_files)
assert re.match(r"Number of created files: 1 \(reg: 1\)$", r_created), r_created
- # FastSync prints only the bare count: no directory accounting and no
- # per-entry "created" knowledge.
- assert re.fullmatch(r"Number of files: 1", f_files), f_files
- assert re.fullmatch(r"Number of created files: 1", f_created), f_created
+ assert f_created == r_created, (r_created, f_created)
+
+ @requires_rsync
+ @pytest.mark.ci
+ @pytest.mark.parametrize("mt", [False, True])
+ def test_stats_created_and_literal_fresh_update_delta(self, shared_server, mt):
+ """The receiver-observed counters must match rsync for the three
+ transfer shapes: a fresh create (created breakdown + whole-file literal),
+ an update (created == 0, whole-file literal), and a delta update (only
+ the literal delta fragments are counted, not the whole file)."""
+ source = os.path.join(TEST_DATA_DIR, "wire_stcd_src")
+ dest = os.path.join(TEST_DATA_DIR, "wire_stcd_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "wire_stcd_rdst")
+ clean_dir(source)
+ clean_dir(dest)
+ clean_dir(rdst)
+ os.makedirs(source, exist_ok=True)
+ os.makedirs(get_dest_received_dir(dest, source), exist_ok=True)
+ with open(os.path.join(source, "big.bin"), "wb") as fh:
+ fh.write(bytes(range(256)) * 4096) # 1 MiB
+ mt_flag = ["--threads"] if mt else []
+
+ def compare(tag):
+ # Pin the delta block size on both ends: rsync's adaptive block size
+ # would otherwise make the literal/matched split non-comparable.
+ rsync_result = _rsync(["-a", "--stats", "--no-whole-file", "-B8192",
+ source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(
+ source, dest,
+ flags=["-a", "--stats", "--incremental", "--delta", "-B8192"] + mt_flag,
+ port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+ keys = ("Number of created files", "Literal data", "Matched data",
+ "Total transferred file size")
+ r = _pick_stats(rsync_result.stdout, keys)
+ f = _pick_stats(result.stdout, keys)
+ assert r == f, f"{tag}: rsync={r} fastsync={f}"
+ return r
+
+ fresh = compare("fresh")
+ assert re.match(r"Number of created files: 1 \(reg: 1\)$",
+ fresh["Number of created files"]), fresh
+
+ # Update the source and re-run: the destination already exists.
+ sleep_mtime = os.path.getmtime(os.path.join(source, "big.bin")) + 2
+ with open(os.path.join(source, "big.bin"), "r+b") as fh:
+ fh.seek(100)
+ fh.write(b"XXXXXXXXXX")
+ os.utime(os.path.join(source, "big.bin"), (sleep_mtime, sleep_mtime))
+ update = compare("update")
+ assert update["Number of created files"] == "Number of created files: 0", update
+
+ # Second delta update: change bytes far apart, so rsync ships only the
+ # literal fragments and FastSync must report the same Literal data.
+ sleep_mtime = os.path.getmtime(os.path.join(source, "big.bin")) + 2
+ with open(os.path.join(source, "big.bin"), "r+b") as fh:
+ fh.seek(500000)
+ fh.write(b"YYYYYYYYYY")
+ os.utime(os.path.join(source, "big.bin"), (sleep_mtime, sleep_mtime))
+ delta = compare("delta")
+ assert delta["Number of created files"] == "Number of created files: 0", delta
+ lit = int(delta["Literal data"].split(":", 1)[1].strip().split()[0].replace(",", ""))
+ assert 0 < lit < 1024 * 1024, delta
+
+ @requires_rsync
+ @pytest.mark.ci
+ @pytest.mark.parametrize("choice", ["xxh128", "xxh64", "xxh3", "md5", "md4", "sha1", "none"])
+ def test_out_format_C_selected_algorithm_matches_rsync(self, shared_server, choice):
+ """`%C` must use the algorithm selected by --checksum-choice, not always
+ xxh128, and render it exactly like rsync (big-endian for the 64-bit
+ hashes, high-then-low for xxh128, standard hex for md5/md4/sha1)."""
+ source = os.path.join(TEST_DATA_DIR, f"wire_cc_{choice}_src")
+ dest = os.path.join(TEST_DATA_DIR, f"wire_cc_{choice}_dst")
+ rdst = os.path.join(TEST_DATA_DIR, f"wire_cc_{choice}_rdst")
+ _make_one_file(source, "f.bin", 200000)
+ clean_dir(dest)
+ clean_dir(rdst)
+ fmt = "%C %l %n"
+ rsync_result = _rsync(["-a", "--checksum-choice=" + choice,
+ "--out-format=" + fmt, source + "/", rdst + "/"])
+ assert rsync_result.returncode == 0, rsync_result.stderr
+ result, _ = run_client(source, dest,
+ flags=["-a", "--checksum-choice=" + choice,
+ "--out-format=" + fmt],
+ port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+
+ def file_lines(text):
+ return [line for line in text.splitlines()
+ if line and not line.rsplit(" ", 1)[-1].endswith("/")]
+
+ assert file_lines(result.stdout) == file_lines(rsync_result.stdout), (
+ f"choice={choice}: rsync={rsync_result.stdout!r} fastsync={result.stdout!r}"
+ )
@requires_rsync
@pytest.mark.ci
diff --git a/tests/integration/test_parity_quickwins.py b/tests/integration/test_parity_quickwins.py
index 9e7a0dc..596b468 100644
--- a/tests/integration/test_parity_quickwins.py
+++ b/tests/integration/test_parity_quickwins.py
@@ -719,13 +719,18 @@ class TestVerifyAndFlip:
source = self._src("cmpd")
dest = self._dst("cmpd")
rdst = self._dst("cmpd_r")
+ # Pin the mtime so rsync's size+mtime quick-check (and FastSync's
+ # default) matches deterministically across a second boundary.
+ OLD = 1_500_000_000
with open(os.path.join(source, "f.txt"), "wb") as fh:
fh.write(b"basis-content\n")
+ os.utime(os.path.join(source, "f.txt"), (OLD, OLD))
# rsync resolves --compare-dest relative to the destination dir; its
# basis file sits at the transfer-relative path.
os.makedirs(os.path.join(rdst, "basis"), exist_ok=True)
with open(os.path.join(rdst, "basis", "f.txt"), "wb") as fh:
fh.write(b"basis-content\n")
+ os.utime(os.path.join(rdst, "basis", "f.txt"), (OLD, OLD))
rs = _rsync(["-a", "--compare-dest=basis", source + "/", rdst + "/"])
assert rs.returncode == 0, rs.stderr
assert not os.path.exists(os.path.join(rdst, "f.txt")), \
@@ -738,6 +743,7 @@ class TestVerifyAndFlip:
os.makedirs(basis, exist_ok=True)
with open(os.path.join(basis, "f.txt"), "wb") as fh:
fh.write(b"basis-content\n")
+ os.utime(os.path.join(basis, "f.txt"), (OLD, OLD))
received = get_dest_received_dir(dest, source)
result, _ = run_client(source, dest,
flags=["--compare-dest=basis", "--incremental"],
@@ -751,14 +757,17 @@ class TestVerifyAndFlip:
def test_link_dest_hardlinks_matches_rsync(self, shared_server):
source = self._src("linkd")
dest = self._dst("linkd")
+ OLD = 1_500_000_000
with open(os.path.join(source, "f.txt"), "wb") as fh:
fh.write(b"link-basis-content\n")
+ os.utime(os.path.join(source, "f.txt"), (OLD, OLD))
rel = os.path.abspath(source).lstrip(os.sep)
basis = os.path.join(dest, "basis", rel)
os.makedirs(basis, exist_ok=True)
basis_file = os.path.join(basis, "f.txt")
with open(basis_file, "wb") as fh:
fh.write(b"link-basis-content\n")
+ os.utime(basis_file, (OLD, OLD))
received = get_dest_received_dir(dest, source)
result, _ = run_client(source, dest,
flags=["--link-dest=basis", "--incremental"],
@@ -769,6 +778,160 @@ class TestVerifyAndFlip:
assert os.stat(dest_file).st_ino == os.stat(basis_file).st_ino, \
"--link-dest must hard-link to the basis file"
+ @requires_rsync
+ def test_basis_dir_size_only_content_residual(self, shared_server):
+ """rsync parity (default): a basis hit is decided by the metadata
+ quick-check alone. With `--size-only`, a same-size, different-content
+ basis is trusted, so rsync links the basis content and FastSync must now
+ do the same instead of xxHash-verifying it. `--verify-basis` restores
+ the stricter content equality (covered by the differential test)."""
+ source = self._src("basissz")
+ rdest = self._dst("basissz_r")
+ fdest = self._dst("basissz_f")
+ with open(os.path.join(source, "f.txt"), "wb") as fh:
+ fh.write(b"AAAA\n")
+ OLD = 1_400_000_000
+ # rsync basis at the transfer-relative path (relative to the dest dir).
+ os.makedirs(os.path.join(rdest, "basis"), exist_ok=True)
+ with open(os.path.join(rdest, "basis", "f.txt"), "wb") as fh:
+ fh.write(b"BBBB\n")
+ os.utime(os.path.join(rdest, "basis", "f.txt"), (OLD, OLD))
+ rs = _rsync(["-a", "--size-only", "--link-dest=basis", source + "/", rdest + "/"])
+ assert rs.returncode == 0, rs.stderr
+ with open(os.path.join(rdest, "f.txt"), "rb") as fh:
+ assert fh.read() == b"BBBB\n", "rsync --size-only did not trust the basis size"
+
+ # FastSync basis is relative to the receive root; the file mirrors the
+ # source path.
+ rel = os.path.abspath(source).lstrip(os.sep)
+ basis = os.path.join(fdest, "basis", rel)
+ os.makedirs(basis, exist_ok=True)
+ with open(os.path.join(basis, "f.txt"), "wb") as fh:
+ fh.write(b"BBBB\n")
+ os.utime(os.path.join(basis, "f.txt"), (OLD, OLD))
+ received = get_dest_received_dir(fdest, source)
+ result, _ = run_client(source, fdest,
+ flags=["-a", "--size-only", "--link-dest=basis", "--incremental"],
+ port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+ with open(os.path.join(received, "f.txt"), "rb") as fh:
+ assert fh.read() == b"BBBB\n", \
+ "FastSync must trust the metadata quick-check exactly like rsync"
+
+ @requires_rsync
+ def test_verify_basis_restores_content_check(self, shared_server):
+ """FastSync-only `--verify-basis`: a same-size, same-mtime basis with
+ different content is rejected by the whole-file digest, so the source is
+ transferred instead of installing the wrong basis bytes. The default
+ (no flag) installs the basis content, matching rsync."""
+ source = self._src("vbasis")
+ fdest = self._dst("vbasis_f")
+ with open(os.path.join(source, "f.txt"), "wb") as fh:
+ fh.write(b"AAAA\n")
+ OLD = 1_400_000_000
+ os.utime(os.path.join(source, "f.txt"), (OLD, OLD))
+ rel = os.path.abspath(source).lstrip(os.sep)
+ basis = os.path.join(fdest, "basis", rel)
+ os.makedirs(basis, exist_ok=True)
+ with open(os.path.join(basis, "f.txt"), "wb") as fh:
+ fh.write(b"BBBB\n")
+ os.utime(os.path.join(basis, "f.txt"), (OLD, OLD))
+ received = get_dest_received_dir(fdest, source)
+ result, _ = run_client(source, fdest,
+ flags=["-a", "--link-dest=basis", "--incremental",
+ "--verify-basis"],
+ port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+ with open(os.path.join(received, "f.txt"), "rb") as fh:
+ assert fh.read() == b"AAAA\n", \
+ "--verify-basis must reject the same-size/different-content basis"
+
+
+def _stat_bytes(output, key):
+ """Parse a --stats byte counter (e.g. ``Matched data: 65,536 bytes``)."""
+ for line in output.splitlines():
+ if line.startswith(key + ":"):
+ raw = line.split(":", 1)[1].strip().split()[0]
+ return int(raw.replace(",", ""))
+ return None
+
+
+class TestFuzzy:
+ """Track 5b: `-y`/`--fuzzy` is an internal bandwidth optimization with a
+ byte-exact result. FastSync ports rsync 3.4.1's weighted-Levenshtein name
+ heuristic, so where both delta engines admit the candidate the tools pick
+ the same basis (the ``fuzzy_basis`` differential asserts the tree and the
+ Matched/Literal counters match with the block size pinned). The residual is
+ candidate ELIGIBILITY: FastSync's delta size gate (both files >= 16 KiB and
+ a <= 10x size ratio) is narrower than rsync's, which empirically uses a
+ fuzzy basis well beyond 10x and below 16 KiB. These tests pin the window
+ boundary and prove the byte-exact fallback on both sides of it."""
+
+ _BASE = b"the quick brown fox jumps over the lazy dog\n" * 4000
+
+ def _src(self, tag):
+ source = os.path.join(TEST_DATA_DIR, f"fz_{tag}_src")
+ clean_dir(source)
+ return source
+
+ def _dst(self, tag):
+ d = os.path.join(TEST_DATA_DIR, f"fz_{tag}_dst")
+ clean_dir(d)
+ return d
+
+ def _run_both(self, shared_server, source, dest, rdst, payload, sibling,
+ rs_extra=(), fs_extra=()):
+ with open(os.path.join(source, "report_v2.txt"), "wb") as fh:
+ fh.write(payload)
+ for root in (rdst, get_dest_received_dir(dest, source)):
+ os.makedirs(root, exist_ok=True)
+ with open(os.path.join(root, "report_v1.txt"), "wb") as fh:
+ fh.write(sibling)
+ rs = _rsync(["-a", "--no-whole-file", "--fuzzy", "--stats"] +
+ list(rs_extra) + [source + "/", rdst + "/"])
+ assert rs.returncode == 0, rs.stderr[:300]
+ result, _ = run_client(
+ source, dest,
+ flags=["-a", "--incremental", "--delta", "--fuzzy", "--stats"] +
+ list(fs_extra),
+ port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+ _assert_same_tree(rdst, get_dest_received_dir(dest, source), "(--fuzzy)")
+ return rs, result
+
+ @requires_rsync
+ def test_fuzzy_above_size_window_declines_but_tree_exact(self, shared_server):
+ """A sibling >10x the source is used by rsync but declined by FastSync's
+ delta size-ratio gate; both destinations stay byte-identical."""
+ n = 65536
+ payload = (self._BASE * ((n // len(self._BASE)) + 1))[:n]
+ sibling = (self._BASE * 200)[: n * 20]
+ source, dest, rdst = (self._src("big"), self._dst("big"),
+ self._dst("big_r"))
+ rs, result = self._run_both(shared_server, source, dest, rdst,
+ payload, sibling)
+ assert _stat_bytes(rs.stdout, "Matched data") > 0, \
+ "rsync should still use a >10x fuzzy basis"
+ assert _stat_bytes(result.stdout, "Matched data") == 0, \
+ "FastSync's 10x delta size-ratio gate must decline the oversized basis"
+ assert _stat_bytes(result.stdout, "Literal data") == n
+
+ @requires_rsync
+ def test_fuzzy_below_delta_minimum_declines_but_tree_exact(self, shared_server):
+ """A sibling below the 16 KiB delta minimum is used by rsync but never
+ enters FastSync's delta/fuzzy path; both trees stay byte-identical."""
+ n = 8192
+ payload = (self._BASE * ((n // len(self._BASE)) + 1))[:n]
+ source, dest, rdst = (self._src("small"), self._dst("small"),
+ self._dst("small_r"))
+ rs, result = self._run_both(shared_server, source, dest, rdst,
+ payload, payload)
+ assert _stat_bytes(rs.stdout, "Matched data") > 0, \
+ "rsync applies --fuzzy below 16 KiB"
+ assert _stat_bytes(result.stdout, "Matched data") == 0, \
+ "FastSync's 16 KiB delta minimum must bypass the fuzzy basis"
+ assert _stat_bytes(result.stdout, "Literal data") == n
+
class TestIgnoreExistingShortCircuit:
"""#9: --ignore-existing is decided by the receiver during the per-file
diff --git a/tests/integration/test_parity_selection.py b/tests/integration/test_parity_selection.py
index 0d240bd..aa4e74a 100644
--- a/tests/integration/test_parity_selection.py
+++ b/tests/integration/test_parity_selection.py
@@ -111,6 +111,45 @@ class TestRelativeGeneral:
assert int(rs.st_mtime) == int(fs.st_mtime), \
f"mtime mismatch for {rel} with {extra}"
+ @requires_rsync
+ @pytest.mark.ci
+ def test_no_implied_dirs_files_from_matches_rsync(self, shared_server):
+ """-R --no-implied-dirs --files-from: a listed file whose parent is not
+ itself listed still transfers; the implied parent is created with
+ default attributes (rsync 3.4.1 parity)."""
+ source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_nidff_src"))
+ # Make the implied parent unmistakably non-default on the source so a
+ # wrongly-applied attribute would be observable.
+ os.chmod(os.path.join(source, "foo"), 0o700)
+ os.chmod(os.path.join(source, "foo", "bar"), 0o711)
+ os.utime(os.path.join(source, "foo"), (978307200, 978307200))
+ os.utime(os.path.join(source, "foo", "bar"), (978307200, 978307200))
+ lst = os.path.join(TEST_DATA_DIR, "sel_nidff_list")
+ with open(lst, "w") as fh:
+ fh.write("foo/bar/baz/f.txt\n")
+ dest = os.path.join(TEST_DATA_DIR, "sel_nidff_dst")
+ rdst = os.path.join(TEST_DATA_DIR, "sel_nidff_rdst")
+ clean_dir(dest)
+ clean_dir(rdst)
+ r = _rsync(["-rlpt", "-R", "--no-implied-dirs", "--files-from=" + lst,
+ source + "/", rdst + "/"])
+ assert r.returncode == 0, r.stderr
+ result, _ = run_client(source, dest, flags=[
+ "-rlpt", "-R", "--no-implied-dirs", "--files-from", lst],
+ port=shared_server.port)
+ assert result.returncode == 0, result.stderr[:300]
+ assert _tree(rdst) == _tree(dest), "implied-parent layout mismatch"
+ # The implied parents exist on both sides and carry the run-time default
+ # attributes, not the source's (non-default) ones.
+ for rel in ("foo", "foo/bar", "foo/bar/baz"):
+ rs = os.stat(os.path.join(rdst, rel))
+ fs = os.stat(os.path.join(dest, rel))
+ assert (rs.st_mode & 0o7777) == (fs.st_mode & 0o7777), \
+ f"mode mismatch for implied {rel}"
+ # The listed file is transferred with its content.
+ with open(os.path.join(dest, "foo", "bar", "baz", "f.txt"), "rb") as fh:
+ assert fh.read() == b"deep\n"
+
class TestDirsOneLevel:
"""#13: -d with a trailing slash (or '.') lists the source's immediate
diff --git a/tests/integration/test_preflight.py b/tests/integration/test_preflight.py
index b78c808..ad47d3b 100644
--- a/tests/integration/test_preflight.py
+++ b/tests/integration/test_preflight.py
@@ -94,14 +94,14 @@ def _seed_protocol_source(source):
class TestProtocol:
@pytest.mark.ci
def test_protocol_current_version_accepted(self, shared_server):
- """--protocol=2.26.0 (the current PROTOCOL_VERSION) is accepted and the
+ """--protocol=2.28.0 (the current PROTOCOL_VERSION) is accepted and the
transfer completes normally."""
source = os.path.join(TEST_DATA_DIR, "proto_ok_src")
dest = os.path.join(TEST_DATA_DIR, "proto_ok_dst")
shutil.rmtree(dest, ignore_errors=True)
os.makedirs(dest)
_seed_protocol_source(source)
- result, _ = run_client(source, dest, flags=["--protocol=2.26.0"],
+ result, _ = run_client(source, dest, flags=["--protocol=2.28.0"],
port=shared_server.port)
assert result.returncode == 0, \
f"--protocol current run failed: {(result.stderr or result.stdout)[:400]}"
@@ -118,8 +118,8 @@ class TestProtocol:
shutil.rmtree(dest, ignore_errors=True)
os.makedirs(dest)
_seed_protocol_source(source)
- for bad in ("2.22.0", "2.21.0", "2.20.0", "2.19.0", "2.18.0", "2.17.0", "2.15.0", "2.16.0",
- "216", "31"):
+ for bad in ("2.27.0", "2.26.0", "2.25.0", "2.24.0", "2.23.0", "2.22.0", "2.21.0", "2.20.0",
+ "2.19.0", "2.18.0", "2.17.0", "2.15.0", "2.16.0", "216", "31"):
result, _ = run_client(source, dest, flags=[f"--protocol={bad}"],
port=shared_server.port)
assert result.returncode != 0, f"--protocol={bad} should be rejected"
diff --git a/tests/runner.c b/tests/runner.c
index 05f583e..de1b4b5 100644
--- a/tests/runner.c
+++ b/tests/runner.c
@@ -11,6 +11,7 @@
#include "test_daemon_conf.h"
#include "test_daemon_limits.h"
#include "test_delay_updates.h"
+#include "test_delete_plan.h"
#include "test_delta.h"
#include "test_file.h"
#include "test_file_list.h"
@@ -42,6 +43,7 @@
#include "test_utils.h"
#include "test_xattr.h"
#include
+#include
#include
// Define global test state variables
@@ -51,6 +53,11 @@ bool current_test_failed = false;
int main() {
signal(SIGPIPE, SIG_IGN);
+ /* The codec/checksum resolvers consult rsync's preference-list environment
+ * variables; clear them so a developer's shell cannot change test outcomes.
+ * The env-specific tests set and restore their own values. */
+ unsetenv("RSYNC_COMPRESS_LIST");
+ unsetenv("RSYNC_CHECKSUM_LIST");
printf("\033[1;36m=== RUNNING UNIT TESTS ===\033[0m\n\n");
RUN_TEST(test_queue);
@@ -66,6 +73,7 @@ int main() {
RUN_TEST(test_scanner);
RUN_TEST(test_checksum);
RUN_TEST(test_delta);
+ RUN_TEST(test_delete_plan);
RUN_TEST(test_data);
RUN_TEST(test_protocol);
RUN_TEST(test_protocol_error);
diff --git a/tests/test_change_list.c b/tests/test_change_list.c
index 1cf91e0..f9ad8a5 100644
--- a/tests/test_change_list.c
+++ b/tests/test_change_list.c
@@ -177,6 +177,40 @@ static void test_change_list_enabled() {
config_delete(config); /* closes config->log_file */
}
+/* %C uses the negotiated TRANSFER checksum's column width (not the pre-transfer
+ * whole-file digest), and `none` renders as a blank 2-char column, matching
+ * rsync. A not-yet-filled checksum renders as spaces. */
+static void test_format_C_padding_uses_transfer_algo() {
+ ChangeEvent event = sample_event();
+ Config* config = config_create();
+ EXPECT_NOT_NULL(config);
+ /* Deliberately different pre-transfer algorithm: the transfer one must win. */
+ config->checksum_algo = (int)CHECKSUM_ALGO_MD4;
+ struct {
+ int algo;
+ int width;
+ } cases[] = {
+ {CHECKSUM_ALGO_XXH128, 32}, {CHECKSUM_ALGO_XXH64, 16}, {CHECKSUM_ALGO_XXH3, 16},
+ {CHECKSUM_ALGO_MD5, 32}, {CHECKSUM_ALGO_MD4, 32}, {CHECKSUM_ALGO_SHA1, 40},
+ {CHECKSUM_ALGO_NONE, 2},
+ };
+ for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) {
+ config->checksum_transfer_algo = cases[i].algo;
+ char expected[64];
+ size_t n = 0;
+ expected[n++] = '[';
+ for (int j = 0; j < cases[i].width; j++)
+ expected[n++] = ' ';
+ expected[n++] = ']';
+ expected[n] = '\0';
+ char* line = change_render_format("[%C]", config, &event);
+ EXPECT_NOT_NULL(line);
+ EXPECT_EQ_STR(line, expected);
+ free(line);
+ }
+ config_delete(config);
+}
+
void test_change_list() {
test_format_tokens();
test_format_unknown_tokens_preserved();
@@ -188,4 +222,5 @@ void test_change_list() {
test_render_itemize_up_to_date_is_empty();
test_render_list_line();
test_change_list_enabled();
+ test_format_C_padding_uses_transfer_algo();
}
diff --git a/tests/test_checksum.c b/tests/test_checksum.c
index bf781ae..81107ea 100644
--- a/tests/test_checksum.c
+++ b/tests/test_checksum.c
@@ -1,7 +1,9 @@
#include "test_checksum.h"
#include "checksum.h"
#include "test_utils.h"
+#include
#include
+#include
/* Known xxHash64 vector (seed 0) for the empty string and a literal.
* The md5 vectors are the standard NIST/RFC1321 test strings. These pin the
@@ -230,6 +232,61 @@ static void test_checksum_null_empty_digest() {
EXPECT_TRUE(memcmp(a, b, alen) == 0);
}
+/* Every algorithm checksum_digest_file() claims to support must produce the
+ * SAME digest as the in-memory one-shot, including the newly added md4/sha1/
+ * none. A 200000-byte payload forces several 64 KiB streaming reads. */
+static void test_checksum_digest_file_matches_oneshot(void) {
+ static const ChecksumAlgo algos[] = {
+ CHECKSUM_ALGO_XXH64, CHECKSUM_ALGO_XXH3, CHECKSUM_ALGO_XXH128, CHECKSUM_ALGO_MD5,
+ CHECKSUM_ALGO_MD4, CHECKSUM_ALGO_SHA1, CHECKSUM_ALGO_NONE,
+ };
+ enum { SIZE = 200000 };
+ uint8_t* data = malloc(SIZE);
+ EXPECT_NOT_NULL(data);
+ for (int i = 0; i < SIZE; i++)
+ data[i] = (uint8_t)((i * 7 + 3) & 0xff);
+ char path[] = "/tmp/fastsync_ck_XXXXXX";
+ int fd = mkstemp(path);
+ EXPECT_TRUE(fd >= 0);
+ ssize_t written = write(fd, data, SIZE);
+ close(fd);
+ EXPECT_EQ_INT((int)written, SIZE);
+ for (size_t a = 0; a < sizeof(algos) / sizeof(algos[0]); a++) {
+ ChecksumAlgo algo = algos[a];
+ uint8_t one[CHECKSUM_MAX_DIGEST_LEN];
+ uint8_t file[CHECKSUM_MAX_DIGEST_LEN];
+ size_t one_len = 0, file_len = 0;
+ EXPECT_TRUE(checksum_digest(algo, 0, data, SIZE, one, sizeof(one), &one_len));
+ EXPECT_TRUE(checksum_digest_file(algo, 0, path, file, sizeof(file), &file_len));
+ EXPECT_EQ_INT((int)file_len, (int)one_len);
+ EXPECT_TRUE(memcmp(one, file, one_len) == 0);
+ }
+ unlink(path);
+ free(data);
+}
+
+/* RSYNC_CHECKSUM_LIST precedence, syntax and fallback. */
+static void test_checksum_choice_env_list() {
+ unsetenv("RSYNC_CHECKSUM_LIST");
+ EXPECT_EQ_INT(checksum_choice_resolve(), (int)CHECKSUM_ALGO_XXH128);
+ EXPECT_EQ_INT((int)checksum_negotiate_default(), (int)CHECKSUM_ALGO_XXH128);
+
+ setenv("RSYNC_CHECKSUM_LIST", "bogus md5 xxh3", 1);
+ EXPECT_EQ_INT(checksum_choice_resolve(), (int)CHECKSUM_ALGO_MD5);
+
+ setenv("RSYNC_CHECKSUM_LIST", "SHA1", 1);
+ EXPECT_EQ_INT(checksum_choice_resolve(), (int)CHECKSUM_ALGO_SHA1);
+
+ /* Whitespace-separated only: comma is not a separator in rsync's syntax. */
+ setenv("RSYNC_CHECKSUM_LIST", "md5,xxh3", 1);
+ EXPECT_EQ_INT(checksum_choice_resolve(), -1);
+
+ setenv("RSYNC_CHECKSUM_LIST", " ", 1);
+ EXPECT_EQ_INT(checksum_choice_resolve(), (int)CHECKSUM_ALGO_XXH128);
+
+ unsetenv("RSYNC_CHECKSUM_LIST");
+}
+
void test_checksum(void) {
test_checksum_xxh64_seed0();
test_checksum_xxh64_empty();
@@ -245,4 +302,6 @@ void test_checksum(void) {
test_checksum_xxh3_xxh128();
test_checksum_truncated_buffer_rejected();
test_checksum_null_empty_digest();
-}
\ No newline at end of file
+ test_checksum_digest_file_matches_oneshot();
+ test_checksum_choice_env_list();
+}
diff --git a/tests/test_client_cli.c b/tests/test_client_cli.c
index 2972a8a..468987b 100644
--- a/tests/test_client_cli.c
+++ b/tests/test_client_cli.c
@@ -7,6 +7,7 @@
#include "delta.h"
#include "file_list.h"
#include "log.h"
+#include "protocol.h"
#include "test_utils.h"
#include "utils.h"
#include
@@ -317,7 +318,7 @@ static void test_parse_args_protocol_accept_current() {
Config* cfg = valid_client_config();
EXPECT_NOT_NULL(cfg);
char* argv_equals[] = {"fastsync", "--source-dir", "/src",
- "--dest-dir", "/dst", "--protocol=2.26.0"};
+ "--dest-dir", "/dst", "--protocol=2.28.0"};
int positional_args[2];
int positional_count = 0;
EXPECT_EQ_INT(parse_args(cfg, 6, argv_equals, positional_args, &positional_count), 0);
@@ -327,7 +328,7 @@ static void test_parse_args_protocol_accept_current() {
cfg = valid_client_config();
EXPECT_NOT_NULL(cfg);
char* argv_space[] = {"fastsync", "--source-dir", "/src", "--dest-dir",
- "/dst", "--protocol", "2.26.0"};
+ "/dst", "--protocol", "2.28.0"};
positional_count = 0;
EXPECT_EQ_INT(parse_args(cfg, 7, argv_space, positional_args, &positional_count), 0);
EXPECT_EQ_STR(cfg->version, PROTOCOL_VERSION);
@@ -339,6 +340,7 @@ static void test_parse_args_protocol_accept_current() {
static void test_parse_args_protocol_rejects_other_versions() {
static const char* const bad_versions[] = {"2.17", "2.16", "2.15.0", "2.16.0", "2.17.0",
"2.18.0", "2.19.0", "2.20.0", "2.21.0", "2.22.0",
+ "2.23.0", "2.24.0", "2.25.0", "2.26.0", "2.27.0",
"216", "31", "abc", ""};
for (size_t i = 0; i < sizeof(bad_versions) / sizeof(bad_versions[0]); i++) {
Config* cfg = valid_client_config();
@@ -1044,6 +1046,23 @@ static void test_parse_args_basis_dirs() {
config_delete(cfg);
}
+/* --verify-basis (FastSync-only, long-only): default off; parses on as a plain
+ boolean and leaves the basis implications intact. */
+static void test_parse_args_verify_basis() {
+ Config* cfg = config_create();
+ EXPECT_FALSE(cfg->verify_basis);
+ config_delete(cfg);
+
+ cfg = config_create();
+ int positional_args[2];
+ int positional_count = 0;
+ char* argv[] = {"fastsync", "--link-dest=prior", "--verify-basis", "/src", "/dst"};
+ EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0);
+ EXPECT_TRUE(cfg->verify_basis);
+ EXPECT_TRUE(config_has_basis(cfg));
+ config_delete(cfg);
+}
+
/* Escaping or degenerate basis-dir values must be rejected up front (they would
resolve outside the destination root on the receiver); an absolute path is
accepted (rsync parity) and canonicalized with its leading '/' preserved. */
@@ -1131,6 +1150,63 @@ static void test_parse_args_delete_timing_flags() {
config_delete(cfg);
}
+/* Plain --delete with no explicit timing defaults to delete-during, matching
+ * rsync's --del default (progressive deletion). --delete-commit is the
+ * FastSync-only long spelling that selects rsync's --delete-after timing (the
+ * late whole-tree commit), and an explicit timing always wins over the default.
+ */
+static void test_parse_args_delete_default_timing_and_commit() {
+ Config* cfg = config_create();
+ char* argv[] = {"fastsync", "--delete", "/src", "/dst"};
+ int positional_args[2];
+ int positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0);
+ EXPECT_TRUE(cfg->use_delete);
+ EXPECT_TRUE(cfg->delete_during);
+ EXPECT_FALSE(cfg->delete_before);
+ EXPECT_FALSE(cfg->delete_delay);
+ EXPECT_FALSE(cfg->delete_after);
+ cfg->send_directory = str_dup("/src");
+ cfg->receive_root_directory = str_dup("/dst");
+ EXPECT_TRUE(validate_config(cfg));
+ config_delete(cfg);
+
+ /* --delete-commit selects the late whole-tree commit (delete_after) and
+ implies --delete. */
+ cfg = config_create();
+ char* argv_commit[] = {"fastsync", "--delete-commit", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, argv_commit, positional_args, &positional_count), 0);
+ EXPECT_TRUE(cfg->use_delete);
+ EXPECT_TRUE(cfg->delete_after);
+ EXPECT_FALSE(cfg->delete_before);
+ EXPECT_FALSE(cfg->delete_during);
+ EXPECT_FALSE(cfg->delete_delay);
+ cfg->send_directory = str_dup("/src");
+ cfg->receive_root_directory = str_dup("/dst");
+ EXPECT_TRUE(validate_config(cfg));
+ config_delete(cfg);
+
+ /* An explicit --delete-after alongside plain --delete keeps the late timing:
+ the default never overwrites an explicit timing. */
+ cfg = config_create();
+ char* argv_after[] = {"fastsync", "--delete", "--delete-after", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 5, argv_after, positional_args, &positional_count), 0);
+ EXPECT_TRUE(cfg->use_delete);
+ EXPECT_TRUE(cfg->delete_after);
+ EXPECT_FALSE(cfg->delete_during);
+ config_delete(cfg);
+
+ /* --delete-commit conflicts with a different timing. */
+ cfg = config_create();
+ char* argv_conflict[] = {"fastsync", "--delete-commit", "--delete-during", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 5, argv_conflict, positional_args, &positional_count), 0);
+ EXPECT_FALSE(validate_config(cfg));
+ config_delete(cfg);
+}
+
/* Two different delete-timing flags on one command line are a conflict, not a
* silent last-one-wins choice. */
static void test_parse_args_delete_timing_conflict_rejected() {
@@ -1327,7 +1403,7 @@ static void test_parse_args_rejects_invalid_info_flag() {
config_delete(cfg);
}
-/* rsync's info "name" category maps to fastsync's per-file name logging, and
+/* rsync's info "name" category maps to fastsync's per-file name output, and
* --info=help prints the flag list and exits without error. */
static void test_parse_args_info_name_and_help() {
Config* cfg = config_create();
@@ -1335,7 +1411,7 @@ static void test_parse_args_info_name_and_help() {
int positional_args[2];
int positional_count = 0;
EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0);
- EXPECT_EQ_INT(cfg->info_level, LOG_INFO_COPY);
+ EXPECT_EQ_INT(cfg->info_level, LOG_INFO_NAME);
config_delete(cfg);
cfg = config_create();
@@ -1345,8 +1421,10 @@ static void test_parse_args_info_name_and_help() {
config_delete(cfg);
}
-/* rsync 3.4.1's remaining --info/--debug categories parse successfully but
- * have no FastSync output wired to them, so they must not set any log flag. */
+/* rsync 3.4.1's full --info/--debug vocabulary parses. The info categories
+ * with a FastSync event set their flag; the remaining rsync-only categories
+ * (backup/mount/symsafe/syms) parse but stay silent. Every --debug category
+ * listed here is FastSync-silent, so debug_level stays 0. */
static void test_parse_args_rsync_flag_vocabulary_accepted() {
Config* cfg = config_create();
char* argv[] = {"fastsync", "--info=backup,del,flist,mount,nonreg,progress,remove,symsafe,syms",
@@ -1358,7 +1436,8 @@ static void test_parse_args_rsync_flag_vocabulary_accepted() {
int positional_count = 0;
EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0);
- EXPECT_EQ_INT(cfg->info_level, 0);
+ EXPECT_EQ_INT(cfg->info_level, LOG_INFO_DEL | LOG_INFO_FLIST | LOG_INFO_NONREG |
+ LOG_INFO_PROGRESS | LOG_INFO_REMOVE);
EXPECT_EQ_INT(cfg->debug_level, 0);
config_delete(cfg);
}
@@ -1389,6 +1468,29 @@ static void test_parse_args_debug_info_levels() {
EXPECT_EQ_INT(cfg->info_level, LOG_INFO_STATS);
config_delete(cfg);
+ /* --info=name level 2 enables the "is uptodate" marker; a later level-1 or
+ level-0 token clears it again. */
+ cfg = config_create();
+ char* name2_argv[] = {"fastsync", "--info=name2", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, name2_argv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->info_level, LOG_INFO_NAME | LOG_INFO_NAME_UPTODATE);
+ config_delete(cfg);
+
+ cfg = config_create();
+ char* name1_argv[] = {"fastsync", "--info=name2,name1", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, name1_argv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->info_level, LOG_INFO_NAME);
+ config_delete(cfg);
+
+ cfg = config_create();
+ char* name0_argv[] = {"fastsync", "--info=name2,name0", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, name0_argv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->info_level, 0);
+ config_delete(cfg);
+
cfg = config_create();
char* bad_argv[] = {"fastsync", "--debug=123", "/src", "/dst"};
positional_count = 0;
@@ -2404,6 +2506,119 @@ static void test_parse_args_rejects_invalid_compression_choice() {
config_delete(cfg);
}
+/* rsync gives each codec its own default --compress-level; an omitted level
+ * resolves to that default and an explicit one is clamped to the codec range. */
+static void test_parse_args_per_codec_compression_level_defaults() {
+ unsetenv("RSYNC_COMPRESS_LIST");
+ struct {
+ const char* choice;
+ int level;
+ } cases[] = {
+ {"zstd", 3},
+ {"zlib", 6},
+ {"zlibx", 6},
+ {"lz4", 1},
+ };
+ for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) {
+ Config* cfg = config_create();
+ char* argv[] = {"fastsync", "-z", "--compress-choice", (char*)cases[i].choice, "/src", "/dst"};
+ int positional_args[2];
+ int positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 6, argv, positional_args, &positional_count), 0);
+ EXPECT_TRUE(cfg->use_compression);
+ EXPECT_EQ_INT(cfg->compression_level, cases[i].level);
+ config_delete(cfg);
+ }
+
+ /* Bare -z resolves to the zstd default. */
+ Config* cfg = config_create();
+ char* bare[] = {"fastsync", "-z", "/src", "/dst"};
+ int positional_args[2];
+ int positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, bare, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->compression_level, 3);
+ config_delete(cfg);
+
+ /* An explicit level wins unchanged for zstd... */
+ cfg = config_create();
+ char* zv[] = {"fastsync", "-z", "--compress-level", "10", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 5, zv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->compression_level, 10);
+ config_delete(cfg);
+
+ /* ...but zlib clamps an over-range level to 9 like rsync. */
+ cfg = config_create();
+ char* zc[] = {"fastsync", "-z", "--compress-choice", "zlib", "--compress-level", "15",
+ "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 8, zc, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->compression_level, 9);
+ config_delete(cfg);
+}
+
+/* RSYNC_COMPRESS_LIST drives the bare -z ("auto") resolution. */
+static void test_parse_args_compression_env_list() {
+ Config* cfg = config_create();
+ char* argv[] = {"fastsync", "-z", "/src", "/dst"};
+ int positional_args[2];
+ int positional_count = 0;
+
+ setenv("RSYNC_COMPRESS_LIST", "zlib lz4", 1);
+ EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->compression_algo, (int)COMPRESSION_ALGO_ZLIB);
+ EXPECT_EQ_INT(cfg->compression_level, 6);
+ config_delete(cfg);
+
+ /* An explicit --compress-choice beats the env list. */
+ cfg = config_create();
+ char* explicit_argv[] = {"fastsync", "-z", "--compress-choice", "zstd", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 6, explicit_argv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->compression_algo, (int)COMPRESSION_ALGO_ZSTD);
+ config_delete(cfg);
+
+ /* A list with no supported name is rsync's failed negotiation (exit 4). */
+ setenv("RSYNC_COMPRESS_LIST", "bogus", 1);
+ cfg = config_create();
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1);
+ EXPECT_EQ_INT(cfg->cli_exit_code, 4);
+ config_delete(cfg);
+ unsetenv("RSYNC_COMPRESS_LIST");
+}
+
+/* RSYNC_CHECKSUM_LIST drives the default checksum choice. */
+static void test_parse_args_checksum_env_list() {
+ Config* cfg = config_create();
+ char* argv[] = {"fastsync", "--checksum", "/src", "/dst"};
+ int positional_args[2];
+ int positional_count = 0;
+
+ setenv("RSYNC_CHECKSUM_LIST", "md5", 1);
+ EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_MD5);
+ EXPECT_EQ_INT(cfg->checksum_transfer_algo, (int)CHECKSUM_ALGO_MD5);
+ config_delete(cfg);
+
+ /* An explicit --cc wins. */
+ cfg = config_create();
+ char* cc_argv[] = {"fastsync", "--checksum", "--cc=sha1", "/src", "/dst"};
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 5, cc_argv, positional_args, &positional_count), 0);
+ EXPECT_EQ_INT(cfg->checksum_algo, (int)CHECKSUM_ALGO_SHA1);
+ config_delete(cfg);
+
+ /* A list with no supported name is rsync's failed negotiation (exit 4). */
+ setenv("RSYNC_CHECKSUM_LIST", "bogus", 1);
+ cfg = config_create();
+ positional_count = 0;
+ EXPECT_EQ_INT(parse_args(cfg, 4, argv, positional_args, &positional_count), -1);
+ EXPECT_EQ_INT(cfg->cli_exit_code, 4);
+ config_delete(cfg);
+ unsetenv("RSYNC_CHECKSUM_LIST");
+}
+
/* Every value-taking table option accepts an inline "--opt=value" form. */
static void test_parse_args_table_equals_size_options() {
Config* cfg = config_create();
@@ -3870,6 +4085,113 @@ static void test_parse_args_unsigned_options_reject_sign() {
config_delete(cfg);
}
+/* --bwlimit must parse with rsync 3.4.1's units and quantization: a bare value
+ * is KiB/s, K/M/G/T/P are binary multipliers, KB/MB are decimal, KiB/MiB are
+ * binary, decimals are rounded to whole KiB like rsync's (size + 512) / 1024,
+ * and 0 (or an empty value) means "no limit". */
+static void test_parse_args_bwlimit_rsync_units() {
+ struct {
+ const char* value;
+ unsigned long long expected; /* bytes/sec */
+ int ok;
+ } cases[] = {
+ {"100", 100ULL * 1024, 1},
+ {"0", 0, 1},
+ {"", 0, 1},
+ {"1.5", 2ULL * 1024, 1},
+ {"100K", 100ULL * 1024, 1},
+ {"100KiB", 100ULL * 1024, 1},
+ {"100KB", (100000ULL + 512) / 1024 * 1024, 1},
+ {"1M", 1024ULL * 1024, 1},
+ {"1MB", (1000000ULL + 512) / 1024 * 1024, 1},
+ {"1.5m", 1536ULL * 1024, 1},
+ {"1G", 1024ULL * 1024 * 1024, 1},
+ {"1000B", (1000ULL + 512) / 1024 * 1024, 1},
+ {"100B", 0, 0}, /* below the 512-byte floor (not 0) */
+ {"0.4", 0, 0}, /* 409 bytes, below the floor */
+ {"511", 511ULL * 1024, 1},
+ {"-1", 0, 0},
+ {"abc", 0, 0},
+ {"1x", 0, 0},
+ };
+ for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) {
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ int positional_args[2];
+ int positional_count = 0;
+ char option[32];
+ snprintf(option, sizeof(option), "--bwlimit=%s", cases[i].value);
+ char* argv[] = {"fastsync", option, "/src", "/dst"};
+ int rc = parse_args(cfg, 4, argv, positional_args, &positional_count);
+ if (cases[i].ok) {
+ EXPECT_EQ_INT(rc, 0);
+ EXPECT_TRUE(io_get_bwlimit() == cases[i].expected);
+ } else {
+ EXPECT_EQ_INT(rc, -1);
+ }
+ config_delete(cfg);
+ }
+
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ int positional_args[2];
+ int positional_count = 0;
+ char* argv[] = {"fastsync", "--bwlimit", "512", "/src", "/dst"};
+ EXPECT_EQ_INT(parse_args(cfg, 5, argv, positional_args, &positional_count), 0);
+ EXPECT_TRUE(io_get_bwlimit() == 512ULL * 1024);
+ config_delete(cfg);
+ io_set_bwlimit(0);
+}
+
+/* Huge/malformed --bwlimit values must be rejected (not accepted or UB) and
+ * oversized-but-representable ones must still be accepted: the scaling used to
+ * be done with an unchecked signed double->long long cast, which is undefined
+ * when the product leaves long long's range. */
+static void test_parse_args_bwlimit_huge_and_boundary() {
+ struct {
+ const char* value;
+ unsigned long long expected; /* bytes/sec, ignored when !ok */
+ int ok;
+ } cases[] = {
+ /* Malformed / non-numeric prefixes. */
+ {"1e300", 0, 0},
+ {"99999999999999999999999999e3", 0, 0},
+ /* Products that exceed LLONG_MAX at every suffix. */
+ {"99999999999999999999999999", 0, 0},
+ {"99999999999999999999999999K", 0, 0},
+ {"100000000000000000000P", 0, 0},
+ {"99999999999999999999999999999999999999999999999999B", 0, 0},
+ /* `strtod` overflow to +inf must be caught by the isfinite() guard. */
+ {"9999999999999999999999999999999999999999999999999999999999999999999999"
+ "9999999999999999999999999999999999999999999999999999999999999999999999"
+ "99999999999999999999999999999999999999999999999999999999999999999999999",
+ 0, 0},
+ /* 2^52 KiB/s: the largest power-of-two scaling that still fits. */
+ {"4503599627370496", 4611686018427387904ULL, 1},
+ /* The +/-1 suffix forms accepted by rsync. */
+ {"1+1", 1024ULL, 1},
+ {"1-1", 1024ULL, 1},
+ };
+ for (size_t i = 0; i < sizeof(cases) / sizeof(cases[0]); i++) {
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ int positional_args[2];
+ int positional_count = 0;
+ char option[1024];
+ snprintf(option, sizeof(option), "--bwlimit=%s", cases[i].value);
+ char* argv[] = {"fastsync", option, "/src", "/dst"};
+ int rc = parse_args(cfg, 4, argv, positional_args, &positional_count);
+ if (cases[i].ok) {
+ EXPECT_EQ_INT(rc, 0);
+ EXPECT_TRUE(io_get_bwlimit() == cases[i].expected);
+ } else {
+ EXPECT_EQ_INT(rc, -1);
+ }
+ config_delete(cfg);
+ io_set_bwlimit(0);
+ }
+}
+
/* --dry-run must not emit a batch file, so it is rejected alongside
* --read-batch/--only-write-batch. */
static void test_validate_config_dry_run_rejects_write_batch() {
@@ -4394,6 +4716,26 @@ static void test_parse_args_rejects_unsupported_short() {
}
}
+/* The --ignore-errors deletion gate, unit-tested without a privileged source
+ * directory: a clean scan always deletes; an I/O error suppresses deletion
+ * unless --ignore-errors is set. (The end-to-end mode-000 differential lives in
+ * the setpriv integration test; this pins the decision itself in the PR gate.) */
+static void test_ignore_errors_allows_delete(void) {
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ EXPECT_TRUE(ignore_errors_allows_delete(cfg, false));
+ EXPECT_FALSE(ignore_errors_allows_delete(cfg, true));
+
+ cfg->ignore_errors = true;
+ EXPECT_TRUE(ignore_errors_allows_delete(cfg, false));
+ EXPECT_TRUE(ignore_errors_allows_delete(cfg, true));
+
+ /* A NULL config cannot opt into --ignore-errors. */
+ EXPECT_TRUE(ignore_errors_allows_delete(NULL, false));
+ EXPECT_FALSE(ignore_errors_allows_delete(NULL, true));
+ config_delete(cfg);
+}
+
void test_client_cli() {
test_validate_config_required_paths();
test_parse_args_numeric_ids();
@@ -4473,6 +4815,7 @@ void test_client_cli() {
test_parse_args_relative_no_implied_mkpath();
test_parse_args_delete_during_alias();
test_parse_args_delete_timing_flags();
+ test_parse_args_delete_default_timing_and_commit();
test_parse_args_delete_timing_conflict_rejected();
test_parse_args_delete_timing_without_delete_rejected();
test_parse_args_rejects_unimplemented_options();
@@ -4528,6 +4871,9 @@ void test_client_cli() {
test_parse_args_compression_alias_equals();
test_parse_args_rejects_invalid_compression_level_equals();
test_parse_args_rejects_invalid_compression_choice();
+ test_parse_args_per_codec_compression_level_defaults();
+ test_parse_args_compression_env_list();
+ test_parse_args_checksum_env_list();
test_parse_args_table_equals_size_options();
test_parse_args_table_equals_string_and_int_options();
test_parse_args_missing_argument_diagnostic();
@@ -4561,6 +4907,7 @@ void test_client_cli() {
test_parse_args_filter_rules();
test_parse_args_from0_cvs_filter_file_flags();
test_parse_args_basis_dirs();
+ test_parse_args_verify_basis();
test_parse_args_basis_invalid_paths();
test_validate_config_basis_rejects_chunk_serialization();
test_parse_args_delete_policy_flags();
@@ -4579,6 +4926,8 @@ void test_client_cli() {
test_parse_args_password_file();
test_parse_args_pattern_file_oversized_rejected();
test_parse_args_unsigned_options_reject_sign();
+ test_parse_args_bwlimit_rsync_units();
+ test_parse_args_bwlimit_huge_and_boundary();
test_validate_config_dry_run_rejects_write_batch();
test_parse_args_short_clustering();
test_parse_args_attached_short_values();
@@ -4587,4 +4936,5 @@ void test_client_cli() {
test_parse_args_noop_does_not_consume_argv();
test_parse_args_backup_copy_links_shorts();
test_parse_args_rejects_unsupported_short();
+ test_ignore_errors_allows_delete();
}
diff --git a/tests/test_compression.c b/tests/test_compression.c
index 833835e..1e35293 100644
--- a/tests/test_compression.c
+++ b/tests/test_compression.c
@@ -4,6 +4,7 @@
#include "data.h"
#include "file.h"
#include "utils.h"
+#include
#include
#include
#include
@@ -390,6 +391,56 @@ static void test_codec_name_mapping() {
EXPECT_TRUE(compression_algo_enabled(COMPRESSION_ALGO_ZSTD));
}
+/* rsync 3.4.1's per-codec default levels and its clamping ranges. */
+static void test_codec_level_defaults_and_clamp() {
+ EXPECT_EQ_INT(compression_default_level(COMPRESSION_ALGO_ZSTD), ZSTD_CLEVEL_DEFAULT);
+ EXPECT_EQ_INT(compression_default_level(COMPRESSION_ALGO_ZSTD), 3);
+ EXPECT_EQ_INT(compression_default_level(COMPRESSION_ALGO_ZLIB), 6);
+ EXPECT_EQ_INT(compression_default_level(COMPRESSION_ALGO_ZLIBX), 6);
+ /* lz4 has no tunable level; a positive placeholder keeps the codec engaged. */
+ EXPECT_TRUE(compression_default_level(COMPRESSION_ALGO_LZ4) > 0);
+ EXPECT_EQ_INT(compression_default_level(COMPRESSION_ALGO_NONE), 0);
+
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_ZSTD, 1), 1);
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_ZSTD, 22), 22);
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_ZSTD, 23), 22);
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_ZSTD, 0), 1);
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_ZLIB, 15), 9);
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_ZLIBX, 15), 9);
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_ZLIB, 1), 1);
+ EXPECT_TRUE(compression_clamp_level(COMPRESSION_ALGO_LZ4, 20) > 0);
+ EXPECT_EQ_INT(compression_clamp_level(COMPRESSION_ALGO_NONE, 20), 0);
+}
+
+/* RSYNC_COMPRESS_LIST precedence, syntax and fallback. */
+static void test_codec_choice_env_list() {
+ unsetenv("RSYNC_COMPRESS_LIST");
+ EXPECT_EQ_INT(compression_choice_resolve(), (int)COMPRESSION_ALGO_ZSTD);
+ EXPECT_EQ_INT((int)compression_negotiate_default(), (int)COMPRESSION_ALGO_ZSTD);
+
+ /* Unknown entries are skipped; the first supported wins. */
+ setenv("RSYNC_COMPRESS_LIST", "bogus zlib lz4", 1);
+ EXPECT_EQ_INT(compression_choice_resolve(), (int)COMPRESSION_ALGO_ZLIB);
+
+ /* Case-insensitive. */
+ setenv("RSYNC_COMPRESS_LIST", "ZSTD", 1);
+ EXPECT_EQ_INT(compression_choice_resolve(), (int)COMPRESSION_ALGO_ZSTD);
+
+ /* Whitespace-separated; the client half ends at '&'. */
+ setenv("RSYNC_COMPRESS_LIST", "lz4 zlib & zstd", 1);
+ EXPECT_EQ_INT(compression_choice_resolve(), (int)COMPRESSION_ALGO_LZ4);
+
+ /* Blank falls back to the compiled-in order. */
+ setenv("RSYNC_COMPRESS_LIST", " ", 1);
+ EXPECT_EQ_INT(compression_choice_resolve(), (int)COMPRESSION_ALGO_ZSTD);
+
+ /* rsync's syntax has no comma/colon separator: this is one unknown name. */
+ setenv("RSYNC_COMPRESS_LIST", "bogus,lz4", 1);
+ EXPECT_EQ_INT(compression_choice_resolve(), -1);
+
+ unsetenv("RSYNC_COMPRESS_LIST");
+}
+
/* The process-global codec selects what the legacy wrappers produce. */
static void test_codec_global_selection() {
Data* original = data_create_empty(64);
@@ -423,5 +474,7 @@ void test_compression() {
test_chunk_compress_decompress_roundtrip();
test_codec_roundtrips();
test_codec_name_mapping();
+ test_codec_level_defaults_and_clamp();
+ test_codec_choice_env_list();
test_codec_global_selection();
}
diff --git a/tests/test_config.c b/tests/test_config.c
index b7e6c7b..a801d70 100644
--- a/tests/test_config.c
+++ b/tests/test_config.c
@@ -1,6 +1,7 @@
#include "test_config.h"
#include "config.h"
#include "delta.h"
+#include "filter.h"
#include "identity.h"
#include "multiprocessing.h"
#include "protocol.h"
@@ -2119,6 +2120,48 @@ static void test_config_receive_rejects_oversized_string_budget() {
config_delete(over_bytes);
}
+/* Pre-auth bounds for the receiver-side filter rule block (protocol 2.28.0).
+ A peer may send `protect`/`risk` rules; the receiver must reject an over-cap
+ count or an over-long pattern before evaluating anything, so a crafted config
+ cannot drive unbounded glob work or install a rule that silently never
+ matches. */
+static void test_config_receive_rejects_bad_protect_rules() {
+ if (is_running_under_valgrind())
+ return;
+
+ /* Over-cap rule count: one more than MAX_FILTER_RULES rules. */
+ Config* c = config_create();
+ EXPECT_NOT_NULL(c);
+ c->send_directory = str_dup("/src");
+ c->receive_root_directory = str_dup("/dst");
+ c->filters = array_list_create(free);
+ EXPECT_NOT_NULL(c->filters);
+ for (int i = 0; i <= MAX_FILTER_RULES; i++)
+ EXPECT_TRUE(array_list_add(c->filters, str_dup("- *.tmp")));
+ EXPECT_TRUE(roundtrip_config_rejected(c));
+ config_delete(c);
+
+ /* A pattern longer than the receiver's evaluation bound is rejected by the
+ sender (mirroring the receiver's guard) instead of being sent as an inert
+ rule. */
+ c = config_create();
+ EXPECT_NOT_NULL(c);
+ c->send_directory = str_dup("/src");
+ c->receive_root_directory = str_dup("/dst");
+ c->filters = array_list_create(free);
+ EXPECT_NOT_NULL(c->filters);
+ size_t big = MAX_PROTECT_PATTERN_LEN + 1;
+ char* long_rule = malloc(big + 3);
+ EXPECT_NOT_NULL(long_rule);
+ long_rule[0] = '-';
+ long_rule[1] = ' ';
+ memset(long_rule + 2, 'x', big);
+ long_rule[big + 2] = '\0';
+ EXPECT_TRUE(array_list_add(c->filters, long_rule));
+ EXPECT_TRUE(roundtrip_config_rejected(c));
+ config_delete(c);
+}
+
/* identity_copy_as_refused() is the pure, pre-snapshot refusal predicate: a
--copy-as is refused when the receiver is not root OR the effective super
mode is OFF (an operator veto), and never when --copy-as is unset. */
@@ -2590,6 +2633,46 @@ static bool basis_equal(const Config* a, const Config* b) {
return true;
}
+/* The receiver reconstructs its delete-protection list from the sender's
+ * compiled base rules, so compare the received list against a fresh
+ * filter_base_build() of the sender's raw --filter texts. */
+static bool filter_rules_equal(const Config* a, const FilterRuleList* got) {
+ int count = a->filters ? a->filters->size : 0;
+ const char** texts = NULL;
+ if (count > 0) {
+ texts = calloc((size_t)count, sizeof(char*));
+ if (!texts)
+ return false;
+ for (int i = 0; i < count; i++)
+ texts[i] = (const char*)a->filters->items[i];
+ }
+ char err[160];
+ FilterRuleList* expected =
+ filter_base_build(texts, count, a->cvs_exclude, a->delete_excluded, err, sizeof(err));
+ free(texts);
+ if (!expected)
+ return false;
+ bool equal = true;
+ int want = expected->count;
+ int have = got ? got->count : 0;
+ if (want != have) {
+ equal = false;
+ } else {
+ for (int i = 0; i < want; i++) {
+ const FilterRule* x = expected->items[i];
+ const FilterRule* y = got->items[i];
+ if (x->action != y->action || x->sides != y->sides || x->anchored != y->anchored ||
+ x->dir_only != y->dir_only || x->negate != y->negate ||
+ !str_opt_equal(x->owner, y->owner) || !str_opt_equal(x->pattern, y->pattern)) {
+ equal = false;
+ break;
+ }
+ }
+ }
+ filter_rule_list_free(expected);
+ return equal;
+}
+
#define CONFIG_CMP_BOOL(a, b, name) ((a)->name == (b)->name)
#define CONFIG_CMP_INT(a, b, name) ((a)->name == (b)->name)
#define CONFIG_CMP_RAW(a, b, name) ((a)->name == (b)->name)
@@ -2615,6 +2698,7 @@ static bool basis_equal(const Config* a, const Config* b) {
#define CONFIG_CMP_COPY_AS_ID(a, b, name) (!(a)->copy_as_set || (a)->name == (b)->name)
#define CONFIG_CMP_BLOCK_SKIP_SUFFIXES(a, b, name) skip_suffixes_equal((a), (b))
#define CONFIG_CMP_BLOCK_BASIS(a, b, name) basis_equal((a), (b))
+#define CONFIG_CMP_BLOCK_PROTECT_RULES(a, b, name) filter_rules_equal((a), (b)->name)
#define CONFIG_CMP_BLOCK_IDMAP(a, b, name) \
idmap_equal((a)->name, (a)->name##_count, (b)->name, (b)->name##_count)
@@ -2784,6 +2868,7 @@ static void golden_config_populate(Config* c) {
c->skip_compress_suffixes[1] = str_dup(".xz");
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_COMPARE, "compare"), 0);
EXPECT_EQ_INT(config_basis_append(c, BASIS_DEST_LINK, "link"), 0);
+ c->verify_basis = true;
c->fuzzy = true;
c->checksum_algo = CHECKSUM_ALGO_MD5;
c->checksum_seed = 0x1122334455667788ULL;
@@ -2829,16 +2914,29 @@ static void golden_config_populate(Config* c) {
c->copy_as_set = true;
c->copy_as_uid = 111;
c->copy_as_gid = 222;
+ /* Compile-through delete-protection rules (protocol 2.28.0). The golden
+ * sender serializes its compiled base rules, so populate a diverse set that
+ * exercises both sides, negate, anchoring and dir-only. */
+ c->filters = array_list_create(free);
+ array_list_add(c->filters, str_dup("P *.log"));
+ array_list_add(c->filters, str_dup("+r **/*.txt"));
+ array_list_add(c->filters, str_dup("H,!secret"));
+ array_list_add(c->filters, str_dup("- /sub/dir/"));
}
-/* The pinned golden frame (protocol 2.26.0). The values below are the only
+/* The pinned golden frame (protocol 2.28.0). The values below are the only
* thing that ties the generated table to the historical wire format; update
* them ONLY with a PROTOCOL_VERSION bump and a documented reason. The 2.24.0
* delete-plan wave changed only the version string; 2.25.0 appended the
- * report_stats bool and 2.26.0 appended the compression_algo int. The
+ * report_stats bool, 2.26.0 appended the compression_algo int, 2.27.0 appended
+ * the report_deletes bool, and 2.28.0 changed only the version string and
+ * appended the receiver-side delete-protection rule block (the STATUS_STATS
+ * body also grew, but that is not part of this frame). Track 5a appends the
+ * FastSync-only verify_basis bool to the basis block WITHOUT a version bump
+ * (project decision), so the frame grew by one int to 886 bytes. The
* byte-exact values are recomputed for the merged layout. */
-#define GOLDEN_WIRE_LEN 705
-#define GOLDEN_WIRE_HASH 4673424031554175633ULL
+#define GOLDEN_WIRE_LEN 886
+#define GOLDEN_WIRE_HASH 5809509022716816757ULL
static unsigned long long fnv1a_64(const unsigned char* buf, size_t len) {
unsigned long long h = 1469598103934665603ULL;
@@ -2920,7 +3018,7 @@ static unsigned long long capture_wire_hash(const Config* cfg, size_t* out_len)
return h;
}
-/* Byte-for-byte wire compatibility guard (protocol 2.26.0). The expected hash
+/* Byte-for-byte wire compatibility guard (protocol 2.28.0). The expected hash
* pins the pre-X-macro byte stream; the refactor MUST NOT change it. */
static void test_config_wire_golden() {
if (is_running_under_valgrind())
@@ -2987,6 +3085,7 @@ static void test_config_wire_golden_receive() {
recv->groupmap[0].to_name != NULL && strcmp(recv->groupmap[0].to_name, "root") == 0;
ok = ok && recv->basis_count == 2 && recv->basis_dirs[0].type == BASIS_DEST_COMPARE &&
recv->basis_dirs[1].type == BASIS_DEST_LINK;
+ ok = ok && recv->verify_basis;
ok = ok && recv->module != NULL && strcmp(recv->module, "goldenmod") == 0;
ok = ok && recv->copy_as_set && recv->copy_as_uid == 111 && recv->copy_as_gid == 222;
}
@@ -3247,6 +3346,7 @@ void test_config() {
test_config_wire_golden_receive();
test_config_wire_receive_bounds();
test_config_receive_rejects_overcap_counts();
+ test_config_receive_rejects_bad_protect_rules();
test_config_wire_roundtrip_all_fields();
test_config_preserve_attribute_wire_roundtrip();
}
diff --git a/tests/test_delete_plan.c b/tests/test_delete_plan.c
new file mode 100644
index 0000000..9d67567
--- /dev/null
+++ b/tests/test_delete_plan.c
@@ -0,0 +1,274 @@
+#include "test_delete_plan.h"
+#include "charset.h"
+#include "config.h"
+#include "delete_plan.h"
+#include "protocol.h"
+#include "test_utils.h"
+#include "utils.h"
+#include
+#include
+#include
+#include
+#include
+#include
+#include
+
+/* Send one STATUS_DELETE_PLAN body (the leading status is consumed by the
+ * caller/receiver entry point) describing `dir` with no kept children. */
+static void send_plan_frame(int fd, const char* dir) {
+ EXPECT_TRUE(send_int(fd, 0)); /* has_config */
+ EXPECT_TRUE(send_int(fd, 1)); /* apply: a real plan */
+ EXPECT_TRUE(send_wire_str(fd, dir));
+ EXPECT_TRUE(send_int(fd, 0)); /* kept child dirs */
+ EXPECT_TRUE(send_int(fd, 0)); /* kept child files */
+}
+
+/* --delete-delay: a directory snapshotted into the plan that is refilled before
+ * the commit is re-scanned and removed recursively (rsync parity). Regression
+ * for the old single-unlink ENOTEMPTY path that left the directory behind. */
+static void test_delete_delay_refilled_dir_removed_recursively(void) {
+ char root[] = "/tmp/fastsync_dp_refill_XXXXXX";
+ EXPECT_TRUE(mkdtemp(root) != NULL);
+ char extra[1024];
+ snprintf(extra, sizeof(extra), "%s/extra", root);
+ EXPECT_EQ_INT(mkdir(extra, 0700), 0);
+
+ Config* config = config_create();
+ EXPECT_NOT_NULL(config);
+ config->receive_root_directory = str_dup(root);
+ config->use_delete = true;
+ config->delete_delay = true;
+
+ int p[2];
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+
+ DeletePlanSession* session = delete_plan_session_create(config);
+ EXPECT_NOT_NULL(session);
+ send_plan_frame(p[1], ".");
+ EXPECT_EQ_INT(delete_plan_session_receive(session, config, p[0]), 0);
+ /* The empty extra directory was snapshotted, not removed yet. */
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 0);
+
+ /* Refill the directory while the deferred commit is pending. */
+ char refill[1200];
+ snprintf(refill, sizeof(refill), "%s/new.txt", extra);
+ int fd = open(refill, O_WRONLY | O_CREAT | O_TRUNC, 0600);
+ EXPECT_TRUE(fd >= 0);
+ close(fd);
+
+ EXPECT_EQ_INT(delete_plan_session_commit(session, config), DELETE_COMMIT_OK);
+ /* The late content and the directory itself are both removed. */
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 2);
+ struct stat st;
+ EXPECT_TRUE(lstat(extra, &st) != 0);
+
+ delete_plan_session_destroy(session);
+ close(p[0]);
+ close(p[1]);
+ rmdir(root);
+ config_delete(config);
+}
+
+/* The complement: a deferred regular extra that DOES get removed is counted. */
+static void test_delete_delay_removed_file_counted(void) {
+ char root[] = "/tmp/fastsync_dp_file_XXXXXX";
+ EXPECT_TRUE(mkdtemp(root) != NULL);
+ char extra[1024];
+ snprintf(extra, sizeof(extra), "%s/extra.txt", root);
+ int fd = open(extra, O_WRONLY | O_CREAT | O_TRUNC, 0600);
+ EXPECT_TRUE(fd >= 0);
+ close(fd);
+
+ Config* config = config_create();
+ EXPECT_NOT_NULL(config);
+ config->receive_root_directory = str_dup(root);
+ config->use_delete = true;
+ config->delete_delay = true;
+
+ int p[2];
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+
+ DeletePlanSession* session = delete_plan_session_create(config);
+ EXPECT_NOT_NULL(session);
+ send_plan_frame(p[1], ".");
+ EXPECT_EQ_INT(delete_plan_session_receive(session, config, p[0]), 0);
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 0);
+ EXPECT_EQ_INT(delete_plan_session_commit(session, config), DELETE_COMMIT_OK);
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 1);
+ EXPECT_TRUE(lstat(extra, &(struct stat){0}) != 0);
+
+ delete_plan_session_destroy(session);
+ close(p[0]);
+ close(p[1]);
+ rmdir(root);
+ config_delete(config);
+}
+
+/* --max-delete is charged on actual removals, not at plan/snapshot time: after
+ * receiving the plans the budget is untouched, and only the commit removes up to
+ * the limit. */
+static void test_delete_delay_max_delete_bounds_actual(void) {
+ char root[] = "/tmp/fastsync_dp_max_XXXXXX";
+ EXPECT_TRUE(mkdtemp(root) != NULL);
+ for (int i = 0; i < 3; i++) {
+ char path[1024];
+ snprintf(path, sizeof(path), "%s/e%d.txt", root, i);
+ int fd = open(path, O_WRONLY | O_CREAT | O_TRUNC, 0600);
+ EXPECT_TRUE(fd >= 0);
+ close(fd);
+ }
+
+ Config* config = config_create();
+ EXPECT_NOT_NULL(config);
+ config->receive_root_directory = str_dup(root);
+ config->use_delete = true;
+ config->delete_delay = true;
+ config->max_delete = 1;
+
+ int p[2];
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+
+ DeletePlanSession* session = delete_plan_session_create(config);
+ EXPECT_NOT_NULL(session);
+ send_plan_frame(p[1], ".");
+ EXPECT_EQ_INT(delete_plan_session_receive(session, config, p[0]), 0);
+ /* Nothing removed yet, so the budget is not consumed at snapshot time. */
+ EXPECT_FALSE(delete_plan_session_limit_reached(session));
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 0);
+ EXPECT_EQ_INT(delete_plan_session_commit(session, config), DELETE_COMMIT_LIMIT_REACHED);
+ EXPECT_TRUE(delete_plan_session_limit_reached(session));
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 1);
+
+ delete_plan_session_destroy(session);
+ close(p[0]);
+ close(p[1]);
+ for (int i = 0; i < 3; i++) {
+ char path[1024];
+ snprintf(path, sizeof(path), "%s/e%d.txt", root, i);
+ unlink(path);
+ }
+ rmdir(root);
+ config_delete(config);
+}
+
+/* --max-delete is charged on ACTUAL removals: the refilled directory's late
+ * content is removed first (consuming the single budget slot), so the directory
+ * itself and the later extra are skipped, matching rsync. The two plans are
+ * sent as separate frames for "a" then "b", so the ordering that decides which
+ * entry gets the budget is deterministic (unlike readdir order). */
+static void test_delete_delay_actual_removal_charges_budget(void) {
+ char root[] = "/tmp/fastsync_dp_planbudget_XXXXXX";
+ EXPECT_TRUE(mkdtemp(root) != NULL);
+ char adir[1024], bdir[1024], xdir[1024], ydir[1024];
+ snprintf(adir, sizeof(adir), "%s/a", root);
+ snprintf(bdir, sizeof(bdir), "%s/b", root);
+ snprintf(xdir, sizeof(xdir), "%s/a/x", root);
+ snprintf(ydir, sizeof(ydir), "%s/b/y", root);
+ EXPECT_EQ_INT(mkdir(adir, 0700), 0);
+ EXPECT_EQ_INT(mkdir(bdir, 0700), 0);
+ EXPECT_EQ_INT(mkdir(xdir, 0700), 0);
+ EXPECT_EQ_INT(mkdir(ydir, 0700), 0);
+
+ Config* config = config_create();
+ EXPECT_NOT_NULL(config);
+ config->receive_root_directory = str_dup(root);
+ config->use_delete = true;
+ config->delete_delay = true;
+ config->max_delete = 1;
+
+ int p[2];
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+
+ DeletePlanSession* session = delete_plan_session_create(config);
+ EXPECT_NOT_NULL(session);
+ /* Both plan snapshots are taken; neither consumes budget yet. */
+ send_plan_frame(p[1], "a");
+ EXPECT_EQ_INT(delete_plan_session_receive(session, config, p[0]), 0);
+ EXPECT_FALSE(delete_plan_session_limit_reached(session));
+ send_plan_frame(p[1], "b");
+ EXPECT_EQ_INT(delete_plan_session_receive(session, config, p[0]), 0);
+ EXPECT_FALSE(delete_plan_session_limit_reached(session));
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 0);
+
+ /* Refill a/x after its plan: the recursive commit must remove this content. */
+ char refill[1200];
+ snprintf(refill, sizeof(refill), "%s/new.txt", xdir);
+ int fd = open(refill, O_WRONLY | O_CREAT | O_TRUNC, 0600);
+ EXPECT_TRUE(fd >= 0);
+ close(fd);
+
+ EXPECT_EQ_INT(delete_plan_session_commit(session, config), DELETE_COMMIT_LIMIT_REACHED);
+ /* The one budget slot removed the late content; the two directories survive. */
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 1);
+ EXPECT_TRUE(delete_plan_session_limit_reached(session));
+ struct stat st;
+ EXPECT_TRUE(lstat(refill, &st) != 0);
+ EXPECT_EQ_INT(lstat(xdir, &st), 0);
+ EXPECT_EQ_INT(lstat(ydir, &st), 0);
+
+ delete_plan_session_destroy(session);
+ close(p[0]);
+ close(p[1]);
+ rmdir(xdir);
+ rmdir(ydir);
+ rmdir(adir);
+ rmdir(bdir);
+ rmdir(root);
+ config_delete(config);
+}
+
+/* Send a config-only carrier frame (apply=false): the per-run config block with
+ * one --delete-missing-args exact path, and no directory walk. */
+static void send_config_only_frame(int fd, const char* missing_path) {
+ EXPECT_TRUE(send_int(fd, 1)); /* has_config */
+ EXPECT_TRUE(send_int(fd, 0)); /* protected prefixes */
+ EXPECT_TRUE(send_int(fd, 0)); /* size-skipped */
+ EXPECT_TRUE(send_int(fd, 1)); /* missing args */
+ EXPECT_TRUE(send_wire_str(fd, missing_path));
+ EXPECT_TRUE(send_int(fd, 0)); /* apply = false */
+ EXPECT_TRUE(send_wire_str(fd, "."));
+ EXPECT_TRUE(send_int(fd, 0));
+ EXPECT_TRUE(send_int(fd, 0));
+}
+
+/* The config-only carrier frame (apply=false) still applies the
+ * --delete-missing-args exact deletions even though it walks no directory. This
+ * is the fix for a --files-from list that synchronizes no directory. */
+static void test_config_only_frame_applies_missing_args(void) {
+ char root[] = "/tmp/fastsync_dp_cfgonly_XXXXXX";
+ EXPECT_TRUE(mkdtemp(root) != NULL);
+ char gone[1024];
+ snprintf(gone, sizeof(gone), "%s/gone.txt", root);
+ int fd = open(gone, O_WRONLY | O_CREAT | O_TRUNC, 0600);
+ EXPECT_TRUE(fd >= 0);
+ close(fd);
+
+ Config* config = config_create();
+ EXPECT_NOT_NULL(config);
+ config->receive_root_directory = str_dup(root);
+ config->delete_missing_args = true;
+
+ int p[2];
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+
+ DeletePlanSession* session = delete_plan_session_create(config);
+ EXPECT_NOT_NULL(session);
+ send_config_only_frame(p[1], "gone.txt");
+ EXPECT_EQ_INT(delete_plan_session_receive(session, config, p[0]), 0);
+ EXPECT_EQ_INT((int)delete_plan_session_deleted(session), 1);
+ EXPECT_TRUE(lstat(gone, &(struct stat){0}) != 0);
+
+ delete_plan_session_destroy(session);
+ close(p[0]);
+ close(p[1]);
+ rmdir(root);
+ config_delete(config);
+}
+
+void test_delete_plan(void) {
+ test_delete_delay_refilled_dir_removed_recursively();
+ test_delete_delay_removed_file_counted();
+ test_delete_delay_max_delete_bounds_actual();
+ test_delete_delay_actual_removal_charges_budget();
+ test_config_only_frame_applies_missing_args();
+}
diff --git a/tests/test_delete_plan.h b/tests/test_delete_plan.h
new file mode 100644
index 0000000..b17617f
--- /dev/null
+++ b/tests/test_delete_plan.h
@@ -0,0 +1,6 @@
+#ifndef TEST_DELETE_PLAN_H
+#define TEST_DELETE_PLAN_H
+
+void test_delete_plan(void);
+
+#endif
diff --git a/tests/test_file.c b/tests/test_file.c
index 3404b6f..611b4cc 100644
--- a/tests/test_file.c
+++ b/tests/test_file.c
@@ -6,6 +6,7 @@
#include "file_receive.h"
#include "data.h"
#include "config.h"
+#include "charset.h"
#include "utils.h"
#include "protocol.h"
#include "test_utils.h"
@@ -1807,6 +1808,159 @@ static void test_receive_incremental_check_empty_path() {
config_delete(cfg);
}
+/* Pure policy helpers behind the basis quick-check / --verify-basis decision. */
+static void test_file_basis_quick_match_decision() {
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ EXPECT_FALSE(file_basis_content_required(cfg));
+ cfg->verify_basis = true;
+ EXPECT_TRUE(file_basis_content_required(cfg));
+ cfg->verify_basis = false;
+
+ struct stat st;
+ memset(&st, 0, sizeof(st));
+ st.st_mtime = 1500000000;
+#ifdef __linux__
+ st.st_mtim.tv_nsec = 500;
+#endif
+ /* Equal size is required by the caller; this leg is the mtime / --size-only
+ rule. Equal mtime matches, a different mtime misses by default. */
+ EXPECT_TRUE(file_basis_quick_match(cfg, &st, 1500000000, 500));
+ EXPECT_FALSE(file_basis_quick_match(cfg, &st, 1500000001, 500));
+ cfg->size_only = true;
+ EXPECT_TRUE(file_basis_quick_match(cfg, &st, 1500000001, 500));
+ cfg->size_only = false;
+ cfg->modify_window = 2;
+ EXPECT_TRUE(file_basis_quick_match(cfg, &st, 1500000002, 500));
+ config_delete(cfg);
+}
+
+/* End-to-end handshake decision for a same-size, same-mtime, DIFFERENT-content
+ basis. Default (rsync parity): the metadata quick-check is trusted, the
+ receiver answers STATUS_OK and materializes the basis bytes. --verify-basis:
+ the whole-file digest is required, the basis is rejected and the receiver
+ asks for the source (STATUS_NEXT + full transfer). */
+static void test_receive_incremental_check_basis_quick_check_and_verify() {
+ const char* root = "test_basis_quick_root";
+ const char* basis_dir = "test_basis_quick_root/basis";
+ const char* basis_file = "test_basis_quick_root/basis/f.txt";
+ unlink(basis_file);
+ unlink("test_basis_quick_root/f.txt");
+ rmdir(basis_dir);
+ rmdir(root);
+ EXPECT_EQ_INT(mkdir(root, 0755), 0);
+ EXPECT_EQ_INT(mkdir(basis_dir, 0755), 0);
+
+ const char* src_bytes = "AAAA";
+ const unsigned long long size = 4;
+ const time_t mtime = 1500000000;
+ {
+ FILE* fh = fopen(basis_file, "wb");
+ EXPECT_NOT_NULL(fh);
+ // cppcheck-suppress knownConditionTrueFalse
+ if (fh) {
+ EXPECT_EQ_INT((int)fwrite("BBBB", 1, (size_t)size, fh), (int)size);
+ fclose(fh);
+ }
+ }
+ struct timespec ts[2] = {{mtime, 0}, {mtime, 0}};
+ EXPECT_EQ_INT(utimensat(AT_FDCWD, basis_file, ts, 0), 0);
+
+ char root_abs[PATH_MAX];
+ EXPECT_NOT_NULL(realpath(root, root_abs));
+ int root_fd = open(root_abs, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
+ EXPECT_TRUE(root_fd >= 0);
+ // cppcheck-suppress knownConditionTrueFalse
+ if (root_fd < 0) {
+ unlink(basis_file);
+ rmdir(basis_dir);
+ rmdir(root);
+ return;
+ }
+ EXPECT_TRUE(utils_set_authorized_root(root_fd, root_abs));
+
+ uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
+ size_t digest_len = 0;
+ EXPECT_TRUE(checksum_digest(CHECKSUM_ALGO_XXH64, 0, src_bytes, size, digest, sizeof(digest),
+ &digest_len));
+
+ /* Route protocol I/O through the explicit descriptors (a previous test group
+ may have left io_set_fds() bound to its own pipe). */
+ io_set_fds(-1, -1);
+ io_set_bwlimit(0);
+
+ for (int verify = 0; verify <= 1; verify++) {
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ cfg->receive_root_directory = str_dup(root_abs);
+ cfg->checksum = false;
+ cfg->checksum_algo = CHECKSUM_ALGO_XXH64;
+ cfg->checksum_seed = 0;
+ cfg->use_incremental = true;
+ cfg->use_delta = false;
+ cfg->use_metadata = false;
+ cfg->verify_basis = (verify != 0);
+ EXPECT_EQ_INT(config_basis_append(cfg, BASIS_DEST_LINK, "basis"), 0);
+
+ int p[2];
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+ EXPECT_TRUE(send_wire_str(p[1], "f.txt"));
+ unsigned long long check_size = size;
+ long long check_mtime = (long long)mtime;
+ long long check_mtime_nsec = 0;
+ EXPECT_TRUE(send_n_data(p[1], &check_size, sizeof(check_size)));
+ EXPECT_TRUE(send_n_data(p[1], &check_mtime, sizeof(check_mtime)));
+ EXPECT_TRUE(send_n_data(p[1], &check_mtime_nsec, sizeof(check_mtime_nsec)));
+ /* Only --verify-basis needs the digest (cfg->checksum is false) and the
+ pre-staged fallback full transfer the receiver will request. */
+ if (verify) {
+ uint8_t wire_len = (uint8_t)digest_len;
+ EXPECT_TRUE(send_n_data(p[1], &wire_len, sizeof(wire_len)));
+ EXPECT_TRUE(send_n_data(p[1], digest, digest_len));
+ char* payload_bytes = str_dup(src_bytes);
+ EXPECT_NOT_NULL(payload_bytes);
+ Data* payload = data_create(payload_bytes, size);
+ EXPECT_NOT_NULL(payload);
+ // cppcheck-suppress knownConditionTrueFalse
+ if (payload)
+ EXPECT_TRUE(send_data(p[1], payload));
+ data_destroy(payload);
+ }
+
+ bool skipped = false;
+ File* file = receive_incremental_check(p[0], cfg, &skipped);
+ EXPECT_NOT_NULL(file);
+ // cppcheck-suppress knownConditionTrueFalse
+ if (file) {
+ EXPECT_FALSE(skipped);
+ Status reply = STATUS_ERROR;
+ EXPECT_TRUE(receive_status(p[1], &reply));
+ if (verify) {
+ EXPECT_EQ_INT((int)reply, (int)STATUS_NEXT);
+ EXPECT_NOT_NULL(file->data->data);
+ EXPECT_TRUE(file->data->data != NULL && memcmp(file->data->data, src_bytes, size) == 0);
+ } else {
+ EXPECT_EQ_INT((int)reply, (int)STATUS_OK);
+ EXPECT_TRUE(file->skip);
+ /* The default quick-check hit materializes from the basis PATH at
+ install time (streaming), so no content is buffered on the File. */
+ EXPECT_NOT_NULL(file->basis_link);
+ EXPECT_NULL(file->data->data);
+ }
+ file_destroy(file);
+ }
+ close(p[0]);
+ close(p[1]);
+ config_delete(cfg);
+ }
+
+ utils_set_authorized_root(-1, NULL);
+ close(root_fd);
+ unlink(basis_file);
+ rmdir(basis_dir);
+ rmdir(root);
+}
+
/* -K/--keep-dirlinks secure open: with an authorized root, a destination path
* component that is a symlink to an IN-ROOT directory is used as that directory
* (its referent is opened through a relative O_NOFOLLOW walk from the root fd,
@@ -2140,6 +2294,8 @@ void test_file() {
test_dir_time_list();
test_dir_time_list_cap();
test_receive_incremental_check_empty_path();
+ test_file_basis_quick_match_decision();
+ test_receive_incremental_check_basis_quick_check_and_verify();
test_keep_dirlinks_secure_open();
test_inplace_overwrite_clears_special_mode_bits();
test_inplace_overwrite_metadata_strips_special_bits();
diff --git a/tests/test_format.c b/tests/test_format.c
index 9b8bd35..7343e26 100644
--- a/tests/test_format.c
+++ b/tests/test_format.c
@@ -86,9 +86,43 @@ static void test_dest_state_roundtrip() {
close(fds[1]);
}
+static void test_stats_roundtrip() {
+ /* STATUS_STATS grew from three counters (2.25.0) to eight (2.28.0); the codec
+ * must carry every field, including the receiver-observed literal/created
+ * counters, across the wire in order. */
+ int fds[2];
+ if (socketpair(AF_UNIX, SOCK_STREAM, 0, fds) != 0)
+ return;
+ ReceiverStats out;
+ memset(&out, 0, sizeof(out));
+ out.matched_data = 111111111ULL;
+ out.deleted_files = 7;
+ out.would_delete_count = 3;
+ out.literal_bytes = 222222222ULL;
+ out.created_reg = 5;
+ out.created_dir = 4;
+ out.created_link = 2;
+ out.created_special = 1;
+ ReceiverStats in;
+ memset(&in, 0, sizeof(in));
+ EXPECT_TRUE(format_stats_send(fds[0], &out));
+ EXPECT_TRUE(format_stats_receive(fds[1], &in));
+ EXPECT_TRUE(in.matched_data == out.matched_data);
+ EXPECT_TRUE(in.deleted_files == out.deleted_files);
+ EXPECT_TRUE(in.would_delete_count == out.would_delete_count);
+ EXPECT_TRUE(in.literal_bytes == out.literal_bytes);
+ EXPECT_TRUE(in.created_reg == out.created_reg);
+ EXPECT_TRUE(in.created_dir == out.created_dir);
+ EXPECT_TRUE(in.created_link == out.created_link);
+ EXPECT_TRUE(in.created_special == out.created_special);
+ close(fds[0]);
+ close(fds[1]);
+}
+
void test_format(void) {
test_human_size_decimal();
test_big_num_grouping();
test_datetime_format();
test_dest_state_roundtrip();
+ test_stats_roundtrip();
}
diff --git a/tests/test_fuzz_smoke.c b/tests/test_fuzz_smoke.c
index ca5d38c..25ae25d 100644
--- a/tests/test_fuzz_smoke.c
+++ b/tests/test_fuzz_smoke.c
@@ -18,10 +18,11 @@
/* P8 config-frame tail: super_mode (4) + copy-as presence (4) + uid (4) + gid (4). */
#define P8_TAIL_BYTES 16
-/* Bytes after the P8 tail: report_dest_info (4), report_stats (4, wire-stats
- * wave) and compression_algo (4, codec wave). The P8 fields sit this many
- * bytes before the end of the frame. */
-#define POST_P8_TAIL_BYTES 12
+/* Bytes after the P8 tail: report_dest_info (4), report_stats (4),
+ * report_deletes (4, --info=del wave), compression_algo (4, codec wave) and the
+ * receiver delete-protection count (4, protocol 2.28.0). The P8 fields sit
+ * this many bytes before the end of the frame. */
+#define POST_P8_TAIL_BYTES 20
/* Smoke test for chunk_deserialize fuzz target */
static void test_fuzz_chunk_deserialize() {
diff --git a/tests/test_iconv.c b/tests/test_iconv.c
index da31243..3e60457 100644
--- a/tests/test_iconv.c
+++ b/tests/test_iconv.c
@@ -148,10 +148,23 @@ static void test_iconv_wire_sender_converts_local_to_remote() {
charset_wire_free();
}
-static void test_iconv_wire_receiver_converts_remote_to_local() {
+/* rsync push parity: with no server --iconv the destination charset is the
+ * client spec's REMOTE half, so the receiver writes the wire bytes verbatim. */
+static void test_iconv_wire_receiver_default_writes_remote() {
EXPECT_TRUE(charset_wire_init_receiver("utf-8,iso-8859-1", NULL));
char* local = charset_wire_apply("caf\xe9");
EXPECT_NOT_NULL(local);
+ EXPECT_EQ_INT(strcmp(local, "caf\xe9"), 0);
+ free(local);
+ charset_wire_free();
+}
+
+/* A server that declares its own --iconv LOCAL converts wire(REMOTE) into that
+ * declared charset (the daemon "charset" analog). */
+static void test_iconv_wire_receiver_server_local_override() {
+ EXPECT_TRUE(charset_wire_init_receiver("utf-8,iso-8859-1", "utf-8"));
+ char* local = charset_wire_apply("caf\xe9");
+ EXPECT_NOT_NULL(local);
EXPECT_EQ_INT(strcmp(local, "caf\xc3\xa9"), 0);
free(local);
charset_wire_free();
@@ -179,7 +192,7 @@ static void test_iconv_wire_str_roundtrip() {
close(p[1]);
io_set_fds(p[0], p[0]);
charset_wire_free();
- charset_wire_init_receiver("utf-8,iso-8859-1", NULL);
+ charset_wire_init_receiver("utf-8,iso-8859-1", "utf-8");
char* got = receive_wire_str(p[0]);
bool ok = got != NULL && strcmp(got, "caf\xc3\xa9") == 0;
free(got);
@@ -211,7 +224,8 @@ void test_iconv() {
test_iconv_exact_fill_no_overflow();
test_iconv_growth_expanding_name();
test_iconv_wire_sender_converts_local_to_remote();
- test_iconv_wire_receiver_converts_remote_to_local();
+ test_iconv_wire_receiver_default_writes_remote();
+ test_iconv_wire_receiver_server_local_override();
test_iconv_wire_disabled_passthrough();
// This subtest forks to exercise the wire string handshake; the instrumented
// parent is too slow under valgrind for the child's blocking reads.
diff --git a/tests/test_multiprocessing.c b/tests/test_multiprocessing.c
index 52c3498..5aa3be3 100644
--- a/tests/test_multiprocessing.c
+++ b/tests/test_multiprocessing.c
@@ -103,6 +103,62 @@ static void test_sender_zero_capacity() {
config_delete(cfg);
}
+/* Regression (blocker): PipelineContextSender.delete_suppressed must be
+ initialized false. A garbage true silently suppresses the --delete keep-set
+ manifest under -m/--threads, so destination extras would never be removed. */
+static void test_sender_delete_suppressed_initialized() {
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ free(cfg->version);
+ cfg->version = str_dup(PROTOCOL_VERSION);
+ cfg->send_directory = str_dup("/src");
+ cfg->receive_root_directory = str_dup("/dst");
+
+ Queue* q1 = queue_create(5, NULL);
+ Queue* q2 = queue_create(5, NULL);
+ EXPECT_NOT_NULL(q1);
+ EXPECT_NOT_NULL(q2);
+ PipelineContextSender* ctx = pipeline_context_sender_create(cfg, q1, q2);
+ EXPECT_NOT_NULL(ctx);
+ EXPECT_FALSE(ctx->delete_suppressed);
+ pipeline_context_sender_destroy(ctx);
+ config_delete(cfg);
+}
+
+/* --info=del (report_deletes) is the only reason the receiver retains the
+ actually-removed paths: a plain --delete receiver must not allocate the list,
+ and with report_deletes armed the observer records into it. */
+static void test_receiver_deleted_paths_gated_by_report_deletes() {
+ Config* plain = config_create();
+ EXPECT_NOT_NULL(plain);
+ free(plain->version);
+ plain->version = str_dup(PROTOCOL_VERSION);
+ plain->receive_root_directory = str_dup("/dst");
+ plain->report_deletes = false;
+ Queue* q_plain = queue_create(5, file_destroy);
+ EXPECT_NOT_NULL(q_plain);
+ PipelineContextReceiver* ctx_plain = pipeline_context_receiver_create(plain, q_plain, -1, NULL);
+ EXPECT_NOT_NULL(ctx_plain);
+ EXPECT_NULL(ctx_plain->deleted_paths);
+ pipeline_context_receiver_destroy(ctx_plain);
+
+ Config* info = config_create();
+ EXPECT_NOT_NULL(info);
+ free(info->version);
+ info->version = str_dup(PROTOCOL_VERSION);
+ info->receive_root_directory = str_dup("/dst");
+ info->report_deletes = true;
+ Queue* q_info = queue_create(5, file_destroy);
+ EXPECT_NOT_NULL(q_info);
+ PipelineContextReceiver* ctx_info = pipeline_context_receiver_create(info, q_info, -1, NULL);
+ EXPECT_NOT_NULL(ctx_info);
+ EXPECT_NOT_NULL(ctx_info->deleted_paths);
+ receiver_record_deleted_path(ctx_info->deleted_paths, "d/old_extra");
+ EXPECT_EQ_INT(ctx_info->deleted_paths->size, 1);
+ EXPECT_EQ_STR((const char*)ctx_info->deleted_paths->items[0], "d/old_extra");
+ pipeline_context_receiver_destroy(ctx_info);
+}
+
/* Test receiver with zero file_descriptor */
static void test_receiver_fd_zero() {
Config* cfg = config_create();
@@ -244,6 +300,7 @@ static void test_write_thread_done() {
* and queue_destroy which would double-free since we created them
* in this test. Let me just free the context directly. */
array_list_delete(ctx->would_delete);
+ array_list_delete(ctx->deleted_paths);
mtx_destroy(&ctx->mutex);
cnd_destroy(&ctx->condition_not_full);
cnd_destroy(&ctx->condition_not_empty);
@@ -329,6 +386,7 @@ static void test_receiver_enqueue_byte_budget() {
/* Tear down: the second file is still queued and is freed by queue_destroy. */
array_list_delete(ctx->would_delete);
+ array_list_delete(ctx->deleted_paths);
mtx_destroy(&ctx->mutex);
cnd_destroy(&ctx->condition_not_full);
cnd_destroy(&ctx->condition_not_empty);
@@ -460,6 +518,8 @@ void test_multiprocessing() {
test_receiver_create_destroy();
test_sender_queue_capacities();
test_sender_zero_capacity();
+ test_sender_delete_suppressed_initialized();
+ test_receiver_deleted_paths_gated_by_report_deletes();
test_receiver_fd_zero();
if (!is_running_under_valgrind()) {
test_receive_thread_finished();
diff --git a/tests/test_scanner.c b/tests/test_scanner.c
index 933372d..a0a2ea1 100644
--- a/tests/test_scanner.c
+++ b/tests/test_scanner.c
@@ -1240,6 +1240,55 @@ static int collect_scan_info_parallel(ParallelScanner* scanner, const char* root
return failed ? -1 : count;
}
+/* The --progress paths-only pre-count relies on `list_dirs` emitting every
+ * directory (including empty ones) exactly once, alongside the files and
+ * symlinks the streaming scanner already emits. */
+static void test_scanner_list_dirs_counts_every_entry() {
+ const char* root = "test_scan_listdirs";
+ EXPECT_EQ_INT(mkdir(root, 0755), 0);
+ EXPECT_EQ_INT(mkdir("test_scan_listdirs/sub1", 0755), 0);
+ EXPECT_EQ_INT(mkdir("test_scan_listdirs/sub1/deep", 0755), 0);
+ EXPECT_EQ_INT(mkdir("test_scan_listdirs/sub2", 0755), 0);
+ EXPECT_EQ_INT(mkdir("test_scan_listdirs/emptydir", 0755), 0);
+ create_test_file("test_scan_listdirs/a.txt", "a");
+ create_test_file("test_scan_listdirs/sub1/c.txt", "c");
+ create_test_file("test_scan_listdirs/sub1/deep/d.txt", "d");
+ create_test_file("test_scan_listdirs/sub2/e.txt", "e");
+ EXPECT_EQ_INT(symlink("a.txt", "test_scan_listdirs/link1"), 0);
+
+ ScannerOptions options = {0};
+ options.list_dirs = true;
+ options.emit_empty_dirs = true;
+ options.follow_symlinks = true; /* -l/--links: carry symlinks, don't skip */
+ ScanInfo infos[16];
+ int count = collect_scan_info(root, &options, infos, 16);
+ EXPECT_EQ_INT(count, 9);
+ int dirs = 0;
+ for (int i = 0; i < count; i++) {
+ if (infos[i].is_dir)
+ dirs++;
+ }
+ EXPECT_EQ_INT(dirs, 4);
+ EXPECT_TRUE(scan_info_present(infos, count, "sub1", true, NULL));
+ EXPECT_TRUE(scan_info_present(infos, count, "sub1/deep", true, NULL));
+ EXPECT_TRUE(scan_info_present(infos, count, "sub2", true, NULL));
+ EXPECT_TRUE(scan_info_present(infos, count, "emptydir", true, NULL));
+ EXPECT_TRUE(scan_info_present(infos, count, "a.txt", false, ""));
+ EXPECT_TRUE(scan_info_present(infos, count, "sub1/c.txt", false, ""));
+ EXPECT_TRUE(scan_info_present(infos, count, "link1", false, ""));
+
+ unlink("test_scan_listdirs/a.txt");
+ unlink("test_scan_listdirs/sub1/c.txt");
+ unlink("test_scan_listdirs/sub1/deep/d.txt");
+ unlink("test_scan_listdirs/sub2/e.txt");
+ unlink("test_scan_listdirs/link1");
+ rmdir("test_scan_listdirs/sub1/deep");
+ rmdir("test_scan_listdirs/sub1");
+ rmdir("test_scan_listdirs/sub2");
+ rmdir("test_scan_listdirs/emptydir");
+ rmdir(root);
+}
+
/* -d without --files-from emits exactly the source-root directory (empty) and
* never descends. */
static void test_dirs_no_descent() {
@@ -1667,6 +1716,7 @@ void test_scanner() {
test_per_dir_filter_override(true);
test_dirs_no_descent();
test_dirs_files_from();
+ test_scanner_list_dirs_counts_every_entry();
test_files_from_relative_send_path();
test_scanner_captures_directory_times();
test_scanner_chunk_ownership();
diff --git a/tests/test_server.c b/tests/test_server.c
index 501c9a3..31c7eb5 100644
--- a/tests/test_server.c
+++ b/tests/test_server.c
@@ -3,6 +3,8 @@
#include "config.h"
#include "delta.h"
#include "file.h"
+#include "file_receive.h"
+#include "format.h"
#include "log.h"
#include "protocol.h"
#include "test_utils.h"
@@ -134,6 +136,124 @@ static void test_receive_files_single_file() {
}
}
+/* Protocol 2.28.0: a fresh single-file transfer over the wire reports the
+ * receiver-observed literal bytes and the created-regular counter through the
+ * terminal STATUS_STATS frame, and an update reports created_reg == 0. */
+static void test_receive_stats_frame_created_and_literal() {
+ const char* content = "stats frame content";
+ size_t len = strlen(content);
+ char root_template[] = "/tmp/fastsync_stats_XXXXXX";
+ char* root = mkdtemp(root_template);
+ EXPECT_NOT_NULL(root);
+
+ int p[2];
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+ io_set_fds(p[0], p[1]);
+ io_set_bwlimit(0);
+
+ Config* cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ free(cfg->version);
+ cfg->version = str_dup(PROTOCOL_VERSION);
+ cfg->send_directory = str_dup("/src");
+ cfg->receive_root_directory = str_dup(root);
+ cfg->save_to_disk = true;
+ cfg->report_stats = true;
+
+ pid_t pid = fork();
+ if (pid == 0) {
+ close(p[1]);
+ io_set_fds(p[0], p[0]);
+ int ret = receiver_receive_files(cfg, p[0]);
+ close(p[0]);
+ config_delete(cfg);
+ _exit(ret == 0 ? 0 : 1);
+ }
+ close(p[0]);
+ io_set_fds(p[1], p[1]);
+ send_status(p[1], STATUS_NEXT);
+ File* file = file_create("created.bin");
+ EXPECT_NOT_NULL(file);
+ file->data->data = malloc(len);
+ EXPECT_NOT_NULL(file->data->data);
+ memcpy(file->data->data, content, len);
+ file->data->size = len;
+ send_str(p[1], file->path);
+ send_data(p[1], file->data);
+ file_destroy(file);
+ send_status(p[1], STATUS_FINISHED);
+
+ Status status;
+ EXPECT_TRUE(receive_status(p[1], &status));
+ EXPECT_EQ_INT(status, STATUS_STATS);
+ ReceiverStats stats;
+ EXPECT_TRUE(format_stats_receive(p[1], &stats));
+ int would = 0;
+ EXPECT_TRUE(receive_int(p[1], &would));
+ EXPECT_EQ_INT(would, 0);
+ EXPECT_TRUE(stats.created_reg == 1);
+ EXPECT_TRUE(stats.created_dir == 0);
+ EXPECT_TRUE(stats.created_link == 0);
+ EXPECT_TRUE(stats.created_special == 0);
+ EXPECT_TRUE(stats.literal_bytes == (unsigned long long)len);
+ EXPECT_TRUE(stats.matched_data == 0);
+
+ Status final;
+ EXPECT_TRUE(receive_status(p[1], &final));
+ EXPECT_EQ_INT(final, STATUS_OK);
+ int wstatus;
+ waitpid(pid, &wstatus, 0);
+ close(p[1]);
+ config_delete(cfg);
+ EXPECT_TRUE(WIFEXITED(wstatus) && WEXITSTATUS(wstatus) == 0);
+
+ /* Second run against the now-existing destination: no created file. */
+ EXPECT_EQ_INT(socketpair(AF_UNIX, SOCK_STREAM, 0, p), 0);
+ io_set_fds(p[0], p[1]);
+ io_set_bwlimit(0);
+ cfg = config_create();
+ EXPECT_NOT_NULL(cfg);
+ free(cfg->version);
+ cfg->version = str_dup(PROTOCOL_VERSION);
+ cfg->send_directory = str_dup("/src");
+ cfg->receive_root_directory = str_dup(root);
+ cfg->save_to_disk = true;
+ cfg->report_stats = true;
+ pid = fork();
+ if (pid == 0) {
+ close(p[1]);
+ io_set_fds(p[0], p[0]);
+ int ret = receiver_receive_files(cfg, p[0]);
+ close(p[0]);
+ config_delete(cfg);
+ _exit(ret == 0 ? 0 : 1);
+ }
+ close(p[0]);
+ io_set_fds(p[1], p[1]);
+ send_status(p[1], STATUS_NEXT);
+ file = file_create("created.bin");
+ EXPECT_NOT_NULL(file);
+ file->data->data = malloc(len);
+ EXPECT_NOT_NULL(file->data->data);
+ memcpy(file->data->data, content, len);
+ file->data->size = len;
+ send_str(p[1], file->path);
+ send_data(p[1], file->data);
+ file_destroy(file);
+ send_status(p[1], STATUS_FINISHED);
+ EXPECT_TRUE(receive_status(p[1], &status));
+ EXPECT_EQ_INT(status, STATUS_STATS);
+ EXPECT_TRUE(format_stats_receive(p[1], &stats));
+ EXPECT_TRUE(receive_int(p[1], &would));
+ EXPECT_TRUE(stats.created_reg == 0);
+ EXPECT_TRUE(stats.literal_bytes == (unsigned long long)len);
+ EXPECT_TRUE(receive_status(p[1], &final));
+ waitpid(pid, &wstatus, 0);
+ close(p[1]);
+ config_delete(cfg);
+ EXPECT_TRUE(WIFEXITED(wstatus) && WEXITSTATUS(wstatus) == 0);
+}
+
/* Test receive_files with STATUS_ABORT */
static void test_receive_files_abort() {
Config* cfg = config_create();
@@ -943,14 +1063,18 @@ static void test_incremental_check_fifo_destination_does_not_hang() {
}
/* A server-contacting --dry-run with an alternate basis dir must never read or
- hash the basis file. An exact (size+mtime+content) basis match would
- otherwise let a client probe the basis bytes against its own supplied digest
- (a 1-bit content oracle). The dry-run decision is metadata-only, so even a
- byte-identical basis is reported as would-transfer, not a compare-dest skip. */
-static void test_incremental_check_dry_run_basis_does_not_read_content() {
+ hash the basis file. Under the default metadata quick-check a hit needs no
+ basis bytes, so a compare-dest match is reported as a skip (STATUS_OK) just
+ like a real run -- and still no content is read. Under --verify-basis a hit
+ would require hashing the basis against the client-supplied digest (a 1-bit
+ content oracle), which a dry-run must never do, so even a byte-identical
+ basis is reported as would-transfer. The destination is never materialized
+ in either arm. */
+static void run_dry_run_basis_check(bool verify, Status expected) {
Config* cfg = config_create();
EXPECT_NOT_NULL(cfg);
cfg->dry_run = true;
+ cfg->verify_basis = verify;
char* root = make_check_root("dryb");
EXPECT_NOT_NULL(root);
cfg->receive_root_directory = str_dup(root);
@@ -966,8 +1090,8 @@ static void test_incremental_check_dry_run_basis_does_not_read_content() {
EXPECT_EQ_INT(stat(basis_path, &bst), 0);
EXPECT_EQ_INT(config_basis_append(cfg, BASIS_DEST_COMPARE, "basis"), 0);
- /* The (correct) source digest for the basis bytes: an unfixed dry-run would
- read+hash the basis and treat this as an exact compare-dest hit. */
+ /* The (correct) source digest for the basis bytes: a buggy dry-run that read
+ and hashed the basis would treat this as an exact compare-dest hit. */
uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
size_t digest_len = 0;
EXPECT_TRUE(checksum_digest((ChecksumAlgo)cfg->checksum_algo, cfg->checksum_seed, content,
@@ -986,7 +1110,8 @@ static void test_incremental_check_dry_run_basis_does_not_read_content() {
bool skipped = false;
bool would_transfer = false;
File* file = receive_incremental_check_ex(p[0], cfg, &skipped, &would_transfer);
- bool ok = file == NULL && !skipped && would_transfer;
+ bool ok = file == NULL && skipped == (expected == STATUS_OK) &&
+ would_transfer == (expected != STATUS_OK);
file_destroy(file);
config_delete(cfg);
close(p[0]);
@@ -1004,13 +1129,16 @@ static void test_incremental_check_dry_run_basis_does_not_read_content() {
EXPECT_TRUE(send_n_data(p[1], &size, sizeof(size)));
EXPECT_TRUE(send_n_data(p[1], &mtime, sizeof(mtime)));
EXPECT_TRUE(send_n_data(p[1], &mtime_nsec, sizeof(mtime_nsec)));
- uint8_t wire_len = (uint8_t)digest_len;
- EXPECT_TRUE(send_n_data(p[1], &wire_len, sizeof(wire_len)));
- EXPECT_TRUE(send_n_data(p[1], digest, digest_len));
+ /* The digest is only on the wire when --checksum or --verify-basis needs it
+ (cfg->checksum is false here); the default quick-check arm sends none. */
+ if (verify) {
+ uint8_t wire_len = (uint8_t)digest_len;
+ EXPECT_TRUE(send_n_data(p[1], &wire_len, sizeof(wire_len)));
+ EXPECT_TRUE(send_n_data(p[1], digest, digest_len));
+ }
Status s;
EXPECT_TRUE(receive_status(p[1], &s));
- /* A skip here would mean the receiver read+hashed the basis file. */
- EXPECT_EQ_INT(s, STATUS_DRY_RUN_TRANSFER);
+ EXPECT_EQ_INT(s, expected);
int status;
waitpid(pid, &status, 0);
@@ -1028,6 +1156,11 @@ static void test_incremental_check_dry_run_basis_does_not_read_content() {
}
}
+static void test_incremental_check_dry_run_basis_does_not_read_content() {
+ run_dry_run_basis_check(false, STATUS_OK);
+ run_dry_run_basis_check(true, STATUS_DRY_RUN_TRANSFER);
+}
+
/* B1: a FIFO planted in a --link-dest basis directory must not block
* basis_open_regular() either; the basis match is simply declined. */
static void test_incremental_check_basis_fifo_does_not_hang() {
@@ -1158,6 +1291,7 @@ static void test_dry_run_delete_plan_commit_does_not_delete() {
EXPECT_TRUE(send_int(p[1], 0)); /* size-skipped prefixes */
EXPECT_TRUE(send_int(p[1], 1)); /* missing-args exact deletions */
EXPECT_TRUE(send_str(p[1], "victim.txt"));
+ EXPECT_TRUE(send_int(p[1], 1)); /* apply: a real plan */
EXPECT_TRUE(send_str(p[1], ".")); /* receive root plan */
EXPECT_TRUE(send_int(p[1], 0)); /* kept child directories */
EXPECT_TRUE(send_int(p[1], 0)); /* kept child files */
@@ -1183,6 +1317,7 @@ void test_server() {
if (!is_running_under_valgrind()) {
test_receive_files_finished();
test_receive_files_single_file();
+ test_receive_stats_frame_created_and_literal();
test_receive_files_abort();
test_receive_manifest_rejects_traversal();
test_receive_incremental_check_rejects_invalid_nanoseconds();
diff --git a/tests/test_shared_utils.c b/tests/test_shared_utils.c
index af4ab15..b822f1e 100644
--- a/tests/test_shared_utils.c
+++ b/tests/test_shared_utils.c
@@ -10,6 +10,7 @@
#include
#include
#include
+#include
#include
#include
#include
@@ -137,7 +138,7 @@ static void test_walker_removes_extras_keeps_manifest_and_protected() {
DeleteSkipEntry skip = {"prot", false};
size_t deleted = 0;
DeleteWalkResult result =
- delete_extras_limited(root, manifest, NULL, 100000, &skip, 1, &deleted, NULL);
+ delete_extras_limited(root, manifest, NULL, 100000, &skip, 1, NULL, &deleted, NULL);
EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK);
EXPECT_FALSE(file_exists(root, "a.txt"));
EXPECT_TRUE(file_exists(root, "keep.txt"));
@@ -172,7 +173,7 @@ static void test_walker_keeps_nested_manifest_dirs() {
EXPECT_NOT_NULL(manifest);
size_t deleted = 0;
DeleteWalkResult result =
- delete_extras_limited(root, manifest, NULL, 100000, NULL, 0, &deleted, NULL);
+ delete_extras_limited(root, manifest, NULL, 100000, NULL, 0, NULL, &deleted, NULL);
EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK);
EXPECT_FALSE(file_exists(root, "extra.txt"));
EXPECT_TRUE(file_exists(root, "keepdir/deep/keep.txt"));
@@ -203,7 +204,7 @@ static void test_walker_max_delete_partial_deletes_up_to_cap() {
size_t deleted = 999;
size_t skipped = 0;
DeleteWalkResult result =
- delete_extras_limited(root, manifest, NULL, 2, NULL, 0, &deleted, &skipped);
+ delete_extras_limited(root, manifest, NULL, 2, NULL, 0, NULL, &deleted, &skipped);
EXPECT_EQ_INT((int)result, (int)DELETE_WALK_LIMIT_REACHED);
EXPECT_EQ_INT((int)deleted, 2);
EXPECT_EQ_INT((int)skipped, 1);
@@ -224,7 +225,8 @@ static void test_walker_max_delete_exact_bound_deletes() {
ArrayList* manifest = make_manifest_strings(keeps, 0);
EXPECT_NOT_NULL(manifest);
size_t deleted = 0;
- DeleteWalkResult result = delete_extras_limited(root, manifest, NULL, 2, NULL, 0, &deleted, NULL);
+ DeleteWalkResult result =
+ delete_extras_limited(root, manifest, NULL, 2, NULL, 0, NULL, &deleted, NULL);
EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK);
EXPECT_EQ_INT((int)deleted, 2);
EXPECT_FALSE(file_exists(root, "a.txt"));
@@ -257,7 +259,7 @@ static void test_walker_removes_extraneous_symlinks() {
EXPECT_NOT_NULL(manifest);
size_t deleted = 0;
DeleteWalkResult result =
- delete_extras_limited(root, manifest, NULL, 100000, NULL, 0, &deleted, NULL);
+ delete_extras_limited(root, manifest, NULL, 100000, NULL, 0, NULL, &deleted, NULL);
EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK);
EXPECT_FALSE(file_exists(root, "link_file"));
EXPECT_FALSE(file_exists(root, "link_dir"));
@@ -293,7 +295,7 @@ static void test_walker_confines_deletion_to_synced_dirs() {
EXPECT_TRUE(array_list_add(dirs, str_dup("inscope")));
size_t deleted = 0;
DeleteWalkResult result =
- delete_extras_limited(root, manifest, dirs, 100000, NULL, 0, &deleted, NULL);
+ delete_extras_limited(root, manifest, dirs, 100000, NULL, 0, NULL, &deleted, NULL);
EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK);
EXPECT_TRUE(file_exists(root, "rootextra.txt"));
EXPECT_FALSE(file_exists(root, "inscope/extra.txt"));
@@ -323,6 +325,45 @@ static void test_walker_unlimited_deletes_all() {
free(root);
}
+/* Receiver-side filter protection (protocol 2.28.0): a compiled protect rule
+ shields a DESTINATION-ONLY extra that never appeared on the sender, a risk
+ rule cancels an earlier/later protect (first match wins), and a dir-only
+ protect rule shields the whole subtree. */
+static void test_walker_protect_rules_shield_dest_only() {
+ char* root = make_walk_root("protectrules");
+ EXPECT_NOT_NULL(root);
+ EXPECT_TRUE(write_file_at(root, "keep.txt", "kept"));
+ EXPECT_TRUE(write_file_at(root, "extra.log", "risk cancels protect"));
+ EXPECT_TRUE(write_file_at(root, "safe.log", "protected"));
+ EXPECT_TRUE(write_file_at(root, "other.txt", "deleted"));
+ EXPECT_EQ_INT(make_subdir(root, "prot"), 0);
+ EXPECT_TRUE(write_file_at(root, "prot/inside.txt", "shielded subtree"));
+ EXPECT_TRUE(write_file_at(root, "prot/deep.log", "shielded subtree"));
+
+ const char* keeps[] = {"keep.txt"};
+ ArrayList* manifest = make_manifest_strings(keeps, 1);
+ EXPECT_NOT_NULL(manifest);
+ const char* rule_text[] = {"R extra.log", "P *.log", "P prot/"};
+ char err[160];
+ FilterRuleList* rules = filter_base_build(rule_text, 3, false, false, err, sizeof(err));
+ EXPECT_NOT_NULL(rules);
+ size_t deleted = 0;
+ DeleteWalkResult result =
+ delete_extras_limited(root, manifest, NULL, 100000, NULL, 0, rules, &deleted, NULL);
+ EXPECT_EQ_INT((int)result, (int)DELETE_WALK_OK);
+ EXPECT_TRUE(file_exists(root, "keep.txt"));
+ EXPECT_FALSE(file_exists(root, "extra.log")); /* risk wins the first match */
+ EXPECT_TRUE(file_exists(root, "safe.log")); /* protect shields the extra */
+ EXPECT_FALSE(file_exists(root, "other.txt"));
+ EXPECT_TRUE(dir_exists(root, "prot"));
+ EXPECT_TRUE(file_exists(root, "prot/inside.txt"));
+ EXPECT_TRUE(file_exists(root, "prot/deep.log"));
+ filter_rule_list_free(rules);
+ array_list_delete(manifest);
+ remove_walk_tree(root);
+ free(root);
+}
+
typedef struct {
bool eight_bit_output;
const char* expected;
@@ -578,7 +619,43 @@ static void test_getdelim_bounded() {
fclose(fp);
}
+static int test_env_resolver(const char* name) {
+ if (strcasecmp(name, "alpha") == 0)
+ return 10;
+ if (strcasecmp(name, "beta") == 0)
+ return 20;
+ return -1;
+}
+
+static void test_env_choice_first_parsing() {
+ const char* var = "FASTSYNC_TEST_CHOICE_LIST";
+ bool specified = true;
+ unsetenv(var);
+ EXPECT_EQ_INT(env_choice_first(var, test_env_resolver, &specified), -1);
+ EXPECT_FALSE(specified);
+
+ /* Unknown entries are skipped, case-insensitive, first supported wins. */
+ setenv(var, "bogus BETA alpha", 1);
+ EXPECT_EQ_INT(env_choice_first(var, test_env_resolver, &specified), 20);
+ EXPECT_TRUE(specified);
+
+ /* The client half ends at '&'. */
+ setenv(var, "alpha & beta", 1);
+ EXPECT_EQ_INT(env_choice_first(var, test_env_resolver, &specified), 10);
+
+ /* Blank means "unspecified"; all-unknown means "specified but no match". */
+ setenv(var, " ", 1);
+ EXPECT_EQ_INT(env_choice_first(var, test_env_resolver, &specified), -1);
+ EXPECT_FALSE(specified);
+ setenv(var, "nope,alpha", 1);
+ EXPECT_EQ_INT(env_choice_first(var, test_env_resolver, &specified), -1);
+ EXPECT_TRUE(specified);
+
+ unsetenv(var);
+}
+
void test_shared_utils() {
+ test_env_choice_first_parsing();
test_path_index_bounded();
test_path_index_semantics();
test_getdelim_bounded();
@@ -589,6 +666,7 @@ void test_shared_utils() {
test_walker_removes_extraneous_symlinks();
test_walker_confines_deletion_to_synced_dirs();
test_walker_unlimited_deletes_all();
+ test_walker_protect_rules_shield_dest_only();
test_loopback_helpers();
test_fd_peer_ip();
diff --git a/tests/test_xattr.c b/tests/test_xattr.c
index ab3981d..4e268e5 100644
--- a/tests/test_xattr.c
+++ b/tests/test_xattr.c
@@ -210,7 +210,8 @@ static void test_xattr_receive_drops_acl_without_preserve_acls() {
/* MINOR-2: a --link-dest / -H copy fallback (linkat refused) must still apply
* the per-file xattrs and --fake-super stat. A DIRECTORY basis forces linkat
- * to fail with EPERM, exercising the byte-copy fallback deterministically.
+ * to fail with EPERM, exercising the byte-copy fallback deterministically (the
+ * basis is not a regular file, so the fallback uses the caller's bytes).
* Guarded on filesystem xattr support. */
static void test_link_copy_fallback_preserves_xattrs() {
const char* dest = "test_link_xattr_dest.txt";