Compare commits
137
Commits
v2.21.0
...
9b05972375
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
9b05972375 | ||
|
|
d119f35066 | ||
|
|
00829fd265 | ||
|
|
ff261bc38a | ||
|
|
b235721f8b | ||
|
|
79a28cdb96 | ||
|
|
402cae80ad | ||
|
|
9691dba6f0 | ||
|
|
558782d339 | ||
|
|
5597e74f6a | ||
|
|
ee6523afac | ||
|
|
4163caa1d3 | ||
|
|
10159dc120 | ||
|
|
cd7b96d0bb | ||
|
|
f6f49d536e | ||
|
|
38d304103c | ||
|
|
4b09213b88 | ||
|
|
36fd0774e8 | ||
|
|
711b7e50b3 | ||
|
|
67076bf218 | ||
|
|
8ec8cb7203 | ||
|
|
82959395fb | ||
|
|
c9f94ea46e | ||
|
|
eff9852038 | ||
|
|
c80098623f | ||
|
|
6119e1e75c | ||
|
|
25062352f6 | ||
|
|
00d628d4ba | ||
|
|
3545d88905 | ||
|
|
596a039454 | ||
|
|
ae037cc27b | ||
|
|
3d0672a721 | ||
|
|
b82aab72c5 | ||
|
|
8e7764007d | ||
|
|
1042d15db7 | ||
|
|
cbe37a77dd | ||
|
|
a5d45ef266 | ||
|
|
a960391b34 | ||
|
|
134f8b027a | ||
|
|
eb7e3fd2e0 | ||
|
|
6a129b54d4 | ||
|
|
c7b2c7eb2b | ||
|
|
e77dfbec70 | ||
|
|
f3ac4df4d0 | ||
|
|
91197fd7cf | ||
|
|
13708352ec | ||
|
|
d162d93570 | ||
|
|
9fa1696eff | ||
|
|
b705fb807f | ||
|
|
cee281ff55 | ||
|
|
5c509831b8 | ||
|
|
ee57aeea4a | ||
|
|
803c1d3385 | ||
|
|
b8ec62beef | ||
|
|
59bfd32b9e | ||
|
|
3432a33d9a | ||
|
|
a1b081d328 | ||
|
|
bbecff9c04 | ||
|
|
c1553bd5d6 | ||
|
|
1a550bda24 | ||
|
|
902f86192d | ||
|
|
6a40ac86e5 | ||
|
|
410ba6e992 | ||
|
|
6c6f02e5dd | ||
|
|
d9006d1fda | ||
|
|
36375010d3 | ||
|
|
5efa0dba7c | ||
|
|
e32733fbf6 | ||
|
|
b02799327d | ||
|
|
946aa934cc | ||
|
|
125921c11b | ||
|
|
9dd5288381 | ||
|
|
3f2c74dd9e | ||
|
|
51e41dee2a | ||
|
|
dbf1b39d47 | ||
|
|
845f20a28d | ||
|
|
a5083776da | ||
|
|
76a81f1684 | ||
|
|
24b81c7e5a | ||
|
|
0e33f84f38 | ||
|
|
c7ac039523 | ||
|
|
e771cc9da6 | ||
|
|
7f9f82a068 | ||
|
|
695b5c8c25 | ||
|
|
5b2188d909 | ||
|
|
e5da916d54 | ||
|
|
de640bba1b | ||
|
|
f0f5719be0 | ||
|
|
6200b298ac | ||
|
|
394a9aae22 | ||
|
|
12d4af1b89 | ||
|
|
9d7c55d3c0 | ||
|
|
5a104bfd88 | ||
|
|
2ada8f9ad5 | ||
|
|
1493f1806d | ||
|
|
ea28e25535 | ||
|
|
d6295d62ce | ||
|
|
a9f416ce44 | ||
|
|
448edc0432 | ||
|
|
4a7703b06a | ||
|
|
6fc297544e | ||
|
|
583d3c8edb | ||
|
|
9883757190 | ||
|
|
a690109975 | ||
|
|
36d4d0e43e | ||
|
|
a0b9d9794b | ||
|
|
3e9f70d9ba | ||
|
|
2b5aaef409 | ||
|
|
478f80be9f | ||
|
|
e674b25213 | ||
|
|
1116da9f64 | ||
|
|
684153350a | ||
|
|
ec206b02d0 | ||
|
|
88bdfeeb58 | ||
|
|
3f5b0250f4 | ||
|
|
c41bfb2cdb | ||
|
|
1b2632f968 | ||
|
|
7dbca70a4b | ||
|
|
1a053f06e5 | ||
|
|
58b3a33e82 | ||
|
|
17b0632098 | ||
|
|
376e6500ab | ||
|
|
82a1d5e240 | ||
|
|
3eec5a4cc3 | ||
|
|
6144c7fc7f | ||
|
|
ea4ab661b4 | ||
|
|
84827ca617 | ||
|
|
23552e823d | ||
|
|
d1a567f7e3 | ||
|
|
93c1fc3c1f | ||
|
|
4815b1b281 | ||
|
|
ad7bc3348b | ||
|
|
34970b961c | ||
|
|
b3f7cad4db | ||
|
|
cd8a84c0a2 | ||
|
|
09c384d7d0 | ||
|
|
81ad313ee5 |
@@ -9,7 +9,7 @@ on:
|
||||
jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
# suite) run on merge to dev/main, so PR CI stays well under ~3 minutes.
|
||||
build-and-test:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
steps:
|
||||
- name: Checkout
|
||||
@@ -49,9 +49,50 @@ jobs:
|
||||
if: github.event_name == 'push'
|
||||
run: python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" --durations=25 --tb=short -q
|
||||
|
||||
# Differential rsync-parity gate: runs real rsync 3.4.1 and FastSync over the
|
||||
# same corpora and compares destinations + normalized output. The fast subset
|
||||
# guards the ✅ surface on every PR; the full set (with FASTSYNC_PARITY_STRICT
|
||||
# so a fixed caveat must be removed from the allowlist) burns the documented
|
||||
# ⚠️/❌ residuals down on push. See tests/integration/README.md.
|
||||
parity-fast:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'pull_request'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build -j$(nproc)
|
||||
|
||||
- name: Differential parity (fast subset)
|
||||
run: python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci -q
|
||||
|
||||
parity-full:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build -j$(nproc)
|
||||
|
||||
- name: Differential parity (full set)
|
||||
run: FASTSYNC_PARITY_STRICT=1 python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity -q
|
||||
|
||||
sanitizers:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
strategy:
|
||||
@@ -72,7 +113,7 @@ jobs:
|
||||
|
||||
fuzz-build:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
@@ -94,7 +135,7 @@ jobs:
|
||||
|
||||
coverage:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
@@ -118,7 +159,7 @@ jobs:
|
||||
|
||||
valgrind:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
|
||||
@@ -55,6 +55,16 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
file(GLOB SHARED_SRCS "src/shared/*.c")
|
||||
@@ -64,15 +74,15 @@ file(GLOB TEST_SRCS "tests/*.c")
|
||||
|
||||
add_executable(server ${SERVER_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(server PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(client PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} src/client/scanner.c)
|
||||
target_include_directories(tests PRIVATE tests src/shared src/server src/client)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
```
|
||||
|
||||
### Source Layout
|
||||
@@ -85,17 +95,23 @@ tests/integration/ — Python pytest integration tests
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)`
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)` (default compression codec)
|
||||
- **zlib** — found via `find_library(ZLIB_LIBRARY z)` (the `zlib`/`zlibx` codecs)
|
||||
- **lz4** — found via `find_library(LZ4_LIBRARY lz4)` (the `lz4` codec)
|
||||
- **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport)
|
||||
- **xxHash** — fetched via `FetchContent` from the upstream repository (delta transfer hashing, v0.8.3)
|
||||
- **pthreads** — found via `find_package(Threads REQUIRED)`
|
||||
- **C11 standard** — required
|
||||
- **CMake 3.22+** — minimum version
|
||||
|
||||
The codec matrix (protocol 2.26.0) uses zstd/zlib/lz4 for compression and
|
||||
xxHash/OpenSSL for the `xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1` checksums
|
||||
(`none` needs no library); both codec families are negotiated per transfer.
|
||||
|
||||
## Conventions
|
||||
|
||||
- Use `file(GLOB ...)` for source collection (existing pattern).
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `${ZLIB_LIBRARY}`, `${LZ4_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- Include directories: `src/shared`, `src/server`, `src/client`, `tests` (for test target).
|
||||
- Sanitizer support: pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (live option in CMakeLists.txt).
|
||||
- Build with `cmake -B build -S . && cmake --build build -j$(nproc)`.
|
||||
|
||||
@@ -116,7 +116,7 @@ The project uses Gitea Actions. Key jobs:
|
||||
jobs:
|
||||
new-job:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Configure
|
||||
|
||||
@@ -16,7 +16,7 @@ Ask the user or determine from context:
|
||||
- **Minor** (x.Y.0) — new features, backward compatible
|
||||
- **Patch** (x.y.Z) — bug fixes, no protocol changes
|
||||
|
||||
Current version: `PROTOCOL_VERSION "2.21.0"` in `src/shared/config.h`
|
||||
Current version: `PROTOCOL_VERSION "2.26.0"` in `src/shared/config.h`
|
||||
|
||||
### Step 2: Check Protocol Version
|
||||
|
||||
|
||||
@@ -4,18 +4,19 @@ FastSync is a high-performance file synchronization system written in C11. It su
|
||||
|
||||
## Dependency installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v10`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, and Node.js.
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v11`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, Node.js, plus `rsync` 3.4.1 (with zstd/xxhash/lz4), `acl` and `attr` (setfacl/getfacl, setfattr/getfattr) for drop-in parity tests.
|
||||
|
||||
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
|
||||
|
||||
```bash
|
||||
# Use the prebuilt CI image directly (faster, guaranteed CI parity)
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v10 fastsync-ci:local
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v11 fastsync-ci:local
|
||||
|
||||
# Or build the image from the repo-root Dockerfile
|
||||
# (Note: the prebuilt :v10 image reflects the previous Dockerfile state;
|
||||
# rebuild from source to pick up any newly added packages like lcov/valgrind.)
|
||||
# (Note: the prebuilt :v11 image is built from the current Dockerfile and
|
||||
# includes rsync 3.4.1 plus acl/attr; rebuild from source after changing
|
||||
# the Dockerfile.)
|
||||
docker build -t fastsync-ci:local .
|
||||
|
||||
# Build, run unit tests, and run integration tests inside the container
|
||||
@@ -55,6 +56,21 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
./build/tests # unit tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # full integration suite (CI excludes env-dependent privilege tests)
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset only
|
||||
|
||||
# Differential rsync-parity gate (real rsync 3.4.1 vs FastSync)
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci # fast PR subset
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity # full set
|
||||
```
|
||||
|
||||
See `tests/integration/README.md` for the differential parity gate and its
|
||||
`parity_caveats.py` allowlist (the residual burn-down mechanism).
|
||||
|
||||
Unit tests under valgrind must set `FASTSYNC_UNDER_VALGRIND=1` (CI does): the
|
||||
tests use it to skip fork-based tests, because valgrind 3.22 does not expose
|
||||
`vgpreload` in the guest's `/proc/self/maps`.
|
||||
|
||||
```bash
|
||||
FASTSYNC_UNDER_VALGRIND=1 valgrind --leak-check=full --show-leak-kinds=definite --error-exitcode=1 ./build/tests
|
||||
```
|
||||
|
||||
## CI Workflow — Waiting for Results
|
||||
@@ -66,14 +82,14 @@ When running the CI workflow via `tea` (the task execution agent), always set a
|
||||
### If lint (clang-format) fails
|
||||
Run clang-format in the CI Docker image to match the exact CI version:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i'
|
||||
```
|
||||
|
||||
### If cppcheck fails
|
||||
Fix reported issues locally, then verify with:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/'
|
||||
```
|
||||
|
||||
|
||||
+307
@@ -4,6 +4,313 @@ All notable changes to FastSync are documented here. Versions match
|
||||
`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must
|
||||
run the same version because the handshake is strict.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0).
|
||||
`RSYNC_COMPAT.md` moves from **116 ✅ / 14 ⚠️ / 27 ❌** to
|
||||
**120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows.
|
||||
|
||||
### Changed
|
||||
|
||||
- **rsync-exact traversal order.** The sequential scanner now walks each
|
||||
directory's entries in rsync 3.4.1's flist order (non-directories ascending,
|
||||
then directories ascending, depth-first), so `--info=name`, the
|
||||
`--delete-during`/`--delete-delay`/`-n` would-delete order and the partial
|
||||
`--max-delete` survivor set match rsync byte-for-byte. `--threads` has no
|
||||
rsync analogue and stays unordered.
|
||||
- **Delete timing.** The complete `--delete-during`/`--delete-delay`
|
||||
per-directory plan set is transmitted before the first data frame, so a
|
||||
mid-transfer abort has already removed every planned extra like rsync's
|
||||
generator; `-d/--dirs` uses per-directory plans (shielded untraversed
|
||||
subdirectories) instead of the end-of-transfer commit. `-n`, `--delete`,
|
||||
`--del`/`--delete-during` and `--delete-delay` are now ✅ Parity.
|
||||
- **Basis directories.** A relative `--compare-dest`/`--copy-dest`/`--link-dest`
|
||||
DIR resolves against the destination directory with the transfer-relative
|
||||
name appended, exactly like rsync 3.4.1.
|
||||
- **`-y`/`--fuzzy`.** The candidate search no longer inherits the ordinary delta
|
||||
engine's 16 KiB minimum or 10× size-ratio bound, so an oversized or
|
||||
sub-16-KiB sibling is reused exactly as rsync reuses it.
|
||||
- `--info=mount` prints rsync's mount-point skip line (repeated `-xx` drops the
|
||||
mount-point directory); `--info=stats` enables the `--stats` block; `-x` is
|
||||
repeatable. `--stats` counts traversed directories for the `Number of files`
|
||||
breakdown under a plain `-r` scan. `--debug` emits real output for
|
||||
`flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send`.
|
||||
|
||||
### Known residuals
|
||||
|
||||
- `--progress` and `--info` still need a receiver→sender event channel for the
|
||||
root `./` line, ancestor-directory suppression, receiver-side `skip`/`backup`
|
||||
wording, and symlink/empty-directory quick-checks.
|
||||
- `--delete-before`'s phase-0 late-file divergence remains (rsync's pre-scan
|
||||
fixes the file list before the data pass).
|
||||
- A single file larger than 256 MiB cannot be streamed in the default path
|
||||
(a general whole-file limit, not basis-specific).
|
||||
- `--stats` byte totals and `--msgs2stderr` stay documented divergences.
|
||||
|
||||
## [2.28.0] - 2026-09-20
|
||||
|
||||
The rsync-parity cycle. `PROTOCOL_VERSION` moves `2.26.0 → 2.27.0 → 2.28.0`;
|
||||
client and server must run the same version (the handshake is strict). See
|
||||
`RSYNC_COMPAT.md` for the per-option matrix, now **116 ✅ / 14 ⚠️ / 27 ❌** of
|
||||
157 rows.
|
||||
|
||||
### Added
|
||||
|
||||
- **Differential rsync 3.4.1 parity gate** (`tests/integration/
|
||||
test_differential_parity.py`, `parity_harness.py`, `parity_caveats.py`): runs
|
||||
real `rsync` and FastSync over generated corpora and diffs the destination
|
||||
tree, normalized stdout and exit code. A fast subset runs on pull requests and
|
||||
the full strict set on push; the residual allowlist is empty.
|
||||
- FastSync-only long option **`--verify-basis`**: require a
|
||||
`--compare-dest`/`--copy-dest`/`--link-dest` hit to match the source by
|
||||
whole-file digest instead of trusting the size+mtime quick-check.
|
||||
- FastSync-only long option **`--delete-commit`** (implies `--delete`): the old
|
||||
atomic late whole-tree commit.
|
||||
- `--bwlimit` now parses rsync's units exactly and paces like rsync's leaky
|
||||
bucket; `--ignore-errors` reproduces rsync's skip-unreadable-subdir and
|
||||
IO-error-suppressed deletion (exit 23).
|
||||
- `--info=name/flist/del/remove/nonreg/progress` emit rsync's line format,
|
||||
including real-run `deleting`/`*deleting` lines carried by a new
|
||||
`report_deletes` wire bool.
|
||||
- Receiver-observed `--stats` counters: `Number of created files` now carries
|
||||
rsync's `(reg/dir/link/special)` breakdown and `Literal data` is exact for a
|
||||
delta transfer (extended `STATUS_STATS`).
|
||||
- `--progress` uses an opt-in paths-only pre-count so the `to-chk` denominator
|
||||
counts every entry like rsync, and emits per-directory/symlink/special names.
|
||||
- Receiver-side `protect`/`risk` filter engine (new bounded filter-rule wire
|
||||
block): `--filter='P ...'` now shields a destination-only entry like rsync.
|
||||
- `auto` for `--compress-choice`/`--checksum-choice` honors
|
||||
`RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST`, and per-codec compression-level
|
||||
defaults match rsync.
|
||||
- Empty source directories are recreated recursively; `-R --no-implied-dirs
|
||||
--files-from` places listed files under missing implied parents; `--iconv`
|
||||
matches rsync's push direction; `--delete-delay` reports actual removals and
|
||||
recursively removes a refilled deferred directory.
|
||||
|
||||
### Changed
|
||||
|
||||
- **`--delete` now defaults to delete-during (rsync `--del`) timing.** With no
|
||||
explicit timing flag, a plain `--delete` removes each directory's extras as
|
||||
that directory is processed instead of committing one whole-tree deletion only
|
||||
after the entire transfer succeeds. This matches rsync, frees destination
|
||||
space progressively, and avoids the whole-old+new-tree peak that could
|
||||
`ENOSPC` a tight destination. The client maps the default onto the existing
|
||||
`delete_during` wire boolean, so `PROTOCOL_VERSION` stays `2.28.0`.
|
||||
- Basis directories (`--compare-dest`/`--copy-dest`/`--link-dest`) now default
|
||||
to rsync's metadata quick-check (equal size and mtime; `--size-only` drops the
|
||||
mtime leg) instead of FastSync's historical always-verify content hash.
|
||||
`--copy-dest` re-applies the source attributes, and basis materialization is
|
||||
streamed so the 256 MiB whole-file cap no longer applies to a basis hit.
|
||||
- The per-directory `STATUS_DELETE_PLAN` frame gained a one-int `apply` flag:
|
||||
the one-shot per-run config block (protected prefixes, size-pruned mirrors,
|
||||
`--delete-missing-args` exact paths) is now always transmitted first on a
|
||||
config-only carrier (`apply=false`), fixing a latent bug where a
|
||||
`--delete-missing-args` run whose `--files-from` list synchronized no directory
|
||||
never sent its exact deletions.
|
||||
|
||||
### Notes
|
||||
|
||||
- `--delete`/`--delete-during` remain caveats for the mid-transfer abort
|
||||
boundary (rsync's generator removes all planned extras ahead of its throttled
|
||||
sender; FastSync removes only reached directories — final trees agree).
|
||||
`--delete-before`, `--progress`, `--stats`, `--fuzzy` and the basis rows keep
|
||||
their documented residuals in `RSYNC_COMPAT.md`; `--filter` and
|
||||
`--delete-excluded` are now parity, including protection of a destination-only
|
||||
excluded entry under default `--delete`.
|
||||
|
||||
### Migration
|
||||
|
||||
- Scripts that relied on plain `--delete` deleting nothing until the transfer
|
||||
fully succeeded must pass **`--delete-commit`** (or `--delete-after`) to keep
|
||||
that behavior. Plain `--delete` now removes reached directories' extras during
|
||||
the transfer, exactly like rsync's default; on a completed run the final tree
|
||||
is unchanged.
|
||||
- Deployments that relied on FastSync's stricter basis verification should pass
|
||||
**`--verify-basis`**; the default now trusts the size+mtime quick-check like
|
||||
rsync.
|
||||
|
||||
## [2.26.0] - 2026-09-17
|
||||
|
||||
|
||||
### Added
|
||||
|
||||
- **Parity-completion wave.** Closed the remaining rsync-parity gaps against
|
||||
rsync 3.4.1 and reclassified the inherently non-rsync rows. It moved the wire
|
||||
protocol three times (`2.23.0 → 2.24.0 → 2.25.0 → 2.26.0`).
|
||||
- **Delete timing (2.24.0):** per-directory delete plans
|
||||
(`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`. An interrupted
|
||||
during-transfer has already removed the reached directories' extras, while a
|
||||
delayed transfer commits per directory only after the whole transfer
|
||||
succeeds (a late-created extra survives `--delete-delay` but not
|
||||
`--delete-after`). `-R --delete` is scoped to the transferred prefix; empty
|
||||
in-scope source directories survive; dry-run never deletes.
|
||||
- **Wire stats (2.25.0):** `STATUS_STATS` carries the receiver counters
|
||||
(matched data, deleted files) and the dry-run would-delete list. `--stats`
|
||||
prints rsync's protocol-independent lines; `--progress`/`-P` print per-file
|
||||
blocks; `--out-format` gains `%b` (wire bytes), `%c` (block-sum bytes) and
|
||||
`%C` (whole-file digest); `-n --delete` prints escaped `*deleting` lines in
|
||||
the sequential and `--threads` paths.
|
||||
- **Codecs (2.26.0):** `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/
|
||||
`none` checksums, with rsync-style `auto` negotiation (default `xxh128` +
|
||||
`zstd`) and exit-4 rejection of unknown names; the resolved `compression_algo`
|
||||
crosses the wire.
|
||||
- General `-R`/`--relative` (including the `/./` cut) and `--no-implied-dirs`;
|
||||
one-level `-d`/`--dirs` listing for `dir`, `dir/` and `.`; the full filter
|
||||
grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` and
|
||||
modifiers) with `-f` bound to `--filter`; a single `-F` transfers
|
||||
`.rsync-filter` and `-FF` excludes it.
|
||||
- Receiver-side `--chown`/`--usermap`/`--groupmap` TO-name resolution; absolute
|
||||
basis directories and a `--link-dest` relink of an up-to-date destination;
|
||||
a receiver-side `--ignore-existing` short-circuit before any payload;
|
||||
`--preallocate` now wins over `--sparse` via `fallocate(2)`.
|
||||
- Client quick wins: `--iconv=.`/`-`/`--no-iconv`, a lone `-h` prints help, an
|
||||
empty `--files-from` succeeds (exit 0), a broken referent under
|
||||
`-L`/`--copy-unsafe-links` exits 23, the full `--info`/`--debug`
|
||||
vocabularies, and the aliases `--ignore-non-existing`, `--protect-args`,
|
||||
`--msgs2stderr`.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.23.0 → 2.24.0` (delete plans),
|
||||
`2.24.0 → 2.25.0` (`STATUS_STATS` + `report_stats`), and
|
||||
`2.25.0 → 2.26.0` (codec negotiation + `md4`/`sha1`/`none`).
|
||||
- `--checksum-choice`/`--cc` now accepts `md4`, `sha1`, `none` and the two-name
|
||||
form; the negotiated whole-file default is `xxh128`.
|
||||
- `--compress-choice`/`--zc` now accepts `lz4`, `zlib`, `zlibx`.
|
||||
- `RSYNC_COMPAT.md` reclassifies the matrix: 9 already-parity rows to ✅, 17
|
||||
inherently non-rsync rows to ❌ (native daemon config/auth, batch, privileged
|
||||
xattr namespaces, and the safe-subset device/privilege flags), and the genuine
|
||||
fixes to ✅; new rows cover `--bwlimit`, `--partial`, `--partial-dir`,
|
||||
`--no-whole-file`, `--inc-recursive`/`--no-inc-recursive`, `--protect-args`
|
||||
and `--msgs2stderr`.
|
||||
- The client `--help` `--max-delete` text now describes the implemented partial
|
||||
semantics (delete up to N, skip the rest, exit 25).
|
||||
|
||||
### Notes
|
||||
|
||||
- Remaining documented divergences include the `--stats` per-type file-count
|
||||
breakdown, `%b`/`%c` being FastSync wire counts, `-n --delete` line ordering,
|
||||
the default `--delete` timing (delete-after, not rsync's delete-during),
|
||||
destination-only exclude protection (still sender-derived), `--temp-dir`
|
||||
absolute paths, basis-dir attribute re-application and the 256 MiB whole-file
|
||||
cap, `--fuzzy` tie-breaking, `--bwlimit=0`/decimal rates, `zlibx`==`zlib`, and
|
||||
recursive empty-directory creation.
|
||||
- Build: adds zlib and lz4 as link dependencies.
|
||||
|
||||
## [2.23.0] - 2026-09-16
|
||||
|
||||
### Added
|
||||
|
||||
- **Rsync-parity wave.** Closed the remaining CLI, filesystem, ownership,
|
||||
deletion, and output gaps against rsync 3.4.1.
|
||||
- Short options `-r` (`--recursive`), `-b` (`--backup`), `-L`
|
||||
(`--copy-links`), and `-B` (`--block-size`/`--delta-block`); rsync
|
||||
short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached/inline values
|
||||
(`--opt=value`, `-B1000`, `-essh`, `-MOPT`). A value that starts with `-`
|
||||
is not mistaken for a cluster.
|
||||
- `-c`/`--checksum` now implies the incremental checksum quick-check (and,
|
||||
like rsync, does not imply `-t`).
|
||||
- `--checksum-choice`/`--cc` accepts `xxh64`/`xxhash`/`xxh3`/`xxh128`/`md5`/
|
||||
`auto` and rejects `md4`/`sha1`/`none` and the two-name form by name;
|
||||
`--checksum-seed=0` (the default) is randomized per transfer and the chosen
|
||||
seed is sent to the receiver.
|
||||
- `--compress-choice`/`--zc` accepts `zstd`/`none`/`auto` and rejects
|
||||
`lz4`/`zlib`/`zlibx` by name; `--skip-compress` defaults to rsync 3.4.1's
|
||||
built-in suffix list; `--no-whole-file` is accepted.
|
||||
- `--timeout` defaults to 0 (disabled) and `--contimeout` to 60 s (both `0`
|
||||
disables), matching rsync; `--max-alloc=0` means no local limit.
|
||||
- `--temp-dir` is confined to the receive root (absolute/`..` rejected by the
|
||||
receiver) and an `EXDEV` install falls back to a non-atomic copy.
|
||||
- `--numeric-ids` is documented as a mapping modifier only;
|
||||
`--usermap`/`--groupmap` support inclusive `LOW-HIGH` ranges, `*`,
|
||||
empty-`FROM` (unnamed ids), and receiver-resolved `TO` names; `--chown`
|
||||
conflicts with a map on the same side are rejected.
|
||||
- `--fake-super` records the *resolved* owner (never a real chown) and replays
|
||||
mode/time; directory ownership and directory xattrs/ACLs are preserved.
|
||||
- `-l`/`--links` stores symlink targets verbatim (absolute and `..`-bearing
|
||||
included), matching rsync; `--safe-links`/`--copy-unsafe-links` are applied
|
||||
sender-side and `--munge-links` uses rsync's `/rsyncd-munged/` marker;
|
||||
`--trust-sender` no longer affects symlink targets.
|
||||
- `--specials` recreates unix sockets with `mknod(S_IFSOCK)` (so `-D` covers
|
||||
the full rsync node set).
|
||||
- Deletion: the manifest carries a synchronized-directory section so
|
||||
`--files-from` subsets no longer delete untransmitted paths;
|
||||
`--delete-excluded` leaves size-pruned mirrors protected; extraneous
|
||||
destination symlinks are unlinked (never followed); `--max-delete=N` is
|
||||
partial (delete up to N, skip the rest, exit 25) and `--delete-missing-args`
|
||||
removals draw from the same budget; `--force` is honored during
|
||||
`--delay-updates` publication.
|
||||
- `-x`/`--one-file-system` emits the mount-point directory entry; the
|
||||
`--include`/`--exclude` layers are an ordered first-match rule list.
|
||||
- `--chmod` is a faithful port of rsync 3.4.1 (numeric/symbolic, `D`/`F`/`X`,
|
||||
`s`/`t`, append semantics, no `-p` implication, no sanitization).
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.22.0 → 2.23.0`: the delete manifest gains a
|
||||
synchronized-directory section and the terminal status gains
|
||||
`STATUS_DELETE_LIMIT` (client exit 25 on a `--max-delete`-capped commit).
|
||||
- **The 2.22.0 mode-masking divergence is removed.** Under `-p` the source mode
|
||||
is copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky;
|
||||
`--chmod` no longer implies `-p`. New files without `-p` still use
|
||||
`source_mode & ~umask` when metadata is present (else `0644`), and new
|
||||
directories without `-p` still use the `0755` creation default.
|
||||
- `--protocol=NUM` accepts only the current `2.23.0` version string.
|
||||
|
||||
### Notes
|
||||
|
||||
- The rsync-compatibility matrix (`RSYNC_COMPAT.md`) now classifies every row
|
||||
as **parity**, **caveat** (works with a documented divergence), or
|
||||
**divergent** (not supported/no-op/impossible), replacing the previous
|
||||
misleading "N implemented / 0 divergence" summary. Durable documented
|
||||
divergences remain: receiver-side symlink target containment is not enforced
|
||||
by default (verbatim storage is rsync parity; use `--safe-links`),
|
||||
`--temp-dir` rejects absolute/foreign-filesystem paths, `--copy-devices`
|
||||
reads a bounded `st_size`, a broken referent under `--copy-links` exits 0,
|
||||
new directories without `-p` use `0755`, `--stats` receiver-only counters are
|
||||
0, and `--password-file`/`--early-input`/`--hash-credentials`/`--iterations`
|
||||
and the batch format are FastSync-native.
|
||||
|
||||
## [2.22.0] - 2026-09-15
|
||||
|
||||
### Added
|
||||
|
||||
- **Per-attribute metadata preservation (protocol 2.22.0).** The former single
|
||||
metadata bundle is split into four independent, rsync-compatible flags:
|
||||
`-p/--perms`, `-t/--times`, `-o/--owner`, and `-g/--group`, each applied
|
||||
independently on the receiver, with negations `--no-perms`/`--no-times`/
|
||||
`--no-owner`/`--no-group` (short `--no-p`/`--no-t`/`--no-o`/`--no-g`) and
|
||||
`--no-preserve` clearing all four. `-a/--archive` is now full rsync
|
||||
`-rlptgoD` (owner and group included; their application stays
|
||||
privilege-gated). `-A/--acls` and `--chmod` imply `-p`, `-X/--xattrs` does
|
||||
not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply
|
||||
`-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the
|
||||
user explicitly negated them.
|
||||
- Receiver applies directory modes under `-p` (at the end of the transfer,
|
||||
alongside the deferred directory times) and symlink mode under `-p`; `-O`
|
||||
suppresses directory times only.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.21.0 → 2.22.0`: the binary config frame gains
|
||||
four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/
|
||||
`preserve_group`) after `omit_link_times`. The fixed-width `FileMetadata`
|
||||
layout is unchanged; the receiver derives the metadata-frame gate
|
||||
(`use_metadata`) from the four attributes.
|
||||
|
||||
### Notes
|
||||
|
||||
- Documented divergences from rsync: a client-supplied mode never grants
|
||||
group/other write (`S_IWGRP|S_IWOTH` are stripped for files, directories,
|
||||
symlinks, and specials; rsync's `-p` preserves them exactly); a brand-new file
|
||||
without `-p` gets `source_mode & ~umask` (sanitized) when metadata is present,
|
||||
else the historical fixed `0644`; `--chmod` implies `-p` (rsync does not);
|
||||
`-o`/`-g` map by name on the receiver with a raw-numeric fallback (only
|
||||
numeric ids cross the wire); and a daemon module without `client owner = yes`
|
||||
does not refuse a plain `-a`/`-o`/`-g` but forces super-user activities off,
|
||||
applies no ownership, and logs a warning (explicit `--chown`/`--usermap`/
|
||||
`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused).
|
||||
|
||||
## [2.21.0] - 2026-09-14
|
||||
|
||||
### Added
|
||||
|
||||
+19
-5
@@ -1,6 +1,6 @@
|
||||
cmake_minimum_required(VERSION 3.22)
|
||||
|
||||
project(FastFileTransfer VERSION 2.21.0)
|
||||
project(FastFileTransfer VERSION 2.28.0)
|
||||
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_C_STANDARD 11)
|
||||
@@ -68,6 +68,16 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
# --- Explicit source lists ---
|
||||
@@ -89,6 +99,7 @@ set(SHARED_SRCS
|
||||
src/shared/daemon_limits.c
|
||||
src/shared/data.c
|
||||
src/shared/delay_updates.c
|
||||
src/shared/delete_plan.c
|
||||
src/shared/delta.c
|
||||
src/shared/file.c
|
||||
src/shared/file_list.c
|
||||
@@ -96,6 +107,7 @@ set(SHARED_SRCS
|
||||
src/shared/file_send.c
|
||||
src/shared/file_store.c
|
||||
src/shared/filter.c
|
||||
src/shared/format.c
|
||||
src/shared/hardlink.c
|
||||
src/shared/identity.c
|
||||
src/shared/log.c
|
||||
@@ -134,8 +146,8 @@ set(CLIENT_MAIN_SRCS src/client/client_cli.c)
|
||||
# --- Library targets ---
|
||||
add_library(fastsync_shared STATIC ${SHARED_SRCS})
|
||||
target_include_directories(fastsync_shared PUBLIC src/shared)
|
||||
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL
|
||||
OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_library(fastsync_client_core STATIC ${CLIENT_CORE_SRCS})
|
||||
target_include_directories(fastsync_client_core PUBLIC src/client)
|
||||
@@ -209,10 +221,12 @@ set(TEST_SRCS
|
||||
tests/test_daemon_limits.c
|
||||
tests/test_data.c
|
||||
tests/test_delay_updates.c
|
||||
tests/test_delete_plan.c
|
||||
tests/test_delta.c
|
||||
tests/test_file.c
|
||||
tests/test_file_list.c
|
||||
tests/test_file_sendfile.c
|
||||
tests/test_format.c
|
||||
tests/test_fuzz_smoke.c
|
||||
tests/test_glob.c
|
||||
tests/test_hardlink.c
|
||||
@@ -273,7 +287,7 @@ if(ENABLE_FUZZ)
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE tests src/shared src/server)
|
||||
target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer)
|
||||
target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL
|
||||
OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
+15
-1
@@ -2,8 +2,22 @@ FROM ubuntu:24.04
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \
|
||||
python3 python3-pip python3-venv openssl openssh-client \
|
||||
lcov valgrind clang libclang-rt-18-dev && \
|
||||
lcov valgrind clang libclang-rt-18-dev \
|
||||
acl attr zlib1g-dev liblz4-dev libxxhash-dev && \
|
||||
pip3 install --break-system-packages pytest pytest-xdist && \
|
||||
curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \
|
||||
apt-get install -y --no-install-recommends nodejs && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# rsync is used as the reference implementation for drop-in parity tests.
|
||||
# Ubuntu 24.04 ships 3.2.7, so build the pinned 3.4.1 reference from source.
|
||||
ARG RSYNC_VERSION=3.4.1
|
||||
ARG RSYNC_SHA256=2924bcb3a1ed8b551fc101f740b9f0fe0a202b115027647cf69850d65fd88c52
|
||||
RUN curl -fsSL "https://download.samba.org/pub/rsync/src/rsync-${RSYNC_VERSION}.tar.gz" -o /tmp/rsync.tar.gz && \
|
||||
echo "${RSYNC_SHA256} /tmp/rsync.tar.gz" | sha256sum -c - && \
|
||||
tar -xzf /tmp/rsync.tar.gz -C /tmp && \
|
||||
cd "/tmp/rsync-${RSYNC_VERSION}" && \
|
||||
./configure --enable-zstd --enable-xxhash --enable-lz4 && \
|
||||
make -j"$(nproc)" && \
|
||||
make install && \
|
||||
rm -rf "/tmp/rsync-${RSYNC_VERSION}" /tmp/rsync.tar.gz
|
||||
|
||||
+225
@@ -0,0 +1,225 @@
|
||||
# FastSync — Session Handoff (2026-09-20)
|
||||
|
||||
## Current status
|
||||
- **Release `v2.28.0`** is tagged and merged to `main` (PR #304, `b4d54504`).
|
||||
`dev` is at `558782d` (the incremental-check flake fix).
|
||||
- **`PROTOCOL_VERSION` = `"2.28.0"`** (`src/shared/config.h`); CMake
|
||||
`project(FastFileTransfer VERSION 2.28.0)`.
|
||||
- **Parity cycle 2.29 on branch `feat/parity-2.29`** (from `dev` @ `558782d`),
|
||||
no wire change. It closes the scanner-order, delete-timing, relative-basis and
|
||||
fuzzy-eligibility residuals and improves the `--info`/`--stats`/`--debug`
|
||||
partials. Parity matrix: **120 ✅ / 10 ⚠️ / 27 ❌ = 157** (was 116/14/27).
|
||||
Remaining ⚠️ rows: `--info`, `--debug`, `--msgs2stderr`, `--stats`,
|
||||
`--progress`, `--delete-before`, `--compare-dest`/`--copy-dest`/`--link-dest`
|
||||
(over-256-MiB basis MISS), `-y`/`--fuzzy` (256 MiB buffer cap).
|
||||
- **Deferred (needs a wire bump):** the `--progress`/`--info` receiver→sender
|
||||
event channel (root `./` line, ancestor suppression, `skip`/`backup` echo,
|
||||
symlink/empty-dir quick-check); `--delete-before` phase-0 keep-set; and the
|
||||
general >256 MiB single-file streaming limit (B4).
|
||||
- Feature branch `feat/parity-2.29`; integration PR to `dev` pending.
|
||||
|
||||
|
||||
## What landed this session
|
||||
1. **Wave 8 (refactors):** Config X-macro wire table; single-owner `authorized_root`;
|
||||
daemon per-module/per-host caps + cross-process auth lockout (`daemon_limits.[ch]`);
|
||||
`Data` charge returns to its owning `ProtocolSession`.
|
||||
2. **Wave 9 (protocol 2.21.0):** optional `STATUS_ERROR_DETAIL` rejection reasons;
|
||||
server-contacting `--dry-run` (`STATUS_DRY_RUN_TRANSFER`, receiver mutates nothing).
|
||||
3. **Security wave:** ran 5 parallel audits (wire parsing; daemon/transport/TLS/auth;
|
||||
receiver confinement; client/CLI/SSH; crypto/memory/limits). Fixed all HIGH and the
|
||||
confirmed MEDIUMs:
|
||||
- SSH `-o ProxyCommand=…` argument injection (RCE) — reject leading `-`, insert `--`.
|
||||
- Truncated zstd frame infinite CPU loop (remote DoS).
|
||||
- FIFO receiver opens lacked `O_NONBLOCK` (indefinite hang).
|
||||
- `--inplace` could write a FIFO/device (bypass of `--write-devices` gate).
|
||||
- `--force` not gated by server `--allow-delete`.
|
||||
- Privileged standalone server defaulted super activities on; added `--allow-super`
|
||||
(never honored with `--stdio`).
|
||||
- `--dry-run` content/hash oracle on `read only`/basis files removed.
|
||||
- Empty `hosts allow`/`deny`/`auth users` now rejected.
|
||||
- TLS: AEAD-only 1.2 + server preference, TOCTOU-safe key load, IP-SAN verify,
|
||||
CN-truncation guard. Glob backtracking bounded; line reads bounded; ACL xattrs
|
||||
gated on `--acls`; decompression/chunk memory charged; pre-auth `basis_count`
|
||||
NULL-deref fixed.
|
||||
4. **Tooling:** benchmark accuracy (data mix, verification, percentiles, `tc`,
|
||||
`build-bench/`, `--warm` mode); `shell.nix` full toolchain and no build-on-entry;
|
||||
docs state push-only / remote-source unsupported.
|
||||
5. **Preserve-attribute split (protocol 2.22.0)** landed on `feat/preserve-attr-split`: per-attribute `-p/-t/-o/-g` + `--no-*` negations, `-a` = `-rlptgoD`, and the 2.21.0 → 2.22.0 wire bump.
|
||||
6. **Rsync-parity wave (protocol 2.23.0)** on `feat/rsync-parity`: rsync short options/clustering/attached values (`-r`/`-b`/`-L`/`-B`, `-av`, `-aAX`, `-B1000`, `-essh`, `-MOPT`), `-c` checksum quick-check, `--checksum-choice`/`--compress-choice` validation and seed randomization, rsync timeout/max-alloc defaults, temp-dir confinement + `EXDEV` fallback, ownership/mapping parity (numeric-ids modifier, map ranges/`*`/empty-FROM, `--chown`+map conflicts, fake-super resolved-owner record), verbatim symlink storage with rsync `--safe-links`/`--munge-links`, socket recreation under `--specials`, `--chmod` 3.4.1 semantics, and delete scoping + `--max-delete` partial/exit-25. Wire: appended delete-manifest synchronized-directory section and `STATUS_DELETE_LIMIT`.
|
||||
7. **Parity-completion wave (protocol 2.24.0 → 2.26.0)** on `feat/parity-completion`: per-directory delete plans (`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`; receiver `STATUS_STATS` counters feeding `--stats`/`--progress` and `--out-format %b/%c/%C`, plus `-n --delete` lines; `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/`none` checksums with `auto` negotiation (default `xxh128`/`zstd`); general `-R`/`--no-implied-dirs`/`-d`; the full filter grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` + modifiers) and corrected `-F`/`-FF`; receiver-side `--chown`/map TO-name resolution; absolute basis dirs + `--link-dest` relink; receiver-side `--ignore-existing` short-circuit; `--preallocate` over `--sparse` via `fallocate(2)`; `--iconv=.`/`-`/`--no-iconv`; lone `-h` help; aliases `--ignore-non-existing`/`--protect-args`/`--msgs2stderr`; and the full `--info`/`--debug` vocabulary. `RSYNC_COMPAT.md` reclassifies the matrix to 106 ✅ / 27 ⚠️ / 23 ❌; the later rsync-parity-stats pass (`fix/parity-stats`) moves it to 107 ✅ / 25 ⚠️ / 24 ❌ (see item 8).
|
||||
8. **rsync-parity-stats pass** on `fix/parity-stats` (no wire change, `PROTOCOL_VERSION` stays `2.26.0`): `--delete-delay` now reports only entries it actually removes, while the `--max-delete` budget is charged at plan/snapshot time (`planned`, via `defer_add`) to bound the deferred list (a refilled deferred directory that survives `ENOTEMPTY` is not reported but still consumes budget); `--stats` gained the `(reg/dir/link/special)` `Number of files` breakdown and now counts only regular files actually stored for `Number of regular files transferred`/transferred size/literal data (up-to-date re-runs report 0); `Total file size` includes symlink target lengths; `--progress` prints the leading `./` root line and counts it in `to-chk` so a single-file transfer matches rsync; and `%C` uses the selected transfer checksum with `checksum_digest_file` supporting md4/sha1/none, byte-identical to rsync for every algorithm. `--out-format` reclassified ❌ (`%b`/delta-`%c` are protocol-specific). Differential + regression tests added; full suite + ASan + clang-format + cppcheck clean.
|
||||
9. **Option-parity wave (protocol 2.26.0 → 2.27.0, on `fix/parity-options`):**
|
||||
`--bwlimit` now ports rsync 3.4.1's units/quantization and paces like its
|
||||
leaky bucket; `--ignore-errors` reproduces rsync's default (an I/O error
|
||||
skips deletion unless the flag is set; the readable tree still transfers and
|
||||
the run exits 23) across every delete timing; the `--info` categories with a
|
||||
FastSync event (`name`/`flist`/`del`/`remove`/`nonreg`/`progress`) emit
|
||||
rsync's line format, with real-run `deleting`/`*deleting` lines carried over
|
||||
the new trailing config bool `report_deletes` (golden wire updated by
|
||||
`tests/test_config.c`). Two residuals were reclassified **divergent**: `-M`
|
||||
over daemon/TCP (no argv channel in FastSync's binary config handshake;
|
||||
rsync-daemon differential pins the rsync behavior) and receiver-side
|
||||
`protect`/`risk` re-derivation for destination-only entries (would need a
|
||||
receiver filter engine; differential pins the divergence — **reversed by
|
||||
track 4a below**, which adds that engine). The options pass
|
||||
stands at **110 ✅ / 21 ⚠️ / 26 ❌**. New `tests/integration/test_option_parity.py`
|
||||
holds the rsync differentials (bwlimit parse+rate, info lines, real-setpriv
|
||||
`--ignore-errors`, rsync-daemon `-M`, filter-protect pin).
|
||||
|
||||
10. **rsync-parity-fs pass** on `fix/parity-fs` (no wire change of its own; integrated
|
||||
on top of the 2.27.0 options wave): recursive transfers now recreate empty source directories (and
|
||||
`-m/--prune-empty-dirs` still suppresses them), a directory entry replaces a
|
||||
blocking destination regular file, and `-R --no-implied-dirs --files-from`
|
||||
places a listed file under a missing implied parent with default attributes
|
||||
instead of refusing (real rsync 3.4.1 parity, differential-tested). `--iconv`
|
||||
now reproduces rsync's push direction (destination charset = the spec's REMOTE
|
||||
half; a server `--iconv` overrides), and `-T/--temp-dir` relative semantics are
|
||||
confirmed identical while the absolute-path confinement is a deliberate
|
||||
divergence. The basis-dir options, `--delay-updates` and `--dry-run` were
|
||||
reclassified to ❌ after a differential test reproduced each exact residual
|
||||
(basis content verification, fixed staging-name collision, and dry-run
|
||||
would-delete over-report). `--fuzzy` was also reclassified to ❌ (deterministic
|
||||
heuristic with a 10× size window, not rsync's matcher), but its residual is the
|
||||
candidate-selection heuristic itself: the final tree is byte-exact by design, so
|
||||
it is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync
|
||||
differential. (Track 5b later found the name heuristic is rsync's own and moved
|
||||
the row ❌ → ⚠️, leaving only the narrower delta size window; see entry 15.) The parity-review pass then moved `--delete-delay` to ⚠️ (the
|
||||
plan-time `--max-delete` charge and non-recursive deferred removal differ from
|
||||
rsync when a snapshotted entry fails removal). Differential-gate allowlist
|
||||
entries `min_size`/`empty_dirs_recursive`/`dirs_plain` were removed. The
|
||||
integrated stats+options+fs branch stands at **111 ✅ / 13 ⚠️ / 33 ❌ = 157**;
|
||||
full suite + ASan + clang-format + cppcheck clean.
|
||||
|
||||
11. **No-wire parity track 1** on `feat/parity-2.28` (no protocol change):
|
||||
`-n --delete` now sends the same filter-excluded + size-pruned protected
|
||||
prefixes and synchronized-directory scope as a real run (dry-run would-delete
|
||||
matches rsync for source-derived protections; the destination-only exclude
|
||||
residual was later closed by track 4a, readdir ordering remains);
|
||||
`--delete-delay` now charges
|
||||
`--max-delete` on actual removals and re-scans a queued directory at commit
|
||||
to remove content created after the plan, with an independent deferred-list
|
||||
cap (only partial-delete ordering remains); and `--info=name2` emits `NAME is
|
||||
uptodate` plus the leading `./` root name line for `--info=name` (only the
|
||||
root-line trigger condition and receiver-side `skip` wording remain). Matrix
|
||||
now **111 ✅ / 14 ⚠️ / 32 ❌ = 157**; differential + unit tests added in
|
||||
`test_features.py`, `test_option_parity.py`, `test_delete_plan.c`,
|
||||
`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`.
|
||||
12. **No-wire parity track 2b** on `feat/parity-2.28` (no protocol change):
|
||||
`--progress`/`-P`/`--info=progress` (when not `--quiet`) now run an opt-in
|
||||
paths-only metadata pre-count (no file reads/hashing) that supplies rsync's
|
||||
full file-list total for the `to-chk` denominator and the directory names,
|
||||
and emits per-directory/symlink/special name lines, in both the sequential
|
||||
and `--threads` paths. `--delete-during`/`--delete-delay` reuse their
|
||||
keep-set pre-scan instead of a second walk; non-progress runs are
|
||||
unaffected. Differential tests (`progress`/`progress_threads` over a new
|
||||
`multidir` corpus) match rsync's name set and `to-chk` denominator on a
|
||||
fresh transfer, and the single-file byte-identical test still passes;
|
||||
emission order (rsync's sorted depth-first vs FastSync's readdir/BFS stream)
|
||||
plus re-run over-naming (unconditional `./`, ancestor dirs named with a
|
||||
transferred child, and no quick-check for symlinks/empty dirs) remain the
|
||||
caveats, so the row stays ⚠️ and the matrix is unchanged at
|
||||
**111 ✅ / 14 ⚠️ / 32 ❌ = 157**.
|
||||
|
||||
13. **Wire parity track 4a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0`): the receiver now has a delete-time filter engine. The sender
|
||||
compiles its root-level selection rules exactly as the scanner does
|
||||
(`filter_base_build`) and streams them as one bounded, self-describing
|
||||
config-frame block (action, sides, anchored, dir-only, negate, owner,
|
||||
pattern; bounded rule count and pattern bytes, unknown action/sides is a
|
||||
protocol error). The receiver reconstructs `protect_rules` and applies them
|
||||
first-match-wins to each extraneous destination path in every delete timing
|
||||
(the whole-tree commit walker, the `--delete-during`/`--delete-delay`
|
||||
per-directory plans, and the `-n` would-delete enumeration), so a
|
||||
`P *.log` rule protects a destination-only `extra.log` like rsync (with
|
||||
`risk` cancelling); the sender-derived protected-prefix behavior is
|
||||
preserved when no rules are sent and `--delete-excluded` semantics are
|
||||
unchanged. Per-directory merge (`:`/`.`) receiver re-derivation remains the
|
||||
residual. `TestFilterProtect` (real + dry-run) plus differential cases
|
||||
`filter_protect`, `filter_protect_during`, `filter_protect_delay` added and
|
||||
the `--filter=RULE` row moves ❌ → ✅: matrix now
|
||||
**115 ✅ / 11 ⚠️ / 31 ❌ = 157**; unit tests, the three named integration
|
||||
files, clang-format and cppcheck clean.
|
||||
|
||||
14. **Wire parity track 5a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0` by project decision): the three basis-dir options now default to
|
||||
rsync's metadata quick-check (equal size + equal mtime, or size alone under
|
||||
`--size-only`; `-I` disables matching) instead of FastSync's historical
|
||||
xxHash64 content equality, so a same-size/different-content basis is trusted
|
||||
exactly as rsync trusts it. A new FastSync-only, long-only `--verify-basis`
|
||||
flag restores the strict whole-file content equality; its bool is appended to
|
||||
the basis block of the config frame (golden wire frame 882 → 886 bytes).
|
||||
`--verify-basis` streams the confined basis descriptor to hash it, and a
|
||||
basis hit is no longer capped at the 256 MiB whole-file payload bound:
|
||||
`--copy-dest` streams the basis through a bounded buffer and `--link-dest`'s
|
||||
copy fallback streams from the basis, so an over-limit hit materializes (a
|
||||
basis MISS still falls back to the normal transfer and keeps its own bound).
|
||||
A `--copy-dest` hit re-applies the SOURCE attributes (the sender transmits
|
||||
the source metadata with the basis check frame), matching rsync's
|
||||
"copy then fix attributes"; a `--link-dest` success keeps the shared inode's
|
||||
attributes (writing through it would mutate the basis). Differential cases
|
||||
`copy_dest` and `verify_basis` added; `test_basis_dir_size_only_content_residual`
|
||||
converted to a passing parity assertion; `TestBasisDestDirs` updated for the
|
||||
new default + `--verify-basis`; unit tests cover the quick-check/verify
|
||||
decision and the same-size/different-content handshake. The
|
||||
`--compare-dest`/`--copy-dest`/`--link-dest` rows move ❌ → ⚠️ (relative-DIR
|
||||
resolution base and over-limit MISS refusal): matrix now
|
||||
**116 ✅ / 13 ⚠️ / 28 ❌ = 157**.
|
||||
|
||||
15. **No-wire parity track 5b** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0` by project decision): `-y`/`--fuzzy` reclassified ❌ → ⚠️. A probe
|
||||
against real rsync 3.4.1 (pinned `-B8192`, repeated-content 64 KiB corpus)
|
||||
showed the name heuristic is already rsync's (`util1.c fuzzy_distance` /
|
||||
`find_filename_suffix` + the exact size+mtime pass) and the output is always
|
||||
byte-exact; the only residual is candidate ELIGIBILITY, because FastSync's
|
||||
`delta_should_attempt` gate caps the size ratio at 10× and requires both
|
||||
files ≥ 16 KiB while rsync will reuse a basis from 0.25× to 10000× and below
|
||||
16 KiB. The choice is observable only as `--stats` bandwidth counters. Added
|
||||
differential case `fuzzy_basis` (same-suffix sibling, one name edit,
|
||||
identical content, block size pinned) asserting tree **and** normalized
|
||||
`--stats` parity where the choices coincide, plus `TestFuzzy` pinning the
|
||||
window boundary on both sides (>10× and <16 KiB siblings declined by
|
||||
FastSync while rsync uses them, both trees byte-identical). Matrix now
|
||||
**116 ✅ / 14 ⚠️ / 27 ❌ = 157**.
|
||||
|
||||
16. **Lockstep delete-default track 6** on `feat/parity-2.28` (`PROTOCOL_VERSION`
|
||||
stays `2.28.0`): plain `--delete` now defaults to rsync's delete-during
|
||||
(`--del`) timing, normalized on the client onto the existing `delete_during`
|
||||
wire bool. The old late whole-tree commit is opt-in via `--delete-after` or
|
||||
the FastSync-only long `--delete-commit` (identical `delete_after` timing).
|
||||
`-d/--dirs` still falls back to the end commit, `--delay-updates` still
|
||||
deletes before publication, and `--files-from`/`-R` scope is unchanged. The
|
||||
`STATUS_DELETE_PLAN` frame gained a one-int `apply` flag so the per-run
|
||||
config block (including `--delete-missing-args` exact paths) is always
|
||||
transmitted, on a config-only carrier when the scope allows no directory
|
||||
plan — fixing a latent bug with a file-only `--files-from` list. Differential
|
||||
cases `delete`/`delete_commit`/`filter_protect_after` plus the extended
|
||||
`test_delete_timing_parity.py` (plain `--delete` mid-abort removes reached
|
||||
extras, `--delete-commit` defers) pass; full `-m "not setpriv"` suite,
|
||||
clang-format and cppcheck clean. Matrix unchanged at
|
||||
**116 ✅ / 14 ⚠️ / 27 ❌ = 157** (the `--delete`/`--delete-during` rows stay
|
||||
⚠️ for the abort boundary; `--delete-after` stays ✅).
|
||||
|
||||
## Next steps
|
||||
1. **Merge PR #284** (`dev` -> `main`) once reviewed (protected branch).
|
||||
2. **Deferred security items** (documented, not implemented):
|
||||
- Pre-auth config/daemon-auth handshake has no aggregate wall-clock deadline
|
||||
(per-message timeout only) — slowloris holds connection slots.
|
||||
- Per-source registry fails open when the shared table is full (per-module/global
|
||||
caps and host ACLs still apply); consider fail-closed or larger/evicting table.
|
||||
- SCRAM-like daemon auth has no TLS channel binding (and is not RFC 5802).
|
||||
- `cleanup()` signal handler calls non-async-signal-safe teardown; daemon `umask(0)`.
|
||||
- Wire protocol assumes homogeneous word size/endianness (lengths are native
|
||||
`size_t`) — document or move to fixed-width framing.
|
||||
3. **Out of scope / intentional:** pull (remote source) mode is **not** planned —
|
||||
FastSync is push-only; see `RSYNC_COMPAT.md#direction`.
|
||||
|
||||
## Key facts / commands
|
||||
- CI image: `gitea.tap-tap.win/taptap/fastsync-ci:v11` (alias `fastsync-ci:local`).
|
||||
- Build/test: `cmake -B build -S . -DSTRICT_WARNINGS=ON && cmake --build build -j$(nproc) && ./build/tests`
|
||||
then `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`.
|
||||
- Dev shell: `nix-shell` (provides clang-format, cppcheck, pytest-xdist, openssh,
|
||||
rsync, iproute2, valgrind, lcov; does not build on entry).
|
||||
- Gitea API token: supplied out-of-band via the `TOKEN` environment variable; it is
|
||||
intentionally **not** recorded in this file.
|
||||
- CI polling: `GET /api/v1/repos/TapTap/FastSync/actions/runs?limit=N`, match `head_sha`,
|
||||
then `/actions/runs/<id>/jobs`.
|
||||
@@ -1,4 +1,4 @@
|
||||
#FastSync
|
||||
# FastSync
|
||||
|
||||
FastSync is a high-performance file synchronization tool designed to become a
|
||||
drop-in replacement for common `rsync` workflows. It keeps the familiar
|
||||
@@ -7,7 +7,7 @@ multithreading, streaming zstd compression, chunking, zero-copy TCP transfers,
|
||||
and native TCP/TLS transports.
|
||||
|
||||
The release version is FastSync's client/server protocol version (printed by
|
||||
`fastsync --version`); client and server must match. See
|
||||
`./build/client --version`); client and server must match. See
|
||||
[CHANGELOG.md](CHANGELOG.md) for the history.
|
||||
|
||||
The compatibility target is straightforward:
|
||||
@@ -28,7 +28,7 @@ FastSync uses a producer-consumer transfer pipeline and can combine several
|
||||
optimizations for large or high-latency transfers:
|
||||
|
||||
- Multithreaded scanning, loading, and sending.
|
||||
- Streaming zstd compression with levels 1 through 22.
|
||||
- Streaming compression (zstd by default, plus lz4/zlib/zlibx) with levels 1 through 22.
|
||||
- Configurable file chunking and compact chunk serialization.
|
||||
- `sendfile()` zero-copy transfers over TCP.
|
||||
- Batched incremental checks to reduce round trips.
|
||||
@@ -51,28 +51,51 @@ replacement for every rsync feature or protocol mode.
|
||||
- Rsync-style source and destination arguments.
|
||||
- SSH transport using `user@host:destination` paths below the remote authorized root.
|
||||
- TCP client/server transfers.
|
||||
- Dry runs, excludes, includes, size filters, backups, statistics, and
|
||||
bandwidth limiting.
|
||||
- Incremental size/mtime checks and optional xxHash64 content checks.
|
||||
- Dry runs (server-contacting since protocol 2.21.0 for server-routed targets),
|
||||
excludes, includes, size filters, backups, statistics, and bandwidth
|
||||
limiting.
|
||||
- Incremental size/mtime checks and optional content checks (`xxh128` by
|
||||
default, selectable with `--checksum-choice`).
|
||||
- FastSync-native delta transfer for changed files.
|
||||
- Optional mode and timestamp preservation.
|
||||
- Delete manifests with server-side delete authorization.
|
||||
- Temporary-file writes with atomic rename by default.
|
||||
- Path traversal checks and destination-root confinement.
|
||||
|
||||
### Not yet equivalent to rsync
|
||||
### Boundaries and documented divergences
|
||||
|
||||
The items below summarize FastSync's rsync compatibility status — recently
|
||||
closed gaps and the remaining known divergences. Each row of the detailed
|
||||
matrix is classified as parity, caveat, or divergent in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
- The FastSync wire protocol is not the rsync wire protocol.
|
||||
- SSH mode requires `fastsync-server` on the remote host.
|
||||
- Archive mode does not yet provide all of rsync's `-rlptgoD` behavior.
|
||||
- Symlink transfer is incomplete; link targets are not yet recreated in all
|
||||
modes.
|
||||
- Owner/group, ACL, xattr, and hard-link handling is incomplete or
|
||||
unavailable.
|
||||
- Archive mode covers rsync's `-rlptgoD` behavior — links, permissions, times,
|
||||
owner, group, devices, and special files — and does not imply compression or
|
||||
multithreading (see [Client](#client)). Ownership application is still
|
||||
privilege-gated: a receiver that cannot `chown` logs a warning and skips it.
|
||||
Under `-p` the source mode is copied exactly, including setuid/setgid/sticky
|
||||
and group/other-write bits (strict rsync parity; see
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)).
|
||||
- Symlink transfer stores targets **verbatim** (`-l`/`--links`), including
|
||||
absolute and `..`-bearing targets, matching rsync. The receiver does not
|
||||
enforce a containment predicate by default; `--safe-links` drops unsafe
|
||||
targets on the sender, and `--munge-links` rewrites them with rsync's
|
||||
`/rsyncd-munged/` marker. `--trust-sender` does not affect symlink targets.
|
||||
A destination later consumed by a link-following tool can therefore follow a
|
||||
link outside the receive root — use `--safe-links` for untrusted sources.
|
||||
- Hard links (`-H`/`--hard-links`), extended attributes (`-X`/`--xattrs`), and
|
||||
POSIX ACLs (`-A`/`--acls`) are preserved; owner/group is applied through
|
||||
`-o`/`-g` (or an `-a`/`--archive` transfer), through the opt-in identity flags
|
||||
(`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`), and only when
|
||||
the receiver has permission. See
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md) for the exact semantics and documented
|
||||
divergences.
|
||||
- Device and special-file preservation is implemented with documented
|
||||
divergences: recreated device nodes require `CAP_MKNOD` on the receiver (a
|
||||
non-root receiver skips the entry), and sockets cannot be recreated (FIFOs
|
||||
are).
|
||||
non-root receiver skips the entry), while FIFOs **and unix sockets** are
|
||||
recreated (`--specials`).
|
||||
- Sparse-file hole preservation (`-S`, `--sparse`) is implemented receiver-side:
|
||||
long all-zero runs are written as holes (no wire change; the full file image
|
||||
is already in memory).
|
||||
@@ -80,78 +103,163 @@ replacement for every rsync feature or protocol mode.
|
||||
the write atomic (temp + rename). With `--partial`, a failed/interrupted write
|
||||
now retains the already-written temp at the destination path (best-effort) so
|
||||
a later `--append`/`--append-verify` run can resume it.
|
||||
- `--dirs` is not implemented. Its compatibility aliases `--old-dirs` and
|
||||
`--old-d` are recognized but rejected explicitly rather than silently using
|
||||
FastSync's recursive directory behavior.
|
||||
- `-d`/`--dirs` and its aliases `--old-dirs`/`--old-d` transfer the named
|
||||
directory entries without recursing into their contents.
|
||||
- Short-option names are now rsync-parity (Phase 7 Wave A): FastSync's former
|
||||
collisions were renamed (`-j`/`--threads`, `--preserve`, `--sendfile`,
|
||||
`--chunk-serialization`, `--timeout`, `--ssh-port`), so `-m`, `-M`, `-f`,
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync. See `RSYNC_COMPAT.md`.
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync.
|
||||
- Short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached values
|
||||
(`-B1000`, `-essh`, `-MOPT`, `--opt=value`) are accepted, matching rsync.
|
||||
- `-r`, `-b`, `-L`, and `-B` are parsed with the rsync short names.
|
||||
- `--stats` prints the counters FastSync can observe plus the receiver-only
|
||||
counters reported over the wire (`Matched data`, deleted files, and the
|
||||
created/literal counters); `Number of files` and `Number of created files`
|
||||
carry rsync's per-type breakdown. `--progress` prints rsync-style per-file
|
||||
blocks including the leading `./` line, and (when progress is requested) a
|
||||
paths-only pre-count supplies rsync's `to-chk` denominator.
|
||||
- Codecs match rsync 3.4.1: `zstd`/`lz4`/`zlib`/`zlibx` compression and
|
||||
`xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1`/`none` checksums. `auto` honors
|
||||
`RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST` and otherwise follows rsync's
|
||||
compiled-in order. An omitted `--compress-level` uses the codec's rsync
|
||||
default (zstd 3, zlib/zlibx 6, lz4 ignored); `zlib`/`zlibx` share the
|
||||
literal-only zlib path (rsync's zlibx semantics), and the transfer checksum is
|
||||
not separately selectable.
|
||||
|
||||
The detailed flag matrix is maintained in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented,
|
||||
partial, alternate, and planned behavior.
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It reports each row as **parity**,
|
||||
**caveat** (works with a documented divergence), or **divergent** (not
|
||||
supported), rather than treating "parsed" as parity.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Build
|
||||
|
||||
`compile_commands.json` is a symlink to `build/compile_commands.json` and is used by clangd/editor tooling; its target is generated by the build, so it dangles until the first build.
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
This produces `./build/client` and `./build/server`. `compile_commands.json` is a symlink to `build/compile_commands.json` and is used by clangd/editor tooling; its target is generated by the build, so it dangles until the first build.
|
||||
|
||||
### Client
|
||||
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| Positional | `<source> <dest>` — automatic SSH detection if dest contains `:` |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime |
|
||||
| `-z, --compress [level]` | Enable streaming zstd compression (level 1–22, default 5) |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials (not compression/multithreading) |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime (implies the incremental checksum quick-check) |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh128` (default), `xxh3`, `xxh64`/`xxhash`, `md5`, `md4`, `sha1`, `none`, or `auto` (plus rsync's two-name `transfer,pre-transfer` form) |
|
||||
| `-z, --compress [level]` | Enable streaming compression (default `zstd`; level 1–22, default 5) |
|
||||
| `--compress-choice <alg>` | Compression algorithm: `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, or `auto` |
|
||||
| `--skip-compress <list>` | Skip compression for suffixes (`/`- or `,`-separated); defaults to rsync 3.4.1's built-in suffix list |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated (not compression/multithreading) |
|
||||
| `-j, --threads[=N]` | Multithreading mode; `N` (1–256) sets the parallel scanner worker count, bare `-j`/`--threads` uses the default |
|
||||
| `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) |
|
||||
| `-r, --recursive` | Recurse into directories (FastSync is always recursive; accepted for rsync compatibility) |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents; aliases `--old-dirs`/`--old-d` |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root |
|
||||
| `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) |
|
||||
| `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) |
|
||||
| `--sendfile` | Sendfile zero-copy. Incompatible with compression / chunk serialization. TCP only. Long form only. |
|
||||
| `--preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) |
|
||||
| `-n, --dry-run` | Scan and print what would be transferred |
|
||||
| `-p, --perms` | Preserve permission bits (part of the metadata bundle) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show real-time transfer speed |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded) |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk) |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`) |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch) |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing; `--no-whole-file` clears it |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`) |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime |
|
||||
| `--preserve` | Preserve mode and mtime (`-p` + `-t`; add `-o`/`-g` for owner/group or `-U`/`--atimes` for atime; `-N`/`--crtimes` captures birth time but cannot apply it) |
|
||||
| `-U, --atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N, --crtimes` | Capture birth time; cannot be applied (documented divergence) |
|
||||
| `-p, --perms` | Preserve permission bits. Strict rsync parity: the source mode is copied exactly, including setuid/setgid/sticky and group/other-write bits |
|
||||
| `-t, --times` | Preserve modification times |
|
||||
| `-o, --owner` | Preserve the source owner (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `-g, --group` | Preserve the source group (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group`, `--no-preserve` | Negate the per-attribute flags (short `--no-p`/`--no-t`/`--no-o`/`--no-g`; `--no-preserve` clears all four) |
|
||||
| `-E, --executability` | Preserve executable permission bits |
|
||||
| `-X, --xattrs` | Preserve user `user.*` extended attributes |
|
||||
| `-A, --acls` | Preserve POSIX ACLs |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax) |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership |
|
||||
| `--numeric-ids` | Apply source numeric uid/gid directly instead of mapping by name |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP] (requires a privileged receiver) |
|
||||
| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes) |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`) |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`) |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--include-from <file>` | Read include patterns from a file |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `-x, --one-file-system` | Do not cross filesystem boundaries; the mount-point directory entry is emitted (empty at the destination) without descending |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G; `0` = no local limit, matching rsync) |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`) |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win) |
|
||||
| `--verify-basis` | FastSync-only: require a basis hit (`--compare-dest`/`--copy-dest`/`--link-dest`) to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync) |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-during, matching rsync, so destination space is freed progressively). Scoped to the synchronized directories, so `--files-from` subsets are safe |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) |
|
||||
| `--delete-after` | Explicit delete-after timing (implies `--delete`) |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G) |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | Positive I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`, built-in default 30 s) and the per-message protocol poll deadline (built-in default 60 s). Omit the option to keep both built-ins; `0` is rejected. The server side keeps the built-in 60 s protocol window (the value is not sent on the wire). |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 10) |
|
||||
| `--backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing) |
|
||||
| `-h, --human-readable` | Format transfer byte sizes with binary units |
|
||||
| `--delete-commit` | FastSync-only: keep the pre-2.28 atomic timing — delete only after the whole transfer succeeded (identical timing to `--delete-after`) |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication) |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install; confined to the receive root (relative only), with an `EXDEV` non-atomic copy fallback |
|
||||
| `-n, --dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters (FastSync does not print rsync's leading `./` line) |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing), including the receiver-only counters reported over the wire; rsync's per-type `Number of files` breakdown is not reproduced |
|
||||
| `-i, --itemize-changes` | Print an rsync-style per-file change line |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`) |
|
||||
| `--list-only` | List source files instead of transferring |
|
||||
| `--fsync` | Fsync every written file before publication |
|
||||
| `-h, --human-readable` | Format transfer byte/rate counts with rsync's decimal (base-1000) units |
|
||||
| `--max-depth <n>` | Maximum directory depth to recurse (0 = unlimited, default: 0) |
|
||||
| `--log-file <path>` | Write log messages to file instead of stderr |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server) |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server) |
|
||||
| `--source-dir <path>` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) |
|
||||
| `--dest-dir <path>` | Server destination directory (overrides `FASTSYNC_DEST_DIR`) |
|
||||
| `--save-to-disk` | Write received files to disk |
|
||||
| `--server-host <ip>` | Server IP address (default: `127.0.0.1`) |
|
||||
| `--server-port <n>` | Server port (default: `8080`) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-e, --rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments, e.g. `-e "ssh -p 2222"`) |
|
||||
| `-M, --remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable) |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address |
|
||||
| `-4, --ipv4` | Force IPv4 for destination resolution |
|
||||
| `-6, --ipv6` | Force IPv6 for destination resolution |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect (`TCP_NODELAY`, `SO_KEEPALIVE`, `SO_RCVBUF`, `SO_SNDBUF`, `SO_REUSEADDR`) |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline. Default `0` = disabled (matching rsync); `0` disables it. `--no-timeout` is the negation. The value is not sent on the wire; the server side keeps its own safe floor. |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 60, matching rsync); `0` disables it (`--no-contimeout` is the negation) |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes (a positive integer); whatever was already transferred is kept |
|
||||
| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`); an early stop skips the late `--delete` keep-set |
|
||||
| `-b, --backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--tls` | Enable TLS encryption |
|
||||
| `--cert <path>` | TLS certificate file (PEM) |
|
||||
| `--key <path>` | TLS private key file (PEM) |
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--client-cn <name>` | TLS client certificate common name; mandatory with `--tls` (a TLS connection always verifies the client CN) |
|
||||
|
||||
The exhaustive rsync flag matrix is in [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
**Per-message vs. connection timeouts.** `--timeout` bounds each individual protocol
|
||||
send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It
|
||||
@@ -190,60 +298,64 @@ transfer is never aborted.
|
||||
| `FASTSYNC_SOURCE_DIR` | — | Source directory fallback |
|
||||
| `FASTSYNC_DEST_DIR` | — | Destination directory fallback |
|
||||
| `FASTSYNC_SAVE_TO_DISK` | `false` | Disk persistence fallback |
|
||||
| `FASTSYNC_SSH_PORT` | `22` | Default SSH port |
|
||||
| `FASTSYNC_SERVER_HOST` | `127.0.0.1` | Default server host |
|
||||
| `FASTSYNC_SERVER_PORT` | `8080` | Default server port |
|
||||
| `FASTSYNC_TLS_CERT` | — | Default TLS certificate path |
|
||||
| `FASTSYNC_TLS_KEY` | — | Default TLS private key path |
|
||||
| `FASTSYNC_TLS_CA` | — | Default TLS CA certificate path |
|
||||
|
||||
## Implementation Details
|
||||
|
||||
### Data Structures
|
||||
1. **Chunk** — collection of files (~10 MB total by default)
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`;
|
||||
uid / gid are advisory wire fields and are never applied by the receiver;
|
||||
atime is unsupported
|
||||
4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement
|
||||
|
||||
1. **Chunk** — collection of files (~10 MB total by default).
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer.
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec` (plus
|
||||
atime/crtime fields). `uid`/`gid` are applied only through the opt-in
|
||||
identity path; atime is preserved with `-U`/`--atimes`; crtime is captured
|
||||
but cannot be set on the destination.
|
||||
4. **Config** — runtime parameters. Most cross the wire (TLS settings
|
||||
excluded); `backup` and `backup_dir` are in the serialized wire table, while
|
||||
`timeout`, `contimeout`, `quiet`, `stats`, `max_depth`, and `log_file` are
|
||||
client-only.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables.
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include
|
||||
pattern support, max-depth enforcement.
|
||||
|
||||
### Key Algorithms
|
||||
1. **File scanning** — BFS directory traversal;
|
||||
entries matched against exclude and include patterns,
|
||||
max - depth enforced 2. * *Chunking ** — files accumulated until `chunk_size` threshold,
|
||||
then flushed 3. *
|
||||
*Compression ** — streaming zstd
|
||||
via `ZSTD_compressStream2` / `ZSTD_decompressStream` 4. *
|
||||
*Network protocol ** — status -
|
||||
code - driven exchange with metadata packing,
|
||||
keep - alive,
|
||||
and abort support 5. * *Incremental check ** — client sends `STATUS_CHECK` + path + size +
|
||||
mtime and,
|
||||
with `--checksum`, XXH64 content checksum; server compares against destination. Can be batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with `nanosleep` throttling on 64 KB write chunks
|
||||
7. **Metadata restoration** — `chmod()`, `chown()`, `utimensat()` on the receiving side
|
||||
8. **`--delete`** — sender tracks all sent paths;
|
||||
receiver walks destination tree and removes unlisted files / directories 9. *
|
||||
*SSH transport *
|
||||
* — `socketpair()` + `fork()` + `execvp("ssh",
|
||||
...)` with `ControlMaster` and port support
|
||||
10. *
|
||||
*TLS transport ** — OpenSSL `SSL_CTX` with TLS
|
||||
1.2 minimum,
|
||||
mutual CA verification,
|
||||
transparent `SSL_read`/`SSL_write` via `io_set_ssl()` 11. *
|
||||
*Path traversal protection ** — `has_path_traversal()` rejects any file path
|
||||
containing `..` components,
|
||||
preventing directory escape attacks 12. *
|
||||
*Connection limiting ** — server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100)13. *
|
||||
*Keep
|
||||
- alive ** — idle connections receive periodic `STATUS_KEEPALIVE` to detect half
|
||||
- open TCP connections 14. * *Abort handling ** — `SIGINT` sets an abort flag; the next protocol operation sends `STATUS_ABORT` for clean server cleanup
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically renamed via `rename()`, preventing partial files
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir` (or same directory with `~` suffix) preserving the original
|
||||
|
||||
1. **File scanning** — BFS directory traversal; entries matched against exclude
|
||||
and include patterns, with max-depth enforced.
|
||||
2. **Chunking** — files accumulated until the `chunk_size` threshold (default
|
||||
10 MiB) is reached, then flushed.
|
||||
3. **Compression** — streaming zstd via `ZSTD_compressStream2()` /
|
||||
`ZSTD_decompressStream()`.
|
||||
4. **Network protocol** — status-code-driven exchange with metadata packing,
|
||||
keep-alive, and abort support.
|
||||
5. **Incremental check** — the client sends `STATUS_CHECK` + path + size +
|
||||
mtime and, with `--checksum`, a whole-file content checksum (`xxh128` by
|
||||
default; selectable via `--checksum-choice`/`--cc`, seeded by
|
||||
`--checksum-seed`); the server compares against the destination. Can be
|
||||
batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with sleep throttling on
|
||||
64 KiB write chunks.
|
||||
7. **Metadata restoration** — mode via `chmod()`/`fchmod()`, times via
|
||||
`utimensat()`/`futimens()`, and ownership only with an identity flag via
|
||||
fd-relative `fchown()`/`fchownat()`.
|
||||
8. **`--delete`** — the sender tracks all sent paths; the receiver walks the
|
||||
destination tree and removes unlisted files and directories.
|
||||
9. **SSH transport** — `socketpair()` + `fork()` + `execvp("ssh", ...)` with
|
||||
`ControlMaster` and port support.
|
||||
10. **TLS transport** — OpenSSL `SSL_CTX` with TLS 1.2 minimum, mutual CA
|
||||
verification, and transparent `SSL_read()`/`SSL_write()` via
|
||||
`io_set_ssl()`.
|
||||
11. **Path traversal protection** — `has_path_traversal()` rejects any file
|
||||
path containing `..` components, preventing directory escape attacks.
|
||||
12. **Connection limiting** — the server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100).
|
||||
13. **Keep-alive** — idle connections receive periodic `STATUS_KEEPALIVE` to
|
||||
detect half-open TCP connections.
|
||||
14. **Abort handling** — `SIGINT` sets an abort flag; the next protocol
|
||||
operation sends `STATUS_ABORT` for clean server cleanup.
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically
|
||||
renamed via `rename()`, preventing partial files.
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir`
|
||||
(or the same directory with a `~` suffix), preserving the original.
|
||||
|
||||
## Security Features
|
||||
|
||||
@@ -300,7 +412,8 @@ cmake --build build -j$(nproc)
|
||||
|
||||
### SSH transfer
|
||||
|
||||
The remote host must have `fastsync-server` available in `PATH`, or use
|
||||
The remote host must have `fastsync-server` available in `PATH` (install or
|
||||
copy the built `./build/server` there as `fastsync-server`), or use
|
||||
`--fastsync-server-path`. SSH starts `fastsync-server --stdio` in its remote
|
||||
working directory, so use a destination below that directory unless the
|
||||
remote server is otherwise configured with a matching authorized root.
|
||||
@@ -341,8 +454,13 @@ Plain TCP requires the explicit `--allow-unauthenticated` server option. Use TLS
|
||||
authenticated network connections.
|
||||
|
||||
### TLS transfer
|
||||
|
||||
Server TLS requires `--cert`, `--key`, `--ca`, and `--client-cn`; the client
|
||||
requires `--cert`, `--key`, and `--ca`.
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem -p 8443
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem \
|
||||
--ca ca.pem --client-cn client -p 8443
|
||||
./build/client --tls --cert client.pem --key client-key.pem --ca ca.pem \
|
||||
--server-host example.com --server-port 8443 \
|
||||
--source-dir /path/to/source --dest-dir /path/to/destination \
|
||||
@@ -355,32 +473,32 @@ These examples show the intended rsync-style workflow. Options marked as
|
||||
FastSync-native are optional performance or transport extensions.
|
||||
|
||||
```bash
|
||||
#Basic synchronization
|
||||
# Basic synchronization
|
||||
./build/client /source/ /destination/
|
||||
|
||||
#Archive - style synchronization(current FastSync archive behavior)
|
||||
# Archive-style synchronization (current FastSync archive behavior)
|
||||
./build/client -a /source/ user@host:destination/
|
||||
|
||||
#Preview a transfer without changing the destination
|
||||
# Preview a transfer without changing the destination
|
||||
./build/client -n /source/ /destination/
|
||||
|
||||
#Exclude temporary and object files
|
||||
# Exclude temporary and object files
|
||||
./build/client --exclude '*.tmp' --exclude '*.o' \
|
||||
/source/ user@host:destination/
|
||||
|
||||
#Remove destination entries not present in the source
|
||||
# Remove destination entries not present in the source
|
||||
./build/client --delete /source/ user@host:destination/
|
||||
|
||||
#Skip unchanged files using size and modification time
|
||||
# Skip unchanged files using size and modification time
|
||||
./build/client --incremental /source/ user@host:destination/
|
||||
|
||||
#Verify content when size and time are not sufficient
|
||||
# Verify content when size and time are not sufficient
|
||||
./build/client --incremental --checksum /source/ user@host:destination/
|
||||
|
||||
#Preserve supported mode and timestamp metadata
|
||||
# Preserve supported mode and timestamp metadata
|
||||
./build/client --preserve /source/ user@host:destination/
|
||||
|
||||
#Keep backups of overwritten destination files
|
||||
# Keep backups of overwritten destination files
|
||||
./build/client --backup --backup-dir backups \
|
||||
/source/ user@host:destination/
|
||||
```
|
||||
@@ -393,11 +511,11 @@ features without changing the meaning of ordinary compatibility options.
|
||||
| Option | Purpose |
|
||||
|---|---|
|
||||
| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming zstd compression, levels 1-22. |
|
||||
| `--compress-level <n>` | Set the zstd compression level. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` and `none`. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming compression (default `zstd`), levels 1-22. |
|
||||
| `--compress-level <n>` | Set the compression level (1-22). Omitted, each codec uses its rsync default: zstd 3, zlib/zlibx 6, lz4 ignored. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, and `auto`; `zlib`/`zlibx` share the same literal-only zlib path. |
|
||||
| `--zl <n>` | Alias for `--compress-level`. |
|
||||
| `--skip-compress <list>` | Skip compression for comma-separated suffixes; incompatible with `--chunk-serialization`. |
|
||||
| `--skip-compress <list>` | Skip compression for `/`- or `,`-separated suffixes; defaults to rsync 3.4.1's built-in list. Incompatible with `--chunk-serialization`. |
|
||||
| `--compress-threads <n>` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. |
|
||||
| `--chunk-size <bytes>` | Set the transfer chunk size. |
|
||||
| `--chunk-serialization` | Enable FastSync chunk serialization (long form only; `-s` is rsync's `--secluded-args`). |
|
||||
@@ -409,10 +527,10 @@ features without changing the meaning of ordinary compatibility options.
|
||||
| `--server-port <port>` | Select the TCP server port (`--port <port>` and `--port=<port>` are rsync-friendly aliases). |
|
||||
| `--tls` | Enable TLS for TCP transport. |
|
||||
| `--bwlimit <KB/s>` | Apply token-bucket bandwidth limiting. |
|
||||
| `--progress` | Show transfer progress and throughput. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout (positive seconds). Omit to keep the built-in 30 s socket / 60 s protocol defaults. |
|
||||
| `--contimeout <seconds>` | Set connection timeout. |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters (FastSync omits rsync's leading `./` line). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; rsync's per-type `Number of files` breakdown is not reproduced. |
|
||||
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout. Default `0` = disabled (matching rsync); `0` disables it. |
|
||||
| `--contimeout <seconds>` | Connection timeout (default 60, matching rsync); `0` disables it. |
|
||||
|
||||
Short-option conflicts with rsync have been resolved for the CLI namespace
|
||||
(Phase 7): `-c` is now rsync's `--checksum`, `-m` is `--prune-empty-dirs`, `-M`
|
||||
@@ -421,7 +539,8 @@ is `--remote-option`, `-f` is `--filter`, `-s` is `--secluded-args`, `-p` is
|
||||
long-form-only or new shorts: multithreading is `-j`/`--threads`, metadata
|
||||
is `--preserve`, sendfile is `--sendfile`, chunk serialization is
|
||||
`--chunk-serialization`, timeout is `--timeout`, and SSH port is `--ssh-port`.
|
||||
`-a`/`--archive` is now real rsync archive (`-rlptgoD`).
|
||||
`-a`/`--archive` is now rsync archive `-rlptgoD` (owner/group implied, but the
|
||||
receiver still needs privilege to apply them).
|
||||
|
||||
`--secluded-args` (and its short form `-s`) is accepted as a compatibility
|
||||
no-op. It does not change FastSync's transport or protocol behavior, because
|
||||
@@ -433,42 +552,96 @@ remote SSH argv is already built injection-safe.
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials. |
|
||||
| `-n`, `--dry-run` | Scan and report without writing files. |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. |
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated. |
|
||||
| `-n`, `--dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer. |
|
||||
| `--incremental` | Skip files matching destination size and mtime. Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `-c, --checksum` | Verify content by checksum (implies the incremental quick-check). Algorithm selectable with `--checksum-choice`. |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh64`/`xxhash` (default), `xxh3`, `xxh128`, `md5`, or `auto`. |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync. |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime. |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match. |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver. |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing (`--no-whole-file` clears it). |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`). |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents (aliases `--old-dirs`/`--old-d`). |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root. |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root). |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`). |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination. |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win). |
|
||||
| `--verify-basis` | FastSync-only: require a basis hit to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync). |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk). |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`). |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch). |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. Scoped to the synchronized directories, so `--files-from` subsets are safe. |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). |
|
||||
| `--delete-commit` | FastSync-only: atomic delete-after timing (only after the whole transfer succeeded). |
|
||||
| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected). |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync. |
|
||||
| `--force` | Allow an incoming file/symlink to replace a destination directory (also during `--delay-updates` publication). |
|
||||
| `--exclude <pattern>` | Exclude matching paths. Repeatable. |
|
||||
| `--include <pattern>` | Include matching paths. Repeatable. |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file. |
|
||||
| `--include-from <file>` | Read include patterns from a file. |
|
||||
| `-f, --filter=RULE` | Add an rsync-style filter rule (`+`/`-`, `include`/`exclude`, `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R`, `clear`/`!`, and modifiers; repeatable). |
|
||||
| `--max-size <bytes>` | Skip files larger than the limit. |
|
||||
| `--min-size <bytes>` | Skip files smaller than the limit. |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth;
|
||||
zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.|
|
||||
| `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` |
|
||||
Back up overwritten files.| | `--backup - dir<dir>` | Store backups under a separate directory.|
|
||||
| `--suffix<suffix>` | Set the backup filename suffix.| | `--partial` |
|
||||
Select partial - transfer handling. On failed/interrupted writes the
|
||||
already-written temp file is retained (best-effort) for resumption.|
|
||||
With `--partial --partial-dir <dir>`, completed files are written under the
|
||||
partial directory and installed atomically. | | `--partial - dir<dir>` |
|
||||
Set a relative partial - transfer directory below the server destination root.
|
||||
Use with `--partial`. |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units; default 1G; `0` = no local limit). |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth; zero means unlimited. |
|
||||
| `-b, --backup` | Back up overwritten files. |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install (confined to the receive root; `EXDEV` falls back to a non-atomic copy). |
|
||||
| `--backup-dir <dir>` | Store backups under a separate directory (requires `--backup`). |
|
||||
| `--suffix <suffix>` | Set the backup filename suffix (default: `~`). |
|
||||
| `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir <dir>`, completed files are written under the partial directory and installed atomically. |
|
||||
| `--partial-dir <dir>` | Set a relative partial-transfer directory below the server destination root. Use with `--partial`. |
|
||||
| `--inplace` | Write directly to the destination instead of using a temporary file. |
|
||||
| `--fsync` | Fsync every written file before publication. |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree. |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server). |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server). |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes; whatever was already transferred is kept. |
|
||||
| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`). An early stop skips the late `--delete` keep-set. |
|
||||
|
||||
### Metadata and links
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--preserve` | Preserve supported file metadata, currently mode and modification time (long form only). |
|
||||
| `-l`, `--links` | Request symlink preservation;
|
||||
link-target transfer remains incomplete. |
|
||||
| `--copy-links` | Copy symlink referents. |
|
||||
| `--safe-links` | Skip symlinks that point outside the transfer tree. |
|
||||
| `--preserve` | Preserve mode and mtime (long form only; equivalent to `-p` + `-t`). Add `-o`/`-g` for owner/group, `-U`/`--atimes` for atime, or an identity flag (`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`) for mapped ownership. |
|
||||
| `-U`, `--atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N`, `--crtimes` | Capture birth time and transmit it; it cannot be applied because no portable filesystem call can set a birth time (documented divergence). |
|
||||
| `-p`, `--perms` | Preserve permission bits. One of the four per-attribute preserve flags (with `-t`/`-o`/`-g`); under `-p` the source mode is copied exactly (setuid/setgid/sticky and group/other-write included), matching rsync. |
|
||||
| `-t`, `--times` | Preserve modification times. Independent of the other attributes; `-O`/`--omit-dir-times` suppresses directories only. |
|
||||
| `-o`, `--owner` | Preserve the source owner (uid). Mapped by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); application is privilege-gated. |
|
||||
| `-g`, `--group` | Preserve the source group (gid). Same name-mapping/numeric-fallback and privilege gating as `-o`. |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group` | Negate each per-attribute flag (also `--no-p`/`--no-t`/`--no-o`/`--no-g`); `--no-preserve` clears all four. |
|
||||
| `-E`, `--executability` | Preserve executable permission bits. |
|
||||
| `-X`, `--xattrs` | Preserve user `user.*` extended attributes. |
|
||||
| `-A`, `--acls` | Preserve POSIX ACLs. |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax, including `D`/`F`/`X` selectors and `s`/`t`); does not imply `-p`. |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files (`USER:GROUP`, `USER`, or `:GROUP`); conflicts with `--usermap`/`--groupmap` on the same side. |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership (`FROM:TO` rules; names, ids, `LOW-HIGH` ranges, `*`, empty-`FROM`). |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership (same syntax as `--usermap`). |
|
||||
| `--numeric-ids` | Mapping modifier: apply the source numeric uid/gid directly instead of mapping by name (combine with `-o`/`-g`, `-a`, or a map). |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP]; requires a privileged receiver. |
|
||||
| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown. |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes). |
|
||||
| `--no-super` | Forbid those super-user activities even when the receiver is root. |
|
||||
| `-l`, `--links` | Copy symlinks as symlinks; the target is stored verbatim (absolute and `..`-bearing targets included), matching rsync. |
|
||||
| `-L`, `--copy-links` | Copy symlink referents (a broken referent makes the run exit 23, matching rsync). |
|
||||
| `--safe-links` | Skip symlinks whose target points outside the transfer tree (applied on the sender). |
|
||||
| `--copy-unsafe-links` | Copy unsafe symlink referents. |
|
||||
| `--munge-links` | Rewrite stored symlink targets with rsync's `/rsyncd-munged/` marker. |
|
||||
| `-k`, `--copy-dirlinks` | Treat a symlink to a directory as a real directory on the sender. |
|
||||
| `-K`, `--keep-dirlinks` | Follow an existing destination symlink-to-directory (confined to the receive root). |
|
||||
| `-H`, `--hard-links` | Preserve hard-link relationships across the transfer. |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`). |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`). |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets. |
|
||||
| `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). |
|
||||
|
||||
### Output and logging
|
||||
@@ -476,8 +649,12 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--progress` | Show live transfer progress. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `-q`, `--quiet` | Suppress non-error output. |
|
||||
| `--progress` | Show rsync-style per-file progress blocks (not rsync's leading `./` line). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire. |
|
||||
| `-i`, `--itemize-changes` | Print an rsync-style per-file change line. |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`). |
|
||||
| `--list-only` | List source files instead of transferring. |
|
||||
| `--log-file <path>` | Write log output to a file. |
|
||||
| `-V`, `--version` | Print the FastSync protocol version. |
|
||||
| `--help` | Print command usage. |
|
||||
@@ -487,31 +664,56 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--ssh-port <port>` | SSH port for the SSH transport (default: 22). Note the short `-p` is now rsync's `--perms`. |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode. |
|
||||
| `-e`, `--rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments). |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode (client-only; never crosses the wire). |
|
||||
| `--rsync-path <path>` | Alias for `--fastsync-server-path`. |
|
||||
| `-M`, `--remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable; rejected for daemon/TCP destinations). |
|
||||
| `--trust-sender` | Receiver-local: trust the remote sender's file list and skip path re-validation (does not affect symlink targets). |
|
||||
| `--timeout <sec>` | Socket + per-message I/O timeout; default `0` = disabled. |
|
||||
| `--contimeout <sec>` | Connection timeout; default 60; `0` disables. |
|
||||
| `--source-dir <path>` | Set the source directory explicitly. |
|
||||
| `--dest-dir <path>` | Set the destination directory explicitly. |
|
||||
| `--save-to-disk` | Enable server-side disk persistence. |
|
||||
| `--server-host <host>` | TCP server address. |
|
||||
| `--server-port <port>` | TCP server port. `--port <port>` / `--port=<port>` is an alias. |
|
||||
| `--tls` | Enable TLS. Requires `--cert` and `--key`. |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address. |
|
||||
| `-4`, `--ipv4` | Force IPv4 for destination resolution. |
|
||||
| `-6`, `--ipv6` | Force IPv6 for destination resolution. |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect. |
|
||||
| `--tls` | Enable TLS. Requires `--cert`, `--key`, and `--ca`. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--ca <path>` | CA file for peer verification (always required with `--tls`). |
|
||||
|
||||
## Server Options
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--stdio` | Serve one SSH connection over standard input/output. |
|
||||
| `-p <port>` | TCP listen port. |
|
||||
| `--daemon` | Run as a persistent daemon listener using a module config file; the daemon default port is 873 (unlike `-p`, which defaults to 8080). |
|
||||
| `--config=FILE` | Daemon config file (default: `~/.config/fastsync/fastsyncd.conf`, else `/etc/fastsyncd.conf`). Requires `--daemon`. |
|
||||
| `--dparam=KEY=VALUE` | Override one global config key on the command line. Requires `--daemon`. |
|
||||
| `--no-detach` | Stay in the foreground (default detaches to the background when running `--daemon`). |
|
||||
| `-p, --port <port>` | TCP listen port (default: 8080, range: 1–65535). |
|
||||
| `--tls` | Enable TLS. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root;
|
||||
defaults to the current directory. |
|
||||
| `--cert <path>` | TLS certificate file (PEM). |
|
||||
| `--key <path>` | TLS private key file (PEM). |
|
||||
| `--ca <path>` | CA file for peer verification (PEM). |
|
||||
| `--client-cn <name>` | TLS client certificate CN; mandatory with `--tls` (the server verifies the client CN). |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root; defaults to the current directory. |
|
||||
| `--address <addr>` | Bind the listening socket to this address. |
|
||||
| `-4`, `--ipv4` | Bind an IPv4 socket (default). |
|
||||
| `-6`, `--ipv6` | Bind an IPv6 socket. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. This also gates `--force` (which can recursively replace/remove a destination directory tree). |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. Rejected with `--stdio` (the SSH remote argv is client-composed; use a forced command if the default must hold). No effect when not root. Daemon modules opt in per module with `client owner = yes`. |
|
||||
| `--trust-sender` | Trust the remote sender's file list: skip the receiver's up-front path-traversal re-validation (fewer checks, faster, potentially unsafe; off by default). It does not affect symlink targets, which are stored verbatim either way. |
|
||||
| `--no-super` | Operator veto: never attempt super-user activities (ownership, device nodes) even as root, and refuse any client `--copy-as`/`--super` request. |
|
||||
| `--allow-unauthenticated` | Permit plaintext/anonymous network clients; an auth-required module still accepts only opted-in loopback plaintext. |
|
||||
| `--iconv=LOCAL[,REMOTE]` | Declare this server's LOCAL charset for file-name conversion. |
|
||||
| `--password-file=FILE` | Credential store for modules that declare `auth users`. Requires `--daemon`. |
|
||||
| `--early-input=FILE` | Second credential store layered over `--password-file`. Requires `--daemon`. |
|
||||
| `--hash-credentials <file>` | Read `<file>`'s `user:password` lines and print PBKDF2 credential-store lines to stdout, then exit. Cannot be combined with `--daemon` or `--stdio`. |
|
||||
| `--iterations N` | PBKDF2 iteration count for `--hash-credentials` (default 600000, range 100000–10000000). Requires `--hash-credentials`. |
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--help` | Print server usage. |
|
||||
|
||||
@@ -540,8 +742,9 @@ and `address`, the global section accepts:
|
||||
- `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access
|
||||
patterns.
|
||||
|
||||
A `[module]` may also set `max connections` (0 = unlimited; enforced per module
|
||||
across all connection children) and its own `hosts allow`/`hosts deny`.
|
||||
A `[module]` requires `path`, and may also set `read only`, `client owner`,
|
||||
`auth users`, `max connections` (0 = unlimited; enforced per module across all
|
||||
connection children), and its own `hosts allow`/`hosts deny`.
|
||||
|
||||
The per-host cap and the shared auth lockout identify a source by its numeric
|
||||
peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local
|
||||
@@ -595,7 +798,7 @@ before the module list, before authentication, and the connecting peer address
|
||||
|
||||
## Protocol and Security
|
||||
|
||||
FastSync protocol version `2.21.0` is shared by the client and server. The
|
||||
FastSync protocol version `2.28.0` is shared by the client and server. The
|
||||
current protocol is sender-driven and includes configuration negotiation,
|
||||
including the maximum allocation limit, incremental checks, checksums,
|
||||
manifests, keep-alives, abort handling, per-file remove-source results, and
|
||||
@@ -647,10 +850,9 @@ mandates `--client-cn`, so a TLS connection to an auth-required module always
|
||||
has its client CN verified (`--client-cn` matches the certificate's CN only, not
|
||||
a subjectAltName, which is acceptable for a private CA).
|
||||
|
||||
TLS provides encrypted TCP transport. Supplying `--ca` enables certificate
|
||||
verification; without it, traffic is encrypted but peer identity is not
|
||||
verified. Use certificate verification for deployments where authentication
|
||||
matters. The default TCP transport is not encrypted.
|
||||
TLS provides encrypted TCP transport. Both the client and the server require
|
||||
`--ca` together with `--tls`, so peer certificates are always verified
|
||||
(`SSL_VERIFY_PEER`, depth 4). The default TCP transport is not encrypted.
|
||||
|
||||
The receiver protects its destination root with path validation, `openat()`
|
||||
directory traversal, `O_NOFOLLOW`, temporary files, and atomic renames. Delete
|
||||
@@ -661,18 +863,27 @@ operations require the server's explicit `--allow-delete` policy.
|
||||
The project will reach the drop-in replacement goal in stages:
|
||||
|
||||
1. Correct rsync option meanings, including short options, combined options,
|
||||
and `--option=value` syntax.
|
||||
and `--option=value` syntax — **done** in the rsync-parity wave: `-r`/`-b`/
|
||||
`-L`/`-B`, short-option clustering (`-av`, `-aAX`, `-rlpt`), and attached
|
||||
values (`-B1000`, `-essh`, `-MOPT`) all parse.
|
||||
2. Add differential tests that compare FastSync and rsync contents, metadata,
|
||||
links, deletes, filters, dry runs, and exit codes.
|
||||
3. Make `-a` implement the expected recursive, links, permissions, times,
|
||||
owner/group, and supported special-file behavior.
|
||||
4. Complete symlink, sparse-file, metadata, delete-policy, and resumable-write
|
||||
semantics.
|
||||
links, deletes, filters, dry runs, and exit codes — **done** for the
|
||||
completion wave's scope; the tests live in `tests/integration/` and skip
|
||||
cleanly when rsync is unavailable.
|
||||
3. `-a` implements full rsync `-rlptgoD`; under `-p` the source mode is copied
|
||||
exactly (no masking). Ownership application stays privilege-gated, as in
|
||||
rsync.
|
||||
4. Symlink (verbatim storage), sparse-file, metadata, delete-policy (including
|
||||
`--max-delete` partial + exit 25, per-directory `--delete-during`/
|
||||
`--delete-delay`), codecs, and resumable-write semantics are implemented;
|
||||
remaining work is the documented edge cases, which the **Parity Completion
|
||||
Wave** section of `RSYNC_COMPAT.md` enumerates honestly.
|
||||
5. Add rsync remote-shell and daemon protocol interoperability.
|
||||
6. Keep FastSync performance options as negotiated, optional extensions.
|
||||
|
||||
The exhaustive implementation matrix and compatibility notes are in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md); each row is classified as parity, caveat,
|
||||
or divergent.
|
||||
|
||||
## Testing
|
||||
|
||||
@@ -709,10 +920,13 @@ rsync protocol or filesystem-semantic compatibility.
|
||||
|
||||
## Performance Guidance
|
||||
|
||||
- Use `-m` for workloads with many files or enough CPU parallelism.
|
||||
- Use `-c` or `-z` when network bandwidth is more constrained than CPU.
|
||||
- Use `-j`/`--threads` for workloads with many files or enough CPU parallelism
|
||||
(`-m` is `--prune-empty-dirs`).
|
||||
- Use `-z` when network bandwidth is more constrained than CPU (`-c` is
|
||||
`--checksum`, not a bandwidth option).
|
||||
- Tune `--chunk-size` for file sizes, memory limits, and network latency.
|
||||
- Use `-f` for large uncompressed TCP transfers where zero-copy I/O helps.
|
||||
- Use `--sendfile` for large uncompressed TCP transfers where zero-copy I/O
|
||||
helps (`-f` is `--filter`).
|
||||
- Use `--incremental` to avoid retransmitting unchanged files.
|
||||
- Use `--delta` for changed files when both endpoints are FastSync peers.
|
||||
- Use `--bwlimit` when sharing a link with other traffic.
|
||||
|
||||
+648
-270
File diff suppressed because one or more lines are too long
@@ -8,3 +8,6 @@ markers =
|
||||
daemon_detach: real double-fork backgrounding path (--daemon without
|
||||
--no-detach); slower/fragile, so it runs in the full suite but not the
|
||||
fast PR gate
|
||||
parity: differential rsync-parity case (full set; runs on push to
|
||||
dev/main)
|
||||
parity_ci: fast differential rsync-parity subset (runs on the PR gate)
|
||||
|
||||
@@ -38,6 +38,8 @@ pkgs.mkShell {
|
||||
|
||||
buildInputs = with pkgs; [
|
||||
zstd
|
||||
zlib
|
||||
lz4
|
||||
openssl
|
||||
];
|
||||
|
||||
@@ -54,6 +56,6 @@ pkgs.mkShell {
|
||||
echo "FastSync dev shell ready."
|
||||
echo " Build: cmake -B build -S . && cmake --build build -j\$(nproc)"
|
||||
echo " Unit: ./build/tests"
|
||||
echo " CI parity: docker run --rm --user \"\$(id -u):\$(id -g)\" -v \"\$PWD:/workspace\" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 ..."
|
||||
echo " CI parity: docker run --rm --user \"\$(id -u):\$(id -g)\" -v \"\$PWD:/workspace\" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 ..."
|
||||
'';
|
||||
}
|
||||
|
||||
+557
-157
@@ -1,5 +1,8 @@
|
||||
#include "change_list.h"
|
||||
#include "checksum.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
@@ -7,21 +10,7 @@
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
/* Itemize code emitted for a transferred regular file.
|
||||
*
|
||||
* Layout (rsync-compatible 11-char item): `>f` marks a regular file that was
|
||||
* transferred to the remote host; the trailing nine markers are, in order,
|
||||
* c(hecksum) s(ize) t(ime) p(erms) o(wner) g(roup) u(ser/acl) a(ttrs) x(attrs).
|
||||
* Every marker is `+` (FastSync does not compare each attribute on the
|
||||
* receiving side, so a sent file is reported as fully updated). Files that
|
||||
* are already up to date print no line at all, matching rsync's single -i
|
||||
* which only itemizes changes.
|
||||
*
|
||||
* Because the scanner only yields regular-file transfer candidates, `>d`
|
||||
* (directory) lines are never produced; directories are not transferred as
|
||||
* items by FastSync. */
|
||||
#define ITEMIZE_SENT_FILE ">f+++++++++"
|
||||
#include <unistd.h>
|
||||
|
||||
typedef struct {
|
||||
char* data;
|
||||
@@ -80,103 +69,23 @@ static bool strbuf_append(StrBuf* buf, const char* text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool strbuf_append_ull(StrBuf* buf, unsigned long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
static bool strbuf_append_longlong(StrBuf* buf, long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%lld", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
bool change_list_enabled(const Config* config) {
|
||||
return config != NULL && (config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL));
|
||||
(config->log_file != NULL && config->log_file_format != NULL) ||
|
||||
(config->info_level & LOG_INFO_NAME) != 0);
|
||||
}
|
||||
|
||||
char* change_render_itemize(const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
const char* code = event->is_directory ? ">d+++++++++" : ITEMIZE_SENT_FILE;
|
||||
StrBuf line = {0};
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append(&line, " ") &&
|
||||
strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
/* Emitted once, lazily, ahead of the first --info=name entry: rsync prints the
|
||||
* transfer-root `./` name line when the root directory is (re)created. */
|
||||
static bool name_root_printed = false;
|
||||
|
||||
void change_reset_name_root(void) {
|
||||
name_root_printed = false;
|
||||
}
|
||||
|
||||
static const char* leaf_name(const char* path) {
|
||||
if (path == NULL)
|
||||
return "";
|
||||
const char* slash = strrchr(path, '/');
|
||||
return slash != NULL && slash[1] != '\0' ? slash + 1 : path;
|
||||
}
|
||||
/* ---- Itemize code ---- */
|
||||
|
||||
char* change_render_format(const char* format, const ChangeEvent* event) {
|
||||
if (format == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = strbuf_append(&line, leaf_name(event->path));
|
||||
break;
|
||||
case 'l':
|
||||
ok = strbuf_append_ull(&line, event->size);
|
||||
break;
|
||||
case 'b':
|
||||
ok = strbuf_append_ull(&line, event->bytes_sent);
|
||||
break;
|
||||
case 'M':
|
||||
ok = strbuf_append_longlong(&line, (long long)event->mtime_sec);
|
||||
break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* Format a mode as an `ls -l` permission string, e.g. `-rw-r--r--`. */
|
||||
/* Format the permission bits as an `ls -l` string, e.g. `-rw-r--r--`. */
|
||||
static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[0] = S_ISDIR(mode) ? 'd'
|
||||
: S_ISLNK(mode) ? 'l'
|
||||
@@ -198,29 +107,94 @@ static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[10] = '\0';
|
||||
}
|
||||
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime,
|
||||
const char* path) {
|
||||
char permission[11];
|
||||
mode_to_ls_string(mode, permission);
|
||||
char date[32];
|
||||
struct tm broken_down;
|
||||
if (localtime_r(&mtime, &broken_down) != NULL) {
|
||||
if (strftime(date, sizeof(date), "%Y/%m/%d %H:%M:%S", &broken_down) == 0)
|
||||
snprintf(date, sizeof(date), "?");
|
||||
} else {
|
||||
snprintf(date, sizeof(date), "?");
|
||||
static char itemize_type_char(const ChangeEvent* event) {
|
||||
if (event->is_directory)
|
||||
return 'd';
|
||||
if (event->is_symlink)
|
||||
return 'L';
|
||||
if (event->is_special) {
|
||||
if (S_ISCHR(event->mode) || S_ISBLK(event->mode))
|
||||
return 'D';
|
||||
return 'S';
|
||||
}
|
||||
return 'f';
|
||||
}
|
||||
|
||||
static bool times_match(const Config* config, const ChangeEvent* event) {
|
||||
if (!event->dest.known || !event->dest.existed)
|
||||
return false;
|
||||
if (event->mtime_sec == event->dest.mtime_sec)
|
||||
return event->mtime_nsec == event->dest.mtime_nsec;
|
||||
long long delta = (long long)event->mtime_sec - (long long)event->dest.mtime_sec;
|
||||
if (delta < 0)
|
||||
delta = -delta;
|
||||
return delta <= (long long)config->modify_window;
|
||||
}
|
||||
|
||||
/* Fill the 11-character itemize code (10 chars + NUL). `created` means the
|
||||
* destination entry did not exist, so every attribute marker is `+`. */
|
||||
static void itemize_code(const Config* config, const ChangeEvent* event, char code[12]) {
|
||||
bool known = event->dest.known;
|
||||
bool created = !known || !event->dest.existed;
|
||||
char update;
|
||||
if (event->is_hardlink)
|
||||
update = 'h';
|
||||
else if (created)
|
||||
update = (event->is_directory || event->is_symlink || event->is_special) ? 'c' : '>';
|
||||
else
|
||||
update = '>';
|
||||
code[0] = update;
|
||||
code[1] = itemize_type_char(event);
|
||||
if (created) {
|
||||
for (int i = 0; i < 9; i++)
|
||||
code[2 + i] = '+';
|
||||
code[11] = '\0';
|
||||
return;
|
||||
}
|
||||
bool size_diff = event->size != event->dest.size;
|
||||
bool time_diff = !times_match(config, event);
|
||||
bool perms_diff = (event->mode & 07777) != (event->dest.mode & 07777);
|
||||
bool owner_diff = event->uid != (uid_t)event->dest.uid;
|
||||
bool group_diff = event->gid != (gid_t)event->dest.gid;
|
||||
code[2] = '.'; /* checksum: no destination digest available */
|
||||
code[3] = size_diff ? 's' : '.';
|
||||
code[4] = time_diff ? 't' : '.';
|
||||
code[5] = (config->preserve_perms && perms_diff) ? 'p' : '.';
|
||||
code[6] = (config->preserve_owner && owner_diff) ? 'o' : '.';
|
||||
code[7] = (config->preserve_group && group_diff) ? 'g' : '.';
|
||||
code[8] = '.'; /* reserved */
|
||||
code[9] = '.'; /* acl: not compared */
|
||||
code[10] = '.';
|
||||
code[11] = '\0';
|
||||
}
|
||||
|
||||
/* rsync %n: the transfer-relative name, with a trailing slash for directories. */
|
||||
static bool append_name(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (!strbuf_append(buf, event->name != NULL ? event->name : ""))
|
||||
return false;
|
||||
if (event->is_directory && (event->name == NULL || event->name[0] == '\0' ||
|
||||
event->name[strlen(event->name) - 1] != '/'))
|
||||
return strbuf_append_char(buf, '/');
|
||||
return true;
|
||||
}
|
||||
|
||||
/* rsync %L: " -> target" for a symlink, " => target" for a hard link, else "". */
|
||||
static bool append_link_suffix(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (event->is_symlink && event->symlink_target != NULL)
|
||||
return strbuf_append(buf, " -> ") && strbuf_append(buf, event->symlink_target);
|
||||
if (event->is_hardlink && event->hardlink_target != NULL)
|
||||
return strbuf_append(buf, " => ") && strbuf_append(buf, event->hardlink_target);
|
||||
return true;
|
||||
}
|
||||
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
StrBuf line = {0};
|
||||
char size_field[32];
|
||||
int written = snprintf(size_field, sizeof(size_field), "%llu", size);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, size_field) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, date) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, path != NULL ? path : "");
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append_char(&line, ' ') &&
|
||||
append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
@@ -228,6 +202,276 @@ char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* rsync's `--info=name` line for an updated entry: the transfer-relative name
|
||||
* (trailing slash for directories) plus the ` -> target` / ` => target` link
|
||||
* suffix. `--info=name` does not alter an itemize/out-format run. */
|
||||
static char* change_render_name(const ChangeEvent* event) {
|
||||
StrBuf line = {0};
|
||||
bool ok = append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* rsync's `--info=name2` line for an unchanged entry: `NAME is uptodate`. */
|
||||
static char* change_render_name_uptodate(const ChangeEvent* event) {
|
||||
char* name = change_render_name(event);
|
||||
if (name == NULL)
|
||||
return NULL;
|
||||
size_t length = strlen(name);
|
||||
char* line = malloc(length + sizeof(" is uptodate"));
|
||||
if (line == NULL) {
|
||||
free(name);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(line, name, length);
|
||||
memcpy(line + length, " is uptodate", sizeof(" is uptodate"));
|
||||
free(name);
|
||||
return line;
|
||||
}
|
||||
|
||||
/* ---- --out-format / --log-file-format ---- */
|
||||
|
||||
/* rsync 3.4.1's `%C` uses the negotiated TRANSFER checksum (the first name of a
|
||||
* two-name "transfer,pre-transfer" --checksum-choice), not the pre-transfer
|
||||
* whole-file digest FastSync compares against on the wire. The default "auto"
|
||||
* resolves to xxh128, so an explicit selection and the default both render the
|
||||
* selected algorithm's digest. */
|
||||
static ChecksumAlgo out_format_checksum_algo(const Config* config) {
|
||||
return (ChecksumAlgo)config->checksum_transfer_algo;
|
||||
}
|
||||
|
||||
/* Render a digest as rsync's sum_as_hex: xxh128 prints the HIGH 64-bit half
|
||||
* before the low half, and xxh64/xxh3 print their 64-bit value big-endian; every
|
||||
* other algorithm prints its bytes in order. */
|
||||
static void digest_to_hex(ChecksumAlgo algo, const uint8_t* digest, size_t len, char* out) {
|
||||
if (algo == CHECKSUM_ALGO_XXH128 && len == 16) {
|
||||
uint64_t low = 0;
|
||||
uint64_t high = 0;
|
||||
memcpy(&low, digest, sizeof(low));
|
||||
memcpy(&high, digest + 8, sizeof(high));
|
||||
snprintf(out, len * 2 + 1, "%016llx%016llx", (unsigned long long)high, (unsigned long long)low);
|
||||
return;
|
||||
}
|
||||
if ((algo == CHECKSUM_ALGO_XXH64 || algo == CHECKSUM_ALGO_XXH3) && len == 8) {
|
||||
uint64_t value = 0;
|
||||
memcpy(&value, digest, sizeof(value));
|
||||
snprintf(out, len * 2 + 1, "%016llx", (unsigned long long)value);
|
||||
return;
|
||||
}
|
||||
static const char hex[] = "0123456789abcdef";
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
out[i * 2] = hex[(digest[i] >> 4) & 0xf];
|
||||
out[i * 2 + 1] = hex[digest[i] & 0xf];
|
||||
}
|
||||
out[len * 2] = '\0';
|
||||
}
|
||||
|
||||
static bool format_uses_checksum(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0')
|
||||
break;
|
||||
if (token == 'C')
|
||||
return true;
|
||||
p += 2;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Fill event->checksum/checksum_known for a transferred regular file. A
|
||||
* non-regular entry (or a hard-link sibling) leaves checksum_known false, which
|
||||
* renders as spaces like rsync. */
|
||||
static void fill_event_checksum(const Config* config, const File* file, ChangeEvent* event) {
|
||||
if (file == NULL || file->is_dir || file->is_symlink || file->is_special ||
|
||||
(file->link_group != 0 && !file->link_first))
|
||||
return;
|
||||
if (!format_uses_checksum(config->out_format) && !format_uses_checksum(config->log_file_format))
|
||||
return;
|
||||
if (file->path == NULL)
|
||||
return;
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
/* rsync renders `--checksum-choice=none` as a blank 2-character column. */
|
||||
if (algo == CHECKSUM_ALGO_NONE)
|
||||
return;
|
||||
uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
|
||||
size_t len = 0;
|
||||
/* rsync's %C is the transfer checksum, which is always seeded with 0 (it is
|
||||
* independent of --checksum-seed, as rsync 3.4.1 demonstrates). */
|
||||
if (!checksum_digest_file(algo, 0, file->path, digest, sizeof(digest), &len))
|
||||
return;
|
||||
digest_to_hex(algo, digest, len, event->checksum);
|
||||
event->checksum_known = true;
|
||||
}
|
||||
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event) {
|
||||
if (format == NULL || event == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'i': {
|
||||
if (event->deleted) {
|
||||
/* rsync's ITEM_DELETED itemize code: `*deleting ` (11 chars). */
|
||||
ok = strbuf_append(&line, "*deleting ");
|
||||
break;
|
||||
}
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
ok = strbuf_append(&line, code);
|
||||
break;
|
||||
}
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = append_name(&line, event);
|
||||
break;
|
||||
case 'L':
|
||||
ok = append_link_suffix(&line, event);
|
||||
break;
|
||||
case 'l': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->size);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'b': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_sent);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'c': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_read);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'C': {
|
||||
if (event->checksum_known) {
|
||||
ok = strbuf_append(&line, event->checksum);
|
||||
} else {
|
||||
/* rsync pads a non-regular / untransferred / `none` entry with spaces;
|
||||
`none` renders as a blank 2-character column. */
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
int width = algo == CHECKSUM_ALGO_NONE ? 2 : checksum_digest_len(algo) * 2;
|
||||
for (int i = 0; i < width && ok; i++)
|
||||
ok = strbuf_append_char(&line, ' ');
|
||||
}
|
||||
} break;
|
||||
case 'M': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(event->mtime_sec, true, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 't': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(time(NULL), false, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 'o':
|
||||
ok = strbuf_append(&line, "send");
|
||||
break;
|
||||
case 'p': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%ld", (long)getpid());
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'B': {
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
ok = strbuf_append(&line, permission + 1);
|
||||
} break;
|
||||
case 'U': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->uid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'G': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->gid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- --list-only ---- */
|
||||
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event) {
|
||||
(void)config;
|
||||
if (event == NULL)
|
||||
return NULL;
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
char date[32];
|
||||
if (!format_rsync_datetime(event->mtime_sec, false, date, sizeof(date)))
|
||||
snprintf(date, sizeof(date), "?");
|
||||
StrBuf line = {0};
|
||||
char size_field[40];
|
||||
char grouped[32];
|
||||
if (!format_big_num(event->size, false, grouped, sizeof(grouped))) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
int written = snprintf(size_field, sizeof(size_field), "%15s", grouped);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
const char* name = event->name != NULL && event->name[0] != '\0' ? event->name : ".";
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append(&line, size_field) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, date) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, name);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- Event emission ---- */
|
||||
|
||||
static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_output) {
|
||||
char* escaped = output_escape(line, eight_bit_output);
|
||||
if (escaped != NULL) {
|
||||
@@ -242,20 +486,49 @@ static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_ou
|
||||
void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
if (event->decision == CHANGE_UP_TO_DATE)
|
||||
return;
|
||||
bool to_stdout = config->itemize_changes || config->out_format != NULL;
|
||||
bool to_log = config->log_file != NULL && config->log_file_format != NULL;
|
||||
bool progress_active = config->show_progress || (config->info_level & LOG_INFO_PROGRESS);
|
||||
if (event->decision == CHANGE_UP_TO_DATE) {
|
||||
/* --info=name2 prints `NAME is uptodate` for entries the receiver already
|
||||
had. An itemize/out-format run reports them through its own format (or
|
||||
not at all), the progress stream has no frame for them, and neither the
|
||||
itemize nor the log-file stream previously reported an up-to-date entry,
|
||||
so nothing else here changes. */
|
||||
if (!to_stdout && (config->info_level & LOG_INFO_NAME_UPTODATE) != 0 && !progress_active) {
|
||||
char* line = change_render_name_uptodate(event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (to_stdout) {
|
||||
char* line = config->out_format != NULL ? change_render_format(config->out_format, event)
|
||||
: change_render_itemize(event);
|
||||
char* line = config->out_format != NULL
|
||||
? change_render_format(config->out_format, config, event)
|
||||
: change_render_itemize(config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
} else if ((config->info_level & LOG_INFO_NAME) != 0 && !progress_active) {
|
||||
/* --info=name without -i/--out-format: print the updated entry's name. The
|
||||
--progress path owns the name line when progress output is active (it
|
||||
emits the same names before the progress frames), so do not duplicate.
|
||||
The transfer-root `./` line precedes the first such name. */
|
||||
if (!name_root_printed) {
|
||||
name_root_printed = true;
|
||||
fputs("./\n", stdout);
|
||||
}
|
||||
char* line = change_render_name(event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
if (to_log) {
|
||||
char* line = change_render_format(config->log_file_format, event);
|
||||
char* line = change_render_format(config->log_file_format, config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(config->log_file, line, config->eight_bit_output);
|
||||
free(line);
|
||||
@@ -266,9 +539,6 @@ void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
static bool format_uses_mtime(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
/* Mirror change_render_format's tokenizer: "%%" is a literal percent (so
|
||||
* "%%M" does NOT expand %M) and unknown "%X" escapes consume both chars.
|
||||
* This keeps the optional stat() fallback below in step with the renderer. */
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
@@ -284,46 +554,176 @@ static bool format_uses_mtime(const char* format) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
/* Relative path of an entry below the transfer root (no leading slash). Uses
|
||||
* the sender-side send_path override when present (bare-relative -R layout). */
|
||||
static char* relative_name(const Config* config, const File* file) {
|
||||
const char* full = file_wire_path(file);
|
||||
if (file->send_path != NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL || full == NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, full, root_len) == 0) {
|
||||
if (full[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (full[root_len] == '/')
|
||||
return str_dup(full + root_len + 1);
|
||||
}
|
||||
return str_dup(full);
|
||||
}
|
||||
|
||||
/* rsync %f long form: the source argument as typed (leading '/' removed,
|
||||
* trailing '/' removed, leading "./" removed) joined to the relative name. */
|
||||
static char* display_name(const Config* config, const char* name) {
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL)
|
||||
return str_dup(name != NULL ? name : "");
|
||||
const char* p = root;
|
||||
while (*p == '/')
|
||||
p++;
|
||||
if (p[0] == '.' && p[1] == '/')
|
||||
p += 2;
|
||||
size_t root_len = strlen(p);
|
||||
while (root_len > 0 && p[root_len - 1] == '/')
|
||||
root_len--;
|
||||
size_t name_len = name != NULL ? strlen(name) : 0;
|
||||
if (root_len == 0 && name_len == 0)
|
||||
return str_dup("");
|
||||
char* out = malloc(root_len + (root_len > 0 && name_len > 0 ? 1 : 0) + name_len + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t offset = 0;
|
||||
if (root_len > 0) {
|
||||
memcpy(out, p, root_len);
|
||||
offset = root_len;
|
||||
}
|
||||
if (root_len > 0 && name_len > 0)
|
||||
out[offset++] = '/';
|
||||
if (name_len > 0)
|
||||
memcpy(out + offset, name, name_len);
|
||||
out[offset + name_len] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
static void fill_event_from_file(const Config* config, const File* file, ChangeEvent* event,
|
||||
char** name_out, char** path_out) {
|
||||
char* name = relative_name(config, file);
|
||||
char* path = display_name(config, name);
|
||||
event->name = name;
|
||||
event->path = path;
|
||||
*name_out = name;
|
||||
*path_out = path;
|
||||
if (file->metadata != NULL) {
|
||||
event->mtime_sec = file->metadata->mtime_sec;
|
||||
event->mtime_nsec = file->metadata->mtime_nsec;
|
||||
event->mode = file->metadata->mode;
|
||||
event->uid = file->metadata->uid;
|
||||
event->gid = file->metadata->gid;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0) {
|
||||
event->mtime_sec = st.st_mtime;
|
||||
event->mtime_nsec = st.st_mtim.tv_nsec;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
/* The displayed path is the one transmitted (with -R + --files-from this is
|
||||
the bare relative destination path); the metadata fallback below still
|
||||
stats the local absolute path. */
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = false;
|
||||
event.is_special = false;
|
||||
event.is_hardlink = false;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
/* FastSync has no wire-byte counter yet, so %b reports the source length
|
||||
* that had to be delivered (always equal to %l); the actual bytes written
|
||||
* to the socket (compressed/delta) are not measured. */
|
||||
event.bytes_sent = event.size;
|
||||
if (file->metadata != NULL) {
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
/* Best-effort fallback for %M when no metadata was captured (no -M): the
|
||||
* path is stat()ed just to fill the field, and any failure leaves 0. */
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0)
|
||||
event.mtime_sec = st.st_mtime;
|
||||
event.dest = file->dest_state;
|
||||
if (file->is_symlink) {
|
||||
event.is_symlink = true;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.size = file->symlink_target != NULL ? strlen(file->symlink_target) : 0;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->is_special) {
|
||||
event.is_special = true;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->link_group != 0 && !file->link_first) {
|
||||
event.is_hardlink = true;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.bytes_sent = 0;
|
||||
} else {
|
||||
event.bytes_sent = bytes_sent;
|
||||
/* rsync's %c is the block-checksum bytes received for the file. Even a
|
||||
* whole-file transfer (no basis; --append/--inplace included) receives
|
||||
* rsync's 16-byte sum header, so rsync reports 16; a dry run transfers
|
||||
* nothing and reports 0. FastSync's whole-file path has no sum header, so
|
||||
* report rsync's value for parity. With delta enabled the real received
|
||||
* bytes are kept, but FastSync's signature framing differs from rsync's so
|
||||
* those stay numerically divergent. */
|
||||
bool delta_active = config->use_delta && !config->whole_file;
|
||||
event.bytes_read = (!config->dry_run && !delta_active) ? 16 : bytes_read;
|
||||
}
|
||||
change_emit(config, &event);
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL) {
|
||||
fill_event_checksum(config, file, &event);
|
||||
change_emit(config, &event);
|
||||
}
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
if (file == NULL)
|
||||
return;
|
||||
unsigned long long payload = file->data != NULL ? file->data->size : 0;
|
||||
change_emit_file_sent_bytes(config, file, payload, 0);
|
||||
}
|
||||
|
||||
void change_emit_file_uptodate(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.decision = CHANGE_UP_TO_DATE;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = file->is_symlink;
|
||||
event.is_special = file->is_special;
|
||||
event.is_hardlink = file->link_group != 0 && !file->link_first;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
void change_emit_dir_sent(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = true;
|
||||
event.size = 0;
|
||||
event.bytes_sent = 0;
|
||||
if (file->metadata != NULL)
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
change_emit(config, &event);
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
+61
-25
@@ -2,7 +2,9 @@
|
||||
#define CHANGE_LIST_H
|
||||
|
||||
#include "config.h"
|
||||
#include "checksum.h"
|
||||
#include "file_types.h"
|
||||
#include "format.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
@@ -26,42 +28,59 @@ typedef enum {
|
||||
} ChangeDecision;
|
||||
|
||||
typedef struct {
|
||||
const char* path; /* full source path */
|
||||
const char* path; /* long-form display path (rsync %f) */
|
||||
const char* name; /* transfer-relative path (rsync %n), no trailing slash */
|
||||
ChangeDecision decision;
|
||||
bool is_directory;
|
||||
unsigned long long size; /* source file length in bytes */
|
||||
/* The number of bytes reported for a sent file. FastSync has no wire-byte
|
||||
* counter, so this is always the source length (== size / %l); actual
|
||||
* post-compression/delta bytes on the wire are not counted. */
|
||||
unsigned long long bytes_sent;
|
||||
time_t mtime_sec; /* 0 when unknown */
|
||||
bool is_symlink;
|
||||
bool is_special;
|
||||
bool is_hardlink; /* a hard-link sibling (linked, no data sent) */
|
||||
bool deleted; /* a would-delete report (-n --delete); no source file */
|
||||
const char* symlink_target;
|
||||
const char* hardlink_target;
|
||||
unsigned long long size; /* source file length in bytes */
|
||||
unsigned long long bytes_sent; /* wire bytes actually transferred (rsync %b) */
|
||||
unsigned long long bytes_read; /* wire bytes read back for this file (rsync %c) */
|
||||
/* rsync %C: whole-file checksum hex for a transferred regular file. Only
|
||||
* filled when the active format uses %C (checksum_known == false otherwise,
|
||||
* which renders as spaces like rsync for non-regular entries). */
|
||||
bool checksum_known;
|
||||
char checksum[CHECKSUM_MAX_DIGEST_LEN * 2 + 1];
|
||||
time_t mtime_sec;
|
||||
long mtime_nsec;
|
||||
mode_t mode;
|
||||
uid_t uid;
|
||||
gid_t gid;
|
||||
/* Receiver-reported pre-transfer destination state (OutputDestState.known is
|
||||
* false when no report was requested/received). */
|
||||
OutputDestState dest;
|
||||
} ChangeEvent;
|
||||
|
||||
/* True when any output mode is active and per-file events matter. */
|
||||
bool change_list_enabled(const Config* config);
|
||||
|
||||
/* Render the rsync-style itemize line for a transferred file:
|
||||
* `>f+++++++++ <path>`
|
||||
* The 11-char code is `>f` (regular file transferred to the remote host)
|
||||
* followed by c/s/t/p/o/g/u/a/x markers that are all `+` (value will be set
|
||||
* / differs) because FastSync does not separately compare checksums, size,
|
||||
* mtime, perms, owner, group, uid, acl, or xattr on the receiving side, so a
|
||||
* sent file is reported as fully updated. Up-to-date files print no line
|
||||
* (rsync single `-i` only shows changes). Caller frees the result. */
|
||||
char* change_render_itemize(const ChangeEvent* event);
|
||||
/* Render the rsync-style itemize line for a transferred item
|
||||
* (`%i %n%L`): `>f+++++++++ sub/b.txt`. Caller frees the result. */
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Expand an --out-format/--log-file-format template. Tokens:
|
||||
* %f full source path %b "bytes sent" == the source length (%l);
|
||||
* %n leaf (base) name actual post-compression/delta wire bytes
|
||||
* %l file length in bytes are not counted
|
||||
* %M mtime in whole seconds %% a literal percent sign
|
||||
/* Expand an --out-format/--log-file-format template. Supported tokens:
|
||||
* %i itemize code %n transfer-relative name (dir: trailing /)
|
||||
* %f long display path %l file length in bytes
|
||||
* %b wire bytes transferred %c block-checksum bytes received (rsync: 16
|
||||
* for a whole-file transfer, 0 for a dry run)
|
||||
* %C whole-file checksum hex (xxh128 by default; spaces for non-regular)
|
||||
* %M mtime (YYYY/MM/DD-HH:MM:SS)
|
||||
* %t current time %o operation ("send"/"del.")
|
||||
* %p pid %B permission bits without the type char
|
||||
* %U uid %G gid
|
||||
* %L " -> target" / " => target" %% a literal percent sign
|
||||
* Unknown %X sequences are preserved verbatim. Caller frees the result. */
|
||||
char* change_render_format(const char* format, const ChangeEvent* event);
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Render one --list-only long-listing entry:
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 <path>`
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 sub/b.txt`
|
||||
* (ls -l style columns; mtime in the local time zone). Caller frees it. */
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime, const char* path);
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Emit an event to every active destination:
|
||||
* stdout: --itemize-changes line, or the --out-format expansion when set;
|
||||
@@ -69,10 +88,27 @@ char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime
|
||||
* CHANGE_UP_TO_DATE events produce no output. */
|
||||
void change_emit(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent. */
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent. `bytes_sent`
|
||||
* is the process-wide wire-byte delta for this file (rsync's %b) and `bytes_read`
|
||||
* the received bytes used for the delta handshake; pass 0 when unknown. For a
|
||||
* whole-file transfer %c is pinned to rsync's 16-byte sum header regardless. */
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent, deriving
|
||||
* the wire byte counts from the source payload length. */
|
||||
void change_emit_file_sent(const Config* config, const File* file);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
void change_emit_dir_sent(const Config* config, const File* file);
|
||||
|
||||
/* Build and emit a CHANGE_UP_TO_DATE event for a file the receiver already had.
|
||||
* With --info=name2 it renders rsync's "NAME is uptodate" line (no output
|
||||
* otherwise). */
|
||||
void change_emit_file_uptodate(const Config* config, const File* file);
|
||||
|
||||
/* Reset the lazy transfer-root `./` line emitted ahead of the first
|
||||
* --info=name entry. Call once at the start of a transfer. */
|
||||
void change_reset_name_root(void);
|
||||
|
||||
#endif
|
||||
|
||||
+1203
-219
File diff suppressed because it is too large
Load Diff
+1570
-385
File diff suppressed because it is too large
Load Diff
@@ -23,6 +23,12 @@ void client_set_abort_armed(bool armed);
|
||||
* config_delete() once the call returns). */
|
||||
int send_files(Config* config);
|
||||
int send_files_multithreaded(Config** config);
|
||||
/* rsync's --ignore-errors deletion gate: with no I/O error during the scan the
|
||||
* deletion phase always proceeds; with one it is suppressed unless
|
||||
* `--ignore-errors` was given. Exposed so the decision can be unit-tested
|
||||
* without a privileged (mode-000) source directory. See client_send.c. */
|
||||
bool ignore_errors_allows_delete(const Config* config, bool had_io_error);
|
||||
|
||||
/* Phase 6 residual-batch (client-only). See client_send.c. */
|
||||
int write_batch_from_source(const Config* config, const char* batch_path);
|
||||
int apply_batch_to_dest(const Config* config, const char* batch_path, const char* dest_root);
|
||||
|
||||
@@ -60,6 +60,16 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport");
|
||||
return false;
|
||||
}
|
||||
/* -M/--remote-option appends an option to the REMOTE server's argv, which
|
||||
* only exists on the SSH (user@host:path) transport. A daemon
|
||||
* (host::module/path) or local TCP destination has no remote command line,
|
||||
* so the option would be silently ignored; reject it by name instead. */
|
||||
if (config->remote_option_count > 0 && config->transport != TRANSPORT_SSH) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"-M/--remote-option is only valid with the SSH transport (user@host:path); it "
|
||||
"cannot be used with a daemon (host::module/path) or local TCP destination");
|
||||
return false;
|
||||
}
|
||||
/* -4 and -6 are mutually exclusive: a socket address family cannot be both. */
|
||||
if (config->ipv4 && config->ipv6) {
|
||||
log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive");
|
||||
|
||||
+1106
-245
File diff suppressed because it is too large
Load Diff
+113
-16
@@ -10,6 +10,7 @@
|
||||
#include "stop_condition.h"
|
||||
#include <dirent.h>
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdatomic.h>
|
||||
#include <sys/types.h>
|
||||
#include <threads.h>
|
||||
@@ -50,7 +51,7 @@ typedef struct {
|
||||
bool copy_dirlinks;
|
||||
bool munge_links;
|
||||
bool checksum;
|
||||
bool one_file_system;
|
||||
int one_file_system;
|
||||
/* Phase 4 special/devices: whether device nodes (--devices) and special files
|
||||
* (--specials) are preserved via recreation, and whether --copy-devices
|
||||
* copies a device's content as an ordinary regular file. */
|
||||
@@ -63,8 +64,23 @@ typedef struct {
|
||||
const FileListSet* file_list; /* --files-from allow-set, or NULL */
|
||||
const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */
|
||||
bool per_dir_filters; /* -F: read .rsync-filter per directory */
|
||||
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
|
||||
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
|
||||
/* --delete-excluded: per-directory plain rules become sender-only, so they no
|
||||
longer protect the receiver from deletion. */
|
||||
bool delete_excluded;
|
||||
/* -FF: also exclude the per-directory filter files themselves from the
|
||||
transfer (single -F transfers them). */
|
||||
bool exclude_per_dir_filter_files;
|
||||
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
|
||||
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
|
||||
/* -R/--relative outside --files-from: the destination-relative path prefix
|
||||
* reconstructed from the source spec (rsync's '/./' cut point), or NULL when
|
||||
* -R is off or --files-from is in use (the bare-relative path then comes from
|
||||
* the listed entry). Borrowed read-only; owned by client_send. */
|
||||
const char* relative_prefix;
|
||||
/* --list-only: emit an is_dir File for every traversed directory (the listing
|
||||
* includes directory entries, matching rsync). Client-only; never set on a
|
||||
* real transfer, which relies on implicit parent creation. */
|
||||
bool list_dirs;
|
||||
/* --prune-empty-dirs (long only): in --dirs mode an empty source directory's
|
||||
explicit entry is omitted from the transfer file list (so nothing is
|
||||
created at the destination and it can be pruned by --delete); explicitly
|
||||
@@ -73,20 +89,62 @@ typedef struct {
|
||||
bool prune_empty_dirs;
|
||||
/* Delete-excluded protection sink (optional): when non-NULL the scanner
|
||||
* appends the destination-relative path of every entry it prunes because a
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules, the legacy
|
||||
* --exclude/--include layer, and --max-size/--min-size). The sender turns
|
||||
* this list into the manifest's protected prefixes so `--delete` leaves the
|
||||
* destination mirror of excluded source paths alone (rsync's default), and
|
||||
* empties it when --delete-excluded opts back into deleting them. NOT
|
||||
* recorded for --files-from subset pruning (whose delete semantics stay
|
||||
* keep-set-only) or for -R/--files-from relative wire paths. When
|
||||
* `excluded_mutex` is non-NULL it is taken around every append (the parallel
|
||||
* scanner shares one list across its worker threads). */
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules and the legacy
|
||||
* --exclude/--include layer). The sender turns this list into the manifest's
|
||||
* protected prefixes so `--delete` leaves the destination mirror of excluded
|
||||
* source paths alone (rsync's default), and drops it when --delete-excluded
|
||||
* opts back into deleting them. NOT recorded for --files-from subset pruning
|
||||
* (whose delete semantics derive from the synchronized-directory set) or for
|
||||
* -R/--files-from relative wire paths. When `excluded_mutex` is non-NULL it
|
||||
* is taken around every append (the parallel scanner shares one list across
|
||||
* its worker threads). */
|
||||
ArrayList* excluded_paths;
|
||||
mtx_t* excluded_mutex;
|
||||
/* --ignore-errors: an unreadable directory during the scan is recorded as an
|
||||
* I/O error and skipped instead of aborting the scan. Client-only. */
|
||||
/* Size-prune protection sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every entry it skipped because of
|
||||
* --max-size/--min-size. rsync never deletes a size-skipped source mirror,
|
||||
* even under --delete-excluded, so the sender always transmits this list as
|
||||
* protected prefixes (unlike excluded_paths, which --delete-excluded drops).
|
||||
* Guarded by `excluded_mutex` like excluded_paths. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Synchronized-directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it is about to traverse
|
||||
* that lies inside a --files-from listed directory (or of every traversed
|
||||
* directory when there is no list). The sender sends this set with the delete
|
||||
* manifest so the receiver confines its extras walk to synchronized
|
||||
* directories, exactly like rsync; the receive root is the "." sentinel.
|
||||
* Guarded by `excluded_mutex`. */
|
||||
ArrayList* synced_dirs;
|
||||
/* Delete-plan directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it traverses (except the
|
||||
* receive root). The per-directory --delete-during/--delete-delay plan
|
||||
* builder uses this to keep an empty in-scope source directory (rsync keeps
|
||||
* it) and to emit its plan after the data stream, when no file frame would
|
||||
* otherwise trigger it. Guarded by `excluded_mutex`. */
|
||||
ArrayList* plan_dirs;
|
||||
/* --ignore-errors: an unreadable subdirectory no longer aborts the scan (it
|
||||
* is always skipped so the rest of the tree transfers); this flag is kept so
|
||||
* the client can distinguish the option state when deciding deletion policy.
|
||||
* Client-only. */
|
||||
bool ignore_io_errors;
|
||||
/* --info=nonreg: print rsync's `skipping non-regular file "NAME"` line for a
|
||||
* non-regular entry that is not being preserved. Client-only. */
|
||||
bool note_nonreg;
|
||||
/* --info=mount: print rsync's `[sender] skipping mount-point dir NAME` when
|
||||
* -xx drops a mount-point directory. Client-only. */
|
||||
bool note_mount;
|
||||
/* --stats directory accounting for a `-r` run (no -t/-p): a shared counter of
|
||||
* traversed directories that are NOT otherwise represented by an inline
|
||||
* directory entry (rsync still counts every directory in `Number of files`).
|
||||
* Incremented when a directory is opened and decremented when an empty
|
||||
* directory is emitted inline (so it is counted exactly once). Atomic
|
||||
* because the parallel scanner's workers share it; NULL disables the
|
||||
* accounting. Client-only. */
|
||||
atomic_ullong* dir_count;
|
||||
/* Source root and 8-bit-output policy used to render a `--info=nonreg` name
|
||||
* relative to the transfer root. Borrowed read-only. */
|
||||
const char* send_directory;
|
||||
bool eight_bit_output;
|
||||
/* --ignore-missing-args (implied by --delete-missing-args): an explicitly
|
||||
* --files-from-listed entry that does not exist under the source is skipped
|
||||
* instead of failing (the --dirs generator is the only scanner path that
|
||||
@@ -114,6 +172,17 @@ typedef struct {
|
||||
bool capture_dir_times;
|
||||
ArrayList* dir_entries;
|
||||
mtx_t* dir_entries_mutex;
|
||||
/* Recreate empty source directories on a recursive transfer: emit a
|
||||
* payload-less directory entry for every traversed directory that produced
|
||||
* no transferred/descended child. Off by default so low-level scanner users
|
||||
* (unit helpers, --list-only) see only the historical file list; the real
|
||||
* sender sets it in prepare_scanner. */
|
||||
bool emit_empty_dirs;
|
||||
/* --no-implied-dirs with -R + --files-from: a directory that is only an
|
||||
* implied parent of a listed entry (not itself listed, nor below a listed
|
||||
* directory) must not carry source metadata; it is created with default
|
||||
* attributes at the destination, matching rsync. */
|
||||
bool no_implied_dirs;
|
||||
} ScannerOptions;
|
||||
|
||||
/* Internal per-scanner filter state. FilterNode chains represent the ordered
|
||||
@@ -131,6 +200,22 @@ typedef struct {
|
||||
int current_depth;
|
||||
dev_t root_dev;
|
||||
bool failed;
|
||||
/* rsync-order traversal: each opened directory's entries are inspected once
|
||||
and buffered (an internal SortedEntry[] owned here) sorted as rsync's flist
|
||||
orders them -- non-directories ascending, then directories ascending. The
|
||||
entries are walked in order and child directories are collected in
|
||||
`pending_dirs` (an ArrayList of DirEntry*, owned here) and pushed onto the
|
||||
LIFO `directories` stack in reverse at directory exhaustion, so the emitted
|
||||
stream is depth-first like rsync. `sorted_*` are reset per directory. */
|
||||
void* sorted_entries;
|
||||
size_t sorted_count;
|
||||
size_t sorted_index;
|
||||
void* pending_dirs;
|
||||
/* Recursive scan: whether the open directory yielded any transferred or
|
||||
descended entry. When it did not, closing it emits a directory entry so
|
||||
the empty source directory is recreated at the destination (rsync
|
||||
parity). */
|
||||
bool current_dir_produced;
|
||||
/* Phase 2 (files-from / filter layer). */
|
||||
char* root_path; /* transfer root (fs path) for rel computation */
|
||||
char* current_rel; /* rel path of the open directory ("" == root) */
|
||||
@@ -149,6 +234,10 @@ typedef struct {
|
||||
--ignore-errors the scan continues past it and the caller decides what to
|
||||
do; `failed` is reserved for fatal errors that always abort the scan. */
|
||||
bool io_error;
|
||||
/* The transfer ROOT could not be opened. It is always fatal, even under
|
||||
--ignore-errors, but the client still maps it to rsync's partial-transfer
|
||||
exit (23) rather than a generic failure. */
|
||||
bool root_io_error;
|
||||
} DirectoryScanner;
|
||||
|
||||
typedef struct {
|
||||
@@ -168,7 +257,8 @@ typedef struct {
|
||||
int completed;
|
||||
Chunk* initial_chunk;
|
||||
ProtocolSession* allocation_session;
|
||||
FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
|
||||
FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
|
||||
const ScannerOptions* options; /* borrowed scan options (--info=nonreg output) */
|
||||
} ParallelScanner;
|
||||
|
||||
DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata,
|
||||
@@ -187,13 +277,20 @@ void directory_scanner_destroy(DirectoryScanner* scanner);
|
||||
/* --one-file-system (-x) decision: a directory entry may be descended into
|
||||
* only when the option is disabled or the entry lives on the same device as
|
||||
* the transfer root. Exposed so tests can exercise the rule directly. */
|
||||
bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entry_device);
|
||||
bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device);
|
||||
|
||||
/* Relative path of an on-disk path below `root` ("" == the root itself, NULL
|
||||
* when `fs_path` is not under `root`). Handles trailing slashes and a root of
|
||||
* "/". Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_path_relative(const char* root, const char* fs_path);
|
||||
|
||||
/* -R/--relative destination-relative prefix reconstructed from a source spec:
|
||||
* the path after rsync's first '.' path component (the '/./' cut point), with
|
||||
* leading/trailing slashes removed, or the whole spec (normalized) when there
|
||||
* is no cut. Returns "" for the receive root, or NULL when `spec` is NULL or
|
||||
* allocation fails. Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_relative_prefix(const char* spec);
|
||||
|
||||
ParallelScanner* parallel_scanner_create_with_options(const char* root_directory,
|
||||
const ScannerOptions* options,
|
||||
ProtocolSession* allocation_session);
|
||||
|
||||
+146
-81
@@ -20,11 +20,16 @@ void print_usage(void) {
|
||||
printf("Options:\n");
|
||||
printf(" -c, --checksum Verify content by checksum instead of size+mtime\n");
|
||||
printf(" -z, --compress [level] Enable compression (level 1-22, default 5)\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, metadata,\n");
|
||||
printf(" devices and specials (not compression/multithreading)\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, perms, times,\n");
|
||||
printf(" owner, group, devices and specials; not\n");
|
||||
printf(" compression/multithreading\n");
|
||||
printf(" -r, --recursive Recurse into directories (FastSync is always recursive)\n");
|
||||
printf(" -n, --dry-run Show what would be transferred\n");
|
||||
printf(" --remove-source-files Remove regular source files after successful transfer\n");
|
||||
printf(" -p, --perms Preserve permission bits (part of the metadata bundle)\n");
|
||||
printf(" -p, --perms Preserve permission bits\n");
|
||||
printf(" -t, --times Preserve modification times\n");
|
||||
printf(" -o, --owner Preserve owner (uid)\n");
|
||||
printf(" -g, --group Preserve group (gid)\n");
|
||||
printf(" --ssh-port <port> SSH port (default: 22)\n");
|
||||
printf(" -e, --rsh <command> Remote shell to launch on the client for the SSH\n");
|
||||
printf(" transport (default: ssh). The command may include\n");
|
||||
@@ -54,24 +59,27 @@ void print_usage(void) {
|
||||
printf(" Emit the batch file only (no destination, no server)\n");
|
||||
printf(" --read-batch=FILE Apply the batch file to the destination (no source, no\n");
|
||||
printf(" server); takes only the destination as an argument\n");
|
||||
printf(" NOTE: the FastSync batch format is NOT interoperable with rsync's batch\n");
|
||||
printf(" files (different container format); do not mix the two tools.\n");
|
||||
printf(" --delete Delete files on receiver not in source\n");
|
||||
printf(" (default timing: delete only after the whole\n");
|
||||
printf(" transfer has succeeded)\n");
|
||||
printf(" (default timing: delete-during, like rsync --del)\n");
|
||||
printf(" --delete-before Delete extras before the transfer starts\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-during Delete extras once the keep-set manifest is known,\n");
|
||||
printf(" before the data is applied (implies --delete)\n");
|
||||
printf(" --delete-during Delete a directory's extras as that directory is\n");
|
||||
printf(" processed (implies --delete)\n");
|
||||
printf(" --del Alias for --delete-during\n");
|
||||
printf(" --delete-delay Delete extras only after a successful transfer\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-delay Record the extras during the scan but remove them\n");
|
||||
printf(" only after a successful transfer (implies --delete)\n");
|
||||
printf(" --delete-after Delete only after the whole transfer succeeded\n");
|
||||
printf(" (the default --delete timing; implies --delete)\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-commit FastSync-only: restore the late whole-tree commit\n");
|
||||
printf(" (identical to --delete-after; implies --delete)\n");
|
||||
printf(" --delete-excluded Also delete destination files that were excluded on\n");
|
||||
printf(" the source (default protects them, matching rsync)\n");
|
||||
printf(" --max-delete=NUM Never delete more than NUM destination entries per run;\n");
|
||||
printf(" if the extras would exceed NUM, nothing is deleted and\n");
|
||||
printf(" the run fails with a clear error (implies --delete only\n");
|
||||
printf(" when used with it)\n");
|
||||
printf(" --max-delete=NUM Delete at most NUM destination entries per run; if the\n");
|
||||
printf(" extras exceed NUM, the rest are skipped and the run is\n");
|
||||
printf(" reported as partial (exit 25, matching rsync). Only\n");
|
||||
printf(" applies together with --delete\n");
|
||||
printf(" --ignore-errors Continue (and still delete) when a source directory is\n");
|
||||
printf(" unreadable during the scan, instead of aborting with no\n");
|
||||
printf(" deletion\n");
|
||||
@@ -83,10 +91,11 @@ void print_usage(void) {
|
||||
printf(" entry's destination mirror receiver-side. Independent of\n");
|
||||
printf(" --delete (it does not imply --delete; a non-empty directory\n");
|
||||
printf(" mirror is removed only with --force or --delete)\n");
|
||||
printf(" -m, --prune-empty-dirs Do not transfer empty directory entries (--dirs mode);\n");
|
||||
printf(" recursive transfers never send empty dirs\n");
|
||||
printf(" -m, --prune-empty-dirs Do not create empty directories (a recursive transfer\n");
|
||||
printf(" otherwise recreates them, like rsync)\n");
|
||||
printf(" Note: each timing flag implies --delete. Combining a timing flag with\n");
|
||||
printf(" --no-delete (in either order) is rejected as a config error.\n");
|
||||
printf(" --no-delete (in either order) is rejected as a config error, as is more\n");
|
||||
printf(" than one timing flag.\n");
|
||||
printf(" --ignore-existing Skip files that already exist on receiver\n");
|
||||
printf(" --delay-updates Put updated files into place only at the end of transfer\n");
|
||||
printf(" --dirs, -d, --old-dirs, --old-d Transfer the named directory entries without\n");
|
||||
@@ -96,24 +105,28 @@ void print_usage(void) {
|
||||
printf(" -R, --relative With --files-from, preserve each listed entry's relative path\n");
|
||||
printf(" below the destination root instead of mirroring the full\n");
|
||||
printf(" source path (no effect without --files-from)\n");
|
||||
printf(" --no-implied-dirs With -R --files-from, refuse to place a listed file whose\n");
|
||||
printf(" parent directory is not itself listed\n");
|
||||
printf(" --no-implied-dirs With -R, do not apply the source metadata of a listed file's\n");
|
||||
printf(" implied parent directories (they are still created with\n");
|
||||
printf(" default attributes)\n");
|
||||
printf(" --mkpath Create the destination root directory on the server when it\n");
|
||||
printf(" does not exist yet\n");
|
||||
printf(" --exclude <pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file> Read include patterns from file\n");
|
||||
printf(" --exclude <pattern>, --exclude=<pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern>, --include=<pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file>, --exclude-from=<file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file>, --include-from=<file> Read include patterns from file\n");
|
||||
printf(" --files-from <file> Read the source file list from FILE (paths relative to the "
|
||||
"source root)\n");
|
||||
printf(" -0, --from0 Entries in --files-from are NUL-delimited\n");
|
||||
printf(" -f, --filter=RULE rsync-style filter rule (+/- include/exclude; repeatable;\n");
|
||||
printf(" both --filter=RULE and the -f RULE / -f=RULE short forms work)\n");
|
||||
printf(" -f, --filter=RULE rsync-style filter rule: exclude/- include/+ hide/H show/S\n");
|
||||
printf(" protect/P risk/R merge/. dir-merge/: clear/! with modifiers\n");
|
||||
printf(" (repeatable; --filter=RULE and -f RULE / -f=RULE both work)\n");
|
||||
printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n");
|
||||
printf(" -F Apply per-directory .rsync-filter files during the scan\n");
|
||||
printf(" -F Apply per-directory .rsync-filter files; repeated -FF also\n");
|
||||
printf(" excludes the .rsync-filter files themselves\n");
|
||||
printf(" --max-size <n> Skip files larger than n bytes\n");
|
||||
printf(" --min-size <n> Skip files smaller than n bytes\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G)\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G; 0 = no limit,\n");
|
||||
printf(" matching rsync)\n");
|
||||
printf(" --incremental Skip files unchanged since last transfer\n");
|
||||
printf(" --size-only Skip incremental files matching in size, ignoring mtime\n");
|
||||
printf(" -I, --ignore-times Transfer files even when size and mtime match\n");
|
||||
@@ -127,21 +140,28 @@ void print_usage(void) {
|
||||
printf(" into the destination instead of transferring its data\n");
|
||||
printf(" --link-dest <dir> Like --copy-dest, but hard-links the unchanged file from DIR\n");
|
||||
printf(" into the destination (repeatable; earlier DIRs win)\n");
|
||||
printf(" --verify-basis FastSync-only: require a basis hit's content to match the\n");
|
||||
printf(" source by whole-file digest instead of trusting rsync's\n");
|
||||
printf(" size+mtime (or --size-only) quick-check\n");
|
||||
printf(" --checksum-choice, --cc <alg> Whole-file checksum algorithm for --incremental/\n");
|
||||
printf(" --checksum compares (xxh64/xxhash or md5; default xxh64 with\n");
|
||||
printf(" seed 0). The seed comes from --checksum-seed\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash64 digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. The\n");
|
||||
printf(" digest algorithm and seed must match on sender and receiver\n");
|
||||
printf(" --checksum compares. Accepted: xxh128 (default), xxh3, xxh64\n");
|
||||
printf(" (aka xxhash), md5, md4, sha1, or none. A two-name\n");
|
||||
printf(" 'transfer,pre-transfer' form is accepted like rsync; 'none' as\n");
|
||||
printf(" the pre-transfer algorithm is rejected with --checksum\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. A seed\n");
|
||||
printf(" of 0 (the default) is randomized per transfer, exactly like\n");
|
||||
printf(" rsync, and the chosen seed is sent to the receiver\n");
|
||||
printf(" --delta Delta transfer for changed files (requires --incremental)\n");
|
||||
printf(" -W, --whole-file Transfer changed files without delta processing\n");
|
||||
printf(" --no-whole-file rsync spelling that clears -W/--whole-file\n");
|
||||
printf(" -y, --fuzzy Use a similar-named file already in the destination\n");
|
||||
printf(" directory as the delta basis when the destination has no\n");
|
||||
printf(" usable file at the exact path (saves bandwidth; implies\n");
|
||||
printf(" --incremental and --delta; inert with --whole-file,\n");
|
||||
printf(" --no-delta, or --no-incremental)\n");
|
||||
printf(" --no-fuzzy Disable --fuzzy\n");
|
||||
printf(" --delta-block <n>, --block-size <n>\n");
|
||||
printf(" -B <n>, --block-size <n>, --delta-block <n>\n");
|
||||
printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT);
|
||||
printf(" --delta-max <n> Max file size for delta transfer (default: %llu)\n",
|
||||
DELTA_MAX_FILE_SIZE);
|
||||
@@ -152,16 +172,26 @@ void print_usage(void) {
|
||||
printf(" --chunk-serialization Enable chunk serialization (long form only)\n");
|
||||
printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n");
|
||||
printf(" SSH argv is already built injection-safe)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm (default: zstd)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only;\n");
|
||||
printf(" -f is bound to --filter, not --sendfile)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm: zstd (default), lz4, zlib,\n");
|
||||
printf(" zlibx, none, or auto\n");
|
||||
printf(" --zc <alg> Alias for --compress-choice\n");
|
||||
printf(" -v, --verbose Enable debug logging\n");
|
||||
printf(" -q, --quiet Suppress non-error output\n");
|
||||
printf(" --debug=FLAGS Fine-grained debug logging (use --debug=help for flags)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,misc,skip,stats,all,none\n");
|
||||
printf(" none suppresses info even with --verbose\n");
|
||||
printf(" --preserve Preserve file metadata (long form only)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,name,misc,skip,stats,all,none\n");
|
||||
printf(" (use --info=help for flags; none suppresses --verbose)\n");
|
||||
printf(" --preserve Preserve permissions and times (= -pt; long form only)\n");
|
||||
printf(" --no-perms Negate -p/--perms\n");
|
||||
printf(" --no-times Negate -t/--times\n");
|
||||
printf(" --no-owner Negate -o/--owner\n");
|
||||
printf(" --no-group Negate -g/--group\n");
|
||||
printf(" --no-preserve Disable metadata preservation (negates --preserve)\n");
|
||||
printf(" -E, --executability Preserve executable permission bits\n");
|
||||
printf(" -U, --atimes Preserve access times\n");
|
||||
printf(" -N, --crtimes Capture birth time; cannot be applied (documented\n");
|
||||
printf(" divergence)\n");
|
||||
printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n");
|
||||
printf(" privileged security.*/trusted.* namespaces are\n");
|
||||
printf(" never captured or applied)\n");
|
||||
@@ -176,28 +206,32 @@ void print_usage(void) {
|
||||
printf(" (char/block device-node creation, --write-devices)\n");
|
||||
printf(" within the confined receive root. Never elevates\n");
|
||||
printf(" privileges and never bypasses confinement; ownership\n");
|
||||
printf(" is still applied only with an explicit identity flag\n");
|
||||
printf(" (--numeric-ids/--chown/--usermap/--groupmap/--copy-as)\n");
|
||||
printf(" is still applied only with -o/--owner, -g/--group, or an\n");
|
||||
printf(" explicit identity flag (--chown/--usermap/--groupmap/\n");
|
||||
printf(" --copy-as); --numeric-ids only changes how ids map\n");
|
||||
printf(" --no-super Forbid those super-user activities even when the\n");
|
||||
printf(" receiver is running as root\n");
|
||||
printf(" --chmod <changes> Modify transferred permissions (rsync syntax)\n");
|
||||
printf(" --numeric-ids Do not map uid/gid by name: use the source numeric\n");
|
||||
printf(" ids directly when applying ownership\n");
|
||||
printf(
|
||||
" --chmod <changes> Modify new/transferred permissions (rsync syntax; implies no -p)\n");
|
||||
printf(" --numeric-ids Map uid/gid by id instead of by name (a modifier, not\n");
|
||||
printf(" an ownership request: combine with -o/-g or a map)\n");
|
||||
printf(" --usermap=MAP Map usernames when applying ownership: comma-separated\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM/TO are names\n");
|
||||
printf(" (resolved on the source machine), * (match any /\n");
|
||||
printf(" current user), or @N numeric ids. e.g. *:nobody\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM is a name (from\n");
|
||||
printf(" the source), an id, an inclusive LOW-HIGH range, *\n");
|
||||
printf(" (any id), or empty (ids with no name). TO is an id, *\n");
|
||||
printf(" (current user), or a name resolved on the receiver.\n");
|
||||
printf(" e.g. 0-99:nobody,*:normal (cannot mix with --chown)\n");
|
||||
printf(" --groupmap=MAP Map group names when applying ownership (same syntax)\n");
|
||||
printf(" --chown=USER:GROUP Override the ownership of transferred files. Forms:\n");
|
||||
printf(" USER:GROUP, USER (owner only), :GROUP (group only); a\n");
|
||||
printf(" value of * means the current/root user as appropriate.\n");
|
||||
printf(" Names resolve on the source machine; @N for numerics.\n");
|
||||
printf(" (Metadata is enabled with --preserve; -M now means\n");
|
||||
printf(" rsync's --remote-option.)\n");
|
||||
printf(" (Implies owner/group metadata; -M now means rsync's\n");
|
||||
printf(" --remote-option.)\n");
|
||||
printf(" --copy-as=USER[:GROUP] Force every written entry (files, dirs, symlinks\n");
|
||||
printf(" and special nodes) to USER[:GROUP], resolved on the\n");
|
||||
printf(" source machine like --chown. Requires a privileged\n");
|
||||
printf(" (root) receiver and implies --preserve; an\n");
|
||||
printf(" (root) receiver and implies owner/group metadata; an\n");
|
||||
printf(" unprivileged receiver refuses the transfer. Never\n");
|
||||
printf(" switches process credentials (safe-subset; see\n");
|
||||
printf(" RSYNC_COMPAT.md). A daemon refuses it.\n");
|
||||
@@ -209,65 +243,82 @@ void print_usage(void) {
|
||||
printf(" --server-port <n> Server port (default: 8080)\n");
|
||||
printf(" --port <n> Alias for --server-port\n");
|
||||
printf(" --password-file <f> Authenticate a host::module/path daemon destination.\n");
|
||||
printf(" The file's first user:password line supplies the\n");
|
||||
printf(" username and password (only a SHA-256 digest of the\n");
|
||||
printf(" password is sent; keep the file mode 0600)\n");
|
||||
printf(" FastSync-native SCRAM/PBKDF2 credential scheme (NOT\n");
|
||||
printf(" rsync's --password-file): the file's first user:password\n");
|
||||
printf(" line supplies the username and password; no password or\n");
|
||||
printf(" reusable digest is sent (keep the file mode 0600)\n");
|
||||
printf(" --no-motd Suppress display of the daemon's MOTD (the server\n");
|
||||
printf(" still sends it; the client just does not show it)\n");
|
||||
printf(" --bwlimit <KB/s> Bandwidth limit in kilobytes per second\n");
|
||||
printf(" --bwlimit=RATE Limit socket I/O bandwidth (default unit KiB/s,\n");
|
||||
printf(" rsync-style: 0 = no limit; K/M/G/T/P suffixes are\n");
|
||||
printf(" binary, KB/MB decimal, KiB/MiB binary; decimals allowed)\n");
|
||||
printf(" --tls Enable TLS encryption\n");
|
||||
printf(" --cert <path> TLS certificate file (PEM)\n");
|
||||
printf(" --key <path> TLS private key file (PEM)\n");
|
||||
printf(" --ca <path> TLS CA certificate file (PEM)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 30; long form only)\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 10)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 0 = disabled, matching\n");
|
||||
printf(" rsync). 0 disables it; --no-timeout is the same\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 60, matching\n");
|
||||
printf(" rsync); 0 disables it (--no-contimeout)\n");
|
||||
printf(" --stop-after=MINS Stop the transfer after MINS minutes (a positive\n");
|
||||
printf(" integer); whatever was already transferred is kept\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time: HH:MM, HH:MM:SS, or\n");
|
||||
printf(" now+N[smhd] (a time already in the past stops the\n");
|
||||
printf(" transfer immediately; client-only). An early stop\n");
|
||||
printf(" skips the late --delete keep-set so it cannot delete\n");
|
||||
printf(" source mirrors that were not yet scanned\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time. Accepts rsync's date form\n");
|
||||
printf(" (Y-M-DTh:m, Y/M/DTh:m, abbreviable fields such as 12-31,\n");
|
||||
printf(" 14:00, :59, 1) plus FastSync's HH:MM[:SS] and now+N[smhd]\n");
|
||||
printf(" (a time already in the past stops the transfer\n");
|
||||
printf(" immediately; client-only). An early stop skips the late\n");
|
||||
printf(" --delete keep-set so it cannot delete source mirrors that\n");
|
||||
printf(" were not yet scanned\n");
|
||||
printf(" --address <ip> Bind the outgoing client socket to this source address\n");
|
||||
printf(" -4, --ipv4 Force IPv4 for destination resolution\n");
|
||||
printf(" -6, --ipv6 Force IPv6 for destination resolution\n");
|
||||
printf(" --sockopts=OPTS Comma-separated OPT=VAL socket options applied before connect:\n");
|
||||
printf(" TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR\n");
|
||||
printf(" --backup Backup existing files before overwriting\n");
|
||||
printf(" -b, --backup Backup existing files before overwriting\n");
|
||||
printf(" --backup-dir <dir> Directory for backups (requires --backup)\n");
|
||||
printf(" --suffix <str> Backup suffix (default: ~)\n");
|
||||
printf(" --stats Print transfer statistics at end\n");
|
||||
printf(" -i, --itemize-changes Print an rsync-style per-file change line\n");
|
||||
printf(" --out-format=FORMAT Output format for changed files (%%f %%n %%l %%b %%M %%%%)\n");
|
||||
printf(" --out-format=FORMAT Output format (%%f %%n %%l %%b %%c %%C %%i %%M %%%%)\n");
|
||||
printf(" --list-only List source files instead of transferring\n");
|
||||
printf(" --log-file-format=FORMAT Per-file log line format (needs --log-file)\n");
|
||||
printf(" -h, --human-readable Print byte sizes in human-readable form\n");
|
||||
printf(" --max-depth <n> Maximum directory depth (0=unlimited)\n");
|
||||
printf(" -x, --one-file-system Do not cross filesystem boundaries\n");
|
||||
printf(" --log-file <path> Write log messages to file\n");
|
||||
printf(" --log-file <path>, --log-file=<path> Write log messages to file\n");
|
||||
printf(" --stderr=MODE Route logging to stderr: errors or all\n");
|
||||
printf(" --partial Keep partial files on interrupted transfer\n");
|
||||
printf(" --partial-dir <dir> Directory for partial files\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install.\n");
|
||||
printf(" Confined to the receive root: a relative dir resolves below\n");
|
||||
printf(" it and an absolute/traversal dir is rejected. The dir must\n");
|
||||
printf(" already exist; a different filesystem falls back to a\n");
|
||||
printf(" non-atomic copy instead of aborting\n");
|
||||
printf(" --fastsync-server-path <path>\n");
|
||||
printf(" Path to fastsync-server on remote (default: fastsync-server)\n");
|
||||
printf(" --old-args Accepted for rsync CLI compatibility; no effect (the\n");
|
||||
printf(" remote server path is always safely quoted now)\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation over SSH\n");
|
||||
printf(" (repeatable; each value is single-quote-escaped on the remote\n");
|
||||
printf(" command line; empty values and values with control characters\n");
|
||||
printf(" are rejected; -M OPT, -M=OPT and --remote-option=OPT work)\n");
|
||||
printf(" --trust-sender Trust the remote sender's file list: the receiver skips its\n");
|
||||
printf(" own up-front path-traversal/containment re-validation of the\n");
|
||||
printf(" incoming file list (fewer checks, faster, potentially unsafe).\n");
|
||||
printf(" Local receiver policy: never sent to the peer, off by default\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation. SSH\n");
|
||||
printf(" transport ONLY (user@host:path): a daemon (host::module) or\n");
|
||||
printf(" local TCP destination rejects it (no remote command line to\n");
|
||||
printf(" append to). Repeatable; each value is single-quote-escaped on\n");
|
||||
printf(" the remote command line; empty values and values with control\n");
|
||||
printf(" characters are rejected; -M OPT, -M=OPT and\n");
|
||||
printf(" --remote-option=OPT work\n");
|
||||
printf(" --trust-sender RECEIVER-LOCAL policy: trust the remote sender's file list\n");
|
||||
printf(" and skip the receiver's own up-front path-traversal/\n");
|
||||
printf(" containment re-validation of the incoming list (fewer checks,\n");
|
||||
printf(" faster, potentially unsafe). It is never sent to the peer, so\n");
|
||||
printf(" for a push it must be enabled on the receiving SERVER\n");
|
||||
printf(" (fastsync-server --trust-sender) or forwarded with\n");
|
||||
printf(" -M--trust-sender; the client flag alone has no effect\n");
|
||||
printf(" -l, --links Copy symlinks as symlinks\n");
|
||||
printf(" --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks that point outside transfer tree\n");
|
||||
printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n");
|
||||
printf(" -L, --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks whose target points outside the tree\n");
|
||||
printf(" --copy-unsafe-links Copy unsafe symlinks (outside tree) as referent files\n");
|
||||
printf(" -k, --copy-dirlinks Transform symlinks to directories into real dirs\n");
|
||||
printf(" -K, --keep-dirlinks Keep an existing symlink-to-dir as that dir\n");
|
||||
printf(" --munge-links Munge symlink targets on the wire (sender)\n");
|
||||
printf(" --munge-links Munge stored symlink targets (/rsyncd-munged/) on the receiver\n");
|
||||
printf(" -H, --hard-links Preserve hard-link relationships across the transfer\n");
|
||||
printf(" -S, --sparse Handle sparse files efficiently\n");
|
||||
printf(
|
||||
@@ -275,8 +326,7 @@ void print_usage(void) {
|
||||
printf(
|
||||
" --devices Recreate device nodes on the destination (privileged; skipped when\n");
|
||||
printf(" the receiver lacks CAP_MKNOD)\n");
|
||||
printf(" --specials Recreate special files (FIFOs) on the destination (sockets "
|
||||
"skipped)\n");
|
||||
printf(" --specials Recreate special files (FIFOs, sockets) on the destination\n");
|
||||
printf(" --copy-devices Copy a source device's content as a regular file instead\n");
|
||||
printf(" --write-devices Write received data into an existing destination device node\n");
|
||||
printf(" --inplace Update files in-place (no temp+rename)\n");
|
||||
@@ -289,7 +339,9 @@ void print_usage(void) {
|
||||
printf(" --fsync Fsync every written file before publication\n");
|
||||
printf(" --compress-level <n> Compression level (default: 5)\n");
|
||||
printf(" --zl <n> Alias for --compress-level\n");
|
||||
printf(" --skip-compress=LIST Skip compression for comma-separated suffixes\n");
|
||||
printf(" --skip-compress=LIST Skip compression for suffixes in LIST (separated by\n");
|
||||
printf(" '/' as in rsync, or ','); a leading dot is optional. The\n");
|
||||
printf(" default is rsync 3.4.1's built-in skip-compress list\n");
|
||||
printf(" --compress-threads <n> Compression worker threads (requires zstd threaded support)\n");
|
||||
printf(" --no-OPTION Disable a supported boolean option\n");
|
||||
printf(" --help Show this help\n");
|
||||
@@ -297,7 +349,20 @@ void print_usage(void) {
|
||||
}
|
||||
|
||||
void print_debug_usage(void) {
|
||||
printf("Supported debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n");
|
||||
printf("Emitting debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): ACL,BACKUP,BIND,CHDIR,\n");
|
||||
printf("CONNECT,CMD,DEL,DELTASUM,DUP,EXIT,FILTER,FLIST,FUZZY,GENR,HASH,HLINK,\n");
|
||||
printf("ICONV,NSTR,OWN,RECV,SEND,TIME.\n");
|
||||
printf("Flags may be comma-separated, for example: --debug=io,proto\n");
|
||||
printf("Other rsync debug flags are unsupported and rejected.\n");
|
||||
printf("An optional level suffix is accepted (e.g. --debug=io2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
|
||||
void print_info_usage(void) {
|
||||
printf("Emitting info flags: COPY,NAME,MISC,SKIP,STATS,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): BACKUP,DEL,FLIST,MOUNT,\n");
|
||||
printf("NONREG,PROGRESS,REMOVE,SYMSAFE.\n");
|
||||
printf("Flags may be comma-separated, for example: --info=name,stats\n");
|
||||
printf("An optional level suffix is accepted (e.g. --info=stats2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
|
||||
@@ -3,5 +3,6 @@
|
||||
|
||||
void print_usage(void);
|
||||
void print_debug_usage(void);
|
||||
void print_info_usage(void);
|
||||
|
||||
#endif
|
||||
|
||||
+235
-37
@@ -3,6 +3,7 @@
|
||||
#include "charset.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "delay_updates.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
@@ -11,6 +12,7 @@
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
@@ -43,17 +45,73 @@ void receiver_outcomes_destroy(ReceiverOutcomes* outcomes) {
|
||||
/* End-of-transfer success frame. When --remove-source-files was negotiated
|
||||
each processed data file is acknowledged first (STATUS_NEXT = written,
|
||||
STATUS_OK = skipped) so the sender never removes a source the receiver did
|
||||
not actually store. The frame always ends with a plain STATUS_OK. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes) {
|
||||
not actually store. The frame ends with `final_status` (STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped). */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status) {
|
||||
if (!config->remove_source_files)
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
size_t count = outcomes ? outcomes->count : 0;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
Status per_file = outcomes->entries[i] == FILE_SAVE_WRITTEN ? STATUS_NEXT : STATUS_OK;
|
||||
if (!send_status(fd, per_file))
|
||||
return false;
|
||||
}
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
}
|
||||
|
||||
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
|
||||
const struct ArrayList* would_delete,
|
||||
const struct ArrayList* deleted_paths) {
|
||||
if (!config->report_stats)
|
||||
return true;
|
||||
ReceiverStats local;
|
||||
memset(&local, 0, sizeof(local));
|
||||
const ReceiverStats* out = stats ? stats : &local;
|
||||
/* The path list carries the dry-run would-delete set for a -n run and the
|
||||
actually-removed set for a real --info=del run. */
|
||||
const struct ArrayList* paths =
|
||||
config->dry_run ? would_delete : (config->report_deletes ? deleted_paths : NULL);
|
||||
size_t count = paths ? (size_t)paths->size : 0;
|
||||
if (count > (size_t)MAX_MANIFEST_ENTRIES)
|
||||
count = MAX_MANIFEST_ENTRIES;
|
||||
ReceiverStats record = *out;
|
||||
record.would_delete_count = count;
|
||||
if (!send_status(fd, STATUS_STATS) || !format_stats_send(fd, &record) ||
|
||||
!send_int(fd, (int)count))
|
||||
return false;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
const char* path = (const char*)paths->items[i];
|
||||
if (!send_wire_str(fd, path ? path : ""))
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Add a delete commit's tally to the sink's end-of-transfer wire counters (when
|
||||
the sink reports them). Runs on the receiving thread, so no locking. */
|
||||
static void receiver_tally_deleted(const ReceiverSink* sink, size_t deleted) {
|
||||
if (sink && sink->stats && deleted > 0)
|
||||
sink->stats->deleted_files += deleted;
|
||||
}
|
||||
|
||||
/* Observer for --info=del: record each truly-removed destination-relative path
|
||||
in the ArrayList passed as the observer context, so the terminal STATUS_STATS
|
||||
frame can list it. A failed append is best-effort (the deletion already
|
||||
happened; output is cosmetic). Shared by the single-threaded receiver and
|
||||
the -m pipeline's deferred commit. */
|
||||
void receiver_record_deleted_path(void* context, const char* rel_path) {
|
||||
ArrayList* paths = context;
|
||||
if (!paths || !rel_path)
|
||||
return;
|
||||
/* Bound the retained list like the keep-set manifest: only MAX_MANIFEST_ENTRIES
|
||||
paths are ever transmitted in the terminal STATUS_STATS frame, so recording
|
||||
more only grows memory. A hostile/huge deletion set is therefore capped. */
|
||||
if ((size_t)paths->size >= (size_t)MAX_MANIFEST_ENTRIES)
|
||||
return;
|
||||
char* copy = str_dup(rel_path);
|
||||
if (copy && !array_list_add(paths, copy))
|
||||
free(copy);
|
||||
}
|
||||
|
||||
static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) {
|
||||
@@ -242,21 +300,23 @@ static bool receiver_note_status(const struct timespec* session_start,
|
||||
}
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) {
|
||||
return receiver_process_pending(config, file_descriptor, sink, NULL);
|
||||
return receiver_process_pending(config, file_descriptor, sink, NULL, NULL);
|
||||
}
|
||||
|
||||
/* Runs the whole receive loop. The delete manifest may legitimately arrive
|
||||
either FIRST (--delete-before / --delete-during: the sender transmits the
|
||||
validated keep-set before any file data) or LAST (plain --delete /
|
||||
--delete-after / --delete-delay: the manifest closes the data stream). In
|
||||
validated keep-set before any file data) or LAST (--delete-after /
|
||||
--delete-commit / --delete-delay: the manifest closes the data stream). In
|
||||
the early modes the receiver deletes as soon as the manifest has been read
|
||||
and acknowledges with STATUS_OK so the sender only starts streaming once the
|
||||
deletion has committed (or failed); in the late modes the manifest is held
|
||||
and the deletion is committed only after the terminal STATUS_FINISHED proves
|
||||
the whole transfer succeeded. See receiver_process_pending() for how the -m
|
||||
receiver defers that commit until its disk writer has drained. */
|
||||
the whole transfer succeeded. A plain --delete defaults to the per-directory
|
||||
delete-during plan mode (no manifest at all). See
|
||||
receiver_process_pending() for how the -m receiver defers that commit until
|
||||
its disk writer has drained. */
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest) {
|
||||
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans) {
|
||||
Status status;
|
||||
if (!receive_status(file_descriptor, &status))
|
||||
return -1;
|
||||
@@ -270,14 +330,22 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
|
||||
return -1;
|
||||
bool early_delete = config_delete_timing_early(config);
|
||||
bool per_dir_delete = config_delete_timing_per_dir(config);
|
||||
/* Parked keep-set for the late/commit timing. Every exit path below frees it
|
||||
exactly once; the only exception is the successful FINISHED handoff, which
|
||||
transfers ownership to *pending_manifest (used by the -m receiver). */
|
||||
DeleteManifest* deferred_manifest = NULL;
|
||||
/* Per-directory delete session for --delete-during/--delete-delay. During the
|
||||
loop it applies plans inline (during) or snapshots their extras (delay); on
|
||||
a successful FINISHED it is either committed here or handed to
|
||||
*pending_plans so the -m caller commits after its disk writer drained. */
|
||||
DeletePlanSession* plan_session = NULL;
|
||||
bool delete_limit_noted = false;
|
||||
while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK ||
|
||||
status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH ||
|
||||
status == STATUS_MKDIR || status == STATUS_MANIFEST || status == STATUS_HARDLINK ||
|
||||
status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES) {
|
||||
status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES ||
|
||||
status == STATUS_DELETE_PLAN) {
|
||||
if (status == STATUS_KEEPALIVE) {
|
||||
if (!send_status(file_descriptor, STATUS_KEEPALIVE))
|
||||
goto fail;
|
||||
@@ -335,32 +403,48 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
if (config->dry_run) {
|
||||
/* Server-contacting --dry-run mutates nothing, so a keep-set manifest
|
||||
is consumed and discarded. The early-delete mode still needs its ACK
|
||||
so a sender blocked on the delete handshake is not left hanging. */
|
||||
so a sender blocked on the delete handshake is not left hanging.
|
||||
When would-delete reporting is armed, enumerate (read-only) the
|
||||
destination extras so the terminal STATUS_STATS frame can list them. */
|
||||
if (config->use_delete && sink->would_delete) {
|
||||
size_t count = 0;
|
||||
if (!manifest_would_delete_list(config, manifest, sink->would_delete, &count))
|
||||
log_message(LOG_LEVEL_WARNING, "dry-run: could not enumerate would-delete paths");
|
||||
}
|
||||
delete_manifest_free(manifest);
|
||||
if (early_delete && !send_status(file_descriptor, STATUS_OK))
|
||||
goto fail;
|
||||
goto next_status;
|
||||
}
|
||||
if (early_delete) {
|
||||
/* --delete-before / --delete-during: the manifest is authoritative the
|
||||
moment it arrives, before any file data. Delete now and acknowledge
|
||||
so the sender only starts streaming once the deletion committed (or
|
||||
failed). This is the rsync delete-before/delete-during window: a
|
||||
later transfer failure does not restore these deletions. */
|
||||
bool deletion_ok = (config->use_delete || config->delete_missing_args)
|
||||
? manifest_delete_all(config, manifest)
|
||||
: true;
|
||||
/* --delete-before: the whole-tree manifest is authoritative the moment
|
||||
it arrives, before any file data. Delete now and acknowledge so the
|
||||
sender only starts streaming once the deletion committed (or failed).
|
||||
A later transfer failure does not restore these deletions. A
|
||||
--max-delete-capped commit still succeeds and the transfer proceeds;
|
||||
the terminal success frame reports the cap. */
|
||||
size_t deleted = 0;
|
||||
DeletePathObserver observer =
|
||||
(config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL;
|
||||
DeleteCommitResult deletion =
|
||||
(config->use_delete || config->delete_missing_args)
|
||||
? manifest_delete_all_observed(config, manifest, &deleted, observer,
|
||||
(void*)sink->deleted_paths)
|
||||
: DELETE_COMMIT_OK;
|
||||
receiver_tally_deleted(sink, deleted);
|
||||
delete_manifest_free(manifest);
|
||||
if (!deletion_ok) {
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
if (!send_status(file_descriptor, STATUS_OK))
|
||||
goto fail;
|
||||
} else if (config->use_delete || config->delete_missing_args) {
|
||||
/* Plain --delete / --delete-after / --delete-delay and the
|
||||
--delete-missing-args exact-path deletions: hold the manifest and
|
||||
commit it only after STATUS_FINISHED. */
|
||||
/* Plain --delete / --delete-after and the --delete-missing-args
|
||||
exact-path deletions: hold the manifest and commit it only after
|
||||
STATUS_FINISHED. The per-directory modes never send this frame. */
|
||||
if (deferred_manifest) {
|
||||
log_message(LOG_LEVEL_ERROR, "Received a second delete manifest");
|
||||
delete_manifest_free(deferred_manifest);
|
||||
@@ -374,6 +458,27 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
delete_manifest_free(manifest);
|
||||
}
|
||||
goto next_status;
|
||||
} else if (status == STATUS_DELETE_PLAN) {
|
||||
if (!per_dir_delete) {
|
||||
log_message(LOG_LEVEL_ERROR, "Received a per-directory delete plan without a per-dir "
|
||||
"delete timing");
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (!plan_session) {
|
||||
plan_session = delete_plan_session_create(config);
|
||||
if (plan_session && config->report_deletes && sink->deleted_paths)
|
||||
delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path,
|
||||
(void*)sink->deleted_paths);
|
||||
}
|
||||
if (!plan_session || delete_plan_session_receive(plan_session, config, file_descriptor) != 0)
|
||||
goto fail;
|
||||
if (delete_plan_session_limit_reached(plan_session) && !delete_limit_noted &&
|
||||
sink->note_delete_limit) {
|
||||
sink->note_delete_limit(sink->context);
|
||||
delete_limit_noted = true;
|
||||
}
|
||||
goto next_status;
|
||||
} else {
|
||||
File* file = file_receive(config, file_descriptor);
|
||||
if (!file) {
|
||||
@@ -407,13 +512,50 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
*pending_manifest = deferred_manifest;
|
||||
deferred_manifest = NULL;
|
||||
} else {
|
||||
bool deletion_ok = manifest_delete_all(config, deferred_manifest);
|
||||
size_t deleted = 0;
|
||||
DeletePathObserver observer =
|
||||
(config->report_deletes && sink->deleted_paths) ? receiver_record_deleted_path : NULL;
|
||||
DeleteCommitResult deletion = manifest_delete_all_observed(
|
||||
config, deferred_manifest, &deleted, observer, (void*)sink->deleted_paths);
|
||||
receiver_tally_deleted(sink, deleted);
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
if (!deletion_ok) {
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
}
|
||||
}
|
||||
/* Per-directory deletion: --delete-during already applied each plan inline, so
|
||||
this only finishes the missing-args deletions; --delete-delay committed
|
||||
nothing yet and applies its decompressed snapshot here. The -m receiver
|
||||
hands the session to its caller instead, which commits after the disk
|
||||
writer drained. */
|
||||
if (plan_session) {
|
||||
if (config->report_deletes && sink->deleted_paths)
|
||||
delete_plan_session_set_delete_observer(plan_session, receiver_record_deleted_path,
|
||||
(void*)sink->deleted_paths);
|
||||
if (pending_plans) {
|
||||
*pending_plans = plan_session;
|
||||
plan_session = NULL;
|
||||
} else if (config->dry_run) {
|
||||
/* Central dry-run no-op: never commit a deletion for a -n run. */
|
||||
delete_plan_session_destroy(plan_session);
|
||||
plan_session = NULL;
|
||||
} else {
|
||||
DeleteCommitResult deletion = delete_plan_session_commit(plan_session, config);
|
||||
bool limit = delete_plan_session_limit_reached(plan_session);
|
||||
receiver_tally_deleted(sink, delete_plan_session_deleted(plan_session));
|
||||
delete_plan_session_destroy(plan_session);
|
||||
plan_session = NULL;
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (limit && !delete_limit_noted && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
}
|
||||
}
|
||||
if (sink->send_success) {
|
||||
@@ -428,11 +570,14 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
|
||||
fail:
|
||||
/* Failure exits that must not (or already did) report a STATUS_ERROR. The
|
||||
parked keep-set is dropped: never commit a deletion for a failed stream. */
|
||||
parked keep-set/session is dropped: never commit a deletion for a failed
|
||||
stream. */
|
||||
if (deferred_manifest) {
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
}
|
||||
if (plan_session)
|
||||
delete_plan_session_destroy(plan_session);
|
||||
return -1;
|
||||
|
||||
receive_error:
|
||||
@@ -440,6 +585,8 @@ receive_error:
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
}
|
||||
if (plan_session)
|
||||
delete_plan_session_destroy(plan_session);
|
||||
if (sink->send_error)
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
return -1;
|
||||
@@ -454,11 +601,22 @@ typedef struct {
|
||||
after the whole transfer (and its delete/publication phases) has run so a
|
||||
child write never clobbers a directory mtime. */
|
||||
DirTimeList dir_times;
|
||||
/* Set when a --max-delete commit was capped; the terminal frame then carries
|
||||
STATUS_DELETE_LIMIT so the sender exits 25 like rsync. */
|
||||
bool delete_limit_reached;
|
||||
/* End-of-transfer wire counters (protocol 2.25.0) and the -n/--dry-run
|
||||
--delete would-delete path list collected while processing the manifest. */
|
||||
ReceiverStats stats;
|
||||
ArrayList* would_delete;
|
||||
/* --info=del: actually-removed paths collected during the delete commit. */
|
||||
ArrayList* deleted_paths;
|
||||
} ReceiverSaveContext;
|
||||
|
||||
static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
FileSaveResult result = FILE_SAVE_ERROR;
|
||||
bool created = false;
|
||||
unsigned created_dirs = 0;
|
||||
if (context->config->dry_run) {
|
||||
/* Defense in depth: a dry-run receiver mutates nothing even if a data
|
||||
frame reaches the sink (the sender is not supposed to send one). */
|
||||
@@ -468,14 +626,24 @@ static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
--remove-source-files sender keeps its source. */
|
||||
result = FILE_SAVE_SKIPPED;
|
||||
} else {
|
||||
result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
|
||||
result = file_save_to_disk_full_ex(context->config->receive_root_directory, file,
|
||||
context->config, &created, &created_dirs);
|
||||
}
|
||||
/* A directory's times are deferred, never applied inline: collect the
|
||||
metadata now and apply it at the end. -O/--omit-dir-times is honored by
|
||||
dir_time_list_apply's caller (see receiver_send_success_frame). */
|
||||
/* Wire-stats tally: bytes reconstructed from the basis file (delta matches)
|
||||
count as matched data in the end-of-transfer report. */
|
||||
if (result != FILE_SAVE_ERROR && file->matched_bytes > 0)
|
||||
context->stats.matched_data += file->matched_bytes;
|
||||
/* Protocol 2.28.0: receiver-observed literal bytes and the created-entry
|
||||
breakdown (regular/dir/link/special) for the `--stats` report. */
|
||||
if (result == FILE_SAVE_WRITTEN)
|
||||
receiver_stats_note_saved(&context->stats, file, created, created_dirs);
|
||||
/* A directory's metadata is deferred, never applied inline: collect it now
|
||||
and apply it at the end. -O/--omit-dir-times and --preserve_perms/-times
|
||||
are honored by dir_metadata_list_apply's caller (see
|
||||
receiver_send_success_frame). */
|
||||
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_times_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
@@ -493,12 +661,21 @@ static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
return result != FILE_SAVE_ERROR;
|
||||
}
|
||||
|
||||
static void receiver_note_delete_limit(void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static bool receiver_send_success_frame(int fd, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK;
|
||||
if (!receiver_send_stats_frame(fd, context->config, &context->stats, context->would_delete,
|
||||
context->deleted_paths))
|
||||
return false;
|
||||
/* Server-contacting --dry-run: nothing was staged or written, so there is
|
||||
nothing to publish and no directory times to stamp. */
|
||||
if (context->config->dry_run)
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
/* --delay-updates: the whole protocol stream (including manifest/delete
|
||||
handling, which ran inside receiver_process) has succeeded and every
|
||||
staged file was fully written. Publish them atomically now, before the
|
||||
@@ -514,18 +691,39 @@ static bool receiver_send_success_frame(int fd, void* context_pointer) {
|
||||
phases have committed, so it is finally safe to stamp directory times.
|
||||
This runs after the deferred deletion because receiver_process commits it
|
||||
before calling this success frame. */
|
||||
dir_time_list_apply(&context->dir_times, context->config->receive_root_directory);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes);
|
||||
dir_metadata_list_apply(&context->dir_times, context->config->receive_root_directory,
|
||||
context->config);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
}
|
||||
|
||||
int receiver_receive_files(Config* config, int file_descriptor) {
|
||||
ReceiverSaveContext context = {.config = config, .outcomes = {0}};
|
||||
dir_time_list_init(&context.dir_times);
|
||||
ReceiverSink sink = {receiver_save_file, &context, true, true, receiver_send_success_frame};
|
||||
context.would_delete = array_list_create(free);
|
||||
/* report_deletes (--info=del / -i / --out-format under --delete) is the only
|
||||
reason to retain the actually-removed paths; a plain --delete must not
|
||||
str_dup every removal. NULL is handled by every consumer. */
|
||||
context.deleted_paths = config->report_deletes ? array_list_create(free) : NULL;
|
||||
if (!context.would_delete || (config->report_deletes && !context.deleted_paths)) {
|
||||
array_list_delete(context.would_delete);
|
||||
array_list_delete(context.deleted_paths);
|
||||
return -1;
|
||||
}
|
||||
ReceiverSink sink = {receiver_save_file,
|
||||
&context,
|
||||
true,
|
||||
true,
|
||||
receiver_send_success_frame,
|
||||
receiver_note_delete_limit,
|
||||
&context.stats,
|
||||
context.would_delete,
|
||||
context.deleted_paths};
|
||||
int ret = receiver_process(config, file_descriptor, &sink);
|
||||
if (ret != 0 && config->delay_updates && config->delay_context)
|
||||
delay_updates_cleanup(config->delay_context);
|
||||
receiver_outcomes_destroy(&context.outcomes);
|
||||
dir_time_list_free(&context.dir_times);
|
||||
array_list_delete(context.would_delete);
|
||||
array_list_delete(context.deleted_paths);
|
||||
return ret;
|
||||
}
|
||||
|
||||
+43
-5
@@ -2,8 +2,10 @@
|
||||
#define RECEIVER_H
|
||||
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include <stdbool.h>
|
||||
#include <time.h>
|
||||
|
||||
@@ -21,6 +23,12 @@ typedef struct {
|
||||
|
||||
typedef bool (*ReceiverSuccessFrame)(int fd, void* context);
|
||||
|
||||
/* Records that a --max-delete commit stopped with extras left over, so the
|
||||
caller's terminal success frame can carry STATUS_DELETE_LIMIT instead of
|
||||
STATUS_OK. The commit runs on the receiver thread, so the flag is stored in
|
||||
the sink's own context rather than in a shared global. */
|
||||
typedef void (*ReceiverNoteDeleteLimit)(void* context);
|
||||
|
||||
typedef struct {
|
||||
ReceiverFileSink store_file;
|
||||
void* context;
|
||||
@@ -28,23 +36,53 @@ typedef struct {
|
||||
bool send_success;
|
||||
/* Emits the end-of-transfer success frame. When the sender requested
|
||||
--remove-source-files this includes one per-file status per processed
|
||||
data file followed by the final STATUS_OK; otherwise just STATUS_OK. */
|
||||
data file followed by the final status; otherwise just the final status. */
|
||||
ReceiverSuccessFrame send_success_frame;
|
||||
/* Optional; may be NULL when the sink has no --max-delete handling. */
|
||||
ReceiverNoteDeleteLimit note_delete_limit;
|
||||
/* Optional end-of-transfer wire counters (protocol 2.25.0). When non-NULL
|
||||
and the wire config carries report_stats, the success frame is preceded by
|
||||
a STATUS_STATS record; `would_delete` (optional, receiver-owned strings)
|
||||
carries the -n/--dry-run --delete path list. */
|
||||
ReceiverStats* stats;
|
||||
struct ArrayList* would_delete;
|
||||
/* When --info=del requested it, receiver-owned strings for every path the
|
||||
deletion commit ACTUALLY removed, sent in the terminal STATUS_STATS frame's
|
||||
path list so the sender can print rsync's `deleting PATH` lines. */
|
||||
struct ArrayList* deleted_paths;
|
||||
} ReceiverSink;
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code);
|
||||
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes);
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes);
|
||||
|
||||
/* DeletePathObserver implementation for --info=del: `context` is an ArrayList*
|
||||
that receives owned copies of every truly-removed destination-relative path.
|
||||
Shared by the single-threaded receiver and the -m pipeline's deferred commit. */
|
||||
void receiver_record_deleted_path(void* context, const char* rel_path);
|
||||
|
||||
/* Send the terminal success frame. `final_status` is usually STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status);
|
||||
|
||||
/* Emit STATUS_STATS (a fixed ReceiverStats record plus, when `would_delete` is
|
||||
non-NULL, a count and that many wire strings) when the wire config requested
|
||||
report_stats. A no-op otherwise. */
|
||||
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
|
||||
const struct ArrayList* would_delete,
|
||||
const struct ArrayList* deleted_paths);
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink);
|
||||
/* receiver_process with an escape hatch for the commit-style (late) deletion:
|
||||
when `pending_manifest` is non-NULL the receiver does NOT delete at
|
||||
STATUS_FINISHED itself; instead it stores the owned keep-set manifest there
|
||||
(leaving *pending_manifest untouched on early modes/errors) so the caller can
|
||||
commit the deletion only after its disk writer has fully drained. Pass NULL
|
||||
to keep the default behaviour (delete before the success frame). */
|
||||
commit the deletion only after its disk writer has fully drained. Likewise,
|
||||
when `pending_plans` is non-NULL the --delete-delay per-directory session is
|
||||
handed to the caller instead of being committed at STATUS_FINISHED. Pass NULL
|
||||
for either to keep the default behaviour (delete before the success frame). */
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest);
|
||||
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans);
|
||||
int receiver_receive_files(Config* config, int file_descriptor);
|
||||
|
||||
/* ---- Connection time bounds (anti-slowloris) ----
|
||||
|
||||
@@ -27,6 +27,11 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue*
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->deferred_manifest = NULL;
|
||||
context->deferred_plans = NULL;
|
||||
context->delete_limit_reached = false;
|
||||
memset(&context->stats, 0, sizeof(context->stats));
|
||||
context->would_delete = NULL;
|
||||
context->deleted_paths = NULL;
|
||||
atomic_init(&context->cancelled, false);
|
||||
int init = 0;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
|
||||
@@ -39,6 +44,18 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue*
|
||||
goto fail;
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
context->would_delete = array_list_create(free);
|
||||
if (!context->would_delete)
|
||||
goto fail;
|
||||
/* The actually-removed path list is only needed to render rsync's
|
||||
`deleting PATH` lines, which the client requests via report_deletes
|
||||
(--info=del / -i / --out-format under --delete). A plain --delete run must
|
||||
not allocate it or observe every removal. */
|
||||
if (config->report_deletes) {
|
||||
context->deleted_paths = array_list_create(free);
|
||||
if (!context->deleted_paths)
|
||||
goto fail;
|
||||
}
|
||||
return context;
|
||||
|
||||
fail:
|
||||
@@ -49,6 +66,12 @@ fail:
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&context->mutex);
|
||||
/* Free every list that was already created before the failing allocation:
|
||||
`context` itself is freed below, so they would otherwise leak. */
|
||||
if (context->would_delete)
|
||||
array_list_delete(context->would_delete);
|
||||
if (context->deleted_paths)
|
||||
array_list_delete(context->deleted_paths);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
@@ -57,9 +80,15 @@ void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
|
||||
config_delete(context->config);
|
||||
if (context->deferred_manifest)
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
if (context->deferred_plans)
|
||||
delete_plan_session_destroy(context->deferred_plans);
|
||||
queue_destroy(context->queue);
|
||||
receiver_outcomes_destroy(&context->outcomes);
|
||||
dir_time_list_free(&context->dir_times);
|
||||
if (context->would_delete)
|
||||
array_list_delete(context->would_delete);
|
||||
if (context->deleted_paths)
|
||||
array_list_delete(context->deleted_paths);
|
||||
mtx_destroy(&context->mutex);
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
@@ -132,9 +161,23 @@ bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, Fi
|
||||
|
||||
static bool receiver_enqueue_file(File* file, void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
if (file && file->matched_bytes > 0) {
|
||||
mtx_lock(&context->mutex);
|
||||
context->stats.matched_data += file->matched_bytes;
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
return pipeline_context_receiver_enqueue_file(context, file);
|
||||
}
|
||||
|
||||
/* Early delete modes (--delete-before/--delete-during) commit the manifest
|
||||
inside receiver_process_pending on this thread; record a capped commit so
|
||||
server.c's terminal frame can report STATUS_DELETE_LIMIT. The plain bool is
|
||||
safe: receive_thread writes it before the main thread joins the thread. */
|
||||
static void receiver_pipeline_note_delete_limit(void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static void receiver_thread_fail(PipelineContextReceiver* context) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
@@ -152,9 +195,17 @@ int receive_thread(void* pipeline_context) {
|
||||
const Config* config = context->config;
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL};
|
||||
if (receiver_process_pending((Config*)config, file_descriptor, &sink,
|
||||
&context->deferred_manifest) != 0) {
|
||||
ReceiverSink sink = {receiver_enqueue_file,
|
||||
context,
|
||||
false,
|
||||
false,
|
||||
NULL,
|
||||
receiver_pipeline_note_delete_limit,
|
||||
&context->stats,
|
||||
context->would_delete,
|
||||
context->deleted_paths};
|
||||
if (receiver_process_pending((Config*)config, file_descriptor, &sink, &context->deferred_manifest,
|
||||
&context->deferred_plans) != 0) {
|
||||
receiver_thread_fail(context);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
@@ -196,12 +247,23 @@ int write_thread(void* pipeline_context) {
|
||||
}
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
FileSaveResult result = FILE_SAVE_SKIPPED;
|
||||
bool created = false;
|
||||
unsigned created_dirs = 0;
|
||||
/* Server-contacting --dry-run: never write. The receiver thread does not
|
||||
enqueue anything on the dry-run path, but this keeps the writer thread
|
||||
provably mutation-free if a data frame ever reached it. */
|
||||
bool dry_run = context->config->dry_run;
|
||||
if (save_to_disk && !dry_run) {
|
||||
result = file_save_to_disk_full(root_directory, file, context->config);
|
||||
result =
|
||||
file_save_to_disk_full_ex(root_directory, file, context->config, &created, &created_dirs);
|
||||
if (result == FILE_SAVE_WRITTEN) {
|
||||
/* Protocol 2.28.0: fold the receiver-observed literal bytes and the
|
||||
created-entry type into the shared stats block under its mutex (the
|
||||
receive thread also writes stats.matched_data). */
|
||||
mtx_lock(&context->mutex);
|
||||
receiver_stats_note_saved(&context->stats, file, created, created_dirs);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
@@ -220,8 +282,8 @@ int write_thread(void* pipeline_context) {
|
||||
write would clobber them); accumulate the metadata here and let the
|
||||
caller apply it once every writer has drained. */
|
||||
if (!dry_run && result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_times_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
|
||||
@@ -41,10 +41,29 @@ typedef struct PipelineContextReceiver {
|
||||
transfer truly succeeded. NULL in the early delete modes (which delete at
|
||||
the manifest). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* Per-directory delete session for --delete-delay: receive_thread snapshots
|
||||
each plan's extras as it arrives and hands the session here instead of
|
||||
committing while the disk writer may still be draining; server.c commits it
|
||||
after both threads joined. NULL for every other timing. */
|
||||
DeletePlanSession* deferred_plans;
|
||||
/* Set by server.c when the deferred delete commit hit the --max-delete
|
||||
budget; the terminal success frame then carries STATUS_DELETE_LIMIT
|
||||
(rsync exit 25) while the transfer itself still succeeds. */
|
||||
bool delete_limit_reached;
|
||||
/* P7 Wave D: directory metadata collected by write_thread from received
|
||||
directory entries. Only write_thread mutates it (before it joins); the
|
||||
caller (server.c) applies it after the delete/delay-updates phase. */
|
||||
DirTimeList dir_times;
|
||||
/* End-of-transfer wire counters (protocol 2.25.0). receive_thread accumulates
|
||||
matched_data under `mutex`; server.c adds the delete-commit tallies after
|
||||
both threads join and emits the STATUS_STATS frame. */
|
||||
ReceiverStats stats;
|
||||
/* -n/--dry-run --delete would-delete path list, collected by receive_thread
|
||||
and reported in the STATUS_STATS frame. */
|
||||
struct ArrayList* would_delete;
|
||||
/* --info=del actually-removed path list, collected by the deferred delete
|
||||
commit in server.c and reported in the STATUS_STATS frame. */
|
||||
struct ArrayList* deleted_paths;
|
||||
} PipelineContextReceiver;
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
|
||||
|
||||
+82
-14
@@ -389,8 +389,12 @@ static const char* module_gate_check_ownership(const Config* config, const Daemo
|
||||
ModuleGateContext* gate_ctx) {
|
||||
if (module->client_owner)
|
||||
return NULL;
|
||||
/* Ownership: refuse the whole transfer up front (a clear failure). */
|
||||
if (identity_ownership_requested(config)) {
|
||||
/* Ownership: refuse the whole transfer up front (a clear failure) for a
|
||||
* client-CHOSEN owner/group request. A plain -o/-g/-a preserve-source
|
||||
* request is deliberately not in this narrow set: it falls through to the
|
||||
* super-mode override below, which forces all ownership activity off for this
|
||||
* connection so no chown happens (the transfer itself still succeeds). */
|
||||
if (identity_explicit_ownership_requested(config)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"daemon module '%s' refuses client-chosen ownership/super-user activities "
|
||||
"(no `client owner = yes` opt-in); refusing",
|
||||
@@ -737,13 +741,26 @@ void handler(int file_descriptor) {
|
||||
* received config. */
|
||||
if (gate_ctx.super_mode_override != -1)
|
||||
config->super_mode = (SuperMode)gate_ctx.super_mode_override;
|
||||
/* Install the codec this connection negotiated before the receiver/writer
|
||||
* threads start (the server forks per connection, so the process-global
|
||||
* codec is private to this session). */
|
||||
compression_set_algo((CompressionAlgo)config->compression_algo);
|
||||
/* If the client requested ownership but the effective super mode forbids it
|
||||
* (operator --no-super, a privileged standalone receiver's secure default, or
|
||||
* a daemon module without `client owner = yes`), say so ONCE per connection so
|
||||
* a successful -a/-o/-g transfer is not mistaken for preserved ownership. */
|
||||
if (config->super_mode == SUPER_MODE_OFF && identity_ownership_requested(config))
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"requested ownership will NOT be applied: super-user activities are disabled "
|
||||
"for this connection (operator veto, or module without `client owner = yes`)");
|
||||
protocol_set_8_bit_output(config->eight_bit_output);
|
||||
/* Server-side per-message protocol deadline for every frame from here on.
|
||||
* `timeout` is not serialized, so this is the server's own config (the server
|
||||
* has no --timeout CLI and defaults it to 0): the built-in 60 s window stays
|
||||
* in effect. A client's --timeout tightens only that client's own protocol
|
||||
* I/O and the server's socket read/write timeout is the transport default. */
|
||||
protocol_session_set_io_timeout(&session, config->timeout);
|
||||
* has no --timeout CLI and defaults it to 0). A client's --timeout tightens
|
||||
* only that client's own protocol I/O; the server floors its own deadline at
|
||||
* SERVER_IO_TIMEOUT_SEC so a silent peer can never hold a session slot
|
||||
* forever (the socket layer gets the same floor at startup). */
|
||||
protocol_session_set_io_timeout(&session, protocol_server_io_timeout_sec(config->timeout));
|
||||
const char* authorized_root = utils_get_authorized_root_path();
|
||||
if (!authorized_root) {
|
||||
log_message(LOG_LEVEL_ERROR, "No server-side destination root configured");
|
||||
@@ -892,7 +909,8 @@ void handler(int file_descriptor) {
|
||||
goto done;
|
||||
}
|
||||
protocol_session_set_max_alloc(&context->session, config->max_alloc);
|
||||
protocol_session_set_io_timeout(&context->session, config->timeout);
|
||||
protocol_session_set_io_timeout(&context->session,
|
||||
protocol_server_io_timeout_sec(config->timeout));
|
||||
atomic_store(&context->session.total_allocated_bytes,
|
||||
atomic_load(&session.total_allocated_bytes));
|
||||
pipeline_context_receiver_set_queue_byte_limit(context, RECEIVER_QUEUE_MAX_BYTES);
|
||||
@@ -937,12 +955,42 @@ void handler(int file_descriptor) {
|
||||
--delay-updates run; the walker skips the staging directory. A
|
||||
server-contacting --dry-run deletes nothing (no manifest is sent). */
|
||||
if (context->deferred_manifest) {
|
||||
if (!manifest_delete_all(config, context->deferred_manifest)) {
|
||||
size_t deleted = 0;
|
||||
DeletePathObserver observer = config->report_deletes ? receiver_record_deleted_path : NULL;
|
||||
DeleteCommitResult deletion = manifest_delete_all_observed(
|
||||
config, context->deferred_manifest, &deleted, observer, (void*)context->deleted_paths);
|
||||
context->stats.deleted_files += deleted;
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
transfer_ok = false;
|
||||
} else if (deletion == DELETE_COMMIT_LIMIT_REACHED) {
|
||||
/* The transfer still succeeds; the terminal frame reports the capped
|
||||
deletion so the sender exits 25 like rsync. */
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
context->deferred_manifest = NULL;
|
||||
}
|
||||
/* --delete-delay: receive_thread snapshotted each plan's extras as it
|
||||
arrived; with the disk writer drained, commit the deferred removals.
|
||||
--delete-during already applied its plans on the receive thread. */
|
||||
if (context->deferred_plans) {
|
||||
/* Defence in depth (the enclosing block already excludes dry-run): a
|
||||
-n run never commits a deletion. */
|
||||
if (config->report_deletes)
|
||||
delete_plan_session_set_delete_observer(
|
||||
context->deferred_plans, receiver_record_deleted_path, (void*)context->deleted_paths);
|
||||
DeleteCommitResult deletion =
|
||||
config->dry_run ? DELETE_COMMIT_OK
|
||||
: delete_plan_session_commit(context->deferred_plans, config);
|
||||
context->stats.deleted_files += delete_plan_session_deleted(context->deferred_plans);
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
transfer_ok = false;
|
||||
} else if (deletion == DELETE_COMMIT_LIMIT_REACHED) {
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
delete_plan_session_destroy(context->deferred_plans);
|
||||
context->deferred_plans = NULL;
|
||||
}
|
||||
}
|
||||
if (transfer_ok && !config->dry_run) {
|
||||
/* --delay-updates: receive_thread has finished the whole protocol stream
|
||||
@@ -959,10 +1007,15 @@ void handler(int file_descriptor) {
|
||||
to stamp directory times; a directory's mtime must not be clobbered by
|
||||
its children or by an extra removal. */
|
||||
if (transfer_ok)
|
||||
dir_time_list_apply(&context->dir_times, config->receive_root_directory);
|
||||
dir_metadata_list_apply(&context->dir_times, config->receive_root_directory, config);
|
||||
}
|
||||
if (transfer_ok) {
|
||||
if (!receiver_send_final_success(file_descriptor, config, &context->outcomes))
|
||||
Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK;
|
||||
/* Emit the optional wire-stats record first (protocol 2.25.0), then the
|
||||
success/outcome frame, exactly like the single-threaded receiver. */
|
||||
if (!receiver_send_stats_frame(file_descriptor, config, &context->stats,
|
||||
context->would_delete, context->deleted_paths) ||
|
||||
!receiver_send_final_success(file_descriptor, config, &context->outcomes, final_status))
|
||||
transfer_ok = false;
|
||||
} else {
|
||||
send_error_detail(file_descriptor, "transfer failed on receiver");
|
||||
@@ -1029,8 +1082,9 @@ static void print_server_usage(void) {
|
||||
printf(" hosts allow, hosts deny)\n");
|
||||
printf(" --no-detach Stay in the foreground (default detaches to\n");
|
||||
printf(" background when running --daemon)\n");
|
||||
printf(" --password-file=FILE Credential store for modules that declare\n");
|
||||
printf(" 'auth users' (line format:\n");
|
||||
printf(" --password-file=FILE FastSync-native SCRAM/PBKDF2 credential store (NOT\n");
|
||||
printf(" rsync's auth scheme) for modules that declare 'auth\n");
|
||||
printf(" users' (line format:\n");
|
||||
printf(" user:$fastsync$1$pbkdf2-sha256$iters$salt$stored$server,\n");
|
||||
printf(" generated by --hash-credentials). Legacy\n");
|
||||
printf(" user:SHA256HEX lines are rejected. Requires\n");
|
||||
@@ -1039,7 +1093,7 @@ static void print_server_usage(void) {
|
||||
printf(" --early-input=FILE Second credential store layered over\n");
|
||||
printf(" --password-file (same format); usually a secrets-\n");
|
||||
printf(" manager/process-substitution file. Requires --daemon\n");
|
||||
printf(" -p <port> TCP port (default: 8080, range: 1-65535)\n");
|
||||
printf(" -p, --port <port> TCP port (default: 8080, range: 1-65535)\n");
|
||||
printf(" --tls Enable TLS encryption\n");
|
||||
printf(" --cert <path> TLS certificate file (PEM)\n");
|
||||
printf(" --key <path> TLS private key file (PEM)\n");
|
||||
@@ -1050,7 +1104,9 @@ static void print_server_usage(void) {
|
||||
printf(" -4, --ipv4 Bind an IPv4 socket (default)\n");
|
||||
printf(" -6, --ipv6 Bind an IPv6 socket\n");
|
||||
printf(" --allow-delete Permit manifest deletion\n");
|
||||
printf(" --trust-sender Trust the remote sender's file list\n");
|
||||
printf(" --trust-sender Trust the remote sender's file list (receiver-local;\n");
|
||||
printf(" this server-side flag is the only one that matters -- a\n");
|
||||
printf(" client --trust-sender is never sent to the server)\n");
|
||||
printf(" --no-super Operator veto: never attempt super-user activities\n");
|
||||
printf(" (ownership, device nodes) even as root, and refuse\n");
|
||||
printf(" any client --copy-as/--super request\n");
|
||||
@@ -1127,10 +1183,18 @@ static bool daemonize(void) {
|
||||
if (chdir("/") != 0)
|
||||
log_message(LOG_LEVEL_WARNING, "daemon: chdir to / failed: %s", strerror(errno));
|
||||
umask(0);
|
||||
/* Refresh the cached umask: main() captured the launch umask before this
|
||||
* (single-threaded) umask(0), and file_mode_base() must see the daemon's
|
||||
* actual umask. */
|
||||
file_umask_capture();
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int argc, char* argv[]) {
|
||||
/* Capture the process umask now, while still single-threaded: the cached
|
||||
* value is what file_mode_base() uses, and reading it later would race with
|
||||
* receiver threads creating files. */
|
||||
file_umask_capture();
|
||||
ServerCliOptions opts;
|
||||
char cli_err[512];
|
||||
int parse_result = server_cli_parse(argc, argv, &opts, cli_err, sizeof(cli_err));
|
||||
@@ -1191,6 +1255,10 @@ int main(int argc, char* argv[]) {
|
||||
server_iconv_spec = opts.iconv_spec;
|
||||
signal(SIGINT, cleanup);
|
||||
signal(SIGTERM, cleanup);
|
||||
/* Server-owned socket deadline floor: the client default --timeout=0 would
|
||||
* otherwise leave accepted sockets without SO_RCVTIMEO/SO_SNDTIMEO and let a
|
||||
* silent peer hold a connection (and its process slot) forever. */
|
||||
tcp_set_timeouts(SERVER_IO_TIMEOUT_SEC, SERVER_IO_TIMEOUT_SEC);
|
||||
|
||||
if (opts.stdio_mode) {
|
||||
/* SSH authenticates the stdio transport outside of FastSync. */
|
||||
|
||||
+13
-7
@@ -192,14 +192,20 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->iconv_spec = inline_value;
|
||||
} else if (arg_is(argv[i], "-p")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for -p");
|
||||
return -1;
|
||||
} else if (arg_is(argv[i], "-p") || arg_has_value(argv[i], "--port", &inline_value)) {
|
||||
if (inline_value) {
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(inline_value, &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
} else {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for %s", argv[i]);
|
||||
return -1;
|
||||
}
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
}
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
} else {
|
||||
if (arg_has_value(argv[i], "--config", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
|
||||
+54
-19
@@ -2,6 +2,7 @@
|
||||
#include "data.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include <errno.h>
|
||||
#include <stdlib.h>
|
||||
@@ -10,11 +11,15 @@
|
||||
|
||||
/* Serialization metadata mode for the batch stream, captured from the config at
|
||||
* batch_write_header time. The header persists it into the file so a batch is
|
||||
* self-describing: batch_read_apply re-reads it from the file (not from the
|
||||
* reading config), so a batch written with -M is applied identically by an
|
||||
* invoking process regardless of its own -M setting. The batch driver is a
|
||||
* single sequential scan pass within one thread, so this module-level flag is
|
||||
* safe. */
|
||||
* self-describing about whether per-entry metadata was CAPTURED in the stream:
|
||||
* batch_read_apply re-reads it from the file (not from the reading config) to
|
||||
* decode the chunk records correctly. Which attributes are actually APPLIED,
|
||||
* however, comes from the INVOKING process's per-attribute config (the
|
||||
* FileAttrPolicy and the dir-metadata gate), so a batch written with -M is NOT
|
||||
* automatically applied identically by an invoking process with a different
|
||||
* -p/-t/-o/-g: --read-batch must be invoked with the same -p/-t/-o/-g as the
|
||||
* write side (rsync requires the same options). The batch driver is a single
|
||||
* sequential scan pass within one thread, so this module-level flag is safe. */
|
||||
static bool batch_metadata_mode = false;
|
||||
|
||||
static bool write_all_bytes(int fd, const void* data, size_t size) {
|
||||
@@ -91,22 +96,37 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
if (fd < 0 || dest_root == NULL || dest_root[0] == '\0')
|
||||
return -1;
|
||||
|
||||
/* Directory metadata is deferred to the end of the apply (a child write would
|
||||
* otherwise clobber its parent's mtime/mode). The batch header's single
|
||||
* metadata bit only says whether metadata is present in the stream; which
|
||||
* attributes are APPLIED comes from the invoking process's config, so
|
||||
* --read-batch must be invoked with the same -p/-t/-o/-g as the write side
|
||||
* (rsync requires the same options). The identity snapshot is activated so
|
||||
* -o/-g and the explicit ownership flags can apply. */
|
||||
DirTimeList dir_times;
|
||||
dir_time_list_init(&dir_times);
|
||||
int result = -1;
|
||||
if (!identity_set_active(config)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not activate the identity policy");
|
||||
goto done;
|
||||
}
|
||||
|
||||
char magic[BATCH_MAGIC_LEN];
|
||||
bool eof = false;
|
||||
if (!read_exact(fd, magic, BATCH_MAGIC_LEN, &eof) || eof ||
|
||||
memcmp(magic, BATCH_MAGIC, BATCH_MAGIC_LEN) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad magic)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char version;
|
||||
if (!read_exact(fd, &version, 1, &eof) || eof || version != BATCH_FORMAT_VERSION) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad or missing format version)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char mode;
|
||||
if (!read_exact(fd, &mode, 1, &eof) || eof || (mode != 0 && mode != 1)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad metadata flag)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
bool use_metadata = mode == 1;
|
||||
|
||||
@@ -114,47 +134,62 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
unsigned long long length;
|
||||
if (!read_exact(fd, &length, sizeof(length), &eof)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated length prefix");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (eof)
|
||||
break; /* clean end of stream */
|
||||
if (length == 0 || length > BATCH_MAX_RECORD) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected record length %llu (valid range 1..%llu)",
|
||||
length, (unsigned long long)BATCH_MAX_RECORD);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
char* record = (char*)malloc((size_t)length);
|
||||
if (record == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not allocate a %llu-byte record", length);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (!read_exact(fd, record, (size_t)length, &eof) || eof) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated chunk record");
|
||||
free(record);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
Data* data = data_create(record, (size_t)length);
|
||||
if (data == NULL)
|
||||
return -1; /* data_create frees `record` on failure */
|
||||
goto done; /* data_create frees `record` on failure */
|
||||
Chunk* chunk = chunk_deserialize(data, use_metadata);
|
||||
data_destroy(data);
|
||||
if (chunk == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected malformed chunk record");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* file = chunk->items[i];
|
||||
chunk->items[i] = NULL;
|
||||
if (file == NULL)
|
||||
continue;
|
||||
FileSaveResult result = file_save_to_disk_full(dest_root, file, config);
|
||||
file_destroy(file);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
FileSaveResult save = file_save_to_disk_full(dest_root, file, config);
|
||||
/* Accumulate directory metadata (when it applies) before the File is
|
||||
* destroyed; applied once the whole stream has been consumed. */
|
||||
if (save != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(config) &&
|
||||
!dir_time_list_add(&dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
chunk_destroy(chunk);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
file_destroy(file);
|
||||
if (save == FILE_SAVE_ERROR) {
|
||||
chunk_destroy(chunk);
|
||||
goto done;
|
||||
}
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
return 0;
|
||||
dir_metadata_list_apply(&dir_times, dest_root, config);
|
||||
result = 0;
|
||||
|
||||
done:
|
||||
identity_clear_active();
|
||||
dir_time_list_free(&dir_times);
|
||||
return result;
|
||||
}
|
||||
+15
-10
@@ -168,9 +168,14 @@ bool charset_spec_valid_direction(const char* from_charset, const char* to_chars
|
||||
return direction_probe_valid(from_charset, to_charset);
|
||||
}
|
||||
|
||||
/* The receiver's real conversion is wire(client REMOTE) -> server-local (the
|
||||
* server's own --iconv LOCAL half, or the client's LOCAL half when the server
|
||||
* has no --iconv). A dedicated pre-ack check so an impossible direction is
|
||||
/* The receiver's conversion is wire charset -> destination charset. rsync's
|
||||
* CONVERT_SPEC is LOCAL,REMOTE and "stays the same whether you're pushing or
|
||||
* pulling", so for a PUSH (FastSync's only direction) the destination end's
|
||||
* charset is the spec's REMOTE half: the client converts LOCAL -> REMOTE on the
|
||||
* sender and the receiver writes the wire bytes verbatim. Only a server that
|
||||
* declares its OWN --iconv (the daemon "charset" analog) has a different local
|
||||
* charset, and then it is that spec's LOCAL half and the receiver converts
|
||||
* wire -> server-local. A dedicated pre-ack check so an impossible direction is
|
||||
* rejected before the connection instead of refusing mid-transfer. */
|
||||
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec) {
|
||||
if (!spec)
|
||||
@@ -180,7 +185,7 @@ bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec)
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
const char* wire = remote;
|
||||
const char* target_local = local;
|
||||
const char* target_local = remote;
|
||||
char* server_local = NULL;
|
||||
char* server_remote = NULL;
|
||||
if (server_spec) {
|
||||
@@ -302,13 +307,13 @@ bool charset_wire_init_receiver(const char* spec, const char* server_spec) {
|
||||
char* remote;
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
/* The wire charset is the client spec's REMOTE half; the local charset is
|
||||
* the client spec's LOCAL half unless the server was itself started with
|
||||
* --iconv naming a different local charset (the server halves above never
|
||||
* travel, so the server's own flag is the only way its local charset can
|
||||
* differ from what the client assumed). */
|
||||
/* The wire charset is the client spec's REMOTE half (rsync's LOCAL,REMOTE
|
||||
* spec stays the same push or pull, so on a push the destination end's
|
||||
* charset is REMOTE and the receiver writes the wire bytes verbatim). Only a
|
||||
* server started with its own --iconv declares a different local charset (the
|
||||
* server halves above never travel), and then it is that spec's LOCAL half. */
|
||||
const char* wire = remote;
|
||||
const char* target_local = local;
|
||||
const char* target_local = remote;
|
||||
char* server_local = NULL;
|
||||
char* server_remote = NULL;
|
||||
if (server_spec) {
|
||||
|
||||
@@ -57,8 +57,9 @@ void charset_conversion_close(void* conversion);
|
||||
/* Process-wide wire conversion. charset_wire_init_sender (client side) opens
|
||||
* LOCAL->REMOTE; charset_wire_init_receiver (server side) opens
|
||||
* wire(REMOTE)->server-local. server_spec is the server's own --iconv, whose
|
||||
* LOCAL half may override the local charset the client assumed; NULL reuses
|
||||
* the client spec's LOCAL half. Both return false on an unsupported spec.
|
||||
* LOCAL half overrides the destination charset; NULL means the destination
|
||||
* charset is the client spec's REMOTE half (rsync's push semantics: the wire
|
||||
* bytes are written verbatim). Both return false on an unsupported spec.
|
||||
* The state is freed with charset_wire_free. */
|
||||
bool charset_wire_init_sender(const char* spec);
|
||||
bool charset_wire_init_receiver(const char* spec, const char* server_spec);
|
||||
@@ -66,9 +67,9 @@ void charset_wire_free(void);
|
||||
bool charset_wire_active(void);
|
||||
|
||||
/* Pre-ack receiver-direction sanity (see charset_wire_init_receiver): true
|
||||
* when the exact wire->server-local conversion the receiver will use (client
|
||||
* spec's REMOTE half into the server's own LOCAL half, or the client's LOCAL
|
||||
* half when the server has no --iconv) opens and produces NUL-free output. */
|
||||
* when the exact wire->destination conversion the receiver will use (client
|
||||
* spec's REMOTE half into the server's own LOCAL half, or REMOTE->REMOTE when
|
||||
* the server has no --iconv) opens and produces NUL-free output. */
|
||||
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec);
|
||||
|
||||
/* Convert a path across the wire in the process direction. Returns a malloc'd
|
||||
|
||||
+369
-17
@@ -1,12 +1,171 @@
|
||||
#include "checksum.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <openssl/evp.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* delta.c owns the single XXH_IMPLEMENTATION that provides the xxHash symbols
|
||||
* for the whole binary; this TU only needs the declarations. */
|
||||
* for the whole binary; this TU only needs the declarations. The streaming
|
||||
* state structs and XXH3_update are exposed only with XXH_STATIC_LINKING_ONLY. */
|
||||
#define XXH_STATIC_LINKING_ONLY
|
||||
#include <xxhash.h>
|
||||
|
||||
/* ---------------------------------------------------------------------------
|
||||
* Self-contained MD4 (RFC 1320). OpenSSL's MD4 lives in the legacy provider
|
||||
* and is not guaranteed present, so FastSync carries its own implementation to
|
||||
* keep --checksum-choice=md4 working on every build.
|
||||
* ------------------------------------------------------------------------- */
|
||||
|
||||
typedef struct {
|
||||
uint32_t state[4];
|
||||
uint64_t bit_count;
|
||||
uint8_t buffer[64];
|
||||
size_t buffer_len;
|
||||
} Md4Ctx;
|
||||
|
||||
static uint32_t md4_rotl(uint32_t x, int n) {
|
||||
return (x << n) | (x >> (32 - n));
|
||||
}
|
||||
|
||||
static void md4_transform(uint32_t state[4], const uint8_t block[64]) {
|
||||
uint32_t x[16];
|
||||
for (int i = 0; i < 16; i++)
|
||||
x[i] = (uint32_t)block[i * 4] | ((uint32_t)block[i * 4 + 1] << 8) |
|
||||
((uint32_t)block[i * 4 + 2] << 16) | ((uint32_t)block[i * 4 + 3] << 24);
|
||||
|
||||
uint32_t a = state[0], b = state[1], c = state[2], d = state[3];
|
||||
|
||||
#define F(x, y, z) (((x) & (y)) | (~(x) & (z)))
|
||||
#define G(x, y, z) (((x) & (y)) | ((x) & (z)) | ((y) & (z)))
|
||||
#define H(x, y, z) ((x) ^ (y) ^ (z))
|
||||
#define ROUND1(a, b, c, d, k, s) a = md4_rotl(a + F(b, c, d) + x[k], s)
|
||||
#define ROUND2(a, b, c, d, k, s) a = md4_rotl(a + G(b, c, d) + x[k] + 0x5a827999u, s)
|
||||
#define ROUND3(a, b, c, d, k, s) a = md4_rotl(a + H(b, c, d) + x[k] + 0x6ed9eba1u, s)
|
||||
|
||||
ROUND1(a, b, c, d, 0, 3);
|
||||
ROUND1(d, a, b, c, 1, 7);
|
||||
ROUND1(c, d, a, b, 2, 11);
|
||||
ROUND1(b, c, d, a, 3, 19);
|
||||
ROUND1(a, b, c, d, 4, 3);
|
||||
ROUND1(d, a, b, c, 5, 7);
|
||||
ROUND1(c, d, a, b, 6, 11);
|
||||
ROUND1(b, c, d, a, 7, 19);
|
||||
ROUND1(a, b, c, d, 8, 3);
|
||||
ROUND1(d, a, b, c, 9, 7);
|
||||
ROUND1(c, d, a, b, 10, 11);
|
||||
ROUND1(b, c, d, a, 11, 19);
|
||||
ROUND1(a, b, c, d, 12, 3);
|
||||
ROUND1(d, a, b, c, 13, 7);
|
||||
ROUND1(c, d, a, b, 14, 11);
|
||||
ROUND1(b, c, d, a, 15, 19);
|
||||
|
||||
ROUND2(a, b, c, d, 0, 3);
|
||||
ROUND2(d, a, b, c, 4, 5);
|
||||
ROUND2(c, d, a, b, 8, 9);
|
||||
ROUND2(b, c, d, a, 12, 13);
|
||||
ROUND2(a, b, c, d, 1, 3);
|
||||
ROUND2(d, a, b, c, 5, 5);
|
||||
ROUND2(c, d, a, b, 9, 9);
|
||||
ROUND2(b, c, d, a, 13, 13);
|
||||
ROUND2(a, b, c, d, 2, 3);
|
||||
ROUND2(d, a, b, c, 6, 5);
|
||||
ROUND2(c, d, a, b, 10, 9);
|
||||
ROUND2(b, c, d, a, 14, 13);
|
||||
ROUND2(a, b, c, d, 3, 3);
|
||||
ROUND2(d, a, b, c, 7, 5);
|
||||
ROUND2(c, d, a, b, 11, 9);
|
||||
ROUND2(b, c, d, a, 15, 13);
|
||||
|
||||
ROUND3(a, b, c, d, 0, 3);
|
||||
ROUND3(d, a, b, c, 8, 9);
|
||||
ROUND3(c, d, a, b, 4, 11);
|
||||
ROUND3(b, c, d, a, 12, 15);
|
||||
ROUND3(a, b, c, d, 2, 3);
|
||||
ROUND3(d, a, b, c, 10, 9);
|
||||
ROUND3(c, d, a, b, 6, 11);
|
||||
ROUND3(b, c, d, a, 14, 15);
|
||||
ROUND3(a, b, c, d, 1, 3);
|
||||
ROUND3(d, a, b, c, 9, 9);
|
||||
ROUND3(c, d, a, b, 5, 11);
|
||||
ROUND3(b, c, d, a, 13, 15);
|
||||
ROUND3(a, b, c, d, 3, 3);
|
||||
ROUND3(d, a, b, c, 11, 9);
|
||||
ROUND3(c, d, a, b, 7, 11);
|
||||
ROUND3(b, c, d, a, 15, 15);
|
||||
|
||||
#undef F
|
||||
#undef G
|
||||
#undef H
|
||||
#undef ROUND1
|
||||
#undef ROUND2
|
||||
#undef ROUND3
|
||||
|
||||
state[0] += a;
|
||||
state[1] += b;
|
||||
state[2] += c;
|
||||
state[3] += d;
|
||||
}
|
||||
|
||||
static void md4_init(Md4Ctx* ctx) {
|
||||
ctx->state[0] = 0x67452301u;
|
||||
ctx->state[1] = 0xefcdab89u;
|
||||
ctx->state[2] = 0x98badcfeu;
|
||||
ctx->state[3] = 0x10325476u;
|
||||
ctx->bit_count = 0;
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
|
||||
static void md4_update(Md4Ctx* ctx, const uint8_t* data, size_t len) {
|
||||
ctx->bit_count += (uint64_t)len * 8;
|
||||
while (len > 0) {
|
||||
size_t space = sizeof(ctx->buffer) - ctx->buffer_len;
|
||||
size_t take = len < space ? len : space;
|
||||
memcpy(ctx->buffer + ctx->buffer_len, data, take);
|
||||
ctx->buffer_len += take;
|
||||
data += take;
|
||||
len -= take;
|
||||
if (ctx->buffer_len == sizeof(ctx->buffer)) {
|
||||
md4_transform(ctx->state, ctx->buffer);
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void md4_final(Md4Ctx* ctx, uint8_t out[16]) {
|
||||
uint64_t bit_count = ctx->bit_count;
|
||||
uint8_t pad = 0x80;
|
||||
md4_update(ctx, &pad, 1);
|
||||
uint8_t zero = 0;
|
||||
while (ctx->buffer_len != 56)
|
||||
md4_update(ctx, &zero, 1);
|
||||
uint8_t length_le[8];
|
||||
for (int i = 0; i < 8; i++)
|
||||
length_le[i] = (uint8_t)((bit_count >> (8 * i)) & 0xff);
|
||||
md4_update(ctx, length_le, sizeof(length_le));
|
||||
for (int i = 0; i < 4; i++) {
|
||||
out[i * 4] = (uint8_t)(ctx->state[i] & 0xff);
|
||||
out[i * 4 + 1] = (uint8_t)((ctx->state[i] >> 8) & 0xff);
|
||||
out[i * 4 + 2] = (uint8_t)((ctx->state[i] >> 16) & 0xff);
|
||||
out[i * 4 + 3] = (uint8_t)((ctx->state[i] >> 24) & 0xff);
|
||||
}
|
||||
}
|
||||
|
||||
/* One-shot EVP digest (md5/sha1). Returns false when OpenSSL refuses. */
|
||||
static bool evp_digest(const EVP_MD* md, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
static const uint8_t empty = 0;
|
||||
const void* input = data ? data : ∅
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_Digest(input, size, out, &digest_len, md, NULL) != 1)
|
||||
return false;
|
||||
if (digest_len > out_capacity)
|
||||
return false;
|
||||
*out_len = digest_len;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
@@ -14,30 +173,166 @@ bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t
|
||||
if (data == NULL && size != 0)
|
||||
return false;
|
||||
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64: {
|
||||
uint64_t digest = XXH64(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5) {
|
||||
case CHECKSUM_ALGO_XXH3: {
|
||||
uint64_t digest = XXH3_64bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_XXH128: {
|
||||
XXH128_hash_t digest = XXH3_128bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
/* md5 takes no seed; the caller's seed is deliberately ignored (documented
|
||||
* in RSYNC_COMPAT.md). OpenSSL's one-shot EVP_Digest needs a non-NULL
|
||||
* buffer even for an empty input, so map a NULL data + size==0 to an empty
|
||||
* buffer. */
|
||||
static const uint8_t empty = 0;
|
||||
const void* input = data ? data : ∅
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_Digest(input, size, out, &digest_len, EVP_md5(), NULL) != 1)
|
||||
return false;
|
||||
if (digest_len > out_capacity)
|
||||
return false;
|
||||
*out_len = digest_len;
|
||||
* in RSYNC_COMPAT.md). */
|
||||
return evp_digest(EVP_md5(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_MD4: {
|
||||
Md4Ctx ctx;
|
||||
md4_init(&ctx);
|
||||
md4_update(&ctx, (const uint8_t*)data, size);
|
||||
md4_final(&ctx, out);
|
||||
*out_len = 16;
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
/* sha1 takes no seed; the caller's seed is deliberately ignored. */
|
||||
return evp_digest(EVP_sha1(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
/* No checksum requested: an empty digest is the successful result. */
|
||||
*out_len = 0;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!path || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
|
||||
int fd = open(path, O_RDONLY | O_CLOEXEC);
|
||||
if (fd < 0)
|
||||
return false;
|
||||
|
||||
bool ok = checksum_digest_fd(algo, seed, fd, out, out_capacity, out_len);
|
||||
close(fd);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len) {
|
||||
if (fd < 0 || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
|
||||
if (algo == CHECKSUM_ALGO_NONE) {
|
||||
/* No checksum requested: nothing to read; an empty digest succeeds. */
|
||||
*out_len = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
uint8_t buffer[64 * 1024];
|
||||
bool ok = false;
|
||||
lseek(fd, 0, SEEK_SET);
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5 || algo == CHECKSUM_ALGO_SHA1) {
|
||||
const EVP_MD* md = algo == CHECKSUM_ALGO_MD5 ? EVP_md5() : EVP_sha1();
|
||||
EVP_MD_CTX* ctx = EVP_MD_CTX_new();
|
||||
if (!ctx)
|
||||
return false;
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_DigestInit_ex(ctx, md, NULL) == 1) {
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (EVP_DigestUpdate(ctx, buffer, (size_t)got) != 1) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
if (ok && EVP_DigestFinal_ex(ctx, out, &digest_len) == 1 && digest_len <= out_capacity)
|
||||
*out_len = digest_len;
|
||||
else
|
||||
ok = false;
|
||||
}
|
||||
EVP_MD_CTX_free(ctx);
|
||||
return ok;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD4) {
|
||||
Md4Ctx ctx;
|
||||
md4_init(&ctx);
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0)
|
||||
md4_update(&ctx, buffer, (size_t)got);
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
if (ok) {
|
||||
md4_final(&ctx, out);
|
||||
*out_len = 16;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
XXH64_state_t xxh64;
|
||||
XXH3_state_t* xxh3 = NULL;
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
XXH64_reset(&xxh64, seed);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3 || algo == CHECKSUM_ALGO_XXH128) {
|
||||
xxh3 = XXH3_createState();
|
||||
if (!xxh3)
|
||||
return false;
|
||||
if (algo == CHECKSUM_ALGO_XXH3)
|
||||
XXH3_64bits_reset_withSeed(xxh3, seed);
|
||||
else
|
||||
XXH3_128bits_reset_withSeed(xxh3, seed);
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64)
|
||||
XXH64_update(&xxh64, buffer, (size_t)got);
|
||||
else if (XXH3_64bits_update(xxh3, buffer, (size_t)got) == XXH_ERROR) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
|
||||
if (ok) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
uint64_t digest = XXH64_digest(&xxh64);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3) {
|
||||
uint64_t digest = XXH3_64bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else {
|
||||
XXH128_hash_t digest = XXH3_128bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
}
|
||||
}
|
||||
if (xxh3)
|
||||
XXH3_freeState(xxh3);
|
||||
return ok;
|
||||
}
|
||||
|
||||
int checksum_algo_from_name(const char* name) {
|
||||
@@ -45,8 +340,18 @@ int checksum_algo_from_name(const char* name) {
|
||||
return -1;
|
||||
if (strcasecmp(name, "xxh64") == 0 || strcasecmp(name, "xxhash") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH64;
|
||||
if (strcasecmp(name, "xxh3") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH3;
|
||||
if (strcasecmp(name, "xxh128") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH128;
|
||||
if (strcasecmp(name, "md5") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD5;
|
||||
if (strcasecmp(name, "md4") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD4;
|
||||
if (strcasecmp(name, "sha1") == 0)
|
||||
return (int)CHECKSUM_ALGO_SHA1;
|
||||
if (strcasecmp(name, "none") == 0)
|
||||
return (int)CHECKSUM_ALGO_NONE;
|
||||
return -1;
|
||||
}
|
||||
|
||||
@@ -54,22 +359,69 @@ const char* checksum_algo_name(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
return "xxh64";
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return "xxh3";
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
return "xxh128";
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
return "md5";
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return "md4";
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return "sha1";
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return "none";
|
||||
}
|
||||
return "<unknown>";
|
||||
}
|
||||
|
||||
bool checksum_algo_valid(int algo) {
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5;
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5 ||
|
||||
algo == (int)CHECKSUM_ALGO_XXH3 || algo == (int)CHECKSUM_ALGO_XXH128 ||
|
||||
algo == (int)CHECKSUM_ALGO_MD4 || algo == (int)CHECKSUM_ALGO_SHA1 ||
|
||||
algo == (int)CHECKSUM_ALGO_NONE;
|
||||
}
|
||||
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return 8;
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return 16;
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return 20;
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static ChecksumAlgo compiled_checksum_preference_first(void) {
|
||||
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
|
||||
* resolves to xxh128. */
|
||||
static const ChecksumAlgo preference[] = {
|
||||
CHECKSUM_ALGO_XXH128, CHECKSUM_ALGO_XXH3, CHECKSUM_ALGO_XXH64, CHECKSUM_ALGO_MD5,
|
||||
CHECKSUM_ALGO_MD4, CHECKSUM_ALGO_SHA1, CHECKSUM_ALGO_NONE,
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(preference) / sizeof(preference[0]); i++) {
|
||||
if (checksum_algo_valid((int)preference[i]))
|
||||
return preference[i];
|
||||
}
|
||||
return CHECKSUM_ALGO_XXH64;
|
||||
}
|
||||
|
||||
int checksum_choice_resolve(void) {
|
||||
bool specified = false;
|
||||
int env = env_choice_first("RSYNC_CHECKSUM_LIST", checksum_algo_from_name, &specified);
|
||||
if (specified)
|
||||
return env; /* -1 = the list named no supported checksum */
|
||||
return (int)compiled_checksum_preference_first();
|
||||
}
|
||||
|
||||
ChecksumAlgo checksum_negotiate_default(void) {
|
||||
int resolved = checksum_choice_resolve();
|
||||
return resolved >= 0 ? (ChecksumAlgo)resolved : compiled_checksum_preference_first();
|
||||
}
|
||||
|
||||
+54
-10
@@ -8,18 +8,35 @@
|
||||
/* Whole-file content-digest algorithms selectable with --checksum-choice and
|
||||
* seeded with --checksum-seed. The ids are the values actually placed on the
|
||||
* wire (config frame), so they must be kept stable and validated on receive.
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the default and is byte-for-byte what FastSync
|
||||
* computed before these options existed (xxHash64 with seed 0). */
|
||||
typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the historical FastSync default and its numeric
|
||||
* value is preserved. The full set mirrors the algorithms rsync 3.4.1 can be
|
||||
* built with; every one of them is implemented here. */
|
||||
typedef enum {
|
||||
CHECKSUM_ALGO_XXH64 = 0,
|
||||
CHECKSUM_ALGO_MD5 = 1,
|
||||
CHECKSUM_ALGO_XXH3 = 2,
|
||||
CHECKSUM_ALGO_XXH128 = 3,
|
||||
CHECKSUM_ALGO_MD4 = 4,
|
||||
CHECKSUM_ALGO_SHA1 = 5,
|
||||
CHECKSUM_ALGO_NONE = 6
|
||||
} ChecksumAlgo;
|
||||
|
||||
/* md5 digest is 16 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 16
|
||||
/* FastSync's negotiated default (rsync 3.4.1 auto-negotiates xxh128 first).
|
||||
* The wire default for Config->checksum_algo is this value. */
|
||||
#define CHECKSUM_ALGO_DEFAULT CHECKSUM_ALGO_XXH128
|
||||
|
||||
/* sha1 digest is 20 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 20
|
||||
|
||||
/* Compute the whole-file digest of the first `size` bytes of `data`.
|
||||
*
|
||||
* - CHECKSUM_ALGO_XXH64: xxHash64(data, size, seed) (full 64-bit seed).
|
||||
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP.
|
||||
* md5 has no seed, so `seed` is ignored (documented).
|
||||
* - CHECKSUM_ALGO_XXH3: XXH3_64bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_XXH128: XXH3_128bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_MD4: md4(data, size), self-contained RFC 1320 (seed ignored).
|
||||
* - CHECKSUM_ALGO_SHA1: sha1(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_NONE: no digest; *out_len is 0 and nothing is written.
|
||||
* - `size == 0` hashes the empty input (plus its seed), not a NULL input.
|
||||
*
|
||||
* Writes up to `out_capacity` bytes into `out`, storing the digest length in
|
||||
@@ -28,9 +45,23 @@ typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Streaming whole-file digest: hash the contents of `path` without holding the
|
||||
* whole file in memory. Same digest/capacity contract as checksum_digest.
|
||||
* Returns false on open/read failure or an undersized buffer. */
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Descriptor form of the streaming digest: rewinds `fd` to the start and hashes
|
||||
* to EOF without closing it. Used by the --verify-basis path to hash an
|
||||
* already-open, root-confined basis descriptor. Same contract as
|
||||
* checksum_digest_file. */
|
||||
bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len);
|
||||
|
||||
/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "xxh64" and "xxhash" (both map to CHECKSUM_ALGO_XXH64, rsync's
|
||||
* xxhash spelling) and "md5". Returns -1 for any unsupported name. */
|
||||
* Accepts "xxh64"/"xxhash", "xxh3", "xxh128", "md5", "md4", "sha1", "none".
|
||||
* "auto" is not an algorithm here; the caller resolves it to the negotiated
|
||||
* default. Returns -1 for any unrecognized name. */
|
||||
int checksum_algo_from_name(const char* name);
|
||||
|
||||
/* Canonical name of an algorithm (used in CLI error messages). */
|
||||
@@ -39,7 +70,20 @@ const char* checksum_algo_name(ChecksumAlgo algo);
|
||||
/* True when `algo` is a supported id (used by config receive validation). */
|
||||
bool checksum_algo_valid(int algo);
|
||||
|
||||
/* Digest length in bytes for an algorithm (xxx64 = 8, md5 = 16). */
|
||||
/* Digest length in bytes for an algorithm (xxh64/xxh3 = 8,
|
||||
* md5/md4/xxh128 = 16, sha1 = 20, none = 0). */
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list that is
|
||||
* supported on this build (rsync 3.4.1's `--version` order:
|
||||
* xxh128 xxh3 xxh64 md5 md4 sha1 none). Used to resolve "auto". */
|
||||
ChecksumAlgo checksum_negotiate_default(void);
|
||||
|
||||
/* Resolve "auto" the way rsync does: the first supported name in
|
||||
* RSYNC_CHECKSUM_LIST (whitespace-separated, client half ends at '&'), then the
|
||||
* compiled-in preference order when the variable is unset/blank. Returns -1
|
||||
* when the variable is set but names no supported checksum (rsync's failed
|
||||
* negotiation), otherwise a valid ChecksumAlgo id. */
|
||||
int checksum_choice_resolve(void);
|
||||
|
||||
#endif /* CHECKSUM_H */
|
||||
+170
-77
@@ -1,90 +1,183 @@
|
||||
#include "chmod.h"
|
||||
#include "file.h"
|
||||
#include <stddef.h>
|
||||
#include <string.h>
|
||||
|
||||
static bool parse_clause(mode_t* mode, const char* begin, const char* end) {
|
||||
const char* p = begin;
|
||||
unsigned who = 0;
|
||||
while (p < end && strchr("ugoa", *p)) {
|
||||
if (*p == 'a')
|
||||
who = 7;
|
||||
else
|
||||
who |= *p == 'u' ? 1U : (*p == 'g' ? 2U : 4U);
|
||||
p++;
|
||||
}
|
||||
if (who == 0)
|
||||
who = 7;
|
||||
if (p == end || (*p != '+' && *p != '-' && *p != '='))
|
||||
return false;
|
||||
char operation = *p++;
|
||||
mode_t bits = 0;
|
||||
while (p < end) {
|
||||
mode_t bit;
|
||||
switch (*p++) {
|
||||
case 'r':
|
||||
bit = 4;
|
||||
break;
|
||||
case 'w':
|
||||
bit = 2;
|
||||
break;
|
||||
case 'x':
|
||||
bit = 1;
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
bits |= bit;
|
||||
}
|
||||
for (unsigned class_index = 0; class_index < 3; class_index++) {
|
||||
unsigned class_bit = 1U << class_index;
|
||||
if (!(who & class_bit))
|
||||
continue;
|
||||
mode_t shift = (mode_t)((2U - class_index) * 3U);
|
||||
mode_t mask = (mode_t)(7U << shift);
|
||||
mode_t class_bits = (mode_t)(bits << shift);
|
||||
if (operation == '+')
|
||||
*mode |= class_bits;
|
||||
else if (operation == '-')
|
||||
*mode &= ~class_bits;
|
||||
else
|
||||
*mode = (*mode & ~mask) | class_bits;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
/* rsync's --chmod parser (parse_chmod + tweak_mode). A single clause is
|
||||
* applied as it is completed, so repeated clauses and repeated --chmod options
|
||||
* (joined with commas by the CLI) accumulate exactly like rsync. The D/F
|
||||
* selectors restrict a clause to directories/files; X adds execute only to
|
||||
* directories or files that were already executable. */
|
||||
|
||||
#define CHMOD_BITS 07777
|
||||
#define CHMOD_FLAG_X_KEEP (1U << 0)
|
||||
#define CHMOD_FLAG_DIRS_ONLY (1U << 1)
|
||||
#define CHMOD_FLAG_FILES_ONLY (1U << 2)
|
||||
|
||||
enum chmod_op { CHMOD_OP_ADD = 1, CHMOD_OP_SUB, CHMOD_OP_EQ, CHMOD_OP_SET };
|
||||
enum chmod_state {
|
||||
CHMOD_STATE_ERROR,
|
||||
CHMOD_STATE_1ST_HALF,
|
||||
CHMOD_STATE_2ND_HALF,
|
||||
CHMOD_STATE_OCTAL
|
||||
};
|
||||
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result) {
|
||||
if (!spec || !*spec || !result)
|
||||
return false;
|
||||
bool numeric = true;
|
||||
size_t length = strlen(spec);
|
||||
if (length > 4)
|
||||
numeric = false;
|
||||
for (size_t i = 0; i < length && numeric; i++)
|
||||
numeric = spec[i] >= '0' && spec[i] <= '7';
|
||||
if (numeric) {
|
||||
if (length == 0 || length > 4)
|
||||
return false;
|
||||
mode_t parsed = 0;
|
||||
for (size_t i = 0; i < length; i++)
|
||||
parsed = (mode_t)((parsed << 3) | (spec[i] - '0'));
|
||||
*result = parsed;
|
||||
return true;
|
||||
}
|
||||
|
||||
const mode_t nonperm = mode & ~(mode_t)CHMOD_BITS;
|
||||
const bool initially_executable = (mode & 0111) != 0;
|
||||
mode_t changed = mode;
|
||||
const char* begin = spec;
|
||||
while (*begin) {
|
||||
const char* end = strchr(begin, ',');
|
||||
if (!end)
|
||||
end = begin + strlen(begin);
|
||||
if (!parse_clause(&changed, begin, end))
|
||||
return false;
|
||||
if (*end == '\0')
|
||||
int state = CHMOD_STATE_1ST_HALF;
|
||||
unsigned where = 0;
|
||||
int what = 0, op = 0, topbits = 0, topoct = 0, flags = 0;
|
||||
const char* p = spec;
|
||||
while (state != CHMOD_STATE_ERROR) {
|
||||
if (*p == '\0' || *p == ',') {
|
||||
int bits;
|
||||
if (!op) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
if (where)
|
||||
bits = (int)(where * (unsigned)what);
|
||||
else {
|
||||
where = 0111;
|
||||
bits = (int)((where * (unsigned)what) & ~(unsigned)file_process_umask());
|
||||
}
|
||||
int mode_and, mode_or;
|
||||
switch (op) {
|
||||
case CHMOD_OP_ADD:
|
||||
mode_and = CHMOD_BITS;
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
case CHMOD_OP_SUB:
|
||||
mode_and = CHMOD_BITS - bits - topoct;
|
||||
mode_or = 0;
|
||||
break;
|
||||
case CHMOD_OP_EQ:
|
||||
mode_and = CHMOD_BITS - (int)(where * 7U) - (topoct ? topbits : 0);
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
default:
|
||||
mode_and = 0;
|
||||
mode_or = bits;
|
||||
break;
|
||||
}
|
||||
bool is_dir = S_ISDIR(nonperm);
|
||||
if (!((flags & CHMOD_FLAG_DIRS_ONLY) && !is_dir) &&
|
||||
!((flags & CHMOD_FLAG_FILES_ONLY) && is_dir)) {
|
||||
changed &= (mode_t)mode_and;
|
||||
if ((flags & CHMOD_FLAG_X_KEEP) && !initially_executable && !is_dir)
|
||||
changed |= (mode_t)(mode_or & ~0111);
|
||||
else
|
||||
changed |= (mode_t)mode_or;
|
||||
}
|
||||
if (*p == '\0')
|
||||
break;
|
||||
p++;
|
||||
state = CHMOD_STATE_1ST_HALF;
|
||||
where = 0;
|
||||
what = op = topoct = topbits = flags = 0;
|
||||
continue;
|
||||
}
|
||||
switch (state) {
|
||||
case CHMOD_STATE_1ST_HALF:
|
||||
switch (*p) {
|
||||
case 'D':
|
||||
if (flags & CHMOD_FLAG_FILES_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_DIRS_ONLY;
|
||||
break;
|
||||
case 'F':
|
||||
if (flags & CHMOD_FLAG_DIRS_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_FILES_ONLY;
|
||||
break;
|
||||
case 'u':
|
||||
where |= 0100;
|
||||
topbits |= 04000;
|
||||
break;
|
||||
case 'g':
|
||||
where |= 0010;
|
||||
topbits |= 02000;
|
||||
break;
|
||||
case 'o':
|
||||
where |= 0001;
|
||||
break;
|
||||
case 'a':
|
||||
where |= 0111;
|
||||
break;
|
||||
case '+':
|
||||
op = CHMOD_OP_ADD;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '-':
|
||||
op = CHMOD_OP_SUB;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '=':
|
||||
op = CHMOD_OP_EQ;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7' && !where) {
|
||||
op = CHMOD_OP_SET;
|
||||
state = CHMOD_STATE_OCTAL;
|
||||
where = 1;
|
||||
what = *p - '0';
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
begin = end + 1;
|
||||
if (!*begin)
|
||||
return false;
|
||||
case CHMOD_STATE_2ND_HALF:
|
||||
switch (*p) {
|
||||
case 'r':
|
||||
what |= 4;
|
||||
break;
|
||||
case 'w':
|
||||
what |= 2;
|
||||
break;
|
||||
case 'X':
|
||||
flags |= CHMOD_FLAG_X_KEEP;
|
||||
/* fall through */
|
||||
case 'x':
|
||||
what |= 1;
|
||||
break;
|
||||
case 's':
|
||||
if (topbits)
|
||||
topoct |= topbits;
|
||||
else
|
||||
topoct = 04000;
|
||||
break;
|
||||
case 't':
|
||||
topoct |= 01000;
|
||||
break;
|
||||
default:
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7') {
|
||||
what = what * 8 + (*p - '0');
|
||||
if (what > CHMOD_BITS)
|
||||
state = CHMOD_STATE_ERROR;
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
p++;
|
||||
}
|
||||
*result = changed;
|
||||
if (state == CHMOD_STATE_ERROR)
|
||||
return false;
|
||||
*result = (changed & (mode_t)CHMOD_BITS) | nonperm;
|
||||
return true;
|
||||
}
|
||||
|
||||
+4
-1
@@ -4,7 +4,10 @@
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Apply the supported rsync --chmod syntax to a permission mode. */
|
||||
/* Apply rsync's --chmod syntax to a permission mode, including the D/F/X
|
||||
* selectors and the s/t special bits. `mode` should carry the file type bits
|
||||
* (S_IFDIR/S_IFREG) so D/F/X can be evaluated; the type bits are preserved in
|
||||
* `result`. A spec may contain comma-separated clauses, which accumulate. */
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result);
|
||||
|
||||
#endif
|
||||
|
||||
+387
-34
@@ -2,40 +2,198 @@
|
||||
#include "data.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <limits.h>
|
||||
#include <lz4.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <zlib.h>
|
||||
#include <zstd.h>
|
||||
|
||||
#define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024)
|
||||
#define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */
|
||||
|
||||
static char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv",
|
||||
".zip", ".gz", ".xz", ".zst", NULL};
|
||||
/* rsync 3.4.1's built-in skip-compress suffix list (the `--skip-compress`
|
||||
* defaults, in the man page's order). rsync stores it as space-separated
|
||||
* "*.suffix" globs; FastSync matches the plain suffix after the final dot, so
|
||||
* the leading "*." is omitted here. A user --skip-compress list replaces this
|
||||
* default entirely (matching rsync). */
|
||||
#define DEFAULT_SKIP_COMPRESS_SUFFIXES \
|
||||
"3g2 3gp 7z aac ace apk avi bz2 deb dmg ear f4v flac flv gpg gz iso jar jpeg jpg lrz lz lz4 " \
|
||||
"lzma " \
|
||||
"lzo m1a m1v m2a m2ts m2v m4a m4b m4p m4r m4v mka mkv mov mp1 mp2 mp3 mp4 mpa mpeg mpg mpv mts " \
|
||||
"odb odf odg odi odm odp ods odt oga ogg ogm ogv ogx opus otg oth otp ots ott oxt png qt rar " \
|
||||
"rpm " \
|
||||
"rz rzip spx squashfs sxc sxd sxg sxm sxw sz tbz tbz2 tgz tlz ts txz tzo vob war webm webp xz " \
|
||||
"z " \
|
||||
"zip zst"
|
||||
|
||||
/* Self-describing compressed frames: the first byte is the CompressionAlgo id.
|
||||
* zlib/lz4 store the uncompressed size as a little-endian uint32 after the
|
||||
* codec byte so decompression can be exactly pre-sized and bounded. */
|
||||
#define LZ4_SIZE_PREFIX_LEN 4
|
||||
|
||||
static _Atomic int g_compression_algo = COMPRESSION_ALGO_ZSTD;
|
||||
|
||||
/* Case-insensitive match of a bare suffix (no leading dot) against a
|
||||
* space-separated suffix list. */
|
||||
static bool suffix_in_list(const char* name, const char* list) {
|
||||
size_t name_len = strlen(name);
|
||||
while (*list) {
|
||||
while (*list == ' ')
|
||||
list++;
|
||||
const char* start = list;
|
||||
while (*list && *list != ' ')
|
||||
list++;
|
||||
size_t len = (size_t)(list - start);
|
||||
if (len == name_len && strncasecmp(name, start, len) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count) {
|
||||
if (!path)
|
||||
return false;
|
||||
const char* dot = strrchr(path, '.');
|
||||
if (!dot)
|
||||
if (!dot || dot[1] == '\0')
|
||||
return false;
|
||||
if (count < 0) {
|
||||
suffixes = SKIP_COMPRESSION_EXTENSIONS;
|
||||
count = 0;
|
||||
while (SKIP_COMPRESSION_EXTENSIONS[count])
|
||||
count++;
|
||||
}
|
||||
const char* name = dot + 1;
|
||||
/* count < 0 (the user gave no --skip-compress) selects rsync's built-in
|
||||
* default list; a non-negative count is the user's explicit list. */
|
||||
if (count < 0)
|
||||
return suffix_in_list(name, DEFAULT_SKIP_COMPRESS_SUFFIXES);
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (strcasecmp(dot, suffixes[i]) == 0)
|
||||
const char* suffix = suffixes[i];
|
||||
if (suffix[0] == '.')
|
||||
suffix++;
|
||||
if (strcasecmp(name, suffix) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
int compression_algo_from_name(const char* name) {
|
||||
if (!name)
|
||||
return -1;
|
||||
if (strcasecmp(name, "zstd") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZSTD;
|
||||
if (strcasecmp(name, "lz4") == 0)
|
||||
return (int)COMPRESSION_ALGO_LZ4;
|
||||
if (strcasecmp(name, "zlib") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZLIB;
|
||||
if (strcasecmp(name, "zlibx") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZLIBX;
|
||||
if (strcasecmp(name, "none") == 0)
|
||||
return (int)COMPRESSION_ALGO_NONE;
|
||||
return -1;
|
||||
}
|
||||
|
||||
const char* compression_algo_name(CompressionAlgo algo) {
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return "none";
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return "zstd";
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return "lz4";
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
return "zlib";
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return "zlibx";
|
||||
}
|
||||
return "<unknown>";
|
||||
}
|
||||
|
||||
bool compression_algo_valid(int algo) {
|
||||
return algo == (int)COMPRESSION_ALGO_NONE || algo == (int)COMPRESSION_ALGO_ZSTD ||
|
||||
algo == (int)COMPRESSION_ALGO_LZ4 || algo == (int)COMPRESSION_ALGO_ZLIB ||
|
||||
algo == (int)COMPRESSION_ALGO_ZLIBX;
|
||||
}
|
||||
|
||||
bool compression_algo_enabled(CompressionAlgo algo) {
|
||||
return algo != COMPRESSION_ALGO_NONE;
|
||||
}
|
||||
|
||||
static CompressionAlgo compiled_preference_first(void) {
|
||||
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
|
||||
* resolves to zstd. */
|
||||
static const CompressionAlgo preference[] = {
|
||||
COMPRESSION_ALGO_ZSTD, COMPRESSION_ALGO_LZ4, COMPRESSION_ALGO_ZLIBX,
|
||||
COMPRESSION_ALGO_ZLIB, COMPRESSION_ALGO_NONE,
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(preference) / sizeof(preference[0]); i++) {
|
||||
if (compression_algo_valid((int)preference[i]))
|
||||
return preference[i];
|
||||
}
|
||||
return COMPRESSION_ALGO_ZSTD;
|
||||
}
|
||||
|
||||
int compression_choice_resolve(void) {
|
||||
bool specified = false;
|
||||
int env = env_choice_first("RSYNC_COMPRESS_LIST", compression_algo_from_name, &specified);
|
||||
if (specified)
|
||||
return env; /* -1 = the list named no supported codec */
|
||||
return (int)compiled_preference_first();
|
||||
}
|
||||
|
||||
CompressionAlgo compression_negotiate_default(void) {
|
||||
int resolved = compression_choice_resolve();
|
||||
return resolved >= 0 ? (CompressionAlgo)resolved : compiled_preference_first();
|
||||
}
|
||||
|
||||
int compression_default_level(CompressionAlgo algo) {
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return ZSTD_CLEVEL_DEFAULT;
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return 6; /* rsync resolves zlib's Z_DEFAULT_COMPRESSION (-1) to 6 */
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return 1; /* rsync lz4 level is 0/ignored; positive keeps the gate on */
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int compression_clamp_level(CompressionAlgo algo, int level) {
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
if (level < 1)
|
||||
return 1;
|
||||
if (level > 22)
|
||||
return 22;
|
||||
return level;
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
if (level < 1)
|
||||
return 1;
|
||||
if (level > 9)
|
||||
return 9;
|
||||
return level;
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return 1; /* ignored by lz4_compress; keeps the "compress" gate on */
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return level;
|
||||
}
|
||||
|
||||
void compression_set_algo(CompressionAlgo algo) {
|
||||
if (compression_algo_valid((int)algo))
|
||||
atomic_store(&g_compression_algo, (int)algo);
|
||||
}
|
||||
|
||||
CompressionAlgo compression_get_algo(void) {
|
||||
return (CompressionAlgo)atomic_load(&g_compression_algo);
|
||||
}
|
||||
|
||||
/* Per-thread cache of zstd contexts plus the grow-only compression scratch
|
||||
* buffer. zstd contexts are stateful and not safe to share between threads,
|
||||
* so each thread keeps its own (see compression_get_thread_ctx). The cache is
|
||||
@@ -126,17 +284,25 @@ static void compression_ctx_put(CompressionThreadCtx* ctx) {
|
||||
compression_ctx_free(ctx);
|
||||
}
|
||||
|
||||
Data* data_compress(Data* data_to_compress, int compression_level) {
|
||||
return data_compress_with_threads(data_to_compress, compression_level, 0);
|
||||
/* Build a frame consisting of a copy of `src` prefixed by `codec`. */
|
||||
static Data* frame_with_codec(const void* src, size_t size, CompressionAlgo codec) {
|
||||
if (size > SIZE_MAX - 1)
|
||||
return NULL;
|
||||
Data* out = data_create_empty(size + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
((uint8_t*)out->data)[0] = (uint8_t)codec;
|
||||
if (size > 0)
|
||||
memcpy((uint8_t*)out->data + 1, src, size);
|
||||
out->size = size + 1;
|
||||
return out;
|
||||
}
|
||||
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads) {
|
||||
if (!data_to_compress || (!data_to_compress->data && data_to_compress->size != 0) ||
|
||||
compression_threads < 0 || compression_threads > COMPRESSION_MAX_THREADS)
|
||||
static Data* zstd_compress(Data* in, int compression_level, int compression_threads) {
|
||||
size_t dst_size = ZSTD_compressBound(in->size);
|
||||
if (dst_size > SIZE_MAX - 1)
|
||||
return NULL;
|
||||
log_message(LOG_LEVEL_DEBUG, "Starting to compress data");
|
||||
size_t dst_size = ZSTD_compressBound(data_to_compress->size);
|
||||
dst_size += 1; /* codec prefix */
|
||||
|
||||
CompressionThreadCtx* ctx = compression_get_thread_ctx();
|
||||
if (ctx == NULL) {
|
||||
@@ -187,7 +353,7 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
|
||||
if (available_threads > 0) {
|
||||
/* Streaming compression needs the source size before threaded mode can end a frame. */
|
||||
size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, data_to_compress->size);
|
||||
size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, in->size);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
@@ -205,8 +371,8 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
ctx->out_cap = dst_size;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0};
|
||||
ZSTD_outBuffer output = {ctx->out_buf, dst_size, 0};
|
||||
ZSTD_inBuffer input = {in->data, in->size, 0};
|
||||
ZSTD_outBuffer output = {(uint8_t*)ctx->out_buf + 1, dst_size - 1, 0};
|
||||
|
||||
size_t ret;
|
||||
do {
|
||||
@@ -219,30 +385,192 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
|
||||
/* Hand off an exactly-sized copy; the scratch buffer stays cached so the next
|
||||
* call does not reallocate a ZSTD_compressBound-sized block. */
|
||||
compressed_data = data_create_empty(output.pos);
|
||||
compressed_data = data_create_empty(output.pos + 1);
|
||||
if (compressed_data == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate compressed data");
|
||||
goto cleanup;
|
||||
}
|
||||
((uint8_t*)compressed_data->data)[0] = (uint8_t)COMPRESSION_ALGO_ZSTD;
|
||||
if (output.pos > 0)
|
||||
memcpy(compressed_data->data, ctx->out_buf, output.pos);
|
||||
compressed_data->size = output.pos;
|
||||
memcpy((uint8_t*)compressed_data->data + 1, (uint8_t*)ctx->out_buf + 1, output.pos);
|
||||
compressed_data->size = output.pos + 1;
|
||||
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu",
|
||||
data_to_compress->size, compressed_data->size);
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu", in->size,
|
||||
compressed_data->size);
|
||||
|
||||
cleanup:
|
||||
compression_ctx_put(ctx);
|
||||
return compressed_data;
|
||||
}
|
||||
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) ||
|
||||
maximum_size == 0)
|
||||
static Data* lz4_compress(Data* in) {
|
||||
int bound = LZ4_compressBound((int)in->size);
|
||||
if (bound < 0 || in->size > (size_t)INT_MAX)
|
||||
return NULL;
|
||||
Data* out = data_create_empty((size_t)bound + 1 + LZ4_SIZE_PREFIX_LEN);
|
||||
if (!out)
|
||||
return NULL;
|
||||
uint32_t raw_size = (uint32_t)in->size;
|
||||
uint8_t* p = (uint8_t*)out->data;
|
||||
p[0] = (uint8_t)COMPRESSION_ALGO_LZ4;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
p[1 + i] = (uint8_t)((raw_size >> (8 * i)) & 0xff);
|
||||
int written = 0;
|
||||
if (in->size > 0) {
|
||||
written = LZ4_compress_default((const char*)in->data, (char*)p + 1 + LZ4_SIZE_PREFIX_LEN,
|
||||
(int)in->size, bound);
|
||||
if (written <= 0) {
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
out->size = (size_t)written + 1 + LZ4_SIZE_PREFIX_LEN;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zlib_compress(Data* in, CompressionAlgo algo, int compression_level) {
|
||||
int level = compression_level;
|
||||
if (level < 1)
|
||||
level = Z_DEFAULT_COMPRESSION;
|
||||
if (level > 9)
|
||||
level = 9;
|
||||
uLong bound = compressBound((uLong)in->size);
|
||||
if (in->size > (size_t)ULONG_MAX)
|
||||
return NULL;
|
||||
Data* out = data_create_empty((size_t)bound + 1 + LZ4_SIZE_PREFIX_LEN);
|
||||
if (!out)
|
||||
return NULL;
|
||||
uint32_t raw_size = (uint32_t)in->size;
|
||||
uint8_t* p = (uint8_t*)out->data;
|
||||
p[0] = (uint8_t)algo;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
p[1 + i] = (uint8_t)((raw_size >> (8 * i)) & 0xff);
|
||||
uLongf dest_len = bound;
|
||||
int rc = compress2(p + 1 + LZ4_SIZE_PREFIX_LEN, &dest_len, (const Bytef*)in->data,
|
||||
(uLong)in->size, level);
|
||||
if (rc != Z_OK) {
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = (size_t)dest_len + 1 + LZ4_SIZE_PREFIX_LEN;
|
||||
return out;
|
||||
}
|
||||
|
||||
Data* data_compress_codec(Data* data_to_compress, CompressionAlgo algo, int compression_level,
|
||||
int compression_threads) {
|
||||
if (!data_to_compress || (!data_to_compress->data && data_to_compress->size != 0) ||
|
||||
compression_threads < 0 || compression_threads > COMPRESSION_MAX_THREADS)
|
||||
return NULL;
|
||||
if (!compression_algo_valid((int)algo))
|
||||
return NULL;
|
||||
log_message(LOG_LEVEL_DEBUG, "Starting to compress data");
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return frame_with_codec(data_to_compress->data, data_to_compress->size, COMPRESSION_ALGO_NONE);
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return zstd_compress(data_to_compress, compression_level, compression_threads);
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return lz4_compress(data_to_compress);
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return zlib_compress(data_to_compress, algo, compression_level);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads) {
|
||||
return data_compress_codec(data_to_compress, compression_get_algo(), compression_level,
|
||||
compression_threads);
|
||||
}
|
||||
|
||||
Data* data_compress(Data* data_to_compress, int compression_level) {
|
||||
return data_compress_codec(data_to_compress, compression_get_algo(), compression_level, 0);
|
||||
}
|
||||
|
||||
static Data* decompress_none(const Data* compressed_data, size_t maximum_size) {
|
||||
size_t size = compressed_data->size - 1;
|
||||
if (size > maximum_size)
|
||||
return NULL;
|
||||
Data* out = data_create_empty(size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (size > 0)
|
||||
memcpy(out->data, (const uint8_t*)compressed_data->data + 1, size);
|
||||
out->size = size;
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Read the 4-byte little-endian raw size stored after the codec byte. */
|
||||
static bool read_raw_size(const Data* in, uint32_t* raw_size) {
|
||||
if (in->size < 1 + LZ4_SIZE_PREFIX_LEN)
|
||||
return false;
|
||||
const uint8_t* p = (const uint8_t*)in->data;
|
||||
uint32_t v = 0;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
v |= (uint32_t)p[1 + i] << (8 * i);
|
||||
*raw_size = v;
|
||||
return true;
|
||||
}
|
||||
|
||||
static Data* lz4_decompress(Data* compressed_data, size_t maximum_size, size_t hard_limit) {
|
||||
uint32_t raw_size = 0;
|
||||
if (!read_raw_size(compressed_data, &raw_size))
|
||||
return NULL;
|
||||
if (raw_size > hard_limit || raw_size > maximum_size)
|
||||
return NULL;
|
||||
size_t comp_size = compressed_data->size - 1 - LZ4_SIZE_PREFIX_LEN;
|
||||
Data* out = data_create_empty(raw_size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (raw_size == 0) {
|
||||
out->size = 0;
|
||||
return out;
|
||||
}
|
||||
int rc = LZ4_decompress_safe((const char*)compressed_data->data + 1 + LZ4_SIZE_PREFIX_LEN,
|
||||
(char*)out->data, (int)comp_size, (int)raw_size);
|
||||
if (rc < 0 || (uint32_t)rc != raw_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "LZ4 decompression failed");
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = raw_size;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zlib_decompress(Data* compressed_data, size_t maximum_size, size_t hard_limit) {
|
||||
uint32_t raw_size = 0;
|
||||
if (!read_raw_size(compressed_data, &raw_size))
|
||||
return NULL;
|
||||
if (raw_size > hard_limit || raw_size > maximum_size)
|
||||
return NULL;
|
||||
size_t comp_size = compressed_data->size - 1 - LZ4_SIZE_PREFIX_LEN;
|
||||
Data* out = data_create_empty(raw_size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (raw_size == 0) {
|
||||
out->size = 0;
|
||||
return out;
|
||||
}
|
||||
uLongf dest_len = raw_size;
|
||||
int rc =
|
||||
uncompress((Bytef*)out->data, &dest_len,
|
||||
(const Bytef*)compressed_data->data + 1 + LZ4_SIZE_PREFIX_LEN, (uLong)comp_size);
|
||||
if (rc != Z_OK || dest_len != raw_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "zlib decompression failed");
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = raw_size;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zstd_decompress(Data* compressed_data, size_t maximum_size) {
|
||||
/* The zstd frame starts after the codec byte. */
|
||||
const void* frame = (const uint8_t*)compressed_data->data + 1;
|
||||
size_t frame_size = compressed_data->size - 1;
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Start to decompress data");
|
||||
unsigned long long dst_size =
|
||||
ZSTD_getFrameContentSize(compressed_data->data, compressed_data->size);
|
||||
unsigned long long dst_size = ZSTD_getFrameContentSize(frame, frame_size);
|
||||
/* ZSTD_isError() is also true for ZSTD_CONTENTSIZE_ERROR and
|
||||
* ZSTD_CONTENTSIZE_UNKNOWN (both are encoded near (size_t)-1), so test the
|
||||
* sentinels explicitly instead of blanket-rejecting every error-ish value:
|
||||
@@ -256,9 +584,9 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
// ZSTD_CONTENTSIZE_UNKNOWN (~2^64) can cause massive allocation;
|
||||
// fall back to a conservative estimate (3x compressed size) when unknown.
|
||||
if (dst_size == ZSTD_CONTENTSIZE_UNKNOWN) {
|
||||
if (compressed_data->size > ULLONG_MAX / 3)
|
||||
if (frame_size > ULLONG_MAX / 3)
|
||||
return NULL;
|
||||
dst_size = compressed_data->size * 3;
|
||||
dst_size = frame_size * 3;
|
||||
if (dst_size < INITIAL_DECOMPRESS_BUF_SIZE)
|
||||
dst_size = INITIAL_DECOMPRESS_BUF_SIZE;
|
||||
}
|
||||
@@ -295,7 +623,7 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
goto cleanup;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {compressed_data->data, compressed_data->size, 0};
|
||||
ZSTD_inBuffer input = {frame, frame_size, 0};
|
||||
ZSTD_outBuffer output = {uncompressed_data->data, buf_size, 0};
|
||||
|
||||
size_t ret;
|
||||
@@ -354,6 +682,31 @@ cleanup:
|
||||
return uncompressed_data;
|
||||
}
|
||||
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) ||
|
||||
maximum_size == 0)
|
||||
return NULL;
|
||||
if (compressed_data->size < 1)
|
||||
return NULL;
|
||||
unsigned long long hard_limit =
|
||||
maximum_size < MAX_DECOMPRESSED_SIZE ? maximum_size : MAX_DECOMPRESSED_SIZE;
|
||||
uint8_t codec = ((const uint8_t*)compressed_data->data)[0];
|
||||
if (!compression_algo_valid(codec))
|
||||
return NULL;
|
||||
switch ((CompressionAlgo)codec) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return decompress_none(compressed_data, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return zstd_decompress(compressed_data, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return lz4_decompress(compressed_data, maximum_size, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return zlib_decompress(compressed_data, maximum_size, (size_t)hard_limit);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* data_decompress(Data* compressed_data) {
|
||||
return data_decompress_limited(compressed_data, MAX_DECOMPRESSED_SIZE);
|
||||
}
|
||||
|
||||
@@ -6,11 +6,79 @@
|
||||
|
||||
#define COMPRESSION_MAX_THREADS 64
|
||||
|
||||
/* Compression algorithms selectable with --compress-choice / -z. The ids are
|
||||
* the values placed on the wire (Config->compression_algo), so they must be
|
||||
* kept stable. NONE is "no compression"; ZSTD is the historical FastSync
|
||||
* default and the negotiated "auto" choice. ZLIBX is rsync's zlib-without-
|
||||
* matched-data variant: FastSync compresses only the delta/token bytes (it does
|
||||
* not put matched file data in the compression stream), so its zlib codec is
|
||||
* already the "x" form and zlib/zlibx share the same implementation, recorded
|
||||
* under distinct ids. */
|
||||
typedef enum {
|
||||
COMPRESSION_ALGO_NONE = 0,
|
||||
COMPRESSION_ALGO_ZSTD = 1,
|
||||
COMPRESSION_ALGO_LZ4 = 2,
|
||||
COMPRESSION_ALGO_ZLIB = 3,
|
||||
COMPRESSION_ALGO_ZLIBX = 4
|
||||
} CompressionAlgo;
|
||||
|
||||
/* Resolve a --compress-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "zstd", "lz4", "zlib", "zlibx", "none". "auto" is not an algorithm
|
||||
* here; the caller resolves it to the negotiated default. Returns -1 for any
|
||||
* unrecognized name. */
|
||||
int compression_algo_from_name(const char* name);
|
||||
const char* compression_algo_name(CompressionAlgo algo);
|
||||
bool compression_algo_valid(int algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list
|
||||
* (rsync 3.4.1's `--version` order: zstd lz4 zlibx zlib none). Resolves
|
||||
* "auto". */
|
||||
CompressionAlgo compression_negotiate_default(void);
|
||||
|
||||
/* Resolve "auto" the way rsync does: the first supported name in
|
||||
* RSYNC_COMPRESS_LIST (whitespace-separated, client half ends at '&'), then the
|
||||
* compiled-in preference order when the variable is unset/blank. Returns -1
|
||||
* when the variable is set but names no supported codec (rsync's failed
|
||||
* negotiation), otherwise a valid CompressionAlgo id. */
|
||||
int compression_choice_resolve(void);
|
||||
|
||||
/* rsync 3.4.1's per-codec default level, applied when the user did not pass
|
||||
* --compress-level/--zl. zstd uses ZSTD_CLEVEL_DEFAULT (3) and zlib/zlibx the
|
||||
* resolved Z_DEFAULT_COMPRESSION (6). lz4 has no tunable level in rsync
|
||||
* (always the default acceleration); FastSync returns a positive placeholder so
|
||||
* its "level > 0" compression gate stays engaged, and lz4_compress ignores the
|
||||
* value, so the output is identical to rsync's. none is 0. */
|
||||
int compression_default_level(CompressionAlgo algo);
|
||||
|
||||
/* Clamp an explicit --compress-level to the codec's accepted range the way
|
||||
* rsync's init_compression_level() does: zstd 1..22, zlib/zlibx 1..9, lz4
|
||||
* ignored (fixed positive placeholder), none 0. */
|
||||
int compression_clamp_level(CompressionAlgo algo, int level);
|
||||
|
||||
/* True when the algorithm actually compresses (i.e. is not NONE). */
|
||||
bool compression_algo_enabled(CompressionAlgo algo);
|
||||
|
||||
/* Select the process-wide codec used by the legacy wrappers below. Each
|
||||
* process serves exactly one transfer config (the server forks per connection,
|
||||
* the client configures itself before spawning transfer threads), so a
|
||||
* process-global default is sufficient and constant for the lifetime of a
|
||||
* transfer. Defaults to ZSTD when never set. Thread-safe. */
|
||||
void compression_set_algo(CompressionAlgo algo);
|
||||
CompressionAlgo compression_get_algo(void);
|
||||
|
||||
/* Codec-aware primitives. The compressed buffer is self-describing: its first
|
||||
* byte is the CompressionAlgo id, so decompression never needs the codec passed
|
||||
* separately (this keeps every existing Decompress call site source-compatible).
|
||||
* `data_compress_codec` returns NULL on invalid input or an unsupported codec. */
|
||||
Data* data_compress_codec(Data* data_to_compress, CompressionAlgo algo, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
|
||||
/* Legacy zstd-default wrappers retained for existing callers/tests. */
|
||||
Data* data_compress(Data* data_to_compress, int compression_level);
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress(Data* compressed_data);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count);
|
||||
|
||||
/* Release the calling thread's cached zstd contexts (compressor, decompressor
|
||||
|
||||
+305
-47
@@ -14,12 +14,15 @@
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <limits.h>
|
||||
#include <errno.h>
|
||||
|
||||
static void config_set_defaults(Config* config) {
|
||||
config->scanner_threads = 0;
|
||||
config->metadata_explicitly_disabled = false;
|
||||
config->preserve_perms_explicit_off = false;
|
||||
config->preserve_times_explicit_off = false;
|
||||
config->show_progress = false;
|
||||
config->compression_threads = 0;
|
||||
config->ssh_port = 22;
|
||||
@@ -43,12 +46,14 @@ static void config_set_defaults(Config* config) {
|
||||
config->server_port = 8080;
|
||||
config->server_port_set = false;
|
||||
config->server_host_set = false;
|
||||
/* 0 means "--timeout not given": the transport keeps its own built-in 30 s
|
||||
* socket timeout (tcp_set_timeouts ignores non-positive values) and the
|
||||
* protocol layer keeps its built-in 60 s per-message deadline. A positive
|
||||
* value overrides BOTH (see protocol_session_set_io_timeout). */
|
||||
/* rsync defaults: --timeout=0 (I/O timeouts disabled) and --contimeout=60.
|
||||
* A value of 0 disables the client's own deadline on both the socket layer
|
||||
* (tcp_set_timeouts) and the protocol layer
|
||||
* (protocol_session_set_io_timeout); a positive value sets it. A server
|
||||
* session floors the deadline at SERVER_IO_TIMEOUT_SEC so 0 can never hold a
|
||||
* connection open forever. */
|
||||
config->timeout = 0;
|
||||
config->contimeout = 10;
|
||||
config->contimeout = 60;
|
||||
config->quiet = false;
|
||||
config->stats = false;
|
||||
config->max_depth = 0;
|
||||
@@ -63,13 +68,18 @@ static void config_set_defaults(Config* config) {
|
||||
config->human_readable = false;
|
||||
config->ignore_errors = false;
|
||||
config->ignore_missing_args = false;
|
||||
config->checksum_transfer_algo = CHECKSUM_ALGO_DEFAULT;
|
||||
config->cli_exit_code = 0;
|
||||
config->compression_level_set = false;
|
||||
config->checksum_choice_set = false;
|
||||
config->filters = NULL;
|
||||
config->files_from = NULL;
|
||||
config->files_from_set = NULL;
|
||||
config->from0 = false;
|
||||
config->cvs_exclude = false;
|
||||
config->per_dir_filter = false;
|
||||
config->one_file_system = false;
|
||||
config->per_dir_filter_count = 0;
|
||||
config->one_file_system = 0;
|
||||
config->no_implied_dirs = false;
|
||||
config->dirs = false;
|
||||
config->rsh_command = NULL;
|
||||
@@ -192,10 +202,14 @@ static bool validate_received_config(const Config* config) {
|
||||
valid_wire_bool(config->partial) && valid_wire_bool(config->delete_before) &&
|
||||
valid_wire_bool(config->checksum) && valid_wire_bool(config->eight_bit_output) &&
|
||||
valid_wire_bool(config->dry_run) && checksum_algo_valid(config->checksum_algo) &&
|
||||
identity_wire_valid(config) && valid_wire_bool(config->preserve_atimes) &&
|
||||
valid_wire_bool(config->preserve_crtimes) && valid_wire_bool(config->omit_dir_times) &&
|
||||
valid_wire_bool(config->omit_link_times) && valid_wire_bool(config->munge_links) &&
|
||||
valid_wire_bool(config->keep_dirlinks) && valid_wire_bool(config->fake_super) &&
|
||||
compression_algo_valid(config->compression_algo) && identity_wire_valid(config) &&
|
||||
valid_wire_bool(config->preserve_atimes) && valid_wire_bool(config->preserve_crtimes) &&
|
||||
valid_wire_bool(config->omit_dir_times) && valid_wire_bool(config->omit_link_times) &&
|
||||
valid_wire_bool(config->preserve_perms) && valid_wire_bool(config->preserve_times) &&
|
||||
valid_wire_bool(config->preserve_owner) && valid_wire_bool(config->preserve_group) &&
|
||||
valid_wire_bool(config->munge_links) && valid_wire_bool(config->keep_dirlinks) &&
|
||||
valid_wire_bool(config->fake_super) && valid_wire_bool(config->report_dest_info) &&
|
||||
valid_wire_bool(config->report_stats) && valid_wire_bool(config->report_deletes) &&
|
||||
(!config->copy_as_set || (config->copy_as_uid >= 0 && config->copy_as_gid >= 0)) &&
|
||||
(!config->use_compression ||
|
||||
(config->compression_level >= 1 && config->compression_level <= 22)) &&
|
||||
@@ -203,8 +217,9 @@ static bool validate_received_config(const Config* config) {
|
||||
config->delta_block_size >= DELTA_BLOCK_SIZE_MIN &&
|
||||
config->delta_block_size <= DELTA_BLOCK_SIZE_MAX &&
|
||||
config->delta_max_file_size <= DELTA_MAX_FILE_SIZE && config->modify_window >= 0 &&
|
||||
config->max_delete >= -1 && config->skip_compress_count >= 0 &&
|
||||
config->skip_compress_count <= MAX_SKIP_COMPRESS_SUFFIXES && config->max_alloc > 0 &&
|
||||
config->max_delete >= -1 && config->max_alloc <= MAX_SERVER_ALLOC &&
|
||||
config->skip_compress_count >= 0 &&
|
||||
config->skip_compress_count <= MAX_SKIP_COMPRESS_SUFFIXES &&
|
||||
(!config->chmod_spec || !*config->chmod_spec ||
|
||||
chmod_apply(0, config->chmod_spec, &(mode_t){0})) &&
|
||||
config->super_mode >= SUPER_MODE_AUTO && config->super_mode <= SUPER_MODE_OFF;
|
||||
@@ -221,7 +236,13 @@ Config* config_create(void) {
|
||||
bool config_delete_timing_early(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
return config->delete_before || config->delete_during;
|
||||
return config->delete_before;
|
||||
}
|
||||
|
||||
bool config_delete_timing_per_dir(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
return config->delete_during || config->delete_delay;
|
||||
}
|
||||
|
||||
/* A delete-timing flag is only meaningful together with --delete. At most one
|
||||
@@ -292,40 +313,68 @@ const char* config_invariants_error(const Config* config) {
|
||||
"timing; at most one may be given and each implies --delete";
|
||||
if (config->iconv_spec && !charset_spec_valid(config->iconv_spec))
|
||||
return "--iconv requires LOCAL[,REMOTE] charset names supported by iconv";
|
||||
if ((config->preserve_perms || config->preserve_times || config->preserve_owner ||
|
||||
config->preserve_group || config->preserve_atimes || config->preserve_crtimes ||
|
||||
config->use_executability) &&
|
||||
!config->use_metadata)
|
||||
return "a preservation attribute requires metadata transmission";
|
||||
if (config->copy_as_set && !config->use_metadata)
|
||||
return "--copy-as requires metadata preservation and cannot be combined with --no-preserve";
|
||||
return NULL;
|
||||
}
|
||||
|
||||
bool config_derived_use_metadata(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
if (config->preserve_perms || config->preserve_times || config->preserve_owner ||
|
||||
config->preserve_group || config->preserve_atimes || config->preserve_crtimes ||
|
||||
config->use_executability || config->preserve_xattrs || config->preserve_acls ||
|
||||
config->fake_super || config->preserve_devices || config->preserve_specials ||
|
||||
config->copy_devices || config->write_devices ||
|
||||
(config->chmod_spec && config->chmod_spec[0]) || config->copy_as_set ||
|
||||
config->chown_uid_set || config->chown_gid_set || config->usermap_count > 0 ||
|
||||
config->groupmap_count > 0 || config->update)
|
||||
return true;
|
||||
return (config->use_incremental || config->use_delta) && !config->metadata_explicitly_disabled;
|
||||
}
|
||||
|
||||
bool config_has_basis(const Config* config) {
|
||||
return config && config->basis_count > 0;
|
||||
}
|
||||
|
||||
/* A basis-dir path travels from the client to the receiver and is resolved
|
||||
* below the destination root, so it must be a non-empty relative path with no
|
||||
* "." or ".." component and no traversal: an absolute or escaping path would
|
||||
* make the receiver read or link files outside its authorized root.
|
||||
* below the destination root when relative, or used verbatim when absolute
|
||||
* (matching rsync). Either form must be non-empty, traversal-free (no "..")
|
||||
* and free of "." components: an escaping path would make the receiver read or
|
||||
* link files outside its authorized root. An absolute path is still subject to
|
||||
* the receiver's root confinement at open time (file_open_secure_parent), so a
|
||||
* basis outside the authorized root is simply not found rather than an escape.
|
||||
*
|
||||
* Returns a malloc'd CANONICAL copy of an accepted path, or NULL when the path
|
||||
* is rejected. Canonicalization collapses interior empty components ("a//b" ->
|
||||
* "a/b"), drops "." components and trailing "/"s, so validation, the delete
|
||||
* walker prefix match and the receiver's basis lookup all agree on one form.
|
||||
* The normalizer is the single source of truth for both config_basis_path_valid
|
||||
* and config_basis_append. */
|
||||
* "a/b"), drops "." components and trailing "/"s, and preserves a leading '/'
|
||||
* for absolute paths, so validation, the delete walker prefix match and the
|
||||
* receiver's basis lookup all agree on one form. The normalizer is the single
|
||||
* source of truth for both config_basis_path_valid and config_basis_append. */
|
||||
static char* basis_path_normalize(const char* path) {
|
||||
if (!path || path[0] == '\0' || path[0] == '/' || has_path_traversal(path))
|
||||
if (!path || path[0] == '\0' || has_path_traversal(path))
|
||||
return NULL;
|
||||
if (strcmp(path, ".") == 0)
|
||||
bool absolute = path[0] == '/';
|
||||
if (!absolute && strcmp(path, ".") == 0)
|
||||
return NULL;
|
||||
if (absolute && strcmp(path, "/") == 0)
|
||||
return NULL;
|
||||
char* dup = str_dup(path);
|
||||
if (!dup)
|
||||
return NULL;
|
||||
size_t out_len = 0;
|
||||
char* out = malloc(strlen(path) + 1);
|
||||
char* out = malloc(strlen(path) + 2);
|
||||
if (!out) {
|
||||
free(dup);
|
||||
return NULL;
|
||||
}
|
||||
if (absolute)
|
||||
out[out_len++] = '/';
|
||||
char* saveptr = NULL;
|
||||
bool ok = true;
|
||||
for (char* part = strtok_r(dup, "/", &saveptr); part; part = strtok_r(NULL, "/", &saveptr)) {
|
||||
@@ -335,14 +384,14 @@ static char* basis_path_normalize(const char* path) {
|
||||
}
|
||||
if (strcmp(part, ".") == 0)
|
||||
continue;
|
||||
if (out_len > 0)
|
||||
if (out_len > 0 && out[out_len - 1] != '/')
|
||||
out[out_len++] = '/';
|
||||
size_t len = strlen(part);
|
||||
memcpy(out + out_len, part, len);
|
||||
out_len += len;
|
||||
}
|
||||
free(dup);
|
||||
if (!ok || out_len == 0) {
|
||||
if (!ok || out_len == 0 || (absolute && out_len == 1)) {
|
||||
free(out);
|
||||
return NULL;
|
||||
}
|
||||
@@ -725,15 +774,25 @@ void config_delete(Config* config) {
|
||||
free(config->skip_compress_suffixes[i]);
|
||||
free(config->skip_compress_suffixes);
|
||||
}
|
||||
free(config->usermap);
|
||||
if (config->usermap) {
|
||||
for (int i = 0; i < config->usermap_count; i++)
|
||||
free(config->usermap[i].to_name);
|
||||
free(config->usermap);
|
||||
}
|
||||
config->usermap = NULL;
|
||||
config->usermap_count = 0;
|
||||
free(config->groupmap);
|
||||
if (config->groupmap) {
|
||||
for (int i = 0; i < config->groupmap_count; i++)
|
||||
free(config->groupmap[i].to_name);
|
||||
free(config->groupmap);
|
||||
}
|
||||
config->groupmap = NULL;
|
||||
config->groupmap_count = 0;
|
||||
if (config->filters) {
|
||||
array_list_delete(config->filters);
|
||||
}
|
||||
filter_rule_list_free(config->protect_rules);
|
||||
config->protect_rules = NULL;
|
||||
/* A --delay-updates staging tree is transient receiver state: remove any
|
||||
leftovers on every exit path (success already emptied it). */
|
||||
if (config->delay_context)
|
||||
@@ -756,11 +815,14 @@ void config_delete(Config* config) {
|
||||
* ------------------------------------------------------------------------- */
|
||||
|
||||
/* --max-alloc: raw 64-bit value, clamped server-side and installed as the
|
||||
* session allocation ceiling. A zero value is rejected. */
|
||||
* session allocation ceiling. A received 0 is rsync's "no alloc limit"; on the
|
||||
* receive path it is mapped to the server ceiling so a client can never disable
|
||||
* it (client-side 0 remains unlimited). Any value above the ceiling is clamped
|
||||
* to it. */
|
||||
static bool config_receive_max_alloc(int fd, unsigned long long* value) {
|
||||
if (!receive_n_data(fd, value, sizeof(*value)) || *value == 0)
|
||||
if (!receive_n_data(fd, value, sizeof(*value)))
|
||||
return false;
|
||||
if (*value > MAX_SERVER_ALLOC)
|
||||
if (*value == 0 || *value > MAX_SERVER_ALLOC)
|
||||
*value = MAX_SERVER_ALLOC;
|
||||
protocol_session_set_max_alloc(NULL, *value);
|
||||
return true;
|
||||
@@ -842,6 +904,14 @@ static bool config_receive_checksum_algo(int fd, int* value) {
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool config_receive_compression_algo(int fd, int* value) {
|
||||
int algo;
|
||||
if (!receive_int(fd, &algo) || !compression_algo_valid(algo))
|
||||
return false;
|
||||
*value = algo;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool config_receive_super_mode(int fd, SuperMode* value) {
|
||||
int mode;
|
||||
if (!receive_int(fd, &mode) || mode < SUPER_MODE_AUTO || mode > SUPER_MODE_OFF)
|
||||
@@ -958,9 +1028,128 @@ static bool receive_basis_entries(int fd, Config* c, ConfigStringBudget* budget)
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Receiver-side delete-protection rules (protocol 2.28.0). The sender compiles
|
||||
* its command-line selection rules exactly as the scanner does and streams the
|
||||
* result as one bounded, self-describing block (count + per-rule records); the
|
||||
* receiver reconstructs a FilterRuleList for the --delete extras walk. owner
|
||||
* and pattern are charged through the shared ConfigStringBudget and the block
|
||||
* additionally enforces MAX_FILTER_RULES / MAX_FILTER_BYTES. */
|
||||
static bool send_protect_entries(int fd, const Config* c) {
|
||||
int count = c->filters ? c->filters->size : 0;
|
||||
const char** texts = NULL;
|
||||
if (count > 0) {
|
||||
texts = malloc((size_t)count * sizeof(char*));
|
||||
if (!texts)
|
||||
return false;
|
||||
for (int i = 0; i < count; i++)
|
||||
texts[i] = (const char*)c->filters->items[i];
|
||||
}
|
||||
char err[160];
|
||||
FilterRuleList* rules =
|
||||
filter_base_build(texts, count, c->cvs_exclude, c->delete_excluded, err, sizeof(err));
|
||||
free(texts);
|
||||
if (!rules) {
|
||||
log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err);
|
||||
return false;
|
||||
}
|
||||
bool ok = send_int(fd, rules->count);
|
||||
for (int i = 0; ok && i < rules->count; i++) {
|
||||
const FilterRule* r = rules->items[i];
|
||||
/* Mirror the receiver's limit so the peer never receives a rule it will
|
||||
reject as a protocol error. */
|
||||
if (r->pattern && strlen(r->pattern) > MAX_PROTECT_PATTERN_LEN) {
|
||||
log_message(LOG_LEVEL_ERROR, "filter pattern exceeds %d bytes", MAX_PROTECT_PATTERN_LEN);
|
||||
filter_rule_list_free(rules);
|
||||
return false;
|
||||
}
|
||||
ok = send_int(fd, (int)r->action) && send_int(fd, (int)r->sides) &&
|
||||
send_int(fd, r->anchored ? 1 : 0) && send_int(fd, r->dir_only ? 1 : 0) &&
|
||||
send_int(fd, r->negate ? 1 : 0) && send_str(fd, r->owner ? r->owner : "") &&
|
||||
send_str(fd, r->pattern ? r->pattern : "");
|
||||
}
|
||||
filter_rule_list_free(rules);
|
||||
return ok;
|
||||
}
|
||||
|
||||
static bool receive_protect_entries(int fd, Config* c, ConfigStringBudget* budget) {
|
||||
int count;
|
||||
if (!receive_int(fd, &count))
|
||||
return false;
|
||||
if (count < 0 || count > MAX_FILTER_RULES)
|
||||
return false;
|
||||
if (count == 0)
|
||||
return true;
|
||||
FilterRuleList* list = filter_rule_list_create();
|
||||
if (!list)
|
||||
return false;
|
||||
size_t pattern_bytes = 0;
|
||||
for (int i = 0; i < count; i++) {
|
||||
int action;
|
||||
int sides;
|
||||
bool anchored;
|
||||
bool dir_only;
|
||||
bool negate;
|
||||
if (!receive_int(fd, &action) ||
|
||||
(action != FILTER_ACTION_EXCLUDE && action != FILTER_ACTION_INCLUDE) ||
|
||||
!receive_int(fd, &sides) || sides < (int)FILTER_SIDE_SENDER ||
|
||||
sides > (int)(FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER) ||
|
||||
!receive_wire_bool(fd, &anchored) || !receive_wire_bool(fd, &dir_only) ||
|
||||
!receive_wire_bool(fd, &negate))
|
||||
goto fail;
|
||||
char* owner = config_receive_str(fd, budget);
|
||||
if (!owner)
|
||||
goto fail;
|
||||
char* pattern = config_receive_str(fd, budget);
|
||||
if (!pattern || pattern[0] == '\0') {
|
||||
free(owner);
|
||||
free(pattern);
|
||||
goto fail;
|
||||
}
|
||||
/* A pattern too long to be evaluated by glob_match against a PATH_MAX path
|
||||
would silently fail to match and leave a protect rule inert (fail-open:
|
||||
the entry is then deleted). Reject it up front as a protocol error
|
||||
rather than accept a rule that can never shield anything. */
|
||||
if (strlen(pattern) > MAX_PROTECT_PATTERN_LEN) {
|
||||
free(owner);
|
||||
free(pattern);
|
||||
goto fail;
|
||||
}
|
||||
size_t bytes = strlen(owner) + strlen(pattern);
|
||||
if (bytes > MAX_FILTER_BYTES - pattern_bytes) {
|
||||
free(owner);
|
||||
free(pattern);
|
||||
goto fail;
|
||||
}
|
||||
pattern_bytes += bytes;
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule) {
|
||||
free(owner);
|
||||
free(pattern);
|
||||
goto fail;
|
||||
}
|
||||
rule->action = (FilterAction)action;
|
||||
rule->sides = (unsigned)sides;
|
||||
rule->anchored = anchored;
|
||||
rule->dir_only = dir_only;
|
||||
rule->negate = negate;
|
||||
rule->owner = owner;
|
||||
rule->pattern = pattern;
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
goto fail;
|
||||
}
|
||||
}
|
||||
c->protect_rules = list;
|
||||
return true;
|
||||
fail:
|
||||
filter_rule_list_free(list);
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool send_identity_entries(int fd, const IdentityMap* map, int count) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (!send_int(fd, map[i].from) || !send_int(fd, map[i].to))
|
||||
if (!send_int(fd, map[i].from) || !send_int(fd, map[i].from_hi) || !send_int(fd, map[i].to) ||
|
||||
!send_str(fd, map[i].to_name ? map[i].to_name : ""))
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
@@ -968,20 +1157,32 @@ static bool send_identity_entries(int fd, const IdentityMap* map, int count) {
|
||||
|
||||
static bool receive_identity_entries(int fd, ConfigStringBudget* budget, int count,
|
||||
IdentityMap** out) {
|
||||
(void)budget;
|
||||
if (count <= 0)
|
||||
return true;
|
||||
IdentityMap* map = calloc((size_t)count, sizeof(IdentityMap));
|
||||
if (!map)
|
||||
return false;
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (!receive_int(fd, &map[i].from) || !receive_int(fd, &map[i].to)) {
|
||||
free(map);
|
||||
return false;
|
||||
if (!receive_int(fd, &map[i].from) || !receive_int(fd, &map[i].from_hi) ||
|
||||
!receive_int(fd, &map[i].to))
|
||||
goto fail;
|
||||
char* name = config_receive_str(fd, budget);
|
||||
if (!name)
|
||||
goto fail;
|
||||
if (name[0] == '\0') {
|
||||
free(name);
|
||||
map[i].to_name = NULL;
|
||||
} else {
|
||||
map[i].to_name = name;
|
||||
}
|
||||
}
|
||||
*out = map;
|
||||
return true;
|
||||
fail:
|
||||
for (int i = 0; i < count; i++)
|
||||
free(map[i].to_name);
|
||||
free(map);
|
||||
return false;
|
||||
}
|
||||
|
||||
/* ---------------------------------------------------------------------------
|
||||
@@ -1030,6 +1231,9 @@ static bool receive_identity_entries(int fd, ConfigStringBudget* budget, int cou
|
||||
#define CONFIG_SEND_INT_CHECKSUM_ALGO(name) send_int(fd, c->name)
|
||||
#define CONFIG_RECV_INT_CHECKSUM_ALGO(name) config_receive_checksum_algo(fd, &c->name)
|
||||
|
||||
#define CONFIG_SEND_INT_COMPRESSION_ALGO(name) send_int(fd, c->name)
|
||||
#define CONFIG_RECV_INT_COMPRESSION_ALGO(name) config_receive_compression_algo(fd, &c->name)
|
||||
|
||||
#define CONFIG_SEND_SUPERMODE(name) send_int(fd, (int)c->name)
|
||||
#define CONFIG_RECV_SUPERMODE(name) config_receive_super_mode(fd, &c->name)
|
||||
|
||||
@@ -1069,6 +1273,9 @@ static bool receive_identity_entries(int fd, ConfigStringBudget* budget, int cou
|
||||
#define CONFIG_RECV_BLOCK_IDMAP(name) \
|
||||
receive_identity_entries(fd, budget, c->name##_count, &c->name)
|
||||
|
||||
#define CONFIG_SEND_BLOCK_PROTECT_RULES(name) send_protect_entries(fd, c)
|
||||
#define CONFIG_RECV_BLOCK_PROTECT_RULES(name) receive_protect_entries(fd, c, budget)
|
||||
|
||||
/* One table entry, applied in sequence. XSEND/XRECV are statement macros so
|
||||
* consecutive entries read as a plain sequence of assignments. */
|
||||
#define XSEND(name, ctype, def, kind) ok = ok && (CONFIG_SEND_##kind(name));
|
||||
@@ -1104,6 +1311,9 @@ CONFIG_DEFINE_SEND(send_daemon_auth, CONFIG_WIRE_DAEMON_AUTH_FIELDS)
|
||||
CONFIG_DEFINE_SEND(send_iconv_spec, CONFIG_WIRE_ICONV_FIELDS)
|
||||
CONFIG_DEFINE_SEND(send_privilege_options, CONFIG_WIRE_PRIVILEGE_FIELDS)
|
||||
CONFIG_DEFINE_SEND(send_copy_as_options, CONFIG_WIRE_COPY_AS_FIELDS)
|
||||
CONFIG_DEFINE_SEND(send_output_options, CONFIG_WIRE_OUTPUT_FIELDS)
|
||||
CONFIG_DEFINE_SEND(send_codec_options, CONFIG_WIRE_CODEC_FIELDS)
|
||||
CONFIG_DEFINE_SEND(send_protect_options, CONFIG_WIRE_PROTECT_FIELDS)
|
||||
|
||||
CONFIG_DEFINE_RECV(receive_core_fields, CONFIG_WIRE_CORE_FIELDS)
|
||||
CONFIG_DEFINE_RECV(receive_delta_fields, CONFIG_WIRE_DELTA_FIELDS)
|
||||
@@ -1122,6 +1332,9 @@ CONFIG_DEFINE_RECV(receive_daemon_auth, CONFIG_WIRE_DAEMON_AUTH_FIELDS)
|
||||
CONFIG_DEFINE_RECV(receive_iconv_spec, CONFIG_WIRE_ICONV_FIELDS)
|
||||
CONFIG_DEFINE_RECV(receive_privilege_options, CONFIG_WIRE_PRIVILEGE_FIELDS)
|
||||
CONFIG_DEFINE_RECV(receive_copy_as_options, CONFIG_WIRE_COPY_AS_FIELDS)
|
||||
CONFIG_DEFINE_RECV(receive_output_options, CONFIG_WIRE_OUTPUT_FIELDS)
|
||||
CONFIG_DEFINE_RECV(receive_codec_options, CONFIG_WIRE_CODEC_FIELDS)
|
||||
CONFIG_DEFINE_RECV(receive_protect_options, CONFIG_WIRE_PROTECT_FIELDS)
|
||||
|
||||
#undef XSEND
|
||||
#undef XRECV
|
||||
@@ -1238,7 +1451,10 @@ bool config_send_wire_block(int file_descriptor, const Config* config) {
|
||||
send_daemon_module(file_descriptor, config) && send_daemon_auth(file_descriptor, config) &&
|
||||
send_iconv_spec(file_descriptor, config) &&
|
||||
send_privilege_options(file_descriptor, config) &&
|
||||
send_copy_as_options(file_descriptor, config);
|
||||
send_copy_as_options(file_descriptor, config) &&
|
||||
send_output_options(file_descriptor, config) &&
|
||||
send_codec_options(file_descriptor, config) &&
|
||||
send_protect_options(file_descriptor, config);
|
||||
}
|
||||
|
||||
bool config_send(int file_descriptor, const Config* config) {
|
||||
@@ -1308,18 +1524,60 @@ Config* config_receive_with_validate(int file_descriptor, ConfigValidateFunc val
|
||||
!receive_daemon_auth(file_descriptor, config, &budget) ||
|
||||
!receive_iconv_spec(file_descriptor, config, &budget) ||
|
||||
!receive_privilege_options(file_descriptor, config, &budget) ||
|
||||
!receive_copy_as_options(file_descriptor, config, &budget))
|
||||
!receive_copy_as_options(file_descriptor, config, &budget) ||
|
||||
!receive_output_options(file_descriptor, config, &budget) ||
|
||||
!receive_codec_options(file_descriptor, config, &budget) ||
|
||||
!receive_protect_options(file_descriptor, config, &budget))
|
||||
goto error;
|
||||
if (config->compress_choice[0] != '\0' && strcmp(config->compress_choice, "zstd") != 0 &&
|
||||
strcmp(config->compress_choice, "none") != 0) {
|
||||
char* escaped_choice = output_escape(config->compress_choice, config->eight_bit_output);
|
||||
log_message(LOG_LEVEL_ERROR, "Unsupported compression choice: %s",
|
||||
escaped_choice ? escaped_choice : "<allocation failed>");
|
||||
char detail[128];
|
||||
snprintf(detail, sizeof(detail), "unsupported compression choice: %s",
|
||||
escaped_choice ? escaped_choice : "<allocation failed>");
|
||||
send_error_detail(file_descriptor, detail);
|
||||
free(escaped_choice);
|
||||
/* Validate/normalize the negotiated codec. compress_choice is the human
|
||||
* spelling (NULL or "" when -z was not given); compression_algo is the
|
||||
* concrete codec id the sender used. They must agree, and "auto" is
|
||||
* canonicalized to FastSync's negotiated default so the stored spelling is
|
||||
* always concrete (a hostile/older client may still send "auto"). */
|
||||
if (config->compress_choice && config->compress_choice[0] != '\0') {
|
||||
int choice_algo = compression_algo_from_name(config->compress_choice);
|
||||
if (choice_algo < 0 && strcasecmp(config->compress_choice, "auto") != 0) {
|
||||
char* escaped_choice = output_escape(config->compress_choice, config->eight_bit_output);
|
||||
log_message(LOG_LEVEL_ERROR, "Unsupported compression choice: %s",
|
||||
escaped_choice ? escaped_choice : "<allocation failed>");
|
||||
char detail[160];
|
||||
snprintf(detail, sizeof(detail), "unsupported compression choice: %s",
|
||||
escaped_choice ? escaped_choice : "<allocation failed>");
|
||||
send_error_detail(file_descriptor, detail);
|
||||
free(escaped_choice);
|
||||
goto error;
|
||||
}
|
||||
if (choice_algo < 0)
|
||||
choice_algo = (int)compression_negotiate_default();
|
||||
if (strcasecmp(config->compress_choice, "auto") == 0 ||
|
||||
choice_algo == (int)COMPRESSION_ALGO_NONE) {
|
||||
const char* canonical = compression_algo_name((CompressionAlgo)choice_algo);
|
||||
char* dup = str_dup(canonical);
|
||||
if (!dup)
|
||||
goto error;
|
||||
free(config->compress_choice);
|
||||
config->compress_choice = dup;
|
||||
}
|
||||
if (config->compression_algo != choice_algo) {
|
||||
log_message(LOG_LEVEL_ERROR, "Compression choice '%s' does not match codec id %d",
|
||||
config->compress_choice, config->compression_algo);
|
||||
send_error_detail(file_descriptor, "compression choice/codec mismatch");
|
||||
goto error;
|
||||
}
|
||||
}
|
||||
/* The concrete codec must exist only when compression is on. A client that
|
||||
* left -z off has no codec in effect, but the field keeps whatever id it
|
||||
* carried (the receiver never dispatches on it without use_compression), so
|
||||
* the wire value round-trips untouched. */
|
||||
if (config->use_compression && config->compression_algo == (int)COMPRESSION_ALGO_NONE) {
|
||||
log_message(LOG_LEVEL_ERROR, "Compression requested with the 'none' codec");
|
||||
send_error_detail(file_descriptor, "compression requested with the none codec");
|
||||
goto error;
|
||||
}
|
||||
/* rsync: "none" as the pre-transfer checksum is invalid with --checksum. */
|
||||
if (config->checksum && config->checksum_algo == (int)CHECKSUM_ALGO_NONE) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid checksum-choice for --checksum: none");
|
||||
send_error_detail(file_descriptor, "checksum-choice 'none' cannot be used with --checksum");
|
||||
goto error;
|
||||
}
|
||||
if (!validate_received_config(config)) {
|
||||
|
||||
+330
-41
@@ -3,6 +3,8 @@
|
||||
|
||||
#include "array_list.h"
|
||||
#include "checksum.h"
|
||||
#include "compression.h"
|
||||
#include "filter.h"
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
@@ -39,15 +41,20 @@ typedef struct BasisDest {
|
||||
char* path; /* relative to the destination root (receiver-confined) */
|
||||
} BasisDest;
|
||||
|
||||
/* One resolved FROM:TO identity-mapping rule (--usermap / --groupmap). Both
|
||||
* fields are numeric ids. IDENTITY_MATCH_ANY (-1) in `from` is rsync's '*'
|
||||
* wildcard (matches any transmitted id); IDENTITY_CURRENT (-1) in `to` makes
|
||||
* the receiver resolve the receiving process's own current euid/egid at apply
|
||||
* time. Names are resolved to numbers at parse time on the client (see
|
||||
* identity.h for the exact subset). */
|
||||
/* One FROM:TO identity-mapping rule (--usermap / --groupmap). `from`/`from_hi`
|
||||
* describe the sender-side FROM matcher (a single id when from_hi == from, an
|
||||
* inclusive LOW-HIGH range, IDENTITY_MATCH_ANY for rsync's '*', or
|
||||
* IDENTITY_MATCH_UNNAMED for rsync's empty FROM). `to` is the receiver-side TO
|
||||
* numeric id (IDENTITY_CURRENT = the receiving process's own euid/egid) UNLESS
|
||||
* `to_name` is non-NULL, in which case the receiver resolves the name against
|
||||
* its own account database at apply time (rsync resolves TO names on the
|
||||
* receiver) and `to` is ignored. FROM names/ranges/globs are resolved on the
|
||||
* client (the sender) exactly as rsync matches them against sender names. */
|
||||
typedef struct {
|
||||
int32_t from;
|
||||
int32_t from_hi;
|
||||
int32_t to;
|
||||
char* to_name;
|
||||
} IdentityMap;
|
||||
|
||||
/* --sockopts=OPTIONS allowlist. Only these option names are accepted; anything
|
||||
@@ -76,7 +83,7 @@ typedef struct {
|
||||
typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF = 2 } SuperMode;
|
||||
|
||||
/* ===========================================================================
|
||||
* Config wire-field table (single source of truth for protocol 2.21.0).
|
||||
* Config wire-field table (single source of truth for protocol 2.28.0).
|
||||
*
|
||||
* Every field below crosses the wire. The table is the ONLY place a
|
||||
* serialized field is named: config.h expands CONFIG_WIRE_FIELDS() to declare
|
||||
@@ -191,14 +198,23 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
|
||||
X(skip_compress_count, int, 0, INT_SKIPCOUNT) \
|
||||
X(skip_compress_suffixes, char**, NULL, BLOCK_SKIP_SUFFIXES)
|
||||
|
||||
/* FastSync-only --verify-basis (protocol 2.28.0, no version bump by project
|
||||
* decision): restores the stricter content equality on a basis hit. By
|
||||
* default a basis hit is accepted on rsync's metadata quick-check alone (equal
|
||||
* size plus equal mtime, or size alone under --size-only); with this flag the
|
||||
* receiver ALSO requires the basis bytes' whole-file digest (the negotiated
|
||||
* --checksum-choice algorithm) to equal the sender's, exactly FastSync's
|
||||
* historical behavior. It is a receiver policy and crosses the wire so the
|
||||
* receiver knows whether to read and hash the basis content. */
|
||||
#define CONFIG_WIRE_BASIS_FIELDS(X) \
|
||||
X(basis_count, int, 0, INT_BASISCOUNT) \
|
||||
X(basis_dirs, BasisDest*, NULL, BLOCK_BASIS)
|
||||
X(basis_dirs, BasisDest*, NULL, BLOCK_BASIS) \
|
||||
X(verify_basis, bool, false, BOOL)
|
||||
|
||||
#define CONFIG_WIRE_FUZZY_FIELDS(X) X(fuzzy, bool, false, BOOL)
|
||||
|
||||
#define CONFIG_WIRE_CHECKSUM_FIELDS(X) \
|
||||
X(checksum_algo, int, CHECKSUM_ALGO_XXH64, INT_CHECKSUM_ALGO) \
|
||||
X(checksum_algo, int, CHECKSUM_ALGO_DEFAULT, INT_CHECKSUM_ALGO) \
|
||||
X(checksum_seed, uint64_t, 0, RAW)
|
||||
|
||||
#define CONFIG_WIRE_IDENTITY_FIELDS(X) \
|
||||
@@ -216,7 +232,11 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
|
||||
X(preserve_atimes, bool, false, BOOL) \
|
||||
X(preserve_crtimes, bool, false, BOOL) \
|
||||
X(omit_dir_times, bool, false, BOOL) \
|
||||
X(omit_link_times, bool, false, BOOL)
|
||||
X(omit_link_times, bool, false, BOOL) \
|
||||
X(preserve_perms, bool, false, BOOL) \
|
||||
X(preserve_times, bool, false, BOOL) \
|
||||
X(preserve_owner, bool, false, BOOL) \
|
||||
X(preserve_group, bool, false, BOOL)
|
||||
|
||||
#define CONFIG_WIRE_SYMLINK_TRUST_FIELDS(X) \
|
||||
X(munge_links, bool, false, BOOL) \
|
||||
@@ -237,6 +257,66 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
|
||||
X(copy_as_uid, int32_t, 0, COPY_AS_ID) \
|
||||
X(copy_as_gid, int32_t, 0, COPY_AS_ID)
|
||||
|
||||
/* Output-parity wave (protocol 2.23.0). report_dest_info tells the receiver to
|
||||
* answer every per-file STATUS_CHECK with a STATUS_DEST_INFO snapshot of the
|
||||
* pre-transfer destination entry (see protocol.h). It is set by the client
|
||||
* only when -i/--itemize-changes or --out-format asks for per-file change
|
||||
* output; the transfer decision itself is unchanged.
|
||||
*
|
||||
* Wire-stats wave (protocol 2.25.0). report_stats tells the receiver to send a
|
||||
* STATUS_STATS frame immediately before its terminal success status carrying
|
||||
* the receiver-only counters (matched data, deleted-file count) and,
|
||||
* for -n/--dry-run --delete, the destination-relative paths it WOULD have
|
||||
* deleted. It is set by the client only when --stats, --progress/-P, an
|
||||
* --out-format token needs a wire counter (%b/%c), or a dry-run carries
|
||||
* --delete; the transfer decision itself is unchanged.
|
||||
*
|
||||
* --info wave (protocol 2.27.0). report_deletes tells the receiver to include
|
||||
* the destination-relative paths it ACTUALLY removed in its terminal
|
||||
* STATUS_STATS record (the same path-list field the dry-run would-delete report
|
||||
* uses), so the sender can print rsync's `deleting PATH`/`*deleting` lines for a
|
||||
* real (non-dry-run) deletion. It is set when --delete is active and any of
|
||||
* --info=del, -i/--itemize-changes or --out-format requests per-file change
|
||||
* output; the transfer decision itself is unchanged. */
|
||||
#define CONFIG_WIRE_OUTPUT_FIELDS(X) \
|
||||
X(report_dest_info, bool, false, BOOL) \
|
||||
X(report_stats, bool, false, BOOL) X(report_deletes, bool, false, BOOL)
|
||||
|
||||
/* Codec-negotiation wave (protocol 2.26.0). compression_algo is the concrete
|
||||
* codec the client selected for this transfer (a CompressionAlgo id) and is the
|
||||
* value the receiver validates and installs. It is the resolved result of
|
||||
* --compress-choice / the "auto" negotiation so both peers agree exactly.
|
||||
*
|
||||
* Negotiation model: FastSync enforces a strict same-version handshake, so both
|
||||
* peers carry the identical compiled-in codec set. The client resolves the
|
||||
* effective algorithm deterministically and serializes it here; "auto" picks
|
||||
* the first entry of the rsync 3.4.1 preference order
|
||||
* (compression: zstd lz4 zlibx zlib none; checksum: xxh128 xxh3 xxh64 md5 md4
|
||||
* sha1 none), and an explicit request wins. The receiver rejects (before
|
||||
* STATUS_OK) any algorithm outside its own supported set, which is rsync's
|
||||
* "no common choice is an error" behavior. The same resolver runs on both
|
||||
* sides (compression_negotiate_default / checksum_negotiate_default), so the
|
||||
* fallback is consistent.
|
||||
*
|
||||
* The field is appended after the output block so every pre-2.26 field keeps
|
||||
* its wire position. */
|
||||
#define CONFIG_WIRE_CODEC_FIELDS(X) \
|
||||
X(compression_algo, int, COMPRESSION_ALGO_ZSTD, INT_COMPRESSION_ALGO)
|
||||
|
||||
/* Receiver-side delete-protection filter rules (protocol 2.28.0). The sender
|
||||
* compiles its root-level selection rules exactly as the scanner does
|
||||
* (filter_base_build over --filter/-f/--exclude/--include/-C) and streams them
|
||||
* as one self-describing, bounded block (count followed by per-rule records).
|
||||
* The receiver reconstructs `protect_rules` and evaluates them against
|
||||
* DESTINATION-ONLY entries during the --delete extras walk, so a
|
||||
* `protect`/`P` rule protects an extra that never appeared on the sender
|
||||
* (rsync re-derives deletion protection from the filter list; FastSync
|
||||
* historically derived it only from the source scan). `protect_rules` is NULL
|
||||
* on the sender and is owned/freed by the receiver Config. Bounded by
|
||||
* MAX_FILTER_RULES and MAX_FILTER_BYTES; an unknown action/sides is a protocol
|
||||
* error. */
|
||||
#define CONFIG_WIRE_PROTECT_FIELDS(X) X(protect_rules, FilterRuleList*, NULL, BLOCK_PROTECT_RULES)
|
||||
|
||||
/* All serialized fields, in exact wire order. Concatenating the per-segment
|
||||
* lists here is what keeps the declaration order = the wire order. */
|
||||
#define CONFIG_WIRE_FIELDS(X) \
|
||||
@@ -257,7 +337,10 @@ typedef enum SuperMode { SUPER_MODE_AUTO = 0, SUPER_MODE_ON = 1, SUPER_MODE_OFF
|
||||
CONFIG_WIRE_DAEMON_AUTH_FIELDS(X) \
|
||||
CONFIG_WIRE_ICONV_FIELDS(X) \
|
||||
CONFIG_WIRE_PRIVILEGE_FIELDS(X) \
|
||||
CONFIG_WIRE_COPY_AS_FIELDS(X)
|
||||
CONFIG_WIRE_COPY_AS_FIELDS(X) \
|
||||
CONFIG_WIRE_OUTPUT_FIELDS(X) \
|
||||
CONFIG_WIRE_CODEC_FIELDS(X) \
|
||||
CONFIG_WIRE_PROTECT_FIELDS(X)
|
||||
|
||||
typedef struct Config {
|
||||
/* -j/--threads=N: number of parallel scanner worker threads for the -m
|
||||
@@ -266,6 +349,15 @@ typedef struct Config {
|
||||
* concern and is NEVER serialized into the wire config frame. */
|
||||
int scanner_threads;
|
||||
bool metadata_explicitly_disabled;
|
||||
/* CLIENT-ONLY (never serialized; not in CONFIG_WIRE_FIELDS). Set when the
|
||||
* user explicitly turned an attribute off with --no-perms / --no-times (long
|
||||
* or short form). --incremental/--delta historically auto-enabled mode and
|
||||
* mtime preservation; these flags let cli_finalize_config restore that
|
||||
* behavior while still honoring the explicit per-attribute negation. A
|
||||
* later -p/-t re-enables the attribute directly, so the flag only prevents
|
||||
* the incremental/delta implication, never a POSITIVE request. */
|
||||
bool preserve_perms_explicit_off;
|
||||
bool preserve_times_explicit_off;
|
||||
bool show_progress;
|
||||
int compression_threads;
|
||||
int ssh_port;
|
||||
@@ -301,12 +393,14 @@ typedef struct Config {
|
||||
char* tls_cert;
|
||||
char* tls_key;
|
||||
char* tls_ca;
|
||||
/* --timeout: per-message I/O deadline in seconds. 0 (the default/unset
|
||||
* sentinel) leaves the transport's built-in 30 s socket timeout and the
|
||||
* protocol's built-in 60 s per-message deadline in place; a positive value
|
||||
* overrides both. See protocol_session_set_io_timeout. */
|
||||
/* --timeout: per-message I/O deadline in seconds. 0 (rsync's default)
|
||||
* disables the deadline entirely on the client's own socket and protocol
|
||||
* layers; a positive value sets it. A server session never inherits the
|
||||
* disabled value: it applies the SERVER_IO_TIMEOUT_SEC floor (see
|
||||
* protocol_server_io_timeout_sec and tcp_set_timeouts). */
|
||||
int timeout;
|
||||
/* --contimeout: connect()/accept timeout, transport layer only. */
|
||||
/* --contimeout: connect()/accept timeout in seconds (rsync's default 60);
|
||||
* 0 disables it. Transport layer only. */
|
||||
int contimeout;
|
||||
bool quiet;
|
||||
bool stats;
|
||||
@@ -341,6 +435,22 @@ typedef struct Config {
|
||||
* enters the keep-set. Implied by --delete-missing-args. */
|
||||
bool ignore_missing_args;
|
||||
|
||||
/* Codec-negotiation CLI state (all client-only, never serialized). The
|
||||
* effective pre-transfer checksum is Config->checksum_algo (serialized);
|
||||
* checksum_transfer_algo is the rsync "transfer" half of a two-name
|
||||
* --checksum-choice form (validated and used only to mirror rsync's
|
||||
* whole-file forcing, since FastSync's per-block strong hash is fixed).
|
||||
* cli_exit_code carries a parser-requested process exit status (rsync uses 4
|
||||
* for an unsupported checksum/compress algorithm) so main() can mirror it. */
|
||||
int checksum_transfer_algo;
|
||||
int cli_exit_code;
|
||||
/* Client-only "the user explicitly chose" bits. They let the per-codec
|
||||
* default level / checksum list be applied only when the corresponding
|
||||
* rsync option was omitted (an explicit --compress-level / --checksum-choice
|
||||
* always wins). Never serialized. */
|
||||
bool compression_level_set;
|
||||
bool checksum_choice_set;
|
||||
|
||||
// Issue #129: Advanced file selection. These fields are CLIENT-ONLY: they are
|
||||
// never serialized to the wire (the receiver must not learn them).
|
||||
ArrayList* filters; /* --filter=RULE rule strings, in order */
|
||||
@@ -349,9 +459,17 @@ typedef struct Config {
|
||||
bool from0; /* -0/--from0: NUL-delimited *-from files */
|
||||
bool cvs_exclude; /* -C/--cvs-exclude: standard CVS ignore set */
|
||||
bool per_dir_filter; /* -F: apply per-directory .rsync-filter files */
|
||||
bool one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries */
|
||||
/* --no-implied-dirs: client-only. With -R + --files-from, refuse to place a
|
||||
* listed file whose ancestor directory is not itself explicitly listed. */
|
||||
/* -F click count. rsync's single -F means --filter='dir-merge
|
||||
* /.rsync-filter' (the .rsync-filter files themselves are transferred); a
|
||||
* repeated -F adds --filter='- .rsync-filter' so they are excluded too. */
|
||||
int per_dir_filter_count;
|
||||
int one_file_system; /* -x/--one-file-system: do not cross filesystem boundaries.
|
||||
Repeated -x (rsync's -xx) drops the mount-point
|
||||
directory entirely instead of recreating it empty. */
|
||||
/* --no-implied-dirs: client-only. With -R, do not transfer the source
|
||||
* metadata of the parent directories implied by a listed path; an unlisted
|
||||
* implied parent is still created (with default attributes) so the listed
|
||||
* file can be placed, matching rsync. */
|
||||
bool no_implied_dirs;
|
||||
/* -d/--dirs: client-only. Transfer the directory entries named by the
|
||||
* source argument / --files-from list without recursing into contents. */
|
||||
@@ -519,13 +637,21 @@ typedef struct Config {
|
||||
/* rsync deletion-timing family (real from Phase 3). At most one of
|
||||
delete_before / delete_during / delete_delay / delete_after may be set, and
|
||||
only together with use_delete (the CLI implies --delete for each of them).
|
||||
delete_before and delete_during select the EARLY engine mode: the keep-set
|
||||
delete_before selects the EARLY engine mode: the whole-tree keep-set
|
||||
manifest is transmitted before any file data and extras are removed then,
|
||||
acknowledged, before the first data byte. delete_delay and delete_after
|
||||
select the LATE commit mode: extras are removed only after the whole
|
||||
transfer has succeeded (plain --delete keeps this mode). The exact
|
||||
semantics and the divergences from rsync are documented in RSYNC_COMPAT.md
|
||||
and in config_delete_timing_early() below. */
|
||||
acknowledged, before the first data byte. delete_during and delete_delay
|
||||
select the per-directory delete-plan mode (protocol 2.24.0): one plan per
|
||||
source directory is streamed in directory order, and the receiver removes
|
||||
each directory's extras when its plan arrives (during) or snapshots them
|
||||
and removes them only after a successful transfer (delay). delete_after
|
||||
keeps the whole-tree commit mode: extras are removed from a fresh
|
||||
end-of-transfer destination scan only after the whole transfer succeeded.
|
||||
A plain --delete with no explicit timing flag defaults to delete_during on
|
||||
the client (cli_finalize_config), matching rsync's --del default; the old
|
||||
late-commit behavior is selected explicitly by --delete-after or the
|
||||
FastSync-only long spelling --delete-commit (an exact alias for
|
||||
--delete-after, mapped onto the same wire field). See
|
||||
config_delete_timing_early()/config_delete_timing_per_dir() below. */
|
||||
/* partial_dir */
|
||||
// PR #174: Partial transfer resumption
|
||||
/* suffix */
|
||||
@@ -563,13 +689,17 @@ typedef struct Config {
|
||||
* targets and, with -K, follows an in-root destination symlink-to-directory);
|
||||
* -k/--copy-dirlinks is sender-only and is never serialized. */
|
||||
/* numeric_ids */
|
||||
/* --numeric-ids: no name lookup, use the transmitted numeric ids raw. */
|
||||
/* --numeric-ids: a mapping MODIFIER only -- no name lookup, use the
|
||||
* transmitted numeric ids raw. It does NOT by itself request ownership. */
|
||||
/* chown_uid_set */
|
||||
/* --chown USER (owner) override; IDENTITY_CURRENT = the receiver's euid. */
|
||||
/* chown_gid_set */
|
||||
/* --chown :GROUP (group) override; IDENTITY_CURRENT = the receiver's egid. */
|
||||
/* usermap */
|
||||
/* --usermap / --groupmap entries, in order (first match wins). */
|
||||
/* --usermap / --groupmap entries, in order (first match wins). Each entry's
|
||||
* from/from_hi are a single id, an inclusive range, IDENTITY_MATCH_ANY ('*'),
|
||||
* or IDENTITY_MATCH_UNNAMED (empty FROM); to_name carries a receiver-resolved
|
||||
* TO name (rsync resolves TO names on the receiving side). */
|
||||
/* preserve_atimes */
|
||||
/* -U/--atimes: preserve source access times on the destination. */
|
||||
/* preserve_crtimes */
|
||||
@@ -579,10 +709,29 @@ typedef struct Config {
|
||||
/* -O/--omit-dir-times: do not apply mtimes to directories. */
|
||||
/* omit_link_times */
|
||||
/* -J/--omit-link-times: do not apply times to symlinks. */
|
||||
/* preserve_perms */
|
||||
/* -p/--perms: preserve the source permission bits (mode). One of the four
|
||||
* per-attribute preservation flags split out of the former single
|
||||
* use_metadata bundle; --chmod and -A/--acls also imply it. */
|
||||
/* preserve_times */
|
||||
/* -t/--times: preserve source modification times. Split out of the former
|
||||
* use_metadata bundle; --preserve and -a/--archive imply it. */
|
||||
/* preserve_owner */
|
||||
/* -o/--owner: preserve the source owner (uid). Split out of the former
|
||||
* use_metadata bundle; --usermap/--chown (and, when a uid is requested,
|
||||
* --copy-as) imply it. Owner application still requires receiver privilege
|
||||
* and is gated separately by the identity flags. */
|
||||
/* preserve_group */
|
||||
/* -g/--group: preserve the source group (gid). Split out of the former
|
||||
* use_metadata bundle; --groupmap/--chown (and, when a gid is requested,
|
||||
* --copy-as) imply it. */
|
||||
/* fake_super */
|
||||
/* --fake-super: receiver-only. When set, each written file additionally gets
|
||||
* a reserved user.fastsync.stat xattr recording the source uid/gid/mode/mtime
|
||||
* so a later privileged restore could re-apply them. Crosses the wire. */
|
||||
* a reserved user.fastsync.stat xattr recording the RESOLVED uid/gid (the
|
||||
* source's own when no ownership request is active, else the --chown/--usermap
|
||||
* result) plus mode/mtime so a later privileged restore could re-apply them.
|
||||
* It NEVER real-chowns: the point is to record the source ownership on an
|
||||
* unprivileged receiver. Crosses the wire. */
|
||||
/* module */
|
||||
/* Daemon module selection (Wave A, protocol 2.15.0). Client-composed from a
|
||||
* host::module/path destination; NULL or "" means "no module" (the ordinary
|
||||
@@ -623,7 +772,7 @@ typedef struct Config {
|
||||
* fd-relative confinement (file_open_secure_parent, O_NOFOLLOW, root checks);
|
||||
* --super only permits an attempt that is already confined. Crosses the wire
|
||||
* as a trailing int so the receiver can enforce the policy. See
|
||||
* privilege_super_permitted() and identity_ownership_requested() in
|
||||
* privilege_super_permitted() and identity_explicit_ownership_requested() in
|
||||
* identity.h. */
|
||||
/* copy_as_set */
|
||||
/* --copy-as=USER[:GROUP] (P7 Wave E, protocol 2.18.0). Safe-subset
|
||||
@@ -803,12 +952,135 @@ typedef struct Config {
|
||||
* unknown status, or the unconsumed detail body, and the strict same-version
|
||||
* handshake (config_receive rejects a mismatched version before parsing
|
||||
* anything else) is what keeps a 2.21 client and a 2.20 server from ever
|
||||
* reaching that state. */
|
||||
#define PROTOCOL_VERSION "2.21.0"
|
||||
* reaching that state.
|
||||
*
|
||||
* Preserve-Attribute Split Wave: 2.21.0 -> 2.22.0.
|
||||
*
|
||||
* WHY the bump, grounded in the wire: this wave splits the former single
|
||||
* use_metadata bundle into four independent rsync-compatible preservation
|
||||
* attributes (preserve_perms / preserve_times / preserve_owner /
|
||||
* preserve_group) so -p/-t/-o/-g (and their --no-* negations) become real
|
||||
* drop-in flags. The binary config frame gains four serialized bools appended
|
||||
* to CONFIG_WIRE_METADATA_TIMES_FIELDS after omit_link_times, in this fixed
|
||||
* order: preserve_perms, preserve_times, preserve_owner, preserve_group. Any
|
||||
* config-frame layout change must bump the protocol version: a peer that does
|
||||
* not parse the new trailing bytes would desynchronize on the frame boundary,
|
||||
* and the strict same-version handshake (config_receive rejects a mismatched
|
||||
* version before parsing anything else) is what keeps a 2.22 client and a 2.21
|
||||
* server from ever reaching that state. The fixed-width FileMetadata layout is
|
||||
* UNCHANGED: the receiver still gates attribute application on use_metadata,
|
||||
* which is now DERIVED from these attributes by config_derived_use_metadata().
|
||||
*
|
||||
* Rsync-Parity Wave: 2.22.0 -> 2.23.0.
|
||||
*
|
||||
* WHY the bump, grounded in the wire. Several independent changes land in this
|
||||
* protocol version:
|
||||
*
|
||||
* (1) Ownership parity (#286/#294): each --usermap/--groupmap wire entry grows
|
||||
* from two int32s to [from][from_hi][to][to_name]; `from_hi` carries an
|
||||
* inclusive LOW-HIGH range (== from for a single/any/unnamed matcher) and the
|
||||
* trailing string carries a TO NAME for the receiver to resolve (rsync resolves
|
||||
* TO names on the receiving side). The STATUS_MKDIR and STATUS_DIR_TIMES frames
|
||||
* also gain a bounded per-entry xattr block when -X/-A is negotiated, so
|
||||
* directory xattrs/ACLs (including default ACLs) are preserved like regular-file
|
||||
* xattrs.
|
||||
*
|
||||
* (2) Delete semantics (#290): the delete-manifest frame gains a fourth trailing
|
||||
* section -- a synchronized-directory count followed by that many
|
||||
* destination-relative directory paths (the receive root is "."). The receiver
|
||||
* confines its extras walk to these directories, so `--files-from` with
|
||||
* `--delete` only removes inside listed directory subtrees (rsync parity)
|
||||
* instead of deleting every untransmitted path under the receive root. The
|
||||
* frame stream also gains STATUS_DELETE_LIMIT, the terminal success status sent
|
||||
* instead of STATUS_OK when a --max-delete commit removes up to the bound and
|
||||
* skips the rest (the sender then exits 25 like rsync).
|
||||
*
|
||||
* Any config-frame layout or frame-sequence change must bump the protocol
|
||||
* version: a 2.22 peer would desynchronize on the new entry bytes, the extra
|
||||
* trailing section or the unknown status, and the strict same-version handshake
|
||||
* (config_receive rejects a mismatched version before parsing anything else) is
|
||||
* what keeps a 2.23 client and a 2.22 server from ever reaching that state.
|
||||
*
|
||||
* (3) Output parity (#291/#292): -i/--itemize-changes and --out-format must
|
||||
* compare the source against the PRE-TRANSFER destination entry (new vs
|
||||
* modified, and which of size/time/perms/owner/group differ), but FastSync's
|
||||
* push sender never sees the destination. The receiver therefore answers a
|
||||
* per-file STATUS_CHECK with a new STATUS_DEST_INFO frame (a fixed-width
|
||||
* snapshot of the old entry) before its ordinary verdict when the config frame
|
||||
* carries the new report_dest_info bool appended after the --copy-as block.
|
||||
* This is both a config-frame layout change (one trailing bool) and a frame
|
||||
* sequence change (the new status).
|
||||
*
|
||||
* (4) Delete timing (protocol 2.24.0): the sender streams one delete plan per
|
||||
* source directory so --delete-during/--delete-delay reproduce rsync's deletion
|
||||
* timing (the plan fields and STATUS_DELETE_PLAN are documented at the keep-set
|
||||
* / delete-plan definitions below).
|
||||
*
|
||||
* (5) Wire-stats parity (protocol 2.25.0): --stats, --progress/-P and the
|
||||
* --out-format %b/%c tokens need receiver-only and wire counters that the push
|
||||
* sender cannot observe, and -n/--dry-run --delete must report the extras it
|
||||
* would have removed without deleting anything. The config frame gains one
|
||||
* trailing report_stats bool and the receiver emits a new STATUS_STATS frame
|
||||
* (carrying matched data, the deleted-file count and the would-delete path
|
||||
* list) immediately before its terminal success status.
|
||||
*
|
||||
* (6) Codec breadth + negotiation (protocol 2.26.0): the config frame gains one
|
||||
* trailing int, compression_algo (a CompressionAlgo id), appended after the
|
||||
* output block. It is the negotiated/effective compression codec and is what
|
||||
* the receiver's self-describing decompressor validates against its own
|
||||
* supported set. The checksum_algo wire value now also accepts md4/sha1/none,
|
||||
* and its default changes to the rsync 3.4.1 auto-negotiated xxh128.
|
||||
*
|
||||
* Any config-frame layout change must bump the protocol version: a peer that
|
||||
* does not parse the new trailing bytes would desynchronize on the frame
|
||||
* boundary, and the strict same-version handshake (config_receive rejects a
|
||||
* mismatched version before parsing anything else) keeps mixed deployments from
|
||||
* ever reaching that state. */
|
||||
/* (7) --info=del report (protocol 2.27.0): the config frame gains one trailing
|
||||
* bool, report_deletes, appended after report_stats. When set, the receiver
|
||||
* lists the paths it actually removed in the terminal STATUS_STATS path list
|
||||
* (the same count-delimited list the -n/--dry-run would-delete report uses), so
|
||||
* the sender can print rsync's `deleting PATH` lines for a real deletion. No
|
||||
* change to the fixed STATUS_STATS record itself; only a new trailing config
|
||||
* bool, which still requires the version bump for the strict lockstep. */
|
||||
/* (8) --stats receiver-observed counters (protocol 2.28.0): the fixed
|
||||
* STATUS_STATS record grows from three counters to eight. The receiver now
|
||||
* reports the bytes it literally stored (`literal_data`) and the count of
|
||||
* destination entries it newly CREATED, split by type
|
||||
* (reg/dir/link/special), so the sender can print rsync's exact
|
||||
* `Number of created files: N (reg: X, dir: Y, link: Z, special: W)` line and
|
||||
* an exact `Literal data` total even for delta transfers. The config-frame
|
||||
* LAYOUT is unchanged (no new config field), but the STATUS_STATS body grows,
|
||||
* so a 2.27 peer that does not consume the five new fixed-width counters would
|
||||
* desynchronize on the trailing would-delete path list; the strict
|
||||
* same-version handshake (config_receive rejects a mismatched version before
|
||||
* parsing anything else) keeps mixed deployments from ever reaching that
|
||||
* state. */
|
||||
/* (9) Receiver-side delete protection (still protocol 2.28.0): the config frame
|
||||
* gains one trailing self-describing block carrying the sender's compiled base
|
||||
* filter rules so the receiver can protect DESTINATION-ONLY entries from
|
||||
* --delete with `protect`/`risk` rules (rsync parity). The block appends after
|
||||
* compression_algo; see CONFIG_WIRE_PROTECT_FIELDS. */
|
||||
#define PROTOCOL_VERSION "2.28.0"
|
||||
#define DEFAULT_CHUNK_SIZE (10 * 1024 * 1024)
|
||||
/* Upper bound on total basis-dir entries (rsync caps --link-dest at 20). */
|
||||
#define MAX_BASIS_DIRS 64
|
||||
|
||||
/* Bounds on the received receiver-side delete-protection rule block. The rule
|
||||
* count and the aggregate pattern+owner bytes are each capped so a hostile
|
||||
* peer cannot pin unbounded pre-auth memory; both are validated strictly on
|
||||
* receive (alongside the per-string ConfigStringBudget). */
|
||||
/* A peer may supply protect rules; cap the list so a crafted config cannot make
|
||||
* the receiver's delete walk evaluate an unbounded number of glob patterns per
|
||||
* destination entry (glob_match is O(pattern x path)). 1024 is far above any
|
||||
* legitimate selection. */
|
||||
#define MAX_FILTER_RULES 1024
|
||||
#define MAX_FILTER_BYTES (256 * 1024)
|
||||
/* glob_match's DP is capped at 64 Mi work units; a pattern longer than this
|
||||
* could exceed the cap against a PATH_MAX path and silently stop matching,
|
||||
* leaving a protect rule inert. Reject such a rule at receive time. */
|
||||
#define MAX_PROTECT_PATTERN_LEN 8192
|
||||
|
||||
/* Upper bound on the number of --skip-compress suffixes accepted from the wire.
|
||||
* Each suffix is an independent wire string (up to MAX_STRING_SIZE = 64 KiB), so
|
||||
* without this a hostile pre-auth client could otherwise retain
|
||||
@@ -829,9 +1101,11 @@ typedef struct Config {
|
||||
|
||||
/* Identity-mapping sentinels and bounds (see identity.h for semantics).
|
||||
* IDENTITY_MATCH_ANY is a usermap/groupmap FROM '*' (matches any id);
|
||||
* IDENTITY_CURRENT is a chown / map TO '*' (resolve to the receiver's current
|
||||
* euid/egid at apply time). */
|
||||
* IDENTITY_MATCH_UNNAMED is a FROM with an empty token (rsync's "ids with no
|
||||
* name on the sender"); IDENTITY_CURRENT is a chown / map TO '*' (resolve to
|
||||
* the receiver's current euid/egid at apply time). */
|
||||
#define IDENTITY_MATCH_ANY (-1)
|
||||
#define IDENTITY_MATCH_UNNAMED (-2)
|
||||
#define IDENTITY_CURRENT (-1)
|
||||
#define MAX_IDENTITY_MAP 128
|
||||
|
||||
@@ -896,16 +1170,23 @@ int config_parse_daemon_dest(Config* config);
|
||||
* 0. */
|
||||
int config_parse_transport_dest(Config* config);
|
||||
|
||||
/* True when the negotiated delete timing performs the extra-file deletion
|
||||
* BEFORE the transfer data (--delete-before / --delete-during). The flag is
|
||||
* a pure function of the config and is used identically on the sender (to pick
|
||||
/* True for the whole-tree delete-before timing: a complete keep-set manifest is
|
||||
* transmitted before any data and committed (with an ack) before the first data
|
||||
* byte. Pure function of the config, used identically on the sender (to pick
|
||||
* the manifest-first frame order) and the receiver (to delete when the early
|
||||
* manifest arrives). When false the deletion is committed only after the whole
|
||||
* transfer succeeded (--delete / --delete-after / --delete-delay). */
|
||||
* manifest arrives). */
|
||||
bool config_delete_timing_early(const Config* config);
|
||||
/* True for the per-directory timings (--delete-during / --delete-delay). The
|
||||
* sender streams a delete plan per source directory in directory order; the
|
||||
* receiver applies each plan on arrival (during) or snapshots its extras and
|
||||
* commits them only after a fully-successful transfer (delay). */
|
||||
bool config_delete_timing_per_dir(const Config* config);
|
||||
/* Delete-timing sanity: with deletion enabled at most one timing flag may be
|
||||
* set (none = the default delete-after commit timing); without deletion no
|
||||
* timing flag may be set (each timing flag implies --delete). */
|
||||
* set; without deletion no timing flag may be set (each timing flag implies
|
||||
* --delete). A plain --delete is normalized to delete_during by
|
||||
* cli_finalize_config on the client, so a transmitted use_delete config always
|
||||
* carries exactly one timing; the zero-timing case remains valid only for a
|
||||
* config that has not been through the CLI. */
|
||||
bool config_has_valid_delete_timing(const Config* config);
|
||||
|
||||
/* Single source of truth for the cross-field ("combination") invariants a
|
||||
@@ -918,6 +1199,14 @@ bool config_has_valid_delete_timing(const Config* config);
|
||||
* validate_received_config() so the receiver enforces exactly the same
|
||||
* invariants it relies on (the server is the trust boundary). */
|
||||
const char* config_invariants_error(const Config* config);
|
||||
/* Single source of truth for the DERIVED transport bit (use_metadata): true
|
||||
* when any configured preservation/ownership option requires the metadata
|
||||
* frame to travel. Returns false when no such option is set (a bare run).
|
||||
* This is a pure predicate over the config; the client lowers it into
|
||||
* Config->use_metadata at the end of parsing so every implication (devices,
|
||||
* executability, identity maps, incremental/delta, ...) is centralized here
|
||||
* rather than scattered as direct writes. */
|
||||
bool config_derived_use_metadata(const Config* config);
|
||||
/* True when at least one --compare-dest/--copy-dest/--link-dest was set. */
|
||||
bool config_has_basis(const Config* config);
|
||||
/* Append one basis-dir entry. Returns 0 on success, -1 on allocation failure. */
|
||||
|
||||
@@ -264,6 +264,20 @@ static bool delay_publish_entry(DelayUpdatesContext* context, const Config* conf
|
||||
const StagedFileEntry* entry) {
|
||||
if (!delay_publish_backup(context, config, entry))
|
||||
return false;
|
||||
/* --force: an incoming regular file/symlink may replace a destination
|
||||
DIRECTORY (possibly non-empty). The immediate-install path handles this in
|
||||
file_receive; a --delay-updates run stages elsewhere and only discovers the
|
||||
blocking directory here, so clear it before the rename (rsync's
|
||||
"could not make way for new regular file" without --force). */
|
||||
if (config && config->force_delete && file_directory_exists_secure(entry->final_path)) {
|
||||
if (!file_remove_tree_secure(entry->final_path)) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not remove destination directory blocking '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!file_rename_secure(entry->staged_path, entry->final_path)) {
|
||||
if (errno == EXDEV) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,105 @@
|
||||
#ifndef DELETE_PLAN_H
|
||||
#define DELETE_PLAN_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Per-directory delete plans (protocol 2.24.0).
|
||||
*
|
||||
* rsync's --delete-during removes a directory's extras while the generator
|
||||
* processes that directory, and --delete-delay records the deletion list during
|
||||
* the scan but applies it only after a fully-successful transfer. FastSync has
|
||||
* no per-directory generator pass; instead the sender streams one plan per
|
||||
* source directory, in directory order, and the receiver applies it when it
|
||||
* arrives (during) or snapshots its extras and commits them at the end (delay).
|
||||
*
|
||||
* The sender side builds a plan set from the path-only pre-scan (it needs every
|
||||
* directory's complete direct-child list before the first data byte of that
|
||||
* directory). The receiver side is a session that carries the global protected
|
||||
* prefixes (filter-excluded and size-skipped source mirrors), the
|
||||
* --delete-missing-args exact deletions, the shared --max-delete budget and,
|
||||
* for --delete-delay, the snapshotted extras. */
|
||||
|
||||
/* ---- Sender: plan builder ---- */
|
||||
|
||||
typedef struct DeletePlanSender DeletePlanSender;
|
||||
|
||||
DeletePlanSender* delete_plan_sender_create(void);
|
||||
void delete_plan_sender_destroy(DeletePlanSender* sender);
|
||||
/* Record one transmitted entry. `path` is the destination-relative wire path;
|
||||
* is_dir marks an explicit directory entry (--dirs, a -x mount point). */
|
||||
bool delete_plan_sender_add(DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Drop plans for directories outside `synced_dirs` (the --files-from
|
||||
* synchronization scope; pass NULL when a full recursive transfer synchronized
|
||||
* every directory). The receive root is the "." sentinel.
|
||||
*
|
||||
* `walk_root` scopes a general -R transfer: when non-NULL it is the
|
||||
* reconstructed destination prefix the run actually transferred, and only the
|
||||
* plan for that prefix (and directories below it) is ever transmitted, so the
|
||||
* prefix's parent-directory siblings are never walked. Pass NULL for a plain
|
||||
* recursive transfer and for --files-from. */
|
||||
void delete_plan_sender_finalize(DeletePlanSender* sender, const ArrayList* synced_dirs,
|
||||
const char* walk_root);
|
||||
/* True when no transmitted FILE entry was recorded (an ambiguous empty scan).
|
||||
Directory keep entries do not count, so an I/O error that hid every file
|
||||
still refuses to delete. */
|
||||
bool delete_plan_sender_empty(const DeletePlanSender* sender);
|
||||
/* Attach the global config sections advertised on the first plan frame. The
|
||||
* block is always transmitted by delete_plan_send_root(), on a config-only
|
||||
* carrier frame when the scope allows no directory plan. */
|
||||
void delete_plan_sender_set_config(DeletePlanSender* sender, const ArrayList* protected_prefixes,
|
||||
const ArrayList* size_skipped, const ArrayList* missing_args);
|
||||
/* Send the root plan (even before any data, so root extras are handled like
|
||||
* rsync's first generator directory), after transmitting the per-run config
|
||||
* block on its own carrier frame. Returns -1 on I/O error. */
|
||||
int delete_plan_send_root(int fd, DeletePlanSender* sender);
|
||||
/* Send the plans for every ancestor of `path` (root-first) and, when is_dir,
|
||||
* for `path` itself; already-sent plans are skipped. */
|
||||
int delete_plan_send_for_path(int fd, DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Send the plan for every directory in `dirs` that has not been transmitted
|
||||
* yet. */
|
||||
int delete_plan_send_remaining(int fd, DeletePlanSender* sender, const ArrayList* dirs);
|
||||
/* Transmit the COMPLETE per-directory plan set in one pass, before any data
|
||||
* frame: the root plan (with the one-shot per-run config block on its carrier
|
||||
* frame) followed by every directory in `dirs`. Because the whole plan set is
|
||||
* known from the path-only pre-scan, sending it all up front means a
|
||||
* mid-transfer abort has already applied every planned removal, matching
|
||||
* rsync's generator (which runs ahead of its throttled sender). A completed
|
||||
* run is unaffected. `dirs` is the set of directories whose direct children
|
||||
* were enumerated (the scanner's plan_dirs sink), so a merely listed but
|
||||
* untraversed directory never gets a plan and its mirror is left intact.
|
||||
* Returns -1 on I/O error. */
|
||||
int delete_plan_send_all(int fd, DeletePlanSender* sender, const ArrayList* dirs);
|
||||
|
||||
/* ---- Receiver: delete session ---- */
|
||||
|
||||
typedef struct DeletePlanSession DeletePlanSession;
|
||||
|
||||
DeletePlanSession* delete_plan_session_create(const Config* config);
|
||||
void delete_plan_session_destroy(DeletePlanSession* session);
|
||||
/* Read one STATUS_DELETE_PLAN frame (the leading status already consumed) and
|
||||
* act on it. Returns 0 on success (including a dry-run/disabled no-op) and -1
|
||||
* after signalling STATUS_ERROR on a malformed frame or a deletion failure. */
|
||||
int delete_plan_session_receive(DeletePlanSession* session, const Config* config, int fd);
|
||||
/* Apply the deferred snapshot (--delete-delay) and the missing-args deletions.
|
||||
* Safe to call once; returns the commit outcome. */
|
||||
DeleteCommitResult delete_plan_session_commit(DeletePlanSession* session, const Config* config);
|
||||
/* True once the shared --max-delete budget stopped part of a deletion. */
|
||||
bool delete_plan_session_limit_reached(const DeletePlanSession* session);
|
||||
/* Number of destination entries the session actually removed, for the
|
||||
end-of-transfer stats. For --delete-delay this excludes a snapshotted entry
|
||||
that survived (e.g. a refilled directory that failed ENOTEMPTY), even though
|
||||
that entry already consumed --max-delete budget at snapshot time. */
|
||||
size_t delete_plan_session_deleted(const DeletePlanSession* session);
|
||||
/* Install an observer invoked for every destination-relative path the session
|
||||
truly removes (including the deferred --delete-delay commit), so the receiver
|
||||
can report rsync's `deleting PATH` lines through the terminal STATUS_STATS
|
||||
record. Pass NULL/0 to clear. */
|
||||
void delete_plan_session_set_delete_observer(DeletePlanSession* session,
|
||||
DeletePathObserver observer, void* context);
|
||||
|
||||
#endif
|
||||
+663
-122
File diff suppressed because it is too large
Load Diff
+87
-28
@@ -28,17 +28,36 @@ void file_metadata_destroy(void* metadata);
|
||||
/* --open-noatime process-wide sender policy; see file.c. */
|
||||
void file_set_open_noatime(bool enable);
|
||||
bool file_get_open_noatime(void);
|
||||
/* Capture the process umask ONCE, before any threads are created. Call this at
|
||||
* the very top of main() in both entry points so the cached value is read while
|
||||
* the process is still single-threaded: reading the umask needs a get+set round
|
||||
* trip (umask(0); umask(old)), which would race against receiver threads
|
||||
* creating files if it happened during the first write. Idempotent and safe to
|
||||
* call more than once. */
|
||||
void file_umask_capture(void);
|
||||
/* Process-wide umask, captured once (thread-safe). Used to derive the mode of
|
||||
* a brand-new destination like rsync: source_mode & 0777 & ~umask. Falls back
|
||||
* to file_umask_capture() (behind pthread_once) if capture was never called. */
|
||||
unsigned file_process_umask(void);
|
||||
/* Open `path` read-only for transfer, honouring --open-noatime when set. */
|
||||
int file_open_for_read(const char* path);
|
||||
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse);
|
||||
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave). --munge-links
|
||||
* sender-side marker: every transmitted symlink target is prefixed with this
|
||||
* while the flag is on; the receiver strips it to restore the real target. */
|
||||
#define SYMLINK_MUNGE_PREFIX "#SYMLINK/"
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave; rsync parity).
|
||||
* --munge-links is a RECEIVER-side rewrite: rsync prefixes every stored symlink
|
||||
* target with this marker, making the link unusable while the referenced
|
||||
* directory does not exist. A SENDER receiving a munged source strips it back
|
||||
* off before transmitting (so a munged tree round-trips through the receiver's
|
||||
* re-munging). */
|
||||
#define SYMLINK_MUNGE_PREFIX "/rsyncd-munged/"
|
||||
|
||||
char* file_symlink_munge(const char* target);
|
||||
/* rsync 3.4.1 unsafe_symlink(): true when `target` escapes the transfer tree
|
||||
* rooted at `link_path` (the symlink's transfer-relative path incl. its name).
|
||||
* Absolute/empty targets and targets climbing above the transfer root (via
|
||||
* "..") are unsafe, as are internal "/../" components and trailing "/..". */
|
||||
bool file_symlink_unsafe(const char* target, const char* link_path);
|
||||
/* True when a lexical target is relative and contains no ".." component, so it
|
||||
* can never escape the receive root once created beneath it. */
|
||||
bool file_symlink_target_contained(const char* target);
|
||||
@@ -46,8 +65,9 @@ bool file_symlink_target_contained(const char* target);
|
||||
* returns true when a marker was removed. */
|
||||
bool file_symlink_unmunge(char* target);
|
||||
/* Create a symlink at `path` -> `target`, confined below the authorized root
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). Returns
|
||||
* false when a directory already occupies `path`. */
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). The link
|
||||
* value is copied verbatim (rsync -l); only the placement path is confined.
|
||||
* Returns false when a directory already occupies `path`. */
|
||||
bool file_symlink_at_secure(const char* path, const char* target);
|
||||
/* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing
|
||||
* symlink-to-directory to be followed as a directory. */
|
||||
@@ -67,6 +87,11 @@ bool file_path_exists_secure(const char* path);
|
||||
bool file_stat_secure(const char* path, struct stat* st);
|
||||
bool file_destination_is_newer_secure(const char* path, const FileMetadata* metadata);
|
||||
int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs);
|
||||
/* Protocol 2.28.0 variant: also increments *dirs_created for every missing
|
||||
* parent directory this walk creates that lies strictly below `count_floor`
|
||||
* (a receive-root-relative path, or NULL to count all of them). */
|
||||
int file_open_secure_parent_counted(const char* path, char** leaf_out, bool create_dirs,
|
||||
unsigned* dirs_created, const char* count_floor);
|
||||
bool file_ensure_directory_secure(const char* path);
|
||||
bool file_directory_exists_secure(const char* path);
|
||||
bool file_rename_secure(const char* old_path, const char* new_path);
|
||||
@@ -75,37 +100,40 @@ bool file_rename_secure(const char* old_path, const char* new_path);
|
||||
regular file. See the .c for the exact success semantics. */
|
||||
bool file_remove_tree_secure(const char* path);
|
||||
/* Open a private 0700 directory (creating it on demand) that must live below
|
||||
the authorized root. Used for the --temp-dir scratch directory and the
|
||||
--delay-updates staging directory. */
|
||||
the authorized root. Used for the --delay-updates staging directory. */
|
||||
int file_open_private_dir(const char* dir_path);
|
||||
|
||||
/* Open an existing --temp-dir scratch directory as-is (absolute or relative;
|
||||
no creation, no root confinement), matching rsync's --temp-dir handling. */
|
||||
int file_open_temp_dir(const char* dir_path);
|
||||
|
||||
/* The file_to_disk_secure* variants write a temporary copy in the destination
|
||||
directory and atomically rename it over `path`. temp_dir is an absolute,
|
||||
root-confined scratch directory (already validated by the caller): when it
|
||||
is non-NULL the temporary copy is instead created there (with a name unique
|
||||
across the whole scratch directory) and atomically renamed into the
|
||||
destination directory once fully written and fsynced. A rename across
|
||||
filesystems (EXDEV) fails the write with an error; the file is never
|
||||
silently copied into place. Pass NULL for the historical same-directory
|
||||
behavior. --inplace writes never use temp_dir. */
|
||||
directory and atomically rename it over `path`. temp_dir is a scratch
|
||||
directory (an absolute path, or one the caller already resolved against the
|
||||
destination root): when it is non-NULL the temporary copy is instead created
|
||||
there (with a name unique across the whole scratch directory) and atomically
|
||||
renamed into the destination directory once fully written and fsynced. When
|
||||
that rename/link fails with EXDEV (the scratch dir is on another filesystem)
|
||||
the write falls back to a non-atomic copy directly in the destination
|
||||
directory, matching rsync. Pass NULL for the same-directory behavior.
|
||||
--inplace writes never use temp_dir. */
|
||||
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, const char* temp_dir);
|
||||
FileAttrPolicy policy, const char* temp_dir);
|
||||
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
FileAttrPolicy policy, bool use_fsync, const char* temp_dir);
|
||||
/* With update enabled, an existing newer destination is left untouched. The
|
||||
check is descriptor-based for inplace writes; atomic replacement still has
|
||||
an unavoidable final rename race without filesystem locking. */
|
||||
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
unsigned long long data_size, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
/* Receiver write-path variant that also applies per-file xattrs (-X/-A) and the
|
||||
* --fake-super stat xattr fd-relative before the final rename. `update` /
|
||||
@@ -113,10 +141,9 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
* enables --partial best-effort retention of a failed write's temp. */
|
||||
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool update, bool no_replace, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super, bool keep_partial,
|
||||
const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
|
||||
bool fake_super, bool keep_partial, const char* temp_dir);
|
||||
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
|
||||
(via a temp name + rename); fall back to a byte-identical local copy from
|
||||
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
|
||||
@@ -125,16 +152,48 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
|
||||
never re-allocated). */
|
||||
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool use_fsync, const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
/* Like file_to_disk_secure_link, but the byte-copy fallback also applies the
|
||||
* per-file xattrs (-X/-A) and --fake-super stat xattr (fd-relative). On a
|
||||
* successful hard link no attributes are applied (the shared inode already
|
||||
* carries the basis's). */
|
||||
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
/* Streaming --copy-dest install: atomically materialize `path` by copying the
|
||||
* bytes of `basis_path` through a bounded buffer (no whole-file buffering, so
|
||||
* an arbitrarily large basis works), applying the SOURCE metadata and the
|
||||
* per-file xattrs / --fake-super record. `update` honors a newer destination;
|
||||
* a --temp-dir scratch location falls back to a direct write on EXDEV. */
|
||||
bool file_copy_basis_stream_attrs(const char* path, const char* basis_path,
|
||||
unsigned long long expected_size, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
/* Protocol 2.28.0 receiver-stat variants: like the two above but additionally
|
||||
* report through `dirs_created` (when non-NULL) how many parent directories the
|
||||
* confined secure walk had to create that lie strictly below `count_floor` (a
|
||||
* receive-root-relative prefix, or NULL for all). Used to reproduce rsync's
|
||||
* `Number of created files` directory count on a fresh destination. */
|
||||
bool file_to_disk_secure_attrs_counted(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
FileAttrPolicy policy, bool update, bool no_replace,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
bool keep_partial, const char* temp_dir,
|
||||
unsigned* dirs_created, const char* count_floor);
|
||||
bool file_to_disk_secure_link_attrs_counted(const char* path, const char* basis_path,
|
||||
const void* data, unsigned long long data_size,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
FileAttrPolicy policy, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir, unsigned* dirs_created,
|
||||
const char* count_floor);
|
||||
/* The logical transfer root expressed receive-root-relative, or NULL when the
|
||||
* wire paths carry no mirror scaffolding above it. Caller frees non-NULL. */
|
||||
char* file_transfer_root_floor(const Config* config);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
#ifndef FILE_ATTR_H
|
||||
#define FILE_ATTR_H
|
||||
|
||||
#include "config.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/*
|
||||
* Per-attribute receiver policy for applying a transmitted FileMetadata. This
|
||||
* is the split-out replacement for the former single use_metadata bundle: each
|
||||
* flag is applied independently, matching rsync's -p/-t/-o/-g/-E/-U semantics.
|
||||
* `use_metadata` remains the transport/presence gate (whether the metadata frame
|
||||
* travelled at all); this struct decides which attributes are ACTUALLY applied.
|
||||
*
|
||||
* It lives in its own header (rather than metadata.h) because xattr.h's
|
||||
* fake_super_restore_fd() takes one and metadata.h <-> file_types.h form an
|
||||
* include cycle that must not be entered from xattr.h.
|
||||
*
|
||||
* The mode leg is: perms wins over executability; an exec-bits-only change is
|
||||
* made only when perms is off; when neither is set the receiver deliberately
|
||||
* sets no source mode. file.c then substitutes the pre-existing destination
|
||||
* mode for a brand-new destination with metadata it uses the sanitized
|
||||
* source-mode-&-umask base (S_IWGRP|S_IWOTH cleared), and the fixed 0644
|
||||
* default only when no metadata is available at all, so a no--p overwrite
|
||||
* does not lose the destination's perms.
|
||||
*/
|
||||
typedef struct FileAttrPolicy {
|
||||
bool perms; /* config->preserve_perms: apply the source mode bits */
|
||||
bool times; /* config->preserve_times: apply the source mtime */
|
||||
bool atimes; /* config->preserve_atimes (-U): apply the source atime */
|
||||
bool executability; /* config->use_executability (-E): exec-bits-only mode */
|
||||
} FileAttrPolicy;
|
||||
|
||||
/* Build the per-attribute policy from a connection's Config. A NULL config
|
||||
* yields the all-off policy (no attribute application). */
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config);
|
||||
|
||||
#endif
|
||||
@@ -240,3 +240,29 @@ bool file_list_affects(const FileListSet* set, const char* rel) {
|
||||
entry (binary search for the first entry at or after `rel` + '/'). */
|
||||
return path_index_has_descendant(&set->index, rel);
|
||||
}
|
||||
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel) {
|
||||
if (!set || set->whole_tree)
|
||||
return true;
|
||||
if (!rel || rel[0] == '\0')
|
||||
return false;
|
||||
/* `rel` itself is listed, or one of its ancestor prefixes is an exact listed
|
||||
directory (a listed prefix of a directory path is necessarily a
|
||||
directory). */
|
||||
size_t len = strlen(rel);
|
||||
while (len > 0) {
|
||||
const char* slash = NULL;
|
||||
for (size_t i = len; i-- > 0;) {
|
||||
if (rel[i] == '/') {
|
||||
slash = rel + i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!slash)
|
||||
break;
|
||||
len = (size_t)(slash - rel);
|
||||
if (path_index_contains_n(&set->index, rel, len))
|
||||
return true;
|
||||
}
|
||||
return path_index_contains(&set->index, rel);
|
||||
}
|
||||
|
||||
@@ -40,4 +40,14 @@ void file_list_destroy(FileListSet* set);
|
||||
* this returns true, files are transferred only when it returns true. */
|
||||
bool file_list_affects(const FileListSet* set, const char* rel);
|
||||
|
||||
/* True when the DIRECTORY `rel` (path relative to the source root) is inside a
|
||||
* listed directory subtree: `rel` itself is a listed entry, or one of `rel`'s
|
||||
* ancestor directory prefixes is an exact listed entry. Unlike
|
||||
* file_list_affects this does NOT treat an ancestor of a listed entry as
|
||||
* affected, so an implied parent directory of a listed file is not synchronized
|
||||
* (rsync deletes nothing in it). With no set or a whole-tree set every
|
||||
* directory is in scope. This is the delete-walker's "synchronized directory"
|
||||
* predicate. */
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel);
|
||||
|
||||
#endif
|
||||
|
||||
+1151
-398
File diff suppressed because it is too large
Load Diff
+110
-21
@@ -3,6 +3,7 @@
|
||||
|
||||
#include "config.h"
|
||||
#include "file_types.h"
|
||||
#include "utils.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Server-side file receive/save path. */
|
||||
@@ -23,6 +24,16 @@ File* file_receive_hardlink(int file_descriptor);
|
||||
File* file_receive_symlink(int file_descriptor, const Config* config);
|
||||
File* file_receive_special(int file_descriptor);
|
||||
bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode);
|
||||
/* Testable basis quick-check / verification policy. file_basis_quick_match is
|
||||
* rsync's metadata quick-check for a basis candidate (equal size is required
|
||||
* separately by the caller; this adds the --size-only / mtime / --modify-window
|
||||
* leg). file_basis_content_required reports whether a hit must ALSO be
|
||||
* confirmed by a whole-file content digest (--verify-basis; false is the
|
||||
* default rsync-parity behavior). */
|
||||
bool file_basis_quick_match(const Config* config, const struct stat* st, time_t check_mtime,
|
||||
long check_mtime_nsec);
|
||||
bool file_basis_content_required(const Config* config);
|
||||
|
||||
File* receive_incremental_check(int fd, const Config* config, bool* skipped);
|
||||
/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set
|
||||
* true only on the server-contacting --dry-run path when the file is not up to
|
||||
@@ -40,31 +51,39 @@ File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped,
|
||||
* parent's mtime). -O/--omit-dir-times skips the application entirely. The
|
||||
* list owns deep copies of the paths and metadata; freed on every path. */
|
||||
typedef struct {
|
||||
char** paths; /* owned, destination-relative wire paths */
|
||||
FileMetadata* entries; /* owned, parallel to paths */
|
||||
char** paths; /* owned, destination-relative wire paths */
|
||||
FileMetadata* entries; /* owned, parallel to paths */
|
||||
FileXattrList** xattrs; /* owned, parallel to paths; NULL when none */
|
||||
size_t count;
|
||||
size_t capacity;
|
||||
size_t bytes; /* cumulative strlen of every retained path */
|
||||
} DirTimeList;
|
||||
|
||||
/* Capture gate shared by the sender-side and receiver-side sinks: directory
|
||||
* metadata is accumulated only when --times/--metadata is in effect and
|
||||
* -O/--omit-dir-times does not suppress it. Kept here, next to the accumulator
|
||||
* it guards, so both call sites express the same condition. */
|
||||
bool dir_times_should_capture(const Config* config);
|
||||
* metadata is accumulated only when a directory attribute is requested
|
||||
* (-p/--perms for directory modes, or -t/--times for directory mtimes with
|
||||
* -O/--omit-dir-times not suppressing them) and metadata rides the wire. Kept
|
||||
* here, next to the accumulator it guards, so both call sites express the same
|
||||
* condition. */
|
||||
bool dir_metadata_should_capture(const Config* config);
|
||||
|
||||
void dir_time_list_init(DirTimeList* list);
|
||||
void dir_time_list_free(DirTimeList* list);
|
||||
/* Deep-copy one directory's path + metadata into the list. Returns false on
|
||||
* allocation failure OR when the cumulative entry/byte caps would be exceeded
|
||||
* (the caller fails the transfer). */
|
||||
bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata);
|
||||
/* Apply every accumulated directory's mtime (and atime when captured) beneath
|
||||
* `root_directory`, confined fd-relative. Best-effort per entry: an absent
|
||||
* directory (an empty/pruned source dir that was deliberately not created) or a
|
||||
* non-directory at the path is skipped QUIETLY, an unreachable one with a
|
||||
* warning, and never fatal. */
|
||||
void dir_time_list_apply(const DirTimeList* list, const char* root_directory);
|
||||
/* Deep-copy one directory's path + metadata (and, when non-NULL, its captured
|
||||
* xattr/ACL block) into the list. Returns false on allocation failure OR when
|
||||
* the cumulative entry/byte caps would be exceeded (the caller fails the
|
||||
* transfer). */
|
||||
bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata,
|
||||
const FileXattrList* xattrs);
|
||||
/* Apply every accumulated directory's metadata beneath `root_directory`,
|
||||
* confined fd-relative: ownership through the negotiated identity policy,
|
||||
* times (mtime, plus atime when -U captured one under -t), the mode (through
|
||||
* --chmod when configured, under -p), and the captured xattrs/ACLs (under
|
||||
* -X/-A). Best-effort per entry: an absent directory (an empty/pruned source
|
||||
* dir that was deliberately not created) or a non-directory at the path is
|
||||
* skipped QUIETLY, an unreachable one with a warning, and never fatal. */
|
||||
void dir_metadata_list_apply(const DirTimeList* list, const char* root_directory,
|
||||
const Config* config);
|
||||
|
||||
/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative
|
||||
paths the sender transferred/keeps) plus `protected`, destination-relative
|
||||
@@ -80,11 +99,18 @@ typedef struct DeleteManifest {
|
||||
ArrayList* keeps;
|
||||
ArrayList* protected;
|
||||
ArrayList* missing;
|
||||
/* Destination-relative paths of the directories the sender synchronized for
|
||||
this run. The extras walker only removes entries directly inside one of
|
||||
these (the receive root is the "." sentinel); `--files-from` runs therefore
|
||||
leave untransmitted directories and the unlisted parts of listed ones
|
||||
alone, matching rsync's "delete only in synchronized directories". */
|
||||
ArrayList* dirs;
|
||||
} DeleteManifest;
|
||||
|
||||
void delete_manifest_free(DeleteManifest* manifest);
|
||||
/* Read a delete-manifest frame: keep count + keeps, then protected count +
|
||||
protected prefixes, then missing count + missing paths (self-delimiting; the
|
||||
/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then
|
||||
protected count + protected prefixes, then missing count + missing paths,
|
||||
then synchronized-directory count + directory paths (self-delimiting; the
|
||||
leading STATUS_MANIFEST code has been consumed). Returns an owned
|
||||
DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */
|
||||
DeleteManifest* receive_manifest_entries(int fd);
|
||||
@@ -103,11 +129,58 @@ bool manifest_delete_extras(const Config* config, DeleteManifest* manifest);
|
||||
confinement or I/O error (the run then fails); tolerated per-path cases are
|
||||
reported and skipped. */
|
||||
bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest);
|
||||
/* Budgeted form of manifest_delete_missing_args for the per-directory delete
|
||||
session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited)
|
||||
and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set
|
||||
when the budget stopped the pass with entries left over. Returns false only
|
||||
on a genuine deletion error. */
|
||||
bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest,
|
||||
size_t max_delete, size_t* deleted, size_t* skipped,
|
||||
bool* limit_hit);
|
||||
/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may
|
||||
be NULL) is invoked for every destination-relative path truly removed. */
|
||||
bool manifest_delete_missing_args_limited_observed(const Config* config, DeleteManifest* manifest,
|
||||
size_t max_delete, size_t* deleted,
|
||||
size_t* skipped, bool* limit_hit,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context);
|
||||
/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's
|
||||
partial --max-delete result: the budget allowed some deletions and the rest
|
||||
were skipped (the run still stores all file data but the client exits 25). */
|
||||
typedef enum {
|
||||
DELETE_COMMIT_OK = 0,
|
||||
DELETE_COMMIT_LIMIT_REACHED,
|
||||
DELETE_COMMIT_ERROR
|
||||
} DeleteCommitResult;
|
||||
|
||||
/* Run every deletion family the manifest carries: the --delete-missing-args
|
||||
exact-path deletions first (user requests are not blocked by exclusion
|
||||
protection), then the ordinary extras walk when --delete is active. Returns
|
||||
true when nothing to do or everything committed. */
|
||||
bool manifest_delete_all(const Config* config, DeleteManifest* manifest);
|
||||
protection), then the ordinary extras walk when --delete is active. Both
|
||||
share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to
|
||||
do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget
|
||||
stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */
|
||||
DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest);
|
||||
/* Like manifest_delete_all, but reports how many destination entries the commit
|
||||
removed (for the end-of-transfer wire stats). `deleted` may be NULL. */
|
||||
DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest,
|
||||
size_t* deleted);
|
||||
/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL)
|
||||
is invoked for every destination-relative path truly removed. */
|
||||
DeleteCommitResult manifest_delete_all_observed(const Config* config, DeleteManifest* manifest,
|
||||
size_t* deleted, DeletePathObserver observer,
|
||||
void* observer_context);
|
||||
|
||||
/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as
|
||||
the delete pass would and append (strdup'd) destination-relative paths that
|
||||
WOULD be removed to `out`, without touching disk. Uses the same staging-dir,
|
||||
basis-dir and protected-prefix skips as the real commit. Returns true on a
|
||||
clean walk; `*count_out` receives the number of paths appended. */
|
||||
bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out,
|
||||
size_t* count_out);
|
||||
/* Convert one basis-directory path to the receive-root-relative protection
|
||||
prefix the delete walker uses (NULL when it lies outside the root). Exposed
|
||||
for unit tests of the root-of-"/" and normalization edge cases. */
|
||||
char* file_receive_basis_delete_relative(const Config* config, const char* path);
|
||||
|
||||
/* Outcome of a single file_save_to_disk operation. The receiver needs to
|
||||
distinguish "written" from "skipped" so --remove-source-files can be told
|
||||
@@ -116,6 +189,22 @@ typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2
|
||||
|
||||
FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file,
|
||||
const Config* config);
|
||||
/* Protocol 2.28.0 variant: also reports through `created` (when non-NULL)
|
||||
* whether the destination entry did not exist before this save, and through
|
||||
* `created_dirs` how many parent directories the confined walk created, so the
|
||||
* receiver can build rsync's `Number of created files` breakdown. The plain
|
||||
* file_save_to_disk_full() is this with both out-params NULL. */
|
||||
FileSaveResult file_save_to_disk_full_ex(const char* root_directory, const File* file,
|
||||
const Config* config, bool* created,
|
||||
unsigned* created_dirs);
|
||||
bool file_save_to_disk(const char* root_directory, const File* file, const Config* config);
|
||||
|
||||
/* Protocol 2.28.0 receiver counter accumulator: fold one successfully saved
|
||||
* entry into `stats`, adding its receiver-observed literal bytes and, when
|
||||
* `created`, the matching created-by-type counter (regular file / symlink /
|
||||
* special) plus `created_dirs` implicitly-created parent directories.
|
||||
* Non-first hardlink siblings contribute no literal bytes. */
|
||||
void receiver_stats_note_saved(ReceiverStats* stats, const File* file, bool created,
|
||||
unsigned created_dirs);
|
||||
|
||||
#endif
|
||||
|
||||
+19
-10
@@ -143,20 +143,28 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
|
||||
}
|
||||
|
||||
off_t offset = 0;
|
||||
/* A non-positive --timeout disables the deadline: poll blocks until the
|
||||
* socket is writable (rsync's --timeout=0 default). */
|
||||
int io_timeout_sec = protocol_get_io_timeout_sec();
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += protocol_get_io_timeout_sec();
|
||||
if (io_timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += io_timeout_sec;
|
||||
}
|
||||
while ((unsigned long long)offset < file_size) {
|
||||
struct timespec now;
|
||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||
long long remaining = (long long)(deadline.tv_sec - now.tv_sec) * 1000LL +
|
||||
(deadline.tv_nsec - now.tv_nsec) / 1000000LL;
|
||||
if (remaining <= 0) {
|
||||
close(fd);
|
||||
return false;
|
||||
int timeout = -1;
|
||||
if (io_timeout_sec > 0) {
|
||||
struct timespec now;
|
||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||
long long remaining = (long long)(deadline.tv_sec - now.tv_sec) * 1000LL +
|
||||
(deadline.tv_nsec - now.tv_nsec) / 1000000LL;
|
||||
if (remaining <= 0) {
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
timeout = remaining > INT_MAX ? INT_MAX : (int)remaining;
|
||||
}
|
||||
struct pollfd pfd = {.fd = file_descriptor, .events = POLLOUT};
|
||||
int timeout = remaining > INT_MAX ? INT_MAX : (int)remaining;
|
||||
int polled = poll(&pfd, 1, timeout);
|
||||
if (polled <= 0 || (pfd.revents & (POLLERR | POLLHUP | POLLNVAL))) {
|
||||
close(fd);
|
||||
@@ -174,6 +182,7 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
protocol_note_bytes_written((unsigned long long)sent);
|
||||
}
|
||||
|
||||
close(fd);
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#define FILE_TYPES_H
|
||||
|
||||
#include "data.h"
|
||||
#include "format.h"
|
||||
#include "xattr.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
@@ -56,6 +57,11 @@ typedef struct {
|
||||
* equals the incoming file, and `data` is kept as the cross-filesystem
|
||||
* fallback (a local copy) if the hard link cannot be created. */
|
||||
char* basis_link;
|
||||
/* Receiver-only, --copy-dest: when set (and basis_link is NULL), stream the
|
||||
* basis file's bytes into the destination instead of `data`/`data->size`.
|
||||
* This lets a basis larger than any whole-file bound materialize without
|
||||
* buffering it; the source metadata on `metadata` is applied afterwards. */
|
||||
char* basis_copy;
|
||||
/* --hard-links (-H), sender + receiver wire state. link_group is a run-local
|
||||
* id shared by every member of one source inode (0 = not part of a group).
|
||||
* The FIRST member (link_first == true) carries its data on the wire and is
|
||||
@@ -86,6 +92,21 @@ typedef struct {
|
||||
* Receiver: parsed off the wire, attached here, and applied fd-relative on
|
||||
* the written file. NULL/0 == the file carries no xattrs. */
|
||||
FileXattrList* xattrs;
|
||||
/* Sender-side output-parity state (never serialized): the receiver-reported
|
||||
* pre-transfer destination snapshot for this entry, filled by the per-file
|
||||
* STATUS_CHECK exchange when report_dest_info is set. `known` is false when
|
||||
* no report was requested/received, in which case -i/--out-format treats the
|
||||
* entry conservatively as newly created. */
|
||||
OutputDestState dest_state;
|
||||
/* Receiver-only wire-stats tally: the number of bytes reconstructed from the
|
||||
* basis file (matched delta blocks) for this entry. 0 when the file was sent
|
||||
* whole. Accumulated into ReceiverStats.matched_data by the receiver sink. */
|
||||
unsigned long long matched_bytes;
|
||||
/* Receiver-only (protocol 2.28.0) wire-stats tally: the literal delta fragment
|
||||
* bytes this entry carried (DELTA_INSTR_LITERAL). 0 when the file was sent
|
||||
* whole; the sink then falls back to the whole payload size. Accumulated
|
||||
* into ReceiverStats.literal_bytes. */
|
||||
unsigned long long literal_bytes;
|
||||
} File;
|
||||
|
||||
/* The path that should be sent on the wire and used for the receiver-side
|
||||
|
||||
+600
-225
@@ -1,163 +1,27 @@
|
||||
#include "filter.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <ctype.h>
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
/* ---- Single rule parsing ---- */
|
||||
|
||||
static bool rule_text_is_unsupported_word(const char* p, size_t len) {
|
||||
static const char* const words[] = {"merge", "dir-merge", "hide", "show",
|
||||
"protect", "risk", "clear"};
|
||||
for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) {
|
||||
size_t wl = strlen(words[i]);
|
||||
if (len == wl && strncmp(p, words[i], wl) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
/* Write a diagnostic message into the caller's optional buffer. A NULL `err`
|
||||
* (or a zero size) is a no-op, so a caller that only needs the boolean status
|
||||
* may pass NULL without the snprintf-on-NULL undefined behaviour. */
|
||||
static void filter_set_error(char* err, size_t err_size, const char* fmt, ...) {
|
||||
if (!err || err_size == 0)
|
||||
return;
|
||||
va_list ap;
|
||||
va_start(ap, fmt);
|
||||
vsnprintf(err, err_size, fmt, ap);
|
||||
va_end(ap);
|
||||
}
|
||||
|
||||
/* rsync include/exclude rule modifiers we do NOT implement. A rule whose +/- is
|
||||
* immediately followed by one of these is rejected instead of being silently
|
||||
* parsed as a literal pattern. */
|
||||
static bool is_unsupported_rule_modifier(char c) {
|
||||
return c == '!' || c == 'C' || c == 's' || c == 'r' || c == 'p' || c == 'x';
|
||||
}
|
||||
|
||||
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (!line)
|
||||
return NULL;
|
||||
char* text = str_dup(line);
|
||||
if (!text) {
|
||||
if (err)
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
size_t len = strlen(text);
|
||||
while (len > 0 && (text[len - 1] == '\n' || text[len - 1] == '\r'))
|
||||
text[--len] = '\0';
|
||||
|
||||
const char* p = text;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
if (*p == '\0') {
|
||||
snprintf(err, err_size, "empty filter rule");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
FilterAction action = FILTER_ACTION_EXCLUDE;
|
||||
if (*p == '+' || *p == '-') {
|
||||
action = *p == '+' ? FILTER_ACTION_INCLUDE : FILTER_ACTION_EXCLUDE;
|
||||
p++;
|
||||
/* rsync attaches rule modifiers directly to the +/- (e.g. "-s foo"). Only
|
||||
* the '/' anchor modifier is supported; anything else is a clear error
|
||||
* rather than a silently-ignored literal. */
|
||||
if (*p != ' ' && *p != '\t' && *p != '\0' && is_unsupported_rule_modifier(*p)) {
|
||||
snprintf(err, err_size,
|
||||
"filter rule modifier '%c' is not supported (only the '/' anchor after +/- "
|
||||
"is implemented; put a space between +/- and the pattern)",
|
||||
*p);
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
} else {
|
||||
/* ':' (dir-merge) and '.' (merge) are rsync filter-rule shorthands. At the
|
||||
* start of a rule they mean "merge this file", so reject them instead of
|
||||
* silently turning them into inert exclude patterns. */
|
||||
if (*p == ':' || *p == '.' || *p == '!') {
|
||||
snprintf(err, err_size,
|
||||
"filter rule starting with '%c' is not supported (merge/dir-merge/list-clear "
|
||||
"shorthands are not implemented; use +/- include/exclude rules)",
|
||||
*p);
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
const char* sp = p;
|
||||
while (*sp != '\0' && *sp != ' ' && *sp != '\t')
|
||||
sp++;
|
||||
size_t word_len = (size_t)(sp - p);
|
||||
if (rule_text_is_unsupported_word(p, word_len)) {
|
||||
snprintf(err, err_size,
|
||||
"'%.*s' filter directives are not supported (only +/- include/exclude rules "
|
||||
"with an optional '/' anchor and trailing '/' dir marker)",
|
||||
(int)word_len, p);
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
if (word_len == strlen("include") && strncmp(p, "include", word_len) == 0) {
|
||||
action = FILTER_ACTION_INCLUDE;
|
||||
p = sp;
|
||||
} else if (word_len == strlen("exclude") && strncmp(p, "exclude", word_len) == 0) {
|
||||
action = FILTER_ACTION_EXCLUDE;
|
||||
p = sp;
|
||||
}
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
}
|
||||
|
||||
if (*p == '\0') {
|
||||
snprintf(err, err_size, "filter rule has no pattern");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* A pattern beginning with '/' is anchored (either as "-/foo" or "- /foo"). */
|
||||
bool anchored = false;
|
||||
if (*p == '/') {
|
||||
anchored = true;
|
||||
p++;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
}
|
||||
if (*p == '\0') {
|
||||
snprintf(err, err_size, "filter rule has no pattern after '/' anchor");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Pattern runs to the end of the rule; a single trailing '/' marks dir-only. */
|
||||
size_t pat_len = strlen(p);
|
||||
bool dir_only = false;
|
||||
if (pat_len > 1 && p[pat_len - 1] == '/') {
|
||||
dir_only = true;
|
||||
pat_len--;
|
||||
} else if (pat_len == 1 && p[0] == '/') {
|
||||
/* "//" anchored with nothing after: meaningless. */
|
||||
snprintf(err, err_size, "filter rule has no pattern");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
rule->pattern = malloc(pat_len + 1);
|
||||
if (!rule->pattern) {
|
||||
free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(rule->pattern, p, pat_len);
|
||||
rule->pattern[pat_len] = '\0';
|
||||
rule->action = action;
|
||||
rule->anchored = anchored;
|
||||
rule->dir_only = dir_only;
|
||||
rule->owner = NULL;
|
||||
free(text);
|
||||
return rule;
|
||||
}
|
||||
/* ---- Ordered rule lists ---- */
|
||||
|
||||
void filter_rule_free(FilterRule* rule) {
|
||||
if (!rule)
|
||||
@@ -167,8 +31,6 @@ void filter_rule_free(FilterRule* rule) {
|
||||
free(rule);
|
||||
}
|
||||
|
||||
/* ---- Ordered rule lists ---- */
|
||||
|
||||
FilterRuleList* filter_rule_list_create(void) {
|
||||
return calloc(1, sizeof(FilterRuleList));
|
||||
}
|
||||
@@ -190,28 +52,42 @@ bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err,
|
||||
size_t err_size) {
|
||||
FilterRule* rule = filter_rule_parse(line, err, err_size);
|
||||
if (!rule)
|
||||
return false;
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void filter_rule_list_free(FilterRuleList* list) {
|
||||
if (!list)
|
||||
return;
|
||||
for (int i = 0; i < list->count; i++)
|
||||
filter_rule_free(list->items[i]);
|
||||
for (int i = 0; i < list->dir_merge_count; i++)
|
||||
free(list->dir_merge_names[i]);
|
||||
free(list->dir_merge_names);
|
||||
free(list->items);
|
||||
free(list);
|
||||
}
|
||||
|
||||
/* Register a per-directory merge-file basename (for "dir-merge NAME"/": NAME"
|
||||
* and -F's .rsync-filter). Duplicate names are ignored. */
|
||||
bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name) {
|
||||
if (!list || !name || name[0] == '\0')
|
||||
return false;
|
||||
for (int i = 0; i < list->dir_merge_count; i++) {
|
||||
if (strcmp(list->dir_merge_names[i], name) == 0)
|
||||
return true;
|
||||
}
|
||||
if (list->dir_merge_count == list->dir_merge_capacity) {
|
||||
int new_cap = list->dir_merge_capacity > 0 ? list->dir_merge_capacity * 2 : 4;
|
||||
char** grown = realloc(list->dir_merge_names, (size_t)new_cap * sizeof(char*));
|
||||
if (!grown)
|
||||
return false;
|
||||
list->dir_merge_names = grown;
|
||||
list->dir_merge_capacity = new_cap;
|
||||
}
|
||||
char* dup = str_dup(name);
|
||||
if (!dup)
|
||||
return false;
|
||||
list->dir_merge_names[list->dir_merge_count++] = dup;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool set_rule_owner(FilterRule* rule, const char* owner) {
|
||||
char* dup = str_dup(owner ? owner : "");
|
||||
if (!dup)
|
||||
@@ -221,7 +97,315 @@ static bool set_rule_owner(FilterRule* rule, const char* owner) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* ---- CVS default excludes (-C) ---- */
|
||||
/* ---- Rule parsing ---- */
|
||||
|
||||
/* A short rule prefix is a single character; a long rule name is alphabetic
|
||||
* (with '-'). `is_short` distinguishes the modifier-attachment rules. */
|
||||
typedef enum {
|
||||
RULE_KIND_EXCLUDE,
|
||||
RULE_KIND_INCLUDE,
|
||||
RULE_KIND_HIDE,
|
||||
RULE_KIND_SHOW,
|
||||
RULE_KIND_PROTECT,
|
||||
RULE_KIND_RISK,
|
||||
RULE_KIND_MERGE,
|
||||
RULE_KIND_DIR_MERGE,
|
||||
RULE_KIND_CLEAR,
|
||||
RULE_KIND_UNKNOWN,
|
||||
} RuleKind;
|
||||
|
||||
static bool short_rule_char(char c, RuleKind* kind) {
|
||||
switch (c) {
|
||||
case '-':
|
||||
*kind = RULE_KIND_EXCLUDE;
|
||||
return true;
|
||||
case '+':
|
||||
*kind = RULE_KIND_INCLUDE;
|
||||
return true;
|
||||
case 'H':
|
||||
*kind = RULE_KIND_HIDE;
|
||||
return true;
|
||||
case 'S':
|
||||
*kind = RULE_KIND_SHOW;
|
||||
return true;
|
||||
case 'P':
|
||||
*kind = RULE_KIND_PROTECT;
|
||||
return true;
|
||||
case 'R':
|
||||
*kind = RULE_KIND_RISK;
|
||||
return true;
|
||||
case '.':
|
||||
*kind = RULE_KIND_MERGE;
|
||||
return true;
|
||||
case ':':
|
||||
*kind = RULE_KIND_DIR_MERGE;
|
||||
return true;
|
||||
case '!':
|
||||
*kind = RULE_KIND_CLEAR;
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
static bool long_rule_name(const char* name, size_t len, RuleKind* kind) {
|
||||
struct {
|
||||
const char* word;
|
||||
RuleKind kind;
|
||||
} table[] = {
|
||||
{"exclude", RULE_KIND_EXCLUDE}, {"include", RULE_KIND_INCLUDE},
|
||||
{"hide", RULE_KIND_HIDE}, {"show", RULE_KIND_SHOW},
|
||||
{"protect", RULE_KIND_PROTECT}, {"risk", RULE_KIND_RISK},
|
||||
{"merge", RULE_KIND_MERGE}, {"dir-merge", RULE_KIND_DIR_MERGE},
|
||||
{"clear", RULE_KIND_CLEAR},
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(table) / sizeof(table[0]); i++) {
|
||||
if (strlen(table[i].word) == len && strncmp(name, table[i].word, len) == 0) {
|
||||
*kind = table[i].kind;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool is_modifier_char(char c) {
|
||||
return c == 's' || c == 'r' || c == 'p' || c == 'x' || c == '/' || c == '!' || c == 'C';
|
||||
}
|
||||
|
||||
/* Parse "RULE[,MODIFIERS] [PATTERN]". On success `kind`, `sides`,
|
||||
* `sides_explicit`, `negate`, `anchored_mod`, `perishable`, `xattr`,
|
||||
* `cvs_inject` and the pattern span (`pat_start`/`pat_len`, possibly 0 for
|
||||
* merge/clear) are filled. Returns true on success. */
|
||||
static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides,
|
||||
bool* sides_explicit, bool* negate, bool* anchored_mod,
|
||||
bool* perishable, bool* xattr, bool* cvs_inject,
|
||||
const char** pat_start, size_t* pat_len) {
|
||||
const char* p = text;
|
||||
*sides = FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER;
|
||||
*sides_explicit = false;
|
||||
*negate = false;
|
||||
*anchored_mod = false;
|
||||
*perishable = false;
|
||||
*xattr = false;
|
||||
*cvs_inject = false;
|
||||
*pat_start = NULL;
|
||||
*pat_len = 0;
|
||||
|
||||
bool is_short = false;
|
||||
if (short_rule_char(*p, kind)) {
|
||||
is_short = true;
|
||||
p++;
|
||||
} else {
|
||||
const char* name_start = p;
|
||||
while (isalpha((unsigned char)*p) || *p == '-')
|
||||
p++;
|
||||
size_t name_len = (size_t)(p - name_start);
|
||||
if (name_len == 0 || !long_rule_name(name_start, name_len, kind))
|
||||
return false;
|
||||
/* A long name must be followed by a separator, a comma or the end. */
|
||||
if (*p != '\0' && *p != ',' && *p != ' ' && *p != '_')
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Modifiers: long names require a comma; short names may attach directly.
|
||||
Only commit a modifier run that terminates at a separator or the end, so a
|
||||
pattern such as "*.tmp" written as "-*.tmp" is not mistaken for modifiers. */
|
||||
const char* mod_start = p;
|
||||
const char* mod_end = p;
|
||||
if (*p == ',') {
|
||||
p++;
|
||||
mod_start = p;
|
||||
while (is_modifier_char(*p))
|
||||
p++;
|
||||
mod_end = p;
|
||||
} else if (is_short) {
|
||||
const char* scan = p;
|
||||
while (is_modifier_char(*scan))
|
||||
scan++;
|
||||
if (*scan == '\0' || *scan == ' ' || *scan == '_') {
|
||||
mod_start = p;
|
||||
mod_end = scan;
|
||||
p = scan;
|
||||
}
|
||||
}
|
||||
for (const char* m = mod_start; m < mod_end; m++) {
|
||||
switch (*m) {
|
||||
case 's':
|
||||
*sides = FILTER_SIDE_SENDER;
|
||||
*sides_explicit = true;
|
||||
break;
|
||||
case 'r':
|
||||
*sides = FILTER_SIDE_RECEIVER;
|
||||
*sides_explicit = true;
|
||||
break;
|
||||
case '!':
|
||||
*negate = true;
|
||||
break;
|
||||
case '/':
|
||||
*anchored_mod = true;
|
||||
break;
|
||||
case 'p':
|
||||
*perishable = true;
|
||||
break;
|
||||
case 'x':
|
||||
*xattr = true;
|
||||
break;
|
||||
case 'C':
|
||||
*cvs_inject = true;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
/* A single space or underscore separates the rule/modifiers from the
|
||||
pattern; further spaces/underscores belong to the pattern. */
|
||||
const char* pat = p;
|
||||
if (*pat == ' ' || *pat == '_')
|
||||
pat++;
|
||||
/* Trim a trailing newline/CR (the caller may pass a raw file line). */
|
||||
*pat_start = pat;
|
||||
*pat_len = strlen(pat);
|
||||
while (*pat_len > 0 && (pat[*pat_len - 1] == '\n' || pat[*pat_len - 1] == '\r'))
|
||||
(*pat_len)--;
|
||||
return true;
|
||||
}
|
||||
|
||||
FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err,
|
||||
size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (!line)
|
||||
return NULL;
|
||||
const char* p = line;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
if (*p == '\0' || *p == '\n' || *p == '\r') {
|
||||
filter_set_error(err, err_size, "empty filter rule");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
RuleKind kind = RULE_KIND_UNKNOWN;
|
||||
unsigned sides;
|
||||
bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject;
|
||||
const char* pat;
|
||||
size_t pat_len;
|
||||
if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable,
|
||||
&xattr, &cvs_inject, &pat, &pat_len)) {
|
||||
filter_set_error(err, err_size, "unrecognized filter rule syntax");
|
||||
return NULL;
|
||||
}
|
||||
if (cvs_inject) {
|
||||
/* The C modifier expands to the CVS defaults in place; the rule itself
|
||||
carries no pattern and is handled by the caller. */
|
||||
filter_set_error(err, err_size, "the C modifier is handled by the rule-list parser");
|
||||
return NULL;
|
||||
}
|
||||
if (xattr) {
|
||||
filter_set_error(err, err_size, "xattr-name filter rules (the x modifier) are not supported");
|
||||
return NULL;
|
||||
}
|
||||
if (kind == RULE_KIND_MERGE || kind == RULE_KIND_DIR_MERGE) {
|
||||
filter_set_error(err, err_size, "merge/dir-merge rules are handled by the rule-list parser");
|
||||
return NULL;
|
||||
}
|
||||
if (kind == RULE_KIND_CLEAR) {
|
||||
if (pat_len != 0) {
|
||||
filter_set_error(err, err_size, "clear takes no pattern");
|
||||
return NULL;
|
||||
}
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
rule->action = FILTER_ACTION_NONE; /* clear marker: no pattern */
|
||||
rule->sides = 0;
|
||||
return rule;
|
||||
}
|
||||
|
||||
FilterAction action;
|
||||
switch (kind) {
|
||||
case RULE_KIND_INCLUDE:
|
||||
case RULE_KIND_SHOW:
|
||||
case RULE_KIND_RISK:
|
||||
action = FILTER_ACTION_INCLUDE;
|
||||
break;
|
||||
default:
|
||||
action = FILTER_ACTION_EXCLUDE;
|
||||
break;
|
||||
}
|
||||
if (kind == RULE_KIND_HIDE)
|
||||
sides = FILTER_SIDE_SENDER;
|
||||
else if (kind == RULE_KIND_SHOW)
|
||||
sides = FILTER_SIDE_SENDER;
|
||||
else if (kind == RULE_KIND_PROTECT)
|
||||
sides = FILTER_SIDE_RECEIVER;
|
||||
else if (kind == RULE_KIND_RISK)
|
||||
sides = FILTER_SIDE_RECEIVER;
|
||||
if (kind == RULE_KIND_HIDE || kind == RULE_KIND_SHOW || kind == RULE_KIND_PROTECT ||
|
||||
kind == RULE_KIND_RISK)
|
||||
sides_explicit = true;
|
||||
/* --delete-excluded turns an unqualified (no explicit s/r) rule into a
|
||||
sender-side-only rule, so it no longer protects the receiver. */
|
||||
if (opts && opts->delete_excluded && !sides_explicit)
|
||||
sides = FILTER_SIDE_SENDER;
|
||||
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "filter rule has no pattern");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
bool anchored = anchored_mod;
|
||||
const char* pat_begin = pat;
|
||||
if (*pat_begin == '/') {
|
||||
anchored = true;
|
||||
pat_begin++;
|
||||
/* Drop the spaces that could follow the anchor in the "-/ foo" form. */
|
||||
while (*pat_begin == ' ' || *pat_begin == '\t')
|
||||
pat_begin++;
|
||||
pat_len = strlen(pat_begin);
|
||||
while (pat_len > 0 && (pat_begin[pat_len - 1] == '\n' || pat_begin[pat_len - 1] == '\r'))
|
||||
pat_len--;
|
||||
}
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "filter rule has no pattern after '/' anchor");
|
||||
return NULL;
|
||||
}
|
||||
bool dir_only = false;
|
||||
if (pat_len > 1 && pat_begin[pat_len - 1] == '/') {
|
||||
dir_only = true;
|
||||
pat_len--;
|
||||
}
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "filter rule has no pattern");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
rule->pattern = malloc(pat_len + 1);
|
||||
if (!rule->pattern) {
|
||||
free(rule);
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
memcpy(rule->pattern, pat_begin, pat_len);
|
||||
rule->pattern[pat_len] = '\0';
|
||||
rule->action = action;
|
||||
rule->sides = sides;
|
||||
rule->anchored = anchored;
|
||||
rule->dir_only = dir_only;
|
||||
rule->negate = negate;
|
||||
rule->perishable = perishable;
|
||||
(void)xattr; /* xattr-name rules never match file/dir names; accepted/ignored */
|
||||
return rule;
|
||||
}
|
||||
|
||||
/* ---- CVS default excludes (-C and the C modifier) ---- */
|
||||
|
||||
typedef struct {
|
||||
const char* pattern;
|
||||
@@ -240,12 +424,13 @@ static const CvsDefaultRule CVS_DEFAULTS[] = {
|
||||
{".svn/", true}, {".git/", true}, {".hg/", true}, {".bzr/", true},
|
||||
};
|
||||
|
||||
static bool cvs_rule_list_append(FilterRuleList* list) {
|
||||
static bool filter_list_append_cvs(FilterRuleList* list, unsigned sides) {
|
||||
for (size_t i = 0; i < sizeof(CVS_DEFAULTS) / sizeof(CVS_DEFAULTS[0]); i++) {
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule)
|
||||
return false;
|
||||
rule->action = FILTER_ACTION_EXCLUDE;
|
||||
rule->sides = sides;
|
||||
rule->dir_only = CVS_DEFAULTS[i].dir_only;
|
||||
size_t plen = strlen(CVS_DEFAULTS[i].pattern);
|
||||
if (rule->dir_only && plen > 0 && CVS_DEFAULTS[i].pattern[plen - 1] == '/')
|
||||
@@ -269,76 +454,236 @@ static bool cvs_rule_list_append(FilterRuleList* list) {
|
||||
return true;
|
||||
}
|
||||
|
||||
#define FILTER_MAX_MERGE_DEPTH 16
|
||||
|
||||
static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* base_dir,
|
||||
int depth, char* err, size_t err_size);
|
||||
|
||||
/* Read a merge file and splice its rules into `list`. A relative path is
|
||||
* resolved below `base_dir` when given, else used as-is (rsync resolves a
|
||||
* command-line merge file relative to the current directory). */
|
||||
static bool filter_list_merge_file(FilterRuleList* list, const char* name,
|
||||
const FilterParseOptions* opts, const char* base_dir, int depth,
|
||||
char* err, size_t err_size) {
|
||||
if (name[0] == '\0') {
|
||||
filter_set_error(err, err_size, "merge requires a filename");
|
||||
return false;
|
||||
}
|
||||
char* path =
|
||||
(base_dir && base_dir[0] && name[0] != '/') ? path_cat(base_dir, name) : str_dup(name);
|
||||
if (!path) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
FILE* fp = fopen(path, "r");
|
||||
if (!fp) {
|
||||
filter_set_error(err, err_size, "could not read merge file '%s': %s", path, strerror(errno));
|
||||
free(path);
|
||||
return false;
|
||||
}
|
||||
char* line = NULL;
|
||||
size_t cap = 0;
|
||||
bool ok = true;
|
||||
while (true) {
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &cap, '\n', UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
filter_set_error(err, err_size, "error reading merge file '%s'", path);
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (n == 0)
|
||||
break;
|
||||
const char* lp = line;
|
||||
while (*lp == ' ' || *lp == '\t')
|
||||
lp++;
|
||||
if (*lp == '\0' || *lp == '\n' || *lp == '\r' || *lp == '#')
|
||||
continue;
|
||||
if (!filter_list_parse_append_depth(list, lp, opts, base_dir, depth + 1, err, err_size)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
free(line);
|
||||
fclose(fp);
|
||||
free(path);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Parse one line and append/merge it into `list`. Handles clear, merge and
|
||||
* dir-merge at the list level. */
|
||||
static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* base_dir,
|
||||
int depth, char* err, size_t err_size) {
|
||||
if (depth > FILTER_MAX_MERGE_DEPTH) {
|
||||
filter_set_error(err, err_size, "merge files nested too deeply");
|
||||
return false;
|
||||
}
|
||||
const char* p = line;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
if (*p == '\0' || *p == '\n' || *p == '\r')
|
||||
return true;
|
||||
|
||||
RuleKind kind = RULE_KIND_UNKNOWN;
|
||||
unsigned sides;
|
||||
bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject;
|
||||
const char* pat;
|
||||
size_t pat_len;
|
||||
if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable,
|
||||
&xattr, &cvs_inject, &pat, &pat_len)) {
|
||||
filter_set_error(err, err_size, "unrecognized filter rule syntax: %s", p);
|
||||
return false;
|
||||
}
|
||||
(void)sides_explicit;
|
||||
(void)negate;
|
||||
(void)anchored_mod;
|
||||
(void)perishable;
|
||||
(void)xattr;
|
||||
|
||||
if (cvs_inject) {
|
||||
/* "C" injects the CVS defaults in place; no pattern is expected. */
|
||||
return filter_list_append_cvs(list, sides);
|
||||
}
|
||||
if (kind == RULE_KIND_CLEAR) {
|
||||
if (pat_len != 0) {
|
||||
filter_set_error(err, err_size, "clear takes no pattern");
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < list->count; i++)
|
||||
filter_rule_free(list->items[i]);
|
||||
list->count = 0;
|
||||
return true;
|
||||
}
|
||||
if (kind == RULE_KIND_MERGE) {
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "merge requires a filename");
|
||||
return false;
|
||||
}
|
||||
char* name = malloc(pat_len + 1);
|
||||
if (!name) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
memcpy(name, pat, pat_len);
|
||||
name[pat_len] = '\0';
|
||||
bool ok = filter_list_merge_file(list, name, opts, base_dir, depth, err, err_size);
|
||||
free(name);
|
||||
return ok;
|
||||
}
|
||||
if (kind == RULE_KIND_DIR_MERGE) {
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "dir-merge requires a filename");
|
||||
return false;
|
||||
}
|
||||
char* name = malloc(pat_len + 1);
|
||||
if (!name) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
memcpy(name, pat, pat_len);
|
||||
name[pat_len] = '\0';
|
||||
bool ok = filter_rule_list_add_dir_merge(list, name);
|
||||
free(name);
|
||||
if (!ok) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
FilterRule* rule = filter_rule_parse(p, opts, err, err_size);
|
||||
if (!rule)
|
||||
return false;
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* merge_base_dir,
|
||||
char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (!list)
|
||||
return false;
|
||||
return filter_list_parse_append_depth(list, line, opts, merge_base_dir, 0, err, err_size);
|
||||
}
|
||||
|
||||
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
|
||||
char* err, size_t err_size) {
|
||||
bool delete_excluded, char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
FilterRuleList* list = filter_rule_list_create();
|
||||
if (!list) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
FilterParseOptions opts = {.delete_excluded = delete_excluded, .cvs_exclude = cvs_exclude};
|
||||
for (int i = 0; i < rule_count; i++) {
|
||||
if (!rule_texts || !rule_texts[i])
|
||||
continue;
|
||||
FilterRule* rule = filter_rule_parse(rule_texts[i], err, err_size);
|
||||
if (!rule) {
|
||||
if (!filter_rule_list_parse_append(list, rule_texts[i], &opts, NULL, err, err_size)) {
|
||||
filter_rule_list_free(list);
|
||||
return NULL;
|
||||
}
|
||||
if (!set_rule_owner(rule, "")) {
|
||||
filter_rule_free(rule);
|
||||
filter_rule_list_free(list);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
filter_rule_list_free(list);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
if (cvs_exclude && !cvs_rule_list_append(list)) {
|
||||
if (cvs_exclude && !filter_list_append_cvs(list, FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER)) {
|
||||
filter_rule_list_free(list);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
return list;
|
||||
}
|
||||
|
||||
/* ---- Per-directory .rsync-filter files ---- */
|
||||
/* ---- Per-directory merge files ---- */
|
||||
|
||||
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
|
||||
char* err, size_t err_size) {
|
||||
/* Undo the rules and dir-merge registrations that one merge file appended,
|
||||
* leaving the caller's earlier content intact. A "clear" rule inside the file
|
||||
* frees every rule, including the caller's; clamp to the surviving count so
|
||||
* those already-freed rules are never resurrected and freed a second time. */
|
||||
static void filter_file_rollback(FilterRuleList* list, int rules_before, int dir_merges_before) {
|
||||
int first = rules_before < list->count ? rules_before : list->count;
|
||||
for (int i = first; i < list->count; i++)
|
||||
filter_rule_free(list->items[i]);
|
||||
list->count = first;
|
||||
for (int i = dir_merges_before; i < list->dir_merge_count; i++)
|
||||
free(list->dir_merge_names[i]);
|
||||
list->dir_merge_count = dir_merges_before;
|
||||
}
|
||||
|
||||
bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts, bool* exists,
|
||||
char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (exists)
|
||||
*exists = false;
|
||||
char* filter_path = path_cat(dir_path, ".rsync-filter");
|
||||
if (!list)
|
||||
return false;
|
||||
char* filter_path = path_cat(dir_path, name);
|
||||
if (!filter_path) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
FILE* fp = fopen(filter_path, "r");
|
||||
free(filter_path);
|
||||
if (!fp) {
|
||||
if (errno == ENOENT || errno == ENOTDIR)
|
||||
return filter_rule_list_create();
|
||||
return true;
|
||||
char* escaped_dir = output_escape(dir_path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s",
|
||||
log_message(LOG_LEVEL_WARNING, "Could not read %s in %s: %s", name,
|
||||
escaped_dir ? escaped_dir : "<allocation failed>", strerror(errno));
|
||||
free(escaped_dir);
|
||||
return filter_rule_list_create();
|
||||
return true;
|
||||
}
|
||||
if (exists)
|
||||
*exists = true;
|
||||
FilterRuleList* list = filter_rule_list_create();
|
||||
if (!list) {
|
||||
fclose(fp);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
int rules_before = list->count;
|
||||
int dir_merges_before = list->dir_merge_count;
|
||||
char* line = NULL;
|
||||
size_t line_cap = 0;
|
||||
bool ok = true;
|
||||
@@ -346,9 +691,10 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, '\n', UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
if (errno == EFBIG) {
|
||||
snprintf(err, err_size, "line in .rsync-filter exceeds %d bytes", (int)UTILS_MAX_LINE_LEN);
|
||||
filter_set_error(err, err_size, "line in %s exceeds %d bytes", name,
|
||||
(int)UTILS_MAX_LINE_LEN);
|
||||
} else {
|
||||
snprintf(err, err_size, "error reading .rsync-filter: %s", strerror(errno));
|
||||
filter_set_error(err, err_size, "error reading %s: %s", name, strerror(errno));
|
||||
}
|
||||
ok = false;
|
||||
break;
|
||||
@@ -360,20 +706,9 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
|
||||
p++;
|
||||
if (*p == '\0' || *p == '\n' || *p == '\r' || *p == '#')
|
||||
continue;
|
||||
FilterRule* rule = filter_rule_parse(p, err, err_size);
|
||||
if (!rule) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (!set_rule_owner(rule, owner_rel)) {
|
||||
filter_rule_free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
/* Merge files inside a per-directory file resolve relative to that
|
||||
directory. */
|
||||
if (!filter_list_parse_append_depth(list, p, opts, dir_path, 0, err, err_size)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
@@ -381,12 +716,40 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
|
||||
free(line);
|
||||
fclose(fp);
|
||||
if (!ok) {
|
||||
filter_file_rollback(list, rules_before, dir_merges_before);
|
||||
return false;
|
||||
}
|
||||
for (int i = rules_before; i < list->count; i++) {
|
||||
if (!set_rule_owner(list->items[i], owner_rel)) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
filter_file_rollback(list, rules_before, dir_merges_before);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
FilterRuleList* filter_file_read_named(const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts,
|
||||
bool* exists, char* err, size_t err_size) {
|
||||
FilterRuleList* list = filter_rule_list_create();
|
||||
if (!list) {
|
||||
if (err && err_size > 0)
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
if (!filter_file_append(list, dir_path, name, owner_rel, opts, exists, err, err_size)) {
|
||||
filter_rule_list_free(list);
|
||||
return NULL;
|
||||
}
|
||||
return list;
|
||||
}
|
||||
|
||||
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
|
||||
char* err, size_t err_size) {
|
||||
return filter_file_read_named(dir_path, ".rsync-filter", owner_rel, NULL, exists, err, err_size);
|
||||
}
|
||||
|
||||
/* ---- Rule matching ---- */
|
||||
|
||||
/* Match a pattern that contains '/' (non-anchored) against the end of the
|
||||
@@ -402,10 +765,10 @@ static bool glob_suffix_match(const char* pattern, const char* str) {
|
||||
}
|
||||
|
||||
static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, const char* leaf,
|
||||
bool is_dir) {
|
||||
bool is_dir, unsigned side) {
|
||||
if (!rule || !rule->pattern)
|
||||
return FILTER_ACTION_NONE;
|
||||
if (rule->dir_only && !is_dir)
|
||||
if (!(rule->sides & side))
|
||||
return FILTER_ACTION_NONE;
|
||||
/* A rule applies only to entries below its owner directory. */
|
||||
const char* rel2 = rel_path;
|
||||
@@ -420,24 +783,36 @@ static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, c
|
||||
if (rel2[0] == '\0')
|
||||
return FILTER_ACTION_NONE;
|
||||
bool matched;
|
||||
if (rule->anchored) {
|
||||
if (rule->dir_only && !is_dir)
|
||||
matched = false;
|
||||
else if (rule->anchored)
|
||||
matched = glob_match(rule->pattern, rel2);
|
||||
} else if (strchr(rule->pattern, '/') != NULL) {
|
||||
else if (strchr(rule->pattern, '/') != NULL)
|
||||
matched = glob_suffix_match(rule->pattern, rel2);
|
||||
} else {
|
||||
else
|
||||
matched = glob_match(rule->pattern, leaf);
|
||||
}
|
||||
return matched ? rule->action : FILTER_ACTION_NONE;
|
||||
if (rule->negate)
|
||||
matched = !matched;
|
||||
if (!matched)
|
||||
return FILTER_ACTION_NONE;
|
||||
if (side == FILTER_SIDE_RECEIVER)
|
||||
return rule->action == FILTER_ACTION_EXCLUDE ? FILTER_ACTION_PROTECT : FILTER_ACTION_RISK;
|
||||
return rule->action;
|
||||
}
|
||||
|
||||
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
|
||||
bool is_dir) {
|
||||
FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path,
|
||||
const char* leaf, bool is_dir, unsigned side) {
|
||||
if (!list)
|
||||
return FILTER_ACTION_NONE;
|
||||
for (int i = 0; i < list->count; i++) {
|
||||
FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir);
|
||||
FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir, side);
|
||||
if (action != FILTER_ACTION_NONE)
|
||||
return action;
|
||||
}
|
||||
return FILTER_ACTION_NONE;
|
||||
}
|
||||
|
||||
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
|
||||
bool is_dir) {
|
||||
return filter_rules_apply_side(list, rel_path, leaf, is_dir, FILTER_SIDE_SENDER);
|
||||
}
|
||||
|
||||
+96
-42
@@ -4,79 +4,133 @@
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
|
||||
/* rsync-style filter rule engine (client-side file selection).
|
||||
/* rsync-style filter rule engine (client-side file selection and the
|
||||
* receiver-side protection set it feeds).
|
||||
*
|
||||
* Supported rule syntax (documented subset):
|
||||
* [+|-] [anchored '/' prefix] pattern [trailing '/' for dir-only]
|
||||
*
|
||||
* "+ PATTERN" include rule (first match wins)
|
||||
* "- PATTERN" exclude rule
|
||||
* "PATTERN" implicit exclude rule (rsync default)
|
||||
* "include PATTERN" / "exclude PATTERN" word forms
|
||||
* leading '/' after the +/- anchors the pattern to its owner directory
|
||||
* (the transfer root for command-line/-C rules, the directory that
|
||||
* contains a .rsync-filter file for per-directory rules)
|
||||
* a trailing '/' makes the rule match directories only
|
||||
*
|
||||
* Rejected explicitly (no silent no-ops): the rsync merge/dir-merge/list-clear
|
||||
* shorthands written as a rule that starts with ':' or '.' or '!', the
|
||||
* merge/dir-merge/hide/show/protect/risk/clear words, and every include/exclude
|
||||
* rule modifier other than '/' (! C s r p x). The pattern must be separated
|
||||
* from +/- by a space (or a single '/' anchor), exactly like rsync's
|
||||
* "-s foo"/"-p ..." modifier syntax is refused.
|
||||
* Rule syntax (see the rsync man page FILTER RULES section):
|
||||
* RULE [PATTERN_OR_FILENAME]
|
||||
* RULE,MODIFIERS [PATTERN_OR_FILENAME]
|
||||
* Short RULE names may attach MODIFIERS directly ("-sr foo"); the long name
|
||||
* form requires the comma. The pattern/filename is separated from the rule by
|
||||
* one space or underscore. Rule names:
|
||||
* exclude/- exclude (by default both sender-hide and receiver-protect)
|
||||
* include/+ include (by default both sender-show and receiver-risk)
|
||||
* hide/H sender-only exclude
|
||||
* show/S sender-only include
|
||||
* protect/P receiver-only exclude (protect from deletion)
|
||||
* risk/R receiver-only include (allow deletion)
|
||||
* merge/. read a client-side merge file for more rules
|
||||
* dir-merge/: per-directory merge file (registered for the scanner)
|
||||
* clear/! clear the current rule list (takes no argument)
|
||||
* Modifiers: '/' absolute anchor, '!' negate match, 'C' inject CVS defaults,
|
||||
* 's' sender side, 'r' receiver side, 'p' perishable, 'x' xattr name rule.
|
||||
* A trailing '/' makes a pattern match directories only. A leading '/' anchors
|
||||
* the pattern to its owner directory.
|
||||
*/
|
||||
|
||||
typedef enum {
|
||||
FILTER_ACTION_NONE = 0, /* no rule matched */
|
||||
FILTER_ACTION_EXCLUDE = -1,
|
||||
FILTER_ACTION_INCLUDE = 1
|
||||
FILTER_ACTION_INCLUDE = 1,
|
||||
/* Receiver-side-only verdicts: the entry is transferred but its destination
|
||||
* mirror is protected from --delete (PROTECT) or explicitly left at risk
|
||||
* (RISK). */
|
||||
FILTER_ACTION_PROTECT = 2,
|
||||
FILTER_ACTION_RISK = 3,
|
||||
} FilterAction;
|
||||
|
||||
typedef struct {
|
||||
FilterAction action;
|
||||
bool anchored; /* pattern anchored to the rule's owner directory */
|
||||
bool dir_only; /* pattern had a trailing '/': matches directories only */
|
||||
char* owner; /* owning directory rel path ("" == transfer root) */
|
||||
char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */
|
||||
} FilterRule;
|
||||
#define FILTER_SIDE_SENDER 1u
|
||||
#define FILTER_SIDE_RECEIVER 2u
|
||||
|
||||
typedef struct {
|
||||
FilterAction action; /* EXCLUDE or INCLUDE (the base pattern action) */
|
||||
unsigned sides; /* FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER */
|
||||
bool anchored; /* pattern anchored to the rule's owner directory */
|
||||
bool dir_only; /* pattern had a trailing '/': matches directories only */
|
||||
bool negate; /* '!' modifier: match succeeds when the pattern does not */
|
||||
bool perishable; /* 'p' modifier (ignored in deleted directories) */
|
||||
char* owner; /* owning directory rel path ("" == transfer root) */
|
||||
char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */
|
||||
} FilterRule;
|
||||
|
||||
typedef struct FilterRuleList {
|
||||
FilterRule** items; /* owned array of rule pointers */
|
||||
int count;
|
||||
int capacity;
|
||||
/* Per-directory merge-file basenames registered by "dir-merge NAME"/": NAME"
|
||||
* or by -F (.rsync-filter). Owned strings; the scanner reads each name in
|
||||
* every directory it traverses. */
|
||||
char** dir_merge_names;
|
||||
int dir_merge_count;
|
||||
int dir_merge_capacity;
|
||||
} FilterRuleList;
|
||||
|
||||
/* Context needed while parsing a rule list (merge files, --delete-excluded). */
|
||||
typedef struct {
|
||||
bool delete_excluded; /* --delete-excluded: default sides become sender-only */
|
||||
bool cvs_exclude; /* -C: expand the CVS default excludes */
|
||||
} FilterParseOptions;
|
||||
|
||||
/* Parse a single filter-rule line (no trailing newline required). Returns an
|
||||
* owned rule, or NULL on unsupported/invalid syntax with a message in `err`. */
|
||||
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size);
|
||||
* owned rule, or NULL on unsupported/invalid syntax with a message in `err`.
|
||||
* `opts` may be NULL (no merge expansion / no delete-excluded). */
|
||||
FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err,
|
||||
size_t err_size);
|
||||
void filter_rule_free(FilterRule* rule);
|
||||
|
||||
FilterRuleList* filter_rule_list_create(void);
|
||||
/* Append a fully-parsed rule (takes ownership). Returns false on OOM. */
|
||||
bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule);
|
||||
/* Parse `line` and append it. Returns false and fills `err` on bad syntax. */
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err,
|
||||
size_t err_size);
|
||||
/* Register a per-directory merge-file basename (idempotent). Returns false on
|
||||
* OOM. Used by the scanner to read custom "dir-merge" files. */
|
||||
bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name);
|
||||
/* Parse `line` and append it. Handles "clear"/"!" (resets the list), "merge
|
||||
* FILE"/". FILE" (splices the file's rules) and "dir-merge NAME"/": NAME"
|
||||
* (registers a per-directory filename). Returns false and fills `err` on bad
|
||||
* syntax or an unreadable merge file. `merge_base_dir` resolves a relative
|
||||
* merge-file path (NULL means the process working directory). */
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* merge_base_dir,
|
||||
char* err, size_t err_size);
|
||||
void filter_rule_list_free(FilterRuleList* list);
|
||||
|
||||
/* Build the command-line filter set: `rule_texts` (--filter=RULE in the order
|
||||
* given, 0..rule_count) followed by the -C CVS default excludes when
|
||||
* cvs_exclude is true. All rules are owned by "" (the transfer root).
|
||||
* cvs_exclude is true. All rules are owned by "" (the transfer root).
|
||||
* Returns NULL on unsupported rule text (message in `err`). */
|
||||
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
|
||||
char* err, size_t err_size);
|
||||
bool delete_excluded, char* err, size_t err_size);
|
||||
|
||||
/* Read "<dir_path>/.rsync-filter" and return its rules, each owned by
|
||||
* `owner_rel`. A missing file yields an empty list with *exists=false; an
|
||||
* unreadable file is treated as missing. Returns NULL only on parse or
|
||||
* allocation failure (message in `err`). */
|
||||
/* Read "<dir_path>/<name>" and return its rules, each owned by `owner_rel`. A
|
||||
* missing file yields an empty list with *exists=false; an unreadable file is
|
||||
* treated as missing. Returns NULL only on parse or allocation failure
|
||||
* (message in `err`). `opts` may be NULL. */
|
||||
FilterRuleList* filter_file_read_named(const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts,
|
||||
bool* exists, char* err, size_t err_size);
|
||||
|
||||
/* Append the rules of "<dir_path>/<name>" into an existing list (each owned by
|
||||
* `owner_rel`). A missing file yields *exists=false and no error. Returns
|
||||
* false only on parse/allocation failure (message in `err`). */
|
||||
bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts, bool* exists,
|
||||
char* err, size_t err_size);
|
||||
|
||||
/* filter_file_read_named with the default ".rsync-filter" name. */
|
||||
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
|
||||
char* err, size_t err_size);
|
||||
|
||||
/* Evaluate an entry against one ordered rule list. Returns FILTER_ACTION_NONE
|
||||
* when no rule matched, otherwise the first matching rule's action.
|
||||
* `rel_path` is the entry's path relative to the transfer root ("" == root),
|
||||
* `leaf` its final name, `is_dir` whether it is a directory. */
|
||||
/* Evaluate an entry against one ordered rule list for one side. Returns
|
||||
* FILTER_ACTION_NONE when no rule matched, otherwise the first matching rule's
|
||||
* action (for the receiver side an EXCLUDE is reported as
|
||||
* FILTER_ACTION_PROTECT and an INCLUDE as FILTER_ACTION_RISK). `rel_path` is
|
||||
* the entry's path relative to the transfer root ("" == root), `leaf` its final
|
||||
* name, `is_dir` whether it is a directory. */
|
||||
FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path,
|
||||
const char* leaf, bool is_dir, unsigned side);
|
||||
|
||||
/* Sender-side convenience wrapper (kept for callers/tests that only need the
|
||||
* transfer decision). */
|
||||
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
|
||||
bool is_dir);
|
||||
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
#include "format.h"
|
||||
#include "protocol.h"
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
|
||||
bool format_human_size_decimal(unsigned long long bytes, char* buffer, size_t buffer_size) {
|
||||
if (!buffer || buffer_size == 0)
|
||||
return false;
|
||||
if (bytes < 1000ULL) {
|
||||
int written = snprintf(buffer, buffer_size, "%llu", bytes);
|
||||
return written >= 0 && (size_t)written < buffer_size;
|
||||
}
|
||||
static const char units[] = "KMGTPE";
|
||||
double value = (double)bytes;
|
||||
size_t divisions = 0;
|
||||
while (value >= 1000.0 && divisions < sizeof(units) - 1) {
|
||||
value /= 1000.0;
|
||||
divisions++;
|
||||
}
|
||||
int written = snprintf(buffer, buffer_size, "%.2f%c", value, units[divisions - 1]);
|
||||
return written >= 0 && (size_t)written < buffer_size;
|
||||
}
|
||||
|
||||
bool format_big_num(unsigned long long value, bool human_readable, char* buffer,
|
||||
size_t buffer_size) {
|
||||
if (human_readable)
|
||||
return format_human_size_decimal(value, buffer, buffer_size);
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
size_t len = (size_t)written;
|
||||
size_t separators = len > 1 ? (len - 1) / 3 : 0;
|
||||
size_t total = len + separators;
|
||||
if (total + 1 > buffer_size)
|
||||
return false;
|
||||
size_t out = total;
|
||||
buffer[out] = '\0';
|
||||
size_t digits_since_sep = 0;
|
||||
for (size_t i = len; i > 0; i--) {
|
||||
buffer[--out] = digits[i - 1];
|
||||
digits_since_sep++;
|
||||
if (digits_since_sep == 3 && i > 1) {
|
||||
buffer[--out] = ',';
|
||||
digits_since_sep = 0;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool format_rsync_datetime(time_t when, bool dash, char* buffer, size_t buffer_size) {
|
||||
if (!buffer || buffer_size == 0)
|
||||
return false;
|
||||
struct tm broken_down;
|
||||
if (localtime_r(&when, &broken_down) == NULL)
|
||||
return false;
|
||||
const char* format = dash ? "%Y/%m/%d-%H:%M:%S" : "%Y/%m/%d %H:%M:%S";
|
||||
return strftime(buffer, buffer_size, format, &broken_down) != 0;
|
||||
}
|
||||
|
||||
bool format_dest_state_send(int fd, const OutputDestState* state) {
|
||||
if (!state)
|
||||
return false;
|
||||
int32_t has_old = state->existed ? 1 : 0;
|
||||
uint64_t size = (uint64_t)state->size;
|
||||
int64_t mtime = (int64_t)state->mtime_sec;
|
||||
int64_t mtime_nsec = state->mtime_nsec;
|
||||
uint32_t mode = state->mode;
|
||||
int32_t uid = state->uid;
|
||||
int32_t gid = state->gid;
|
||||
return send_n_data(fd, &has_old, sizeof(has_old)) && send_n_data(fd, &size, sizeof(size)) &&
|
||||
send_n_data(fd, &mtime, sizeof(mtime)) &&
|
||||
send_n_data(fd, &mtime_nsec, sizeof(mtime_nsec)) && send_n_data(fd, &mode, sizeof(mode)) &&
|
||||
send_n_data(fd, &uid, sizeof(uid)) && send_n_data(fd, &gid, sizeof(gid));
|
||||
}
|
||||
|
||||
bool format_dest_state_receive(int fd, OutputDestState* state) {
|
||||
if (!state)
|
||||
return false;
|
||||
int32_t has_old = 0;
|
||||
uint64_t size = 0;
|
||||
int64_t mtime = 0;
|
||||
int64_t mtime_nsec = 0;
|
||||
uint32_t mode = 0;
|
||||
int32_t uid = 0;
|
||||
int32_t gid = 0;
|
||||
if (!receive_n_data(fd, &has_old, sizeof(has_old)) || !receive_n_data(fd, &size, sizeof(size)) ||
|
||||
!receive_n_data(fd, &mtime, sizeof(mtime)) ||
|
||||
!receive_n_data(fd, &mtime_nsec, sizeof(mtime_nsec)) ||
|
||||
!receive_n_data(fd, &mode, sizeof(mode)) || !receive_n_data(fd, &uid, sizeof(uid)) ||
|
||||
!receive_n_data(fd, &gid, sizeof(gid)))
|
||||
return false;
|
||||
memset(state, 0, sizeof(*state));
|
||||
state->known = true;
|
||||
state->existed = has_old != 0;
|
||||
state->size = size;
|
||||
state->mtime_sec = mtime;
|
||||
state->mtime_nsec = mtime_nsec;
|
||||
state->mode = mode;
|
||||
state->uid = uid;
|
||||
state->gid = gid;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool format_stats_send(int fd, const ReceiverStats* stats) {
|
||||
if (!stats)
|
||||
return false;
|
||||
unsigned long long fields[8] = {
|
||||
stats->matched_data, stats->deleted_files, stats->would_delete_count, stats->literal_bytes,
|
||||
stats->created_reg, stats->created_dir, stats->created_link, stats->created_special,
|
||||
};
|
||||
return send_n_data(fd, fields, sizeof(fields));
|
||||
}
|
||||
|
||||
bool format_stats_receive(int fd, ReceiverStats* stats) {
|
||||
if (!stats)
|
||||
return false;
|
||||
unsigned long long fields[8] = {0};
|
||||
if (!receive_n_data(fd, fields, sizeof(fields)))
|
||||
return false;
|
||||
memset(stats, 0, sizeof(*stats));
|
||||
stats->matched_data = fields[0];
|
||||
stats->deleted_files = fields[1];
|
||||
stats->would_delete_count = fields[2];
|
||||
stats->literal_bytes = fields[3];
|
||||
stats->created_reg = fields[4];
|
||||
stats->created_dir = fields[5];
|
||||
stats->created_link = fields[6];
|
||||
stats->created_special = fields[7];
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
#ifndef FORMAT_H
|
||||
#define FORMAT_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <time.h>
|
||||
|
||||
/* Low-level output-formatting primitives shared by the change-event model
|
||||
* (change_list.c) and the transfer driver (client_send.c).
|
||||
*
|
||||
* The functions here are pure/string-level except for the STATUS_DEST_INFO
|
||||
* codec, which lets the receiver report the pre-transfer destination entry so
|
||||
* the sender can render rsync-accurate --itemize-changes / --out-format
|
||||
* columns (see protocol.h). */
|
||||
|
||||
/* Pre-transfer destination snapshot, reported by the receiver when the wire
|
||||
* config carries report_dest_info. `known` distinguishes "no report was
|
||||
* requested/received" from "the destination did not exist" (`existed == false`
|
||||
* with `known == true`). */
|
||||
typedef struct {
|
||||
bool known;
|
||||
bool existed;
|
||||
unsigned long long size;
|
||||
long long mtime_sec;
|
||||
long long mtime_nsec;
|
||||
uint32_t mode;
|
||||
int32_t uid;
|
||||
int32_t gid;
|
||||
} OutputDestState;
|
||||
|
||||
/* rsync's -h/--human-readable size (decimal, base 1000): integers below 1000
|
||||
* print verbatim; larger values use the largest unit that keeps the value
|
||||
* below 1000 (K/M/G/T/P/E) with exactly two decimals, so 1500000 -> "1.50M"
|
||||
* and 999999 -> "1000.00K" (matching rsync's human_num). Returns false when
|
||||
* the buffer is too small (nothing is written). */
|
||||
bool format_human_size_decimal(unsigned long long bytes, char* buffer, size_t buffer_size);
|
||||
|
||||
/* rsync's general number formatting (big_num). When `human_readable` is true
|
||||
* this is format_human_size_decimal; otherwise the integer is rendered with a
|
||||
* ',' thousands separator every three digits (rsync's separator in the C
|
||||
* locale). Returns false on an undersized buffer. */
|
||||
bool format_big_num(unsigned long long value, bool human_readable, char* buffer,
|
||||
size_t buffer_size);
|
||||
|
||||
/* rsync's %M/%t timestamp. When `dash` is true the separator between the date
|
||||
* and the time is '-' (the %M form: "YYYY/MM/DD-HH:MM:SS"); otherwise it is a
|
||||
* space (the %t form: "YYYY/MM/DD HH:MM:SS"). Local time. Returns false on a
|
||||
* bad time or an undersized buffer. */
|
||||
bool format_rsync_datetime(time_t when, bool dash, char* buffer, size_t buffer_size);
|
||||
|
||||
/* Fixed-width STATUS_DEST_INFO record codec (int32 has_old, uint64 size,
|
||||
* int64 mtime, int64 mtime_nsec, uint32 mode, int32 uid, int32 gid). The
|
||||
* status frame itself is sent/received by the caller. Returns false on I/O
|
||||
* failure. */
|
||||
bool format_dest_state_send(int fd, const OutputDestState* state);
|
||||
bool format_dest_state_receive(int fd, OutputDestState* state);
|
||||
|
||||
/* End-of-transfer receiver counters reported through STATUS_STATS (protocol
|
||||
* 2.25.0, extended in 2.28.0) when the wire config carries report_stats.
|
||||
* `would_delete_count` is the number of destination-relative paths the receiver
|
||||
* would have deleted in a -n/--dry-run --delete run; that many wire strings
|
||||
* immediately follow the fixed record (sent/read by the caller).
|
||||
*
|
||||
* Protocol 2.28.0 adds the receiver-observed counters the sender cannot see:
|
||||
* `literal_bytes` is the file data the receiver actually stored literally
|
||||
* (whole files plus the literal fragments of a delta) and the four `created_*`
|
||||
* counters split the destination entries the receiver newly created by type,
|
||||
* reproducing rsync's `Number of created files` breakdown and an exact
|
||||
* `Literal data` for a delta run. */
|
||||
typedef struct {
|
||||
unsigned long long matched_data;
|
||||
unsigned long long deleted_files;
|
||||
unsigned long long would_delete_count;
|
||||
unsigned long long literal_bytes;
|
||||
unsigned long long created_reg;
|
||||
unsigned long long created_dir;
|
||||
unsigned long long created_link;
|
||||
unsigned long long created_special;
|
||||
} ReceiverStats;
|
||||
|
||||
/* Fixed-width STATUS_STATS counter record. The status frame and the optional
|
||||
* would-delete path list are sent/received by the caller. Returns false on I/O
|
||||
* failure. */
|
||||
bool format_stats_send(int fd, const ReceiverStats* stats);
|
||||
bool format_stats_receive(int fd, ReceiverStats* stats);
|
||||
|
||||
/* Sender-side file-list accounting for rsync's `--stats` block. Filled while
|
||||
* the scan/send loops walk each entry: the flist counters describe every
|
||||
* scanned source entry (transferred or skipped), while the transferred/literal
|
||||
* counters describe only the regular files the receiver actually stored. The
|
||||
* type split lets the client print rsync's `Number of files` breakdown; the
|
||||
* receiver-only counters (matched data, deleted, created) come from
|
||||
* STATUS_STATS. */
|
||||
typedef struct {
|
||||
unsigned long long flist_reg;
|
||||
unsigned long long flist_dir;
|
||||
unsigned long long flist_link;
|
||||
unsigned long long flist_special;
|
||||
unsigned long long total_file_size; /* sum of entry sizes (link target len) */
|
||||
unsigned long long transferred_regular; /* regular files actually stored */
|
||||
unsigned long long transferred_file_size; /* source size of those files */
|
||||
/* Whole-file accuracy: the `--stats` "Literal data" row. The sender counts
|
||||
* the source size of every stored file, so a whole-file transfer matches
|
||||
* rsync. A delta run actually ships only the literal fragments of the diff
|
||||
* (the rest is matched/copied), so here the value is an upper bound, not
|
||||
* rsync's literal-byte total; see RSYNC_COMPAT.md's `--stats` row. */
|
||||
unsigned long long literal_data;
|
||||
} TransferStats;
|
||||
|
||||
#endif
|
||||
+479
-136
@@ -36,14 +36,34 @@ typedef struct {
|
||||
bool copy_as_set;
|
||||
int32_t copy_as_uid;
|
||||
int32_t copy_as_gid;
|
||||
/* -o/--owner and -g/--group: preserve the source owner/group through the
|
||||
* normal name/identity resolution path. Split out of the former
|
||||
* use_metadata bundle; unlike --numeric-ids/--chown/--usermap/--groupmap/-a
|
||||
* these are a preserve-source request, not an arbitrary client-chosen owner,
|
||||
* so they are tracked separately from the explicit ownership gate. */
|
||||
bool preserve_owner;
|
||||
bool preserve_group;
|
||||
/* --fake-super: when active the receiver must only RECORD the (resolved)
|
||||
* ownership in the reserved xattr, never perform a real chown. Snapshotted
|
||||
* so the fd-relative ownership helpers can suppress the chown without a
|
||||
* Config argument. */
|
||||
bool fake_super;
|
||||
bool set;
|
||||
} IdentityActive;
|
||||
|
||||
static IdentityActive g_identity;
|
||||
|
||||
static void identity_active_reset(void) {
|
||||
free(g_identity.usermap);
|
||||
free(g_identity.groupmap);
|
||||
if (g_identity.usermap) {
|
||||
for (int i = 0; i < g_identity.usermap_count; i++)
|
||||
free(g_identity.usermap[i].to_name);
|
||||
free(g_identity.usermap);
|
||||
}
|
||||
if (g_identity.groupmap) {
|
||||
for (int i = 0; i < g_identity.groupmap_count; i++)
|
||||
free(g_identity.groupmap[i].to_name);
|
||||
free(g_identity.groupmap);
|
||||
}
|
||||
g_identity.usermap = NULL;
|
||||
g_identity.groupmap = NULL;
|
||||
g_identity.usermap_count = 0;
|
||||
@@ -57,6 +77,9 @@ static void identity_active_reset(void) {
|
||||
g_identity.copy_as_set = false;
|
||||
g_identity.copy_as_uid = 0;
|
||||
g_identity.copy_as_gid = 0;
|
||||
g_identity.preserve_owner = false;
|
||||
g_identity.preserve_group = false;
|
||||
g_identity.fake_super = false;
|
||||
g_identity.set = false;
|
||||
}
|
||||
|
||||
@@ -77,32 +100,57 @@ bool identity_set_active(const Config* config) {
|
||||
g_identity.copy_as_set = config->copy_as_set;
|
||||
g_identity.copy_as_uid = config->copy_as_uid;
|
||||
g_identity.copy_as_gid = config->copy_as_gid;
|
||||
g_identity.preserve_owner = config->preserve_owner;
|
||||
g_identity.preserve_group = config->preserve_group;
|
||||
g_identity.fake_super = config->fake_super;
|
||||
if (config->usermap_count > 0) {
|
||||
g_identity.usermap = calloc((size_t)config->usermap_count, sizeof(IdentityMap));
|
||||
if (!g_identity.usermap)
|
||||
goto alloc_failed;
|
||||
memcpy(g_identity.usermap, config->usermap,
|
||||
(size_t)config->usermap_count * sizeof(IdentityMap));
|
||||
for (int i = 0; i < config->usermap_count; i++) {
|
||||
g_identity.usermap[i] = config->usermap[i];
|
||||
g_identity.usermap[i].to_name =
|
||||
config->usermap[i].to_name ? str_dup(config->usermap[i].to_name) : NULL;
|
||||
if (config->usermap[i].to_name && !g_identity.usermap[i].to_name) {
|
||||
g_identity.usermap_count = i; /* free only the entries already duplicated */
|
||||
goto alloc_failed;
|
||||
}
|
||||
}
|
||||
g_identity.usermap_count = config->usermap_count;
|
||||
}
|
||||
if (config->groupmap_count > 0) {
|
||||
g_identity.groupmap = calloc((size_t)config->groupmap_count, sizeof(IdentityMap));
|
||||
if (!g_identity.groupmap)
|
||||
goto alloc_failed;
|
||||
memcpy(g_identity.groupmap, config->groupmap,
|
||||
(size_t)config->groupmap_count * sizeof(IdentityMap));
|
||||
for (int i = 0; i < config->groupmap_count; i++) {
|
||||
g_identity.groupmap[i] = config->groupmap[i];
|
||||
g_identity.groupmap[i].to_name =
|
||||
config->groupmap[i].to_name ? str_dup(config->groupmap[i].to_name) : NULL;
|
||||
if (config->groupmap[i].to_name && !g_identity.groupmap[i].to_name) {
|
||||
g_identity.groupmap_count = i;
|
||||
goto alloc_failed;
|
||||
}
|
||||
}
|
||||
g_identity.groupmap_count = config->groupmap_count;
|
||||
}
|
||||
g_identity.set = true;
|
||||
/* A root receiver would honor any client-supplied ownership request (a
|
||||
--usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids).
|
||||
Surface that prominently; a privileged daemon applying arbitrary client
|
||||
ownership is a deliberate, opt-in choice the operator should be aware of. */
|
||||
if (geteuid() == 0)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"identity mapping active and running as root: client-supplied "
|
||||
"ownership (usermap/groupmap/chown/numeric-ids) will be honored; "
|
||||
"run the daemon as an unprivileged user unless intended");
|
||||
--usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids)
|
||||
ONLY when super-user activities are permitted. --no-super (or a daemon
|
||||
veto that forced SUPER_MODE_OFF) forbids the chown even for root, so do
|
||||
not claim the ownership will be honored in that case. */
|
||||
if (geteuid() == 0) {
|
||||
if (privilege_super_mode_permitted(g_identity.super_mode))
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"identity mapping active and running as root: client-supplied "
|
||||
"ownership (usermap/groupmap/chown/numeric-ids) will be honored; "
|
||||
"run the daemon as an unprivileged user unless intended");
|
||||
else
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"identity mapping active and running as root, but super-user activities are "
|
||||
"disabled (--no-super): requested ownership will NOT be applied; run the "
|
||||
"daemon as an unprivileged user unless intended");
|
||||
}
|
||||
/* --super explicitly requests super-user activities, but FastSync never
|
||||
elevates privileges: when the receiver is not already root the kernel will
|
||||
refuse those confined attempts and each is skipped per entry. Warn exactly
|
||||
@@ -138,25 +186,55 @@ bool privilege_super_mode_permitted(SuperMode mode) {
|
||||
}
|
||||
|
||||
bool identity_active_enabled(void) {
|
||||
/* numeric_ids is included: this set only gates identity_apply_ownership,
|
||||
which runs only when metadata is present (a -M/--preserve transfer). A
|
||||
standalone --numeric-ids (no ownership-affecting flag) carries no
|
||||
metadata, never reaches identity_apply_ownership, and therefore correctly
|
||||
stays inert; combined with -M it activates raw-id application. --super /
|
||||
--no-super does NOT enable ownership: it only permits or forbids the
|
||||
already-requested super-user activities, so a --super with no explicit
|
||||
identity flag must never silently apply client-chosen ownership. */
|
||||
/* --numeric-ids is deliberately NOT included: it is a mapping MODIFIER (use
|
||||
* the transmitted numeric id raw instead of a name lookup), not a request to
|
||||
* change ownership. rsync's --numeric-ids on its own never chowns anything;
|
||||
* it only changes how an already-requested -o/-g/map resolves. Ownership is
|
||||
* activated only by an explicit request: --chown/--usermap/--groupmap/
|
||||
* --copy-as or a preserve-source -o/--owner / -g/--group. --super/--no-super
|
||||
* likewise does NOT enable ownership: it only permits or forbids the
|
||||
* already-requested super-user activities. */
|
||||
return g_identity.set &&
|
||||
(g_identity.numeric_ids || g_identity.chown_uid_set || g_identity.chown_gid_set ||
|
||||
g_identity.usermap_count > 0 || g_identity.groupmap_count > 0 || g_identity.copy_as_set);
|
||||
(g_identity.chown_uid_set || g_identity.chown_gid_set || g_identity.usermap_count > 0 ||
|
||||
g_identity.groupmap_count > 0 || g_identity.copy_as_set || g_identity.preserve_owner ||
|
||||
g_identity.preserve_group);
|
||||
}
|
||||
|
||||
bool identity_owner_requested(void) {
|
||||
return g_identity.set && (g_identity.copy_as_set || g_identity.chown_uid_set ||
|
||||
g_identity.preserve_owner || g_identity.usermap_count > 0);
|
||||
}
|
||||
|
||||
bool identity_group_requested(void) {
|
||||
return g_identity.set && (g_identity.copy_as_set || g_identity.chown_gid_set ||
|
||||
g_identity.preserve_group || g_identity.groupmap_count > 0);
|
||||
}
|
||||
|
||||
bool identity_ownership_requested(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
/* Every value that makes the receiver act on a client-chosen owner, plus an
|
||||
* explicit --super (super-user device-node activities). Pure config, so the
|
||||
* daemon gate can evaluate it before identity_set_active(). */
|
||||
/* General-awareness predicate: every value that makes the receiver act on a
|
||||
* client-chosen owner, plus an explicit --super (super-user device-node
|
||||
* activities) and the preserve-source -o/-g requests. Pure config, so callers
|
||||
* can evaluate it before identity_set_active(). The daemon module gate uses
|
||||
* the narrower identity_explicit_ownership_requested() below, which treats a
|
||||
* plain -o/-g/-a as a preserve-source request rather than arbitrary
|
||||
* client-chosen ownership. */
|
||||
return config->numeric_ids || config->chown_uid_set || config->chown_gid_set ||
|
||||
config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set ||
|
||||
config->preserve_owner || config->preserve_group || config->fake_super ||
|
||||
config->super_mode == SUPER_MODE_ON;
|
||||
}
|
||||
|
||||
bool identity_explicit_ownership_requested(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
/* The narrow set the daemon gate refuses for a non-opted module: a request
|
||||
* that lets the CLIENT choose an arbitrary owner/group (rather than preserve
|
||||
* the source's own). Deliberately EXCLUDES preserve_owner/preserve_group so a
|
||||
* plain -a/-o/-g push is not refused; for those the gate instead forces
|
||||
* super-user ownership activity off (no chown happens) unless the module has
|
||||
* `client owner = yes`. */
|
||||
return config->numeric_ids || config->chown_uid_set || config->chown_gid_set ||
|
||||
config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set ||
|
||||
config->fake_super || config->super_mode == SUPER_MODE_ON;
|
||||
@@ -177,6 +255,29 @@ bool identity_copy_as_refused(const Config* config) {
|
||||
return geteuid() != 0 || config->super_mode == SUPER_MODE_OFF;
|
||||
}
|
||||
|
||||
/* Validate one received FROM:TO map rule. `from` is a single id, the LOW end
|
||||
* of an inclusive range, IDENTITY_MATCH_ANY, or IDENTITY_MATCH_UNNAMED; a
|
||||
* sentinel FROM must carry the same value in from_hi. `to` is a non-negative
|
||||
* id, IDENTITY_CURRENT, or ignored when a bounded receiver-resolved `to_name`
|
||||
* is present. */
|
||||
static bool identity_wire_map_valid(const IdentityMap* map) {
|
||||
if (!map)
|
||||
return false;
|
||||
if (map->from < IDENTITY_MATCH_UNNAMED)
|
||||
return false;
|
||||
if (map->from < 0) {
|
||||
if (map->from_hi != map->from)
|
||||
return false;
|
||||
} else if (map->from_hi < map->from) {
|
||||
return false;
|
||||
}
|
||||
if (map->to < IDENTITY_CURRENT)
|
||||
return false;
|
||||
if (map->to_name && strlen(map->to_name) > 255)
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool identity_wire_valid(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
@@ -188,11 +289,11 @@ bool identity_wire_valid(const Config* config) {
|
||||
if (config->chown_gid_set && config->chown_gid < IDENTITY_MATCH_ANY)
|
||||
return false;
|
||||
for (int i = 0; i < config->usermap_count; i++) {
|
||||
if (config->usermap[i].from < IDENTITY_MATCH_ANY || config->usermap[i].to < IDENTITY_CURRENT)
|
||||
if (!identity_wire_map_valid(&config->usermap[i]))
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < config->groupmap_count; i++) {
|
||||
if (config->groupmap[i].from < IDENTITY_MATCH_ANY || config->groupmap[i].to < IDENTITY_CURRENT)
|
||||
if (!identity_wire_map_valid(&config->groupmap[i]))
|
||||
return false;
|
||||
}
|
||||
/* Defense-in-depth: a --copy-as block must never carry a negative (sentinel)
|
||||
@@ -250,15 +351,137 @@ static int identity_resolve_token(const char* token, bool is_group, int32_t* out
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int identity_append_rule(IdentityMap** map, int* count, int32_t from, int32_t to) {
|
||||
static bool identity_all_digits(const char* token) {
|
||||
if (!token || *token == '\0')
|
||||
return false;
|
||||
for (const char* p = token; *p; p++)
|
||||
if (*p < '0' || *p > '9')
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool identity_token_has_glob(const char* token) {
|
||||
return token && (strchr(token, '*') || strchr(token, '?') || strchr(token, '['));
|
||||
}
|
||||
|
||||
/* Parse a --usermap/--groupmap FROM token into a matcher (from/from_hi). rsync
|
||||
* accepts a name, a numeric id, an inclusive LOW-HIGH range, '*' (any id), or an
|
||||
* empty token (ids with no name on the sender). Returns 0 on success, -1 on a
|
||||
* malformed token or an unresolvable sender-side name. */
|
||||
static int identity_parse_from(const char* token, bool is_group, int32_t* out_from,
|
||||
int32_t* out_hi) {
|
||||
if (token[0] == '\0') {
|
||||
*out_from = IDENTITY_MATCH_UNNAMED;
|
||||
*out_hi = IDENTITY_MATCH_UNNAMED;
|
||||
return 0;
|
||||
}
|
||||
if (strcmp(token, "*") == 0) {
|
||||
*out_from = IDENTITY_MATCH_ANY;
|
||||
*out_hi = IDENTITY_MATCH_ANY;
|
||||
return 0;
|
||||
}
|
||||
const char* num = token[0] == '@' ? token + 1 : token;
|
||||
if (identity_all_digits(num)) {
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_from = id;
|
||||
*out_hi = id;
|
||||
return 0;
|
||||
}
|
||||
/* An inclusive LOW-HIGH numeric range. */
|
||||
const char* dash = strchr(num, '-');
|
||||
if (dash && dash != num && dash[1] != '\0' && strchr(dash + 1, '-') == NULL) {
|
||||
size_t lo_len = (size_t)(dash - num);
|
||||
size_t hi_len = strlen(dash + 1);
|
||||
char low[16];
|
||||
char high[16];
|
||||
if (lo_len < sizeof(low) && hi_len < sizeof(high)) {
|
||||
memcpy(low, num, lo_len);
|
||||
low[lo_len] = '\0';
|
||||
memcpy(high, dash + 1, hi_len);
|
||||
high[hi_len] = '\0';
|
||||
if (identity_all_digits(low) && identity_all_digits(high)) {
|
||||
char* endptr = NULL;
|
||||
errno = 0;
|
||||
long lo = strtol(low, &endptr, 10);
|
||||
if (errno != 0 || !endptr || *endptr != '\0')
|
||||
return -1;
|
||||
errno = 0;
|
||||
long hi = strtol(high, &endptr, 10);
|
||||
if (errno != 0 || !endptr || *endptr != '\0' || hi < lo || hi > INT32_MAX)
|
||||
return -1;
|
||||
*out_from = (int32_t)lo;
|
||||
*out_hi = (int32_t)hi;
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
/* Not a numeric LOW-HIGH range: fall through and treat as a name (a
|
||||
* hyphenated account name like "wayne-smith" must still resolve). */
|
||||
}
|
||||
/* A sender-side name. A wildcard other than the bare '*' is matched by rsync
|
||||
* against the sender's names; because FastSync transmits numeric ids only, the
|
||||
* receiver cannot evaluate it, so reject rather than silently mis-match. */
|
||||
if (identity_token_has_glob(token)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%smap FROM '%s': name wildcards other than '*' are not supported "
|
||||
"(FastSync transmits numeric ids, so sender names are unavailable on the "
|
||||
"receiver)",
|
||||
is_group ? "--group" : "--user", token);
|
||||
return -1;
|
||||
}
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_from = id;
|
||||
*out_hi = id;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Parse a --usermap/--groupmap TO token. '*', a bare numeric id, or an @N id is
|
||||
* stored numerically; every other non-empty token is a NAME resolved on the
|
||||
* RECEIVER at apply time (rsync resolves TO names against the receiving side).
|
||||
* Returns 0 on success, -1 on an empty/malformed token. */
|
||||
static int identity_parse_to(const char* token, bool is_group, int32_t* out_to, char** out_name) {
|
||||
if (token[0] == '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "%smap TO value is missing", is_group ? "--group" : "--user");
|
||||
return -1;
|
||||
}
|
||||
if (strcmp(token, "*") == 0) {
|
||||
*out_to = IDENTITY_CURRENT;
|
||||
*out_name = NULL;
|
||||
return 0;
|
||||
}
|
||||
const char* num = token[0] == '@' ? token + 1 : token;
|
||||
if (identity_all_digits(num)) {
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_to = id;
|
||||
*out_name = NULL;
|
||||
return 0;
|
||||
}
|
||||
if (identity_token_has_glob(token)) {
|
||||
log_message(LOG_LEVEL_ERROR, "%smap TO '%s' may not contain a wildcard",
|
||||
is_group ? "--group" : "--user", token);
|
||||
return -1;
|
||||
}
|
||||
char* name = str_dup(token);
|
||||
if (!name)
|
||||
return -1;
|
||||
*out_to = 0;
|
||||
*out_name = name;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int identity_append_rule(IdentityMap** map, int* count, const IdentityMap* rule) {
|
||||
if (*count >= MAX_IDENTITY_MAP)
|
||||
return -1;
|
||||
IdentityMap* grown = realloc(*map, (size_t)(*count + 1) * sizeof(IdentityMap));
|
||||
if (!grown)
|
||||
return -1;
|
||||
*map = grown;
|
||||
(*map)[*count].from = from;
|
||||
(*map)[*count].to = to;
|
||||
(*map)[*count] = *rule;
|
||||
(*count)++;
|
||||
return 0;
|
||||
}
|
||||
@@ -275,7 +498,7 @@ int identity_parse_map(Config* config, const char* value, bool is_group) {
|
||||
char* saveptr = NULL;
|
||||
for (char* rule = strtok_r(list, ",", &saveptr); rule; rule = strtok_r(NULL, ",", &saveptr)) {
|
||||
char* colon = strchr(rule, ':');
|
||||
if (!colon || colon == rule) {
|
||||
if (!colon) {
|
||||
/* Log before freeing: `rule` points into the str_dup'd list. */
|
||||
log_message(LOG_LEVEL_ERROR, "%s rules must be FROM:TO (got '%s')", optname, rule);
|
||||
free(list);
|
||||
@@ -284,25 +507,25 @@ int identity_parse_map(Config* config, const char* value, bool is_group) {
|
||||
*colon = '\0';
|
||||
char* from_token = rule;
|
||||
char* to_token = colon + 1;
|
||||
if (*to_token == '\0') {
|
||||
IdentityMap parsed;
|
||||
memset(&parsed, 0, sizeof(parsed));
|
||||
if (identity_parse_from(from_token, is_group, &parsed.from, &parsed.from_hi) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%s could not resolve FROM '%s' in '%s' (a name must exist on the "
|
||||
"source; use @N for a numeric id)",
|
||||
optname, from_token, value);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR, "%s rule 'FROM:' is missing the TO value (got '%s')", optname,
|
||||
value);
|
||||
return -1;
|
||||
}
|
||||
int32_t from_id, to_id;
|
||||
if (identity_resolve_token(from_token, is_group, &from_id) != 0 ||
|
||||
identity_resolve_token(to_token, is_group, &to_id) != 0) {
|
||||
if (identity_parse_to(to_token, is_group, &parsed.to, &parsed.to_name) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s could not parse TO '%s' in '%s'", optname, to_token, value);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%s could not resolve '%s' (name must exist on the source; use "
|
||||
"@N for a numeric id)",
|
||||
optname, value);
|
||||
return -1;
|
||||
}
|
||||
if (identity_append_rule(is_group ? &config->groupmap : &config->usermap,
|
||||
is_group ? &config->groupmap_count : &config->usermap_count, from_id,
|
||||
to_id) != 0) {
|
||||
is_group ? &config->groupmap_count : &config->usermap_count,
|
||||
&parsed) != 0) {
|
||||
free(parsed.to_name);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR, "%s has too many rules (max %d)", optname, MAX_IDENTITY_MAP);
|
||||
return -1;
|
||||
@@ -366,6 +589,67 @@ static int identity_split_chown(const char* value, char** puser, char** pgroup)
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* --chown is rsync's shorthand for "--usermap=*:USER --groupmap=*:GROUP", so a
|
||||
* name TO value must be resolved on the RECEIVER, not on the sender. Append the
|
||||
* equivalent map rule (FROM matches every id). The numeric/'*' forms are stored
|
||||
* numerically exactly as rsync's id_parse/user_to_uid would. Returns 0 on
|
||||
* success, -1 on a malformed numeric token or allocation failure. */
|
||||
static int identity_append_chown_rule(Config* config, bool is_group, const char* token) {
|
||||
IdentityMap rule;
|
||||
memset(&rule, 0, sizeof(rule));
|
||||
rule.from = IDENTITY_MATCH_ANY;
|
||||
rule.from_hi = IDENTITY_MATCH_ANY;
|
||||
if (strcmp(token, "*") == 0) {
|
||||
rule.to = IDENTITY_CURRENT;
|
||||
} else if (identity_all_digits(token[0] == '@' ? token + 1 : token)) {
|
||||
if (identity_resolve_token(token, is_group, &rule.to) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown numeric id is out of range: %s", token);
|
||||
return -1;
|
||||
}
|
||||
} else {
|
||||
rule.to = 0;
|
||||
rule.to_name = str_dup(token);
|
||||
if (!rule.to_name)
|
||||
return -1;
|
||||
}
|
||||
if (identity_append_rule(is_group ? &config->groupmap : &config->usermap,
|
||||
is_group ? &config->groupmap_count : &config->usermap_count,
|
||||
&rule) != 0) {
|
||||
free(rule.to_name);
|
||||
log_message(LOG_LEVEL_ERROR, "--chown has too many rules (max %d)", MAX_IDENTITY_MAP);
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Resolve/record one --chown side. The source-side numeric value is kept in
|
||||
* chown_uid/chown_gid purely as a fallback (the appended map rule resolves the
|
||||
* name on the receiver and wins); a name that does not exist on the sender is
|
||||
* accepted and left to receiver-side resolution, matching rsync. */
|
||||
static int identity_parse_chown_side(Config* config, bool is_group, const char* token) {
|
||||
if (identity_append_chown_rule(config, is_group, token) != 0)
|
||||
return -1;
|
||||
bool numeric = identity_all_digits(token[0] == '@' ? token + 1 : token);
|
||||
int32_t resolved;
|
||||
if (identity_resolve_token(token, is_group, &resolved) == 0) {
|
||||
if (is_group) {
|
||||
config->chown_gid = resolved;
|
||||
config->chown_gid_set = true;
|
||||
} else {
|
||||
config->chown_uid = resolved;
|
||||
config->chown_uid_set = true;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
if (numeric) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown could not resolve numeric id '%s'", token);
|
||||
return -1;
|
||||
}
|
||||
/* Unknown sender-side name: rsync accepts it and resolves it (or warns) on
|
||||
* the receiver; do the same instead of failing the whole run. */
|
||||
return 0;
|
||||
}
|
||||
|
||||
int identity_parse_chown(Config* config, const char* value) {
|
||||
if (!config || !value || *value == '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown requires a value (USER:GROUP, USER, or :GROUP)");
|
||||
@@ -404,32 +688,18 @@ int identity_parse_chown(Config* config, const char* value) {
|
||||
if (*user == '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown requires a user or group (got '%s')", value);
|
||||
ret = -1;
|
||||
} else if (identity_resolve_token(user, false, &config->chown_uid) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--chown could not resolve user '%s' (use a name that exists "
|
||||
"on the source, '*', or @N)",
|
||||
value);
|
||||
} else if (identity_parse_chown_side(config, false, user) != 0) {
|
||||
ret = -1;
|
||||
} else {
|
||||
config->chown_uid_set = true;
|
||||
}
|
||||
} else {
|
||||
/* --chown=USER:GROUP, --chown=:GROUP, --chown=USER: */
|
||||
if (*user != '\0') {
|
||||
if (identity_resolve_token(user, false, &config->chown_uid) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown could not resolve user '%s'", value);
|
||||
ret = -1;
|
||||
goto done;
|
||||
}
|
||||
config->chown_uid_set = true;
|
||||
if (*user != '\0' && identity_parse_chown_side(config, false, user) != 0) {
|
||||
ret = -1;
|
||||
goto done;
|
||||
}
|
||||
if (*group != '\0') {
|
||||
if (identity_resolve_token(group, true, &config->chown_gid) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown could not resolve group '%s'", value);
|
||||
ret = -1;
|
||||
goto done;
|
||||
}
|
||||
config->chown_gid_set = true;
|
||||
if (*group != '\0' && identity_parse_chown_side(config, true, group) != 0) {
|
||||
ret = -1;
|
||||
goto done;
|
||||
}
|
||||
if (!*user && !*group) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown must set a user, a group, or both (got '%s')", value);
|
||||
@@ -567,9 +837,6 @@ int identity_parse_copy_as(Config* config, const char* value) {
|
||||
config->copy_as_set = true;
|
||||
config->copy_as_uid = uid;
|
||||
config->copy_as_gid = gid;
|
||||
/* Ownership application needs the metadata path (the source uid/gid must be
|
||||
* transmitted); imply it exactly like --chown/--usermap/--groupmap. */
|
||||
config->use_metadata = true;
|
||||
ret = 0;
|
||||
|
||||
done:
|
||||
@@ -580,34 +847,120 @@ done:
|
||||
|
||||
/* ---- Receiver-side ownership application ---- */
|
||||
|
||||
static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id,
|
||||
/* True when a map rule's FROM matcher accepts `id`. A sentinel FROM never
|
||||
* carries a range. IDENTITY_MATCH_UNNAMED mirrors rsync's empty FROM: it
|
||||
* matches only ids that have no name in the account database (rsync matches the
|
||||
* sender's names; FastSync transmits numeric ids only, so it approximates this
|
||||
* with the receiver's database -- documented in RSYNC_COMPAT.md). */
|
||||
static bool identity_map_from_matches(const IdentityMap* map, int32_t id, bool is_group) {
|
||||
if (map->from == IDENTITY_MATCH_ANY)
|
||||
return true;
|
||||
if (map->from == IDENTITY_MATCH_UNNAMED)
|
||||
return is_group ? (getgrgid((gid_t)id) == NULL) : (getpwuid((uid_t)id) == NULL);
|
||||
return id >= map->from && id <= map->from_hi;
|
||||
}
|
||||
|
||||
/* First matching rule wins. A rule whose TO is a receiver-side name resolves it
|
||||
* against the receiver's account database here; an unresolvable TO name is
|
||||
* skipped with a warning and the next rule is considered (rsync prints "Unknown
|
||||
* --usermap name on receiver" and leaves the id unmapped rather than aborting). */
|
||||
static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id, bool is_group,
|
||||
int32_t* out_to) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (map[i].from == IDENTITY_MATCH_ANY || map[i].from == source_id) {
|
||||
if (!identity_map_from_matches(&map[i], source_id, is_group))
|
||||
continue;
|
||||
if (map[i].to_name) {
|
||||
if (is_group) {
|
||||
struct group* gr = getgrnam(map[i].to_name);
|
||||
if (!gr) {
|
||||
log_message(LOG_LEVEL_WARNING, "Unknown --groupmap name on receiver: %s", map[i].to_name);
|
||||
continue;
|
||||
}
|
||||
*out_to = (int32_t)gr->gr_gid;
|
||||
} else {
|
||||
struct passwd* pw = getpwnam(map[i].to_name);
|
||||
if (!pw) {
|
||||
log_message(LOG_LEVEL_WARNING, "Unknown --usermap name on receiver: %s", map[i].to_name);
|
||||
continue;
|
||||
}
|
||||
*out_to = (int32_t)pw->pw_uid;
|
||||
}
|
||||
} else {
|
||||
*out_to = map[i].to;
|
||||
return true;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Resolve the owner side from the negotiated policy. Sets *out and returns
|
||||
* true when an owner-affecting request is active (a usermap, --chown USER, or
|
||||
* -o/--owner); returns false (leaving *out untouched) when the owner side is
|
||||
* not requested, so callers can pass (uid_t)-1 to fchown and leave it as-is.
|
||||
* --numeric-ids only changes the RESOLUTION (raw id instead of a name lookup);
|
||||
* it never makes the side requested. */
|
||||
static bool identity_resolve_owner(int32_t source_uid, uid_t* out) {
|
||||
if (!(g_identity.chown_uid_set || g_identity.preserve_owner || g_identity.usermap_count > 0))
|
||||
return false;
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, false,
|
||||
&target)) {
|
||||
*out = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target;
|
||||
} else if (g_identity.chown_uid_set) {
|
||||
*out = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
*out = (uid_t)source_uid;
|
||||
} else {
|
||||
/* Best-effort name mapping against the receiver's own database. When the
|
||||
* transmitted (numeric) id has no name here, fall back to the raw numeric id
|
||||
* so -o still preserves the source owner. */
|
||||
struct passwd* pw = getpwuid((uid_t)source_uid);
|
||||
if (pw) {
|
||||
const struct passwd* mapped = getpwnam(pw->pw_name);
|
||||
*out = mapped ? mapped->pw_uid : (uid_t)source_uid;
|
||||
} else {
|
||||
*out = (uid_t)source_uid;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Group-side counterpart of identity_resolve_owner(). */
|
||||
static bool identity_resolve_group(int32_t source_gid, gid_t* out) {
|
||||
if (!(g_identity.chown_gid_set || g_identity.preserve_group || g_identity.groupmap_count > 0))
|
||||
return false;
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, true,
|
||||
&target)) {
|
||||
*out = target == IDENTITY_CURRENT ? getegid() : (gid_t)target;
|
||||
} else if (g_identity.chown_gid_set) {
|
||||
*out = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
*out = (gid_t)source_gid;
|
||||
} else {
|
||||
struct group* gr = getgrgid((gid_t)source_gid);
|
||||
if (gr) {
|
||||
const struct group* mapped = getgrnam(gr->gr_name);
|
||||
*out = mapped ? mapped->gr_gid : (gid_t)source_gid;
|
||||
} else {
|
||||
*out = (gid_t)source_gid;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Resolve the target ownership from the negotiated policy against the entry's
|
||||
* current stat. Shared by the fd (regular file) and no-follow (symlink) apply
|
||||
* paths. Returns false when no side is to be changed. */
|
||||
static bool identity_resolve_targets(const struct stat* st, int32_t source_uid, int32_t source_gid,
|
||||
uid_t* out_uid, gid_t* out_gid) {
|
||||
bool set_uid = false;
|
||||
bool set_gid = false;
|
||||
uid_t uid = 0;
|
||||
gid_t gid = 0;
|
||||
|
||||
/* --copy-as (P7 Wave E) has the highest priority: it forces BOTH the owner
|
||||
* and group of every written entry to the requested ids, beating usermap /
|
||||
* groupmap / --chown / --numeric-ids and the best-effort name lookup. Only
|
||||
* skip when the entry already carries exactly those ids. */
|
||||
if (g_identity.copy_as_set) {
|
||||
uid = (uid_t)g_identity.copy_as_uid;
|
||||
gid = (gid_t)g_identity.copy_as_gid;
|
||||
uid_t uid = (uid_t)g_identity.copy_as_uid;
|
||||
gid_t gid = (gid_t)g_identity.copy_as_gid;
|
||||
if (st->st_uid == uid && st->st_gid == gid)
|
||||
return false;
|
||||
*out_uid = uid;
|
||||
@@ -615,67 +968,52 @@ static bool identity_resolve_targets(const struct stat* st, int32_t source_uid,
|
||||
return true;
|
||||
}
|
||||
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, &target)) {
|
||||
uid = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target;
|
||||
set_uid = true;
|
||||
} else if (g_identity.chown_uid_set) {
|
||||
uid = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid;
|
||||
set_uid = true;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
uid = (uid_t)source_uid;
|
||||
set_uid = true;
|
||||
} else {
|
||||
/* Best-effort name mapping against the receiver's own database: if the
|
||||
* transmitted (numeric) id resolves to a name present on this machine,
|
||||
* re-resolve it. On a shared-account host this is the identity operation;
|
||||
* when the id has no name here, the user side is left alone. */
|
||||
struct passwd* pw = getpwuid((uid_t)source_uid);
|
||||
if (pw) {
|
||||
const struct passwd* mapped = getpwnam(pw->pw_name);
|
||||
if (mapped) {
|
||||
uid = mapped->pw_uid;
|
||||
set_uid = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, &target)) {
|
||||
gid = target == IDENTITY_CURRENT ? getegid() : (gid_t)target;
|
||||
set_gid = true;
|
||||
} else if (g_identity.chown_gid_set) {
|
||||
gid = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid;
|
||||
set_gid = true;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
gid = (gid_t)source_gid;
|
||||
set_gid = true;
|
||||
} else {
|
||||
struct group* gr = getgrgid((gid_t)source_gid);
|
||||
if (gr) {
|
||||
const struct group* mapped = getgrnam(gr->gr_name);
|
||||
if (mapped) {
|
||||
gid = mapped->gr_gid;
|
||||
set_gid = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!set_uid && !set_gid)
|
||||
/* Each side is resolved independently: -o/-g and the explicit identity flags
|
||||
* request the owner/group respectively, and a side that is NOT requested must
|
||||
* be left exactly as it is (`-1` to fchown on that side). This is what lets
|
||||
* plain -g change only the group, or -o only the owner. */
|
||||
uid_t uid = (uid_t)-1;
|
||||
gid_t gid = (gid_t)-1;
|
||||
bool owner_requested = identity_resolve_owner(source_uid, &uid);
|
||||
bool group_requested = identity_resolve_group(source_gid, &gid);
|
||||
if (!owner_requested && !group_requested)
|
||||
return false;
|
||||
/* An unset side keeps the file's current id so the other side can change. */
|
||||
if (!set_uid)
|
||||
uid = st->st_uid;
|
||||
if (!set_gid)
|
||||
gid = st->st_gid;
|
||||
/* Only change ownership when the target differs (avoid needless syscalls and
|
||||
* any chance of clearing setuid/setgid on an already-correct entry). */
|
||||
if (st->st_uid == uid && st->st_gid == gid)
|
||||
|
||||
/* Only change ownership when a requested side actually differs (avoid
|
||||
* needless syscalls and any chance of clearing setuid/setgid on an
|
||||
* already-correct entry). */
|
||||
bool changed = (owner_requested && uid != st->st_uid) || (group_requested && gid != st->st_gid);
|
||||
if (!changed)
|
||||
return false;
|
||||
*out_uid = uid;
|
||||
*out_gid = gid;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* --fake-super storage resolution: the receiver records the ownership it WOULD
|
||||
* have applied. A requested side uses the resolved mapping (--copy-as /
|
||||
* usermap / --chown / -o/-g, with --numeric-ids as the raw-id modifier); a side
|
||||
* that was not requested keeps the source's own id, so a plain --fake-super run
|
||||
* records the source owner untouched. */
|
||||
void identity_resolve_storage_ids(int32_t source_uid, int32_t source_gid, uint32_t* out_uid,
|
||||
uint32_t* out_gid) {
|
||||
if (g_identity.copy_as_set) {
|
||||
*out_uid = (uint32_t)g_identity.copy_as_uid;
|
||||
*out_gid = (uint32_t)g_identity.copy_as_gid;
|
||||
return;
|
||||
}
|
||||
uid_t uid = (uid_t)source_uid;
|
||||
gid_t gid = (gid_t)source_gid;
|
||||
uid_t resolved_uid;
|
||||
gid_t resolved_gid;
|
||||
if (identity_resolve_owner(source_uid, &resolved_uid))
|
||||
uid = resolved_uid;
|
||||
if (identity_resolve_group(source_gid, &resolved_gid))
|
||||
gid = resolved_gid;
|
||||
*out_uid = (uint32_t)uid;
|
||||
*out_gid = (uint32_t)gid;
|
||||
}
|
||||
|
||||
static void identity_log_chown_failure(const char* what, uid_t uid, gid_t gid) {
|
||||
/* EPERM/EACCES are expected when the receiver is not privileged (e.g. the CI
|
||||
* `nobody` user): warn and continue, never abort the transfer. Any other
|
||||
@@ -710,8 +1048,12 @@ bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) {
|
||||
/* Ownership application is OFF unless the client requested an identity flag.
|
||||
* This is the controlled gate: a default (or plain -M) transfer never changes
|
||||
* ownership, byte-for-byte preserving FastSync's existing behavior. --no-super
|
||||
* additionally forbids it even when the receiver is root. */
|
||||
if (!identity_active_enabled() || !privilege_super_permitted() || fd < 0)
|
||||
* additionally forbids it even when the receiver is root. --fake-super never
|
||||
* performs a REAL chown: that would defeat the point of the flag (record the
|
||||
* source ownership on an unprivileged receiver for a later privileged
|
||||
* restore); the resolved ownership is stored in the reserved xattr instead by
|
||||
* fake_super_store_fd(). */
|
||||
if (!identity_active_enabled() || g_identity.fake_super || !privilege_super_permitted() || fd < 0)
|
||||
return true;
|
||||
struct stat st;
|
||||
if (fstat(fd, &st) != 0)
|
||||
@@ -731,7 +1073,8 @@ bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) {
|
||||
|
||||
bool identity_apply_ownership_link(int parent_fd, const char* leaf, int32_t source_uid,
|
||||
int32_t source_gid) {
|
||||
if (!identity_active_enabled() || !privilege_super_permitted() || parent_fd < 0 || !leaf)
|
||||
if (!identity_active_enabled() || g_identity.fake_super || !privilege_super_permitted() ||
|
||||
parent_fd < 0 || !leaf)
|
||||
return true;
|
||||
struct stat st;
|
||||
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0)
|
||||
|
||||
+35
-6
@@ -80,16 +80,37 @@ void identity_clear_active(void);
|
||||
* snapshot. Ownership stays OFF ("do not apply") for every transfer that
|
||||
* requests none of them, preserving FastSync's existing behavior. --super /
|
||||
* --no-super alone does NOT enable ownership; an explicit identity flag
|
||||
* (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) is required. */
|
||||
* (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) or a
|
||||
* preserve-source -o/--owner / -g/--group request is required. */
|
||||
bool identity_active_enabled(void);
|
||||
|
||||
/* Pure, config-only predicate: true when the client requested ANY
|
||||
* client-chosen ownership or super-user activity (--numeric-ids, --chown,
|
||||
* --usermap/--groupmap, --copy-as, --fake-super, or an explicit --super). Used
|
||||
* by the daemon module gate to decide whether a module's per-module opt-in is
|
||||
* required; it never reads the per-connection snapshot. */
|
||||
/* Per-side predicates over the ACTIVE per-connection snapshot (call
|
||||
* identity_set_active() first). They mirror the owner_requested /
|
||||
* group_requested conditions inside identity_resolve_targets() exactly, so
|
||||
* callers that must apply only one side (e.g. the --fake-super owner replay)
|
||||
* can pass (uid_t)-1 / (gid_t)-1 for the side that was NOT requested and leave
|
||||
* it untouched. The owner side is requested by --copy-as, --chown USER,
|
||||
* --numeric-ids, -o/--owner, or a non-empty --usermap; the group side by
|
||||
* --copy-as, --chown :GROUP, --numeric-ids, -g/--group, or a non-empty
|
||||
* --groupmap. */
|
||||
bool identity_owner_requested(void);
|
||||
bool identity_group_requested(void);
|
||||
|
||||
/* Pure, config-only predicate: true when the client requested ANY client-chosen
|
||||
* ownership or super-user activity (--numeric-ids, --chown, --usermap/--groupmap,
|
||||
* --copy-as, --fake-super, an explicit --super, or a preserve-source -o/-g).
|
||||
* General awareness only; the daemon module gate uses the narrower
|
||||
* identity_explicit_ownership_requested() below. Never reads the snapshot. */
|
||||
bool identity_ownership_requested(const Config* config);
|
||||
|
||||
/* Pure, config-only predicate for the narrow set that lets the CLIENT choose an
|
||||
* arbitrary owner/group: --numeric-ids, --chown, --usermap/--groupmap,
|
||||
* --copy-as, --fake-super, or an explicit --super. Deliberately EXCLUDES a
|
||||
* plain -o/--owner / -g/--group (or -a) preserve-source request, which the
|
||||
* daemon gate handles by forcing super-user ownership activity off rather than
|
||||
* refusing the whole transfer. Never reads the snapshot. */
|
||||
bool identity_explicit_ownership_requested(const Config* config);
|
||||
|
||||
/* Apply the negotiated ownership to an already-written file descriptor.
|
||||
* source_uid/source_gid are the transmitted numeric ids. Resolution order:
|
||||
* --copy-as (highest priority, forces both ids), then a matching
|
||||
@@ -106,6 +127,14 @@ bool identity_ownership_requested(const Config* config);
|
||||
* is active returns true. */
|
||||
bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid);
|
||||
|
||||
/* Resolve the ownership that --fake-super should RECORD in the reserved xattr
|
||||
* (rather than chown for real). A requested side (--copy-as / usermap /
|
||||
* --chown / -o / -g, with --numeric-ids as the raw-id modifier) yields the
|
||||
* resolved target; a side that was not requested keeps the transmitted source
|
||||
* id. Must be called after identity_set_active(). */
|
||||
void identity_resolve_storage_ids(int32_t source_uid, int32_t source_gid, uint32_t* out_uid,
|
||||
uint32_t* out_gid);
|
||||
|
||||
/* P7 Wave D: the no-follow (symlink) counterpart. Resolves the same
|
||||
* usermap/groupmap/chown/numeric-ids/copy-as policy but applies it with
|
||||
* fchownat(..., AT_SYMLINK_NOFOLLOW) so a symlink's own ownership is changed
|
||||
|
||||
+37
-2
@@ -13,7 +13,19 @@ typedef enum {
|
||||
LOG_DEBUG_PROTO = 1u << 1,
|
||||
LOG_DEBUG_PACK = 1u << 2,
|
||||
LOG_DEBUG_UTIL = 1u << 3,
|
||||
LOG_DEBUG_ALL = (1u << 4) - 1,
|
||||
/* rsync --debug categories that now map to a natural FastSync event:
|
||||
* flist (file-list scan progress), del (deletions), hash/deltasum
|
||||
* (whole-file hashing and delta-sum generation), recv (receiver
|
||||
* responses/signatures), filter (selection/exclusion decisions) and send
|
||||
* (files handed to the sender). Only emitted when the category is
|
||||
* explicitly enabled; a normal run stays silent. */
|
||||
LOG_DEBUG_FLIST = 1u << 4,
|
||||
LOG_DEBUG_DEL = 1u << 5,
|
||||
LOG_DEBUG_HASH = 1u << 6,
|
||||
LOG_DEBUG_RECV = 1u << 7,
|
||||
LOG_DEBUG_FILTER = 1u << 8,
|
||||
LOG_DEBUG_SEND = 1u << 9,
|
||||
LOG_DEBUG_ALL = (1u << 10) - 1,
|
||||
} LogDebugFlag;
|
||||
|
||||
typedef enum {
|
||||
@@ -21,7 +33,30 @@ typedef enum {
|
||||
LOG_INFO_MISC = 1u << 1,
|
||||
LOG_INFO_SKIP = 1u << 2,
|
||||
LOG_INFO_STATS = 1u << 3,
|
||||
LOG_INFO_ALL = LOG_INFO_COPY | LOG_INFO_MISC | LOG_INFO_SKIP | LOG_INFO_STATS,
|
||||
/* rsync categories that map to a FastSync event (emitted in rsync's line
|
||||
* format): del (deletions), remove (sender-side source removal), name
|
||||
* (transferred entry names), flist (file-list header), nonreg (skipped
|
||||
* non-regular files), progress (per-file progress). rsync's `backup`
|
||||
* category is accepted for CLI parity but stays silent: the receiver does the
|
||||
* backing-up and FastSync has no backup event to report from the sender. */
|
||||
LOG_INFO_DEL = 1u << 4,
|
||||
LOG_INFO_REMOVE = 1u << 5,
|
||||
LOG_INFO_NAME = 1u << 6,
|
||||
LOG_INFO_FLIST = 1u << 7,
|
||||
LOG_INFO_NONREG = 1u << 8,
|
||||
LOG_INFO_PROGRESS = 1u << 9,
|
||||
/* Marker for `--info=name2` and higher: also print rsync's
|
||||
"NAME is uptodate" line for entries the receiver already has. It rides in
|
||||
the info_level bitset (there is no separate Config field) and is never set
|
||||
by --info=all (which selects level 1). */
|
||||
LOG_INFO_NAME_UPTODATE = 1u << 10,
|
||||
/* --info=mount: print rsync's `[sender] skipping mount-point dir NAME` when
|
||||
* -xx/--one-file-system drops a mount-point directory (FastSync's client is
|
||||
* the sender). */
|
||||
LOG_INFO_MOUNT = 1u << 11,
|
||||
LOG_INFO_ALL = LOG_INFO_COPY | LOG_INFO_MISC | LOG_INFO_SKIP | LOG_INFO_STATS | LOG_INFO_DEL |
|
||||
LOG_INFO_REMOVE | LOG_INFO_NAME | LOG_INFO_FLIST | LOG_INFO_NONREG |
|
||||
LOG_INFO_PROGRESS | LOG_INFO_MOUNT,
|
||||
} LogInfoFlag;
|
||||
|
||||
void log_message(LogLevel log_level, const char* message, ...);
|
||||
|
||||
+109
-52
@@ -209,22 +209,58 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) {
|
||||
return m;
|
||||
}
|
||||
|
||||
static mode_t metadata_mode(const FileMetadata* metadata, mode_t current_mode,
|
||||
bool preserve_executability) {
|
||||
bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrPolicy policy,
|
||||
mode_t* out_mode) {
|
||||
const mode_t execute_bits = S_IXUSR | S_IXGRP | S_IXOTH;
|
||||
if (preserve_executability)
|
||||
return (current_mode & 0777 & ~execute_bits) | (metadata->mode & execute_bits);
|
||||
return metadata->mode & 0777 & ~(S_IWGRP | S_IWOTH);
|
||||
if (policy.perms) {
|
||||
/* rsync --perms copies the source's permission and special bits exactly,
|
||||
* including group/other write and setuid/setgid/sticky. The kernel may
|
||||
* still clear setgid when the receiver is not in the file's group; the
|
||||
* caller logs a failed chmod rather than silently masking the bits here. */
|
||||
*out_mode = source_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
return true;
|
||||
}
|
||||
if (policy.executability) {
|
||||
/* -E/--executability (rsync 3.4 rule): do NOT copy the source's execute
|
||||
* bits per class. If the source is executable at all, derive the execute
|
||||
* bits from the DESTINATION's own read bits (so a class that can read may
|
||||
* execute); otherwise clear every execute bit. This runs on the
|
||||
* destination-derived base (pre-existing dest mode, or source&~umask for a
|
||||
* new file), and leaves the special bits untouched. --perms wins when both
|
||||
* are set (handled above). */
|
||||
mode_t base = current_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
if (source_mode & 0111)
|
||||
*out_mode = base | ((base & 0444) >> 2);
|
||||
else
|
||||
*out_mode = base & ~execute_bits;
|
||||
return true;
|
||||
}
|
||||
/* Neither requested: no source mode is applied at all. */
|
||||
return false;
|
||||
}
|
||||
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool preserve_executability) {
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config) {
|
||||
FileAttrPolicy policy = {false, false, false, false};
|
||||
if (config) {
|
||||
policy.perms = config->preserve_perms;
|
||||
policy.times = config->preserve_times;
|
||||
policy.atimes = config->preserve_atimes;
|
||||
policy.executability = config->use_executability;
|
||||
}
|
||||
return policy;
|
||||
}
|
||||
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata, FileAttrPolicy policy) {
|
||||
if (metadata == NULL)
|
||||
return;
|
||||
struct stat current;
|
||||
mode_t current_mode = stat(path, ¤t) == 0 ? current.st_mode : 0;
|
||||
mode_t safe_mode = metadata_mode(metadata, current_mode, preserve_executability);
|
||||
if (chmod(path, safe_mode) != 0) {
|
||||
bool apply_mode = false;
|
||||
mode_t safe_mode = 0;
|
||||
if (policy.perms || policy.executability) {
|
||||
struct stat current;
|
||||
mode_t current_mode = stat(path, ¤t) == 0 ? current.st_mode : 0;
|
||||
apply_mode = metadata_mode_for_policy(metadata->mode, current_mode, policy, &safe_mode);
|
||||
}
|
||||
if (apply_mode && chmod(path, safe_mode) != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Failed to chmod %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
|
||||
@@ -232,14 +268,23 @@ void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
}
|
||||
/* Never apply client-supplied ownership. The descriptor API below is the
|
||||
receiver write path; retain this legacy API only for compatibility. */
|
||||
struct timespec times[2];
|
||||
times[0].tv_sec = 0;
|
||||
times[0].tv_nsec = UTIME_OMIT;
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
if (metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
if (policy.times || (policy.atimes && metadata->atime_valid)) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
if (utimensat(AT_FDCWD, path, times, 0) != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
|
||||
free(escaped_path);
|
||||
}
|
||||
}
|
||||
if (metadata->crtime_valid) {
|
||||
log_message(LOG_LEVEL_DEBUG,
|
||||
@@ -247,16 +292,10 @@ void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
"setter exists",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec, path);
|
||||
}
|
||||
if (utimensat(AT_FDCWD, path, times, 0) != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
|
||||
free(escaped_path);
|
||||
}
|
||||
}
|
||||
|
||||
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool omit_link_times) {
|
||||
FileAttrPolicy policy, bool omit_link_times) {
|
||||
if (path == NULL || metadata == NULL)
|
||||
return !identity_copy_as_active();
|
||||
char* leaf = NULL;
|
||||
@@ -269,18 +308,25 @@ bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadat
|
||||
best-effort. */
|
||||
bool owned = identity_apply_ownership_link(parent_fd, leaf, (int32_t)metadata->uid,
|
||||
(int32_t)metadata->gid);
|
||||
/* Symlink mode: not settable on Linux (fchmodat AT_SYMLINK_NOFOLLOW returns
|
||||
EOPNOTSUPP/ENOTSUP); attempt it for platforms that support it and quietly
|
||||
ignore the unsupported case so the transfer never fails over it. */
|
||||
mode_t link_mode = metadata->mode & 0777;
|
||||
if (fchmodat(parent_fd, leaf, link_mode, AT_SYMLINK_NOFOLLOW) != 0 && errno != EOPNOTSUPP &&
|
||||
errno != ENOTSUP && errno != ENOSYS) {
|
||||
log_message(LOG_LEVEL_DEBUG, "Could not set symlink mode on %s: %s", path, strerror(errno));
|
||||
/* Symlink mode: only when -p is in effect. It is not settable on Linux
|
||||
(fchmodat AT_SYMLINK_NOFOLLOW returns EOPNOTSUPP/ENOTSUP); attempt it for
|
||||
platforms that support it and quietly ignore the unsupported case so the
|
||||
transfer never fails over it. */
|
||||
if (policy.perms) {
|
||||
mode_t link_mode = metadata->mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
if (fchmodat(parent_fd, leaf, link_mode, AT_SYMLINK_NOFOLLOW) != 0 && errno != EOPNOTSUPP &&
|
||||
errno != ENOTSUP && errno != ENOSYS) {
|
||||
log_message(LOG_LEVEL_DEBUG, "Could not set symlink mode on %s: %s", path, strerror(errno));
|
||||
}
|
||||
}
|
||||
if (!omit_link_times) {
|
||||
if (!omit_link_times && (policy.times || (policy.atimes && metadata->atime_valid))) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
|
||||
if (metadata->atime_valid) {
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
@@ -296,19 +342,13 @@ bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadat
|
||||
return owned;
|
||||
}
|
||||
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability) {
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, FileAttrPolicy policy) {
|
||||
if (fd < 0 || metadata == NULL)
|
||||
return metadata == NULL;
|
||||
bool ok = true;
|
||||
struct stat current;
|
||||
if (fstat(fd, ¤t) != 0)
|
||||
return false;
|
||||
mode_t safe_mode = metadata_mode(metadata, current.st_mode, preserve_executability);
|
||||
if (fchmod(fd, safe_mode) != 0)
|
||||
ok = false;
|
||||
/* Client uid/gid values are deliberately not authoritative UNLESS the client
|
||||
explicitly opted in with an identity flag (--numeric-ids / --usermap /
|
||||
--groupmap / --chown). identity_apply_ownership is the controlled,
|
||||
--groupmap / --chown / -o/-g). identity_apply_ownership is the controlled,
|
||||
privilege-gated path: it consults the negotiated policy, resolves the
|
||||
target ids, and applies them via an fd-relative fchown() that is confined
|
||||
to the just-written file (EPERM/EACCES are logged, never fatal) -- EXCEPT
|
||||
@@ -316,14 +356,19 @@ bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserv
|
||||
marks this entry as failed instead of reporting a wrong-owner write as
|
||||
success. With no identity flag set it is a no-op, so a default or plain -M
|
||||
transfer keeps FastSync's existing behavior of never applying client
|
||||
ownership. */
|
||||
ownership. Ownership runs BEFORE the mode because a chown clears
|
||||
setuid/setgid; rsync likewise chowns first and then restores the source
|
||||
mode (including its special bits). */
|
||||
if (!identity_apply_ownership(fd, (int32_t)metadata->uid, (int32_t)metadata->gid))
|
||||
ok = false;
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
|
||||
if (metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
if (policy.perms || policy.executability) {
|
||||
struct stat current;
|
||||
if (fstat(fd, ¤t) != 0)
|
||||
return false;
|
||||
mode_t safe_mode = 0;
|
||||
bool apply_mode = metadata_mode_for_policy(metadata->mode, current.st_mode, policy, &safe_mode);
|
||||
if (apply_mode && fchmod(fd, safe_mode) != 0)
|
||||
ok = false;
|
||||
}
|
||||
/* --crtimes captures and transmits the source birth time, but there is no
|
||||
* portable way to set a birth time (utimensat can only set atime/mtime), so
|
||||
@@ -335,7 +380,19 @@ bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserv
|
||||
"crtime (birth time) %lld.%09ld transmitted but not applied: no portable setter",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec);
|
||||
}
|
||||
if (futimens(fd, times) != 0)
|
||||
ok = false;
|
||||
if (policy.times || (policy.atimes && metadata->atime_valid)) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
if (futimens(fd, times) != 0)
|
||||
ok = false;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
+20
-7
@@ -2,6 +2,7 @@
|
||||
#define METADATA_H
|
||||
|
||||
#include "file.h"
|
||||
#include "file_attr.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
@@ -49,19 +50,31 @@ void metadata_to_buf(char** buf, const FileMetadata* m);
|
||||
FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len);
|
||||
bool metadata_send(int file_descriptor, const FileMetadata* m);
|
||||
FileMetadata* metadata_receive(int file_descriptor, int* ok);
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool preserve_executability);
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability);
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata, FileAttrPolicy policy);
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, FileAttrPolicy policy);
|
||||
|
||||
/* Shared mode-policy helper: the single source of truth for the receiver's
|
||||
* mode rule. Given a source mode and the destination's CURRENT mode, returns
|
||||
* true and stores the exact mode to apply in *out_mode when `policy` requests
|
||||
* a change, or false when it requests neither --perms nor --executability (the
|
||||
* caller then leaves the destination mode alone). --perms wins over -E; the
|
||||
* -E rule derives exec bits from the destination's read bits (rsync 3.4);
|
||||
* group/other write is never granted from a client-supplied mode. Shared by
|
||||
* file_restore_metadata_fd() and the --fake-super replay so the two cannot
|
||||
* diverge. */
|
||||
bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrPolicy policy,
|
||||
mode_t* out_mode);
|
||||
/* P7 Wave D: apply a SYMLINK's own metadata using no-follow primitives only
|
||||
* (utimensat/lchown/fchmodat with AT_SYMLINK_NOFOLLOW), confined fd-relative
|
||||
* under the authorized root. `omit_link_times` (-J/--omit-link-times)
|
||||
* suppresses the timestamps; the link's mode/ownership are still attempted
|
||||
* (ownership stays gated by the identity policy and by default is not applied).
|
||||
* under the authorized root. The link's mode is applied only when policy.perms;
|
||||
* policy.times (further suppressed by `omit_link_times` for -J) applies the
|
||||
* mtime with policy.atimes controlling the atime slot; ownership stays gated by
|
||||
* the identity policy and by default is not applied.
|
||||
* A null metadata or an unfollowable parent is a harmless no-op. Returns false
|
||||
* only when a REQUIRED --copy-as ownership application failed, so the caller can
|
||||
* report the entry as failed instead of claiming a wrong-owner success. */
|
||||
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool omit_link_times);
|
||||
FileAttrPolicy policy, bool omit_link_times);
|
||||
|
||||
/* Compare timestamps using rsync's whole-second modification window. */
|
||||
bool metadata_mtime_matches(time_t left_sec, long left_nsec, time_t right_sec, long right_nsec,
|
||||
|
||||
@@ -30,20 +30,28 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que
|
||||
context->max_queue_bytes = 0;
|
||||
context->manifest = NULL;
|
||||
context->excluded_paths = NULL;
|
||||
context->size_skipped_paths = NULL;
|
||||
context->synced_dirs = NULL;
|
||||
context->plan_dirs = NULL;
|
||||
context->missing_args = NULL;
|
||||
context->scan_had_io_error = false;
|
||||
context->remove_source_files = NULL;
|
||||
context->early_delete = false;
|
||||
context->delete_plans = NULL;
|
||||
context->delete_suppressed = false;
|
||||
context->scan_stopped_early = false;
|
||||
context->total_files = 0;
|
||||
context->progress_bytes = 0;
|
||||
context->total_bytes = 0;
|
||||
memset(&context->stats, 0, sizeof(context->stats));
|
||||
context->sender_done = false;
|
||||
atomic_init(&context->cancelled, false);
|
||||
protocol_session_init(&context->allocation_session, -1, -1);
|
||||
protocol_session_set_max_alloc(&context->allocation_session, config->max_alloc);
|
||||
context->dir_entries = NULL;
|
||||
context->dir_entries_mutex_init = false;
|
||||
atomic_init(&context->dir_count, 0);
|
||||
context->delete_limit = false;
|
||||
int init = 0;
|
||||
if (config->use_metadata) {
|
||||
context->dir_entries = array_list_create(file_destroy);
|
||||
@@ -184,8 +192,16 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) {
|
||||
if (context->manifest) {
|
||||
array_list_delete(context->manifest);
|
||||
}
|
||||
if (context->delete_plans)
|
||||
delete_plan_sender_destroy(context->delete_plans);
|
||||
if (context->excluded_paths)
|
||||
array_list_delete(context->excluded_paths);
|
||||
if (context->size_skipped_paths)
|
||||
array_list_delete(context->size_skipped_paths);
|
||||
if (context->synced_dirs)
|
||||
array_list_delete(context->synced_dirs);
|
||||
if (context->plan_dirs)
|
||||
array_list_delete(context->plan_dirs);
|
||||
if (context->missing_args)
|
||||
array_list_delete(context->missing_args);
|
||||
if (context->remove_source_files)
|
||||
|
||||
@@ -7,7 +7,9 @@
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "format.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "stop_condition.h"
|
||||
@@ -42,6 +44,23 @@ typedef struct {
|
||||
scanner's exclusion sink) or, in the early modes, by the path-only pre-scan
|
||||
on the calling thread before the pipeline starts. */
|
||||
ArrayList* excluded_paths;
|
||||
/* --max-size/--min-size pruned source paths. These are ALWAYS sent as
|
||||
protected prefixes (even with --delete-excluded), so the destination
|
||||
mirrors of size-skipped files survive --delete like rsync. Populated by
|
||||
the scanner thread (workers append under mutex_scanner) or, in the early
|
||||
modes, by the path-only pre-scan on the calling thread. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Destination-relative paths of the directories the source scan synchronized
|
||||
for this run (the receive root is the "." sentinel). Sent with the
|
||||
manifest so the receiver confines its extras walk to them, matching rsync's
|
||||
"delete only in synchronized directories" (notably for --files-from).
|
||||
Populated by the scanner thread or the early pre-scan. */
|
||||
ArrayList* synced_dirs;
|
||||
/* Destination-relative paths of every traversed source directory, for the
|
||||
per-directory delete plan keep set (so an empty source directory survives
|
||||
--delete rather than being removed as an extra). Prebuilt by the path-only
|
||||
pre-scan on the calling thread. */
|
||||
ArrayList* plan_dirs;
|
||||
/* --delete-missing-args: the destination-relative mirrors of the --files-from
|
||||
entries that are missing under the source. Computed by the preflight on
|
||||
the calling thread before the pipeline starts; the sender thread transmits
|
||||
@@ -54,15 +73,32 @@ typedef struct {
|
||||
--ignore-errors kept the run going. */
|
||||
bool scan_had_io_error;
|
||||
ArrayList* remove_source_files;
|
||||
/* True when --delete-before/--delete-during require the keep-set manifest to
|
||||
be transmitted before any file data: context->manifest is then prebuilt by
|
||||
a path-only pre-scan on the calling thread and the pipeline scanner must
|
||||
not append to it. Set once before the worker threads start. */
|
||||
/* True when --delete-before requires the whole-tree keep-set manifest to be
|
||||
transmitted before any file data: context->manifest is then prebuilt by a
|
||||
path-only pre-scan on the calling thread and the pipeline scanner must not
|
||||
append to it. Set once before the worker threads start. */
|
||||
bool early_delete;
|
||||
/* Non-NULL for --delete-during/--delete-delay: the per-directory plan set
|
||||
prebuilt by the path-only pre-scan on the calling thread. The sender
|
||||
thread transmits the root plan before any data and the remaining plans
|
||||
alongside the chunks. Set once before the worker threads start. */
|
||||
DeletePlanSender* delete_plans;
|
||||
/* A scan I/O error without --ignore-errors suppressed deletion: the prebuilt
|
||||
keep-set/plans were dropped, and the streaming scanner must not build a
|
||||
fresh manifest or re-send the per-directory plans. Set once before the
|
||||
worker threads start. */
|
||||
bool delete_suppressed;
|
||||
mtx_t mutex_progress;
|
||||
int total_files;
|
||||
unsigned long long progress_bytes;
|
||||
unsigned long long total_bytes;
|
||||
/* Per-type flist / transferred accounting for the rsync --stats breakdown and
|
||||
the progress `to-chk` denominator. Owned by the sender thread: it is the
|
||||
only writer (the entry/transfer notes in send_chunks_multithreaded) and it
|
||||
reads the totals in its completion tail, so no lock is needed. This is NOT
|
||||
guarded by mutex_progress (which covers total_files/progress_bytes/
|
||||
total_bytes/sender_done). */
|
||||
TransferStats stats;
|
||||
bool sender_done;
|
||||
atomic_bool cancelled;
|
||||
ProtocolSession allocation_session;
|
||||
@@ -83,6 +119,14 @@ typedef struct {
|
||||
ArrayList* dir_entries;
|
||||
mtx_t dir_entries_mutex;
|
||||
bool dir_entries_mutex_init;
|
||||
/* --stats directory accounting for a `-r` scan (no directory metadata):
|
||||
shared by the parallel scanner workers, read by the sender thread once the
|
||||
scanner is done. See ScannerOptions.dir_count. */
|
||||
atomic_ullong dir_count;
|
||||
/* Set by the sender thread when the receiver reported a --max-delete-capped
|
||||
deletion (STATUS_DELETE_LIMIT): the transfer succeeded and the process must
|
||||
exit 25 like rsync. Read by the caller after the sender thread is joined. */
|
||||
bool delete_limit;
|
||||
} PipelineContextSender;
|
||||
|
||||
/* `config` is borrowed and must outlive the context: destroy does NOT free it,
|
||||
|
||||
+94
-25
@@ -12,8 +12,7 @@
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define RECEIVE_TIMEOUT_SEC 60 /* 60 second per-message timeout */
|
||||
#define SEND_TIMEOUT_SEC 60
|
||||
#define RECEIVE_TIMEOUT_SEC 60 /* built-in fallback for explicit -timed calls only */
|
||||
|
||||
static __thread int io_read_fd = -1;
|
||||
static __thread int io_write_fd = -1;
|
||||
@@ -30,6 +29,13 @@ static unsigned long long io_bwlimit = 0;
|
||||
static mtx_t bw_mutex;
|
||||
static once_flag bw_mutex_once = ONCE_FLAG_INIT;
|
||||
|
||||
/* Process-wide wire byte counters, used by the client to render rsync's
|
||||
* --stats/--progress totals and the --out-format %b/%c tokens. The zero-copy
|
||||
* sendfile path bypasses protocol_send_n_data, so it reports its bytes through
|
||||
* protocol_note_bytes_written. */
|
||||
static atomic_ullong io_bytes_written = 0;
|
||||
static atomic_ullong io_bytes_read = 0;
|
||||
|
||||
static unsigned long long global_bwlimit(void);
|
||||
|
||||
static bool protocol_reserve_memory(ProtocolSession* session, size_t charge) {
|
||||
@@ -99,8 +105,14 @@ void protocol_session_set_io_timeout(ProtocolSession* session, int sec) {
|
||||
|
||||
int protocol_get_io_timeout_sec(void) {
|
||||
const ProtocolSession* session = bound_session ? bound_session : &legacy_io_session;
|
||||
int sec = session->io_timeout_sec;
|
||||
return sec > 0 ? sec : RECEIVE_TIMEOUT_SEC;
|
||||
/* 0 (or negative) means the session timeout is disabled, matching rsync's
|
||||
* --timeout=0 default. Callers must treat a non-positive result as "wait
|
||||
* without a deadline" instead of substituting a built-in window. */
|
||||
return session->io_timeout_sec > 0 ? session->io_timeout_sec : 0;
|
||||
}
|
||||
|
||||
int protocol_server_io_timeout_sec(int client_timeout) {
|
||||
return client_timeout > 0 ? client_timeout : SERVER_IO_TIMEOUT_SEC;
|
||||
}
|
||||
|
||||
void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc) {
|
||||
@@ -110,7 +122,8 @@ void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long
|
||||
}
|
||||
|
||||
static bool allocation_allowed(const ProtocolSession* session, size_t size) {
|
||||
return (unsigned long long)size <= session->max_alloc;
|
||||
/* max_alloc == 0 is rsync's --max-alloc=0 "no limit". */
|
||||
return session->max_alloc == 0 || (unsigned long long)size <= session->max_alloc;
|
||||
}
|
||||
|
||||
static void* protocol_alloc_for_session(const ProtocolSession* session, size_t size) {
|
||||
@@ -170,12 +183,28 @@ void io_set_bwlimit(unsigned long long bytes_per_sec) {
|
||||
mtx_unlock(&bw_mutex);
|
||||
}
|
||||
|
||||
unsigned long long io_get_bwlimit(void) {
|
||||
return global_bwlimit();
|
||||
}
|
||||
|
||||
/* rsync's throttle (io.c sleep_for_bwlimit) sleeps once its unslept debt
|
||||
* reaches ~100 ms of bandwidth, so its effective initial burst is about 0.1 s
|
||||
* worth of bytes, not a full second. FastSync models the same with a token
|
||||
* bucket whose capacity is bwlimit/10, so a throttled run paces like rsync
|
||||
* instead of sending a full second's worth up front. */
|
||||
static long long bw_burst_capacity(unsigned long long bwlimit) {
|
||||
if (bwlimit == 0)
|
||||
return 0;
|
||||
long long burst = (long long)(bwlimit / 10);
|
||||
return burst > 0 ? burst : 1;
|
||||
}
|
||||
|
||||
void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec) {
|
||||
if (!session)
|
||||
return;
|
||||
session->bwlimit =
|
||||
bytes_per_sec > (unsigned long long)LLONG_MAX ? (unsigned long long)LLONG_MAX : bytes_per_sec;
|
||||
session->bw_tokens = (long long)session->bwlimit;
|
||||
session->bw_tokens = bw_burst_capacity(session->bwlimit);
|
||||
struct timespec now;
|
||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||
session->bw_last_refill_sec = now.tv_sec;
|
||||
@@ -209,8 +238,9 @@ static void bw_throttle_session(ProtocolSession* session, size_t bytes_written)
|
||||
|
||||
long long tokens_to_add = (long long)((double)session->bwlimit * elapsed_ns / 1000000000.0);
|
||||
session->bw_tokens += tokens_to_add;
|
||||
if (session->bw_tokens > (long long)session->bwlimit)
|
||||
session->bw_tokens = (long long)session->bwlimit;
|
||||
long long burst = bw_burst_capacity(session->bwlimit);
|
||||
if (session->bw_tokens > burst)
|
||||
session->bw_tokens = burst;
|
||||
|
||||
session->bw_tokens -= bytes_written;
|
||||
|
||||
@@ -221,7 +251,11 @@ static void bw_throttle_session(ProtocolSession* session, size_t bytes_written)
|
||||
poll(NULL, 0, (int)(deficit_us / 1000));
|
||||
else
|
||||
usleep((useconds_t)deficit_us);
|
||||
/* Reset the bucket AFTER the sleep: crediting the sleep duration as elapsed
|
||||
refill time would cancel half the throttle (the next call would see the
|
||||
whole sleep as refill and immediately grant a fresh burst). */
|
||||
session->bw_tokens = 0;
|
||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||
session->bw_last_refill_sec = now.tv_sec;
|
||||
session->bw_last_refill_nsec = now.tv_nsec;
|
||||
}
|
||||
@@ -236,6 +270,18 @@ SSL* io_get_ssl(void) {
|
||||
return io_ssl;
|
||||
}
|
||||
|
||||
unsigned long long protocol_bytes_written(void) {
|
||||
return atomic_load(&io_bytes_written);
|
||||
}
|
||||
|
||||
unsigned long long protocol_bytes_read(void) {
|
||||
return atomic_load(&io_bytes_read);
|
||||
}
|
||||
|
||||
void protocol_note_bytes_written(unsigned long long bytes) {
|
||||
atomic_fetch_add(&io_bytes_written, bytes);
|
||||
}
|
||||
|
||||
static ProtocolSession* legacy_session(int read_fd, int write_fd) {
|
||||
if (bound_session)
|
||||
return bound_session;
|
||||
@@ -280,11 +326,15 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
log_debug_message(LOG_DEBUG_IO, " Sending n Data: %zu", data_size);
|
||||
if (!session)
|
||||
return false;
|
||||
int timeout_sec = session->io_timeout_sec > 0 ? session->io_timeout_sec : SEND_TIMEOUT_SEC;
|
||||
/* A non-positive session timeout disables the deadline entirely (rsync's
|
||||
* --timeout=0 default); poll then blocks until the socket becomes writable. */
|
||||
int timeout_sec = session->io_timeout_sec > 0 ? session->io_timeout_sec : 0;
|
||||
int fd = session->write_fd;
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
if (timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
}
|
||||
short wait_events = POLLOUT;
|
||||
ssize_t total_bytes_send = 0;
|
||||
while ((size_t)total_bytes_send < data_size) {
|
||||
@@ -292,7 +342,7 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
if (session->bwlimit > 0 && chunk > 65536)
|
||||
chunk = 65536;
|
||||
struct pollfd pfd = {.fd = fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, deadline_remaining_ms(&deadline));
|
||||
int poll_result = poll(&pfd, 1, timeout_sec > 0 ? deadline_remaining_ms(&deadline) : -1);
|
||||
if (poll_result == 0 || (poll_result < 0 && errno != EINTR)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Send timeout or poll failure");
|
||||
return false;
|
||||
@@ -333,17 +383,27 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
wait_events = POLLOUT;
|
||||
}
|
||||
log_debug_message(LOG_DEBUG_IO, " Send n Data: %zu", total_bytes_send);
|
||||
atomic_fetch_add(&io_bytes_written, (unsigned long long)total_bytes_send);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size,
|
||||
int timeout_sec);
|
||||
static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, size_t data_size,
|
||||
const struct timespec* deadline);
|
||||
|
||||
bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_size) {
|
||||
/* Honor the session's configured deadline; protocol_receive_n_data_timed
|
||||
* re-applies the built-in 60 s default when the value is <= 0. */
|
||||
int timeout_sec = session ? session->io_timeout_sec : 0;
|
||||
return protocol_receive_n_data_timed(session, data, data_size, timeout_sec);
|
||||
/* Honor the session's configured deadline. A non-positive value disables the
|
||||
* deadline (rsync's --timeout=0 default): wait without a poll timeout. The
|
||||
* explicit _timed variants keep their own 0 -> built-in-default contract. */
|
||||
if (!session)
|
||||
return false;
|
||||
if (session->io_timeout_sec <= 0)
|
||||
return protocol_receive_n_data_until(session, data, data_size, NULL);
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += session->io_timeout_sec;
|
||||
return protocol_receive_n_data_until(session, data, data_size, &deadline);
|
||||
}
|
||||
|
||||
/* Read exactly `data_size` bytes from `session` before `deadline` elapses
|
||||
@@ -353,7 +413,7 @@ bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_s
|
||||
static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, size_t data_size,
|
||||
const struct timespec* deadline) {
|
||||
log_debug_message(LOG_DEBUG_IO, " Receiving n Data: %zu", data_size);
|
||||
if (!session || !deadline)
|
||||
if (!session)
|
||||
return false;
|
||||
int fd = session->read_fd;
|
||||
|
||||
@@ -362,7 +422,8 @@ static bool protocol_receive_n_data_until(ProtocolSession* session, void* data,
|
||||
while (total_bytes_received < data_size) {
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
struct pollfd pfd = {.fd = fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, deadline_remaining_ms(deadline));
|
||||
/* A NULL deadline means "wait indefinitely" (timeout disabled). */
|
||||
int poll_result = poll(&pfd, 1, deadline ? deadline_remaining_ms(deadline) : -1);
|
||||
if (poll_result == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout");
|
||||
return false;
|
||||
@@ -410,6 +471,7 @@ static bool protocol_receive_n_data_until(ProtocolSession* session, void* data,
|
||||
wait_events = POLLIN;
|
||||
}
|
||||
log_debug_message(LOG_DEBUG_IO, " Received n Data: %zu", total_bytes_received);
|
||||
atomic_fetch_add(&io_bytes_read, (unsigned long long)total_bytes_received);
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -479,6 +541,10 @@ static const char* status_to_string(Status status) {
|
||||
return "ERROR_DETAIL";
|
||||
case STATUS_DRY_RUN_TRANSFER:
|
||||
return "DRY_RUN_TRANSFER";
|
||||
case STATUS_DELETE_LIMIT:
|
||||
return "DELETE_LIMIT";
|
||||
case STATUS_DEST_INFO:
|
||||
return "DEST_INFO";
|
||||
default:
|
||||
return "UNKNOWN";
|
||||
}
|
||||
@@ -702,13 +768,16 @@ static bool protocol_capture_error_detail(ProtocolSession* session, Status* stat
|
||||
bool protocol_receive_status(ProtocolSession* session, Status* status) {
|
||||
if (!session || !status)
|
||||
return false;
|
||||
int timeout_sec = session->io_timeout_sec > 0 ? session->io_timeout_sec : RECEIVE_TIMEOUT_SEC;
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
if (!protocol_receive_n_data_until(session, status, sizeof(Status), &deadline))
|
||||
const struct timespec* deadline_ptr = NULL;
|
||||
if (session->io_timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += session->io_timeout_sec;
|
||||
deadline_ptr = &deadline;
|
||||
}
|
||||
if (!protocol_receive_n_data_until(session, status, sizeof(Status), deadline_ptr))
|
||||
return false;
|
||||
if (!protocol_capture_error_detail(session, status, &deadline, NULL))
|
||||
if (!protocol_capture_error_detail(session, status, deadline_ptr, NULL))
|
||||
return false;
|
||||
log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status));
|
||||
return true;
|
||||
@@ -747,8 +816,8 @@ static bool protocol_read_status_until(ProtocolSession* session, Status* status,
|
||||
short wait_events = POLLIN;
|
||||
while (got < sizeof(Status)) {
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
int remaining_ms = deadline_remaining_ms(deadline);
|
||||
if (remaining_ms <= 0) {
|
||||
int remaining_ms = deadline ? deadline_remaining_ms(deadline) : -1;
|
||||
if (remaining_ms == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status");
|
||||
return false;
|
||||
}
|
||||
|
||||
+75
-12
@@ -34,6 +34,11 @@
|
||||
#define DEFAULT_MAX_ALLOC (1ULL * 1024 * 1024 * 1024)
|
||||
/* Server policy ceiling for a client-provided allocation limit. */
|
||||
#define MAX_SERVER_ALLOC (256ULL * 1024 * 1024)
|
||||
/* Server-owned floor for the per-message I/O deadline. A client --timeout=0
|
||||
(rsync's default) disables the client's own deadlines, but a server session
|
||||
must never be held open forever by a silent peer (slow-loris), so the server
|
||||
floors the effective deadline at this value. */
|
||||
#define SERVER_IO_TIMEOUT_SEC 60
|
||||
/* Bounded cumulative per-connection receive budget. In-flight wire buffers,
|
||||
decompression buffers and queued (not yet written) file payloads for a
|
||||
connection must stay within this ceiling. */
|
||||
@@ -59,10 +64,12 @@ typedef struct ProtocolSession {
|
||||
bool eight_bit_output;
|
||||
unsigned long long max_alloc;
|
||||
/* Per-session deadline (seconds) applied to every protocol send/receive by
|
||||
* protocol_send_n_data / protocol_receive_n_data. Defaults to the built-in
|
||||
* 60 s window; a value <= 0 falls back to that default. Set from the
|
||||
* negotiated Config->timeout so --timeout is honored by the poll()-driven
|
||||
* protocol I/O, not just the socket SO_RCVTIMEO/SO_SNDTIMEO. */
|
||||
* protocol_send_n_data / protocol_receive_n_data. The initialized default is
|
||||
* the built-in 60 s window; a value <= 0 disables the deadline (rsync's
|
||||
* --timeout=0). Set from the negotiated Config->timeout so --timeout is
|
||||
* honored by the poll()-driven protocol I/O, not just the socket
|
||||
* SO_RCVTIMEO/SO_SNDTIMEO. The server does not propagate a client 0 here: it
|
||||
* installs protocol_server_io_timeout_sec() so its sessions keep a floor. */
|
||||
int io_timeout_sec;
|
||||
} ProtocolSession;
|
||||
|
||||
@@ -155,14 +162,65 @@ enum NET_STATUS {
|
||||
* (the receiver reads none in dry-run). STATUS_OK keeps its meaning in this
|
||||
* path ("already up to date / nothing to do"). Appended after
|
||||
* STATUS_ERROR_DETAIL so no existing status is renumbered. */
|
||||
STATUS_DRY_RUN_TRANSFER
|
||||
STATUS_DRY_RUN_TRANSFER,
|
||||
/* --max-delete budget exhausted (protocol 2.23.0). Sent by the receiver as
|
||||
* the terminal success status INSTEAD of STATUS_OK when a --delete/
|
||||
* --delete-missing-args commit removed up to the --max-delete bound but had
|
||||
* to skip further extras. The transfer itself succeeded and all file data is
|
||||
* stored; the sender maps this to rsync's exit code 25 ("the --max-delete
|
||||
* limit stopped deletions"). Appended after STATUS_DRY_RUN_TRANSFER so no
|
||||
* existing status is renumbered. */
|
||||
STATUS_DELETE_LIMIT,
|
||||
/* Destination-state report for output parity (protocol 2.23.0). When the
|
||||
* wire config carries report_dest_info=true, the receiver answers every
|
||||
* per-file STATUS_CHECK request with STATUS_DEST_INFO FIRST, followed by a
|
||||
* fixed record describing the pre-transfer destination entry
|
||||
* (int32 has_old; uint64 size; int64 mtime; int64 mtime_nsec; uint32 mode;
|
||||
* int32 uid; int32 gid). The ordinary STATUS_OK/STATUS_NEXT/... verdict
|
||||
* follows, so the sender can render rsync-accurate -i/--out-format columns
|
||||
* (new vs modified, and which of size/time/perms/owner/group differ) without
|
||||
* changing the transfer decision itself. Appended after
|
||||
* STATUS_DELETE_LIMIT so no existing status is renumbered. */
|
||||
STATUS_DEST_INFO,
|
||||
/* Per-directory delete plan (protocol 2.24.0). The sender of a
|
||||
* --delete-during/--delete-delay transfer streams one frame per source
|
||||
* directory in directory order instead of a single whole-tree keep-set
|
||||
* manifest. The receiver applies the plan when it arrives
|
||||
* (--delete-during removes that directory's extras immediately) or records
|
||||
* the extras and applies them only after the whole transfer succeeded
|
||||
* (--delete-delay). Payload: an int32 has_config flag (1 on the first plan
|
||||
* of the run, 0 afterwards); when set, the three global config sections
|
||||
* (protected-prefix count+paths, size-skipped count+paths, missing-args
|
||||
* count+paths); then an int32 apply flag (1 for a real plan, 0 for a
|
||||
* config-only carrier frame that must not walk a directory); then the
|
||||
* destination-relative directory path wire string
|
||||
* ("." for the receive root); then the child-directory count + names and the
|
||||
* child-file count + names that must be kept. Appended after
|
||||
* STATUS_DEST_INFO so no existing status is renumbered. */
|
||||
STATUS_DELETE_PLAN,
|
||||
/* End-of-transfer receiver counter report (protocol 2.25.0). When the wire
|
||||
* config carries report_stats=true, the receiver sends this status once,
|
||||
* immediately before its terminal success status, followed by a fixed stats
|
||||
* record (see format_stats_send/receive in format.h) and, when the run is a
|
||||
* --dry-run with --delete, the would-delete path list. Appended after
|
||||
* STATUS_DELETE_PLAN so no existing status is renumbered. */
|
||||
STATUS_STATS
|
||||
};
|
||||
|
||||
void io_set_fds(int read_fd, int write_fd);
|
||||
void io_set_bwlimit(unsigned long long bytes_per_sec);
|
||||
unsigned long long io_get_bwlimit(void);
|
||||
void io_set_ssl(SSL* ssl);
|
||||
SSL* io_get_ssl(void);
|
||||
|
||||
/* Process-wide wire byte counters. protocol_send_n_data/protocol_receive_n_data
|
||||
* update them; the zero-copy sendfile path reports through
|
||||
* protocol_note_bytes_written. Used by the client to render rsync's
|
||||
* --stats/--progress totals and the --out-format %b/%c tokens. */
|
||||
unsigned long long protocol_bytes_written(void);
|
||||
unsigned long long protocol_bytes_read(void);
|
||||
void protocol_note_bytes_written(unsigned long long bytes);
|
||||
|
||||
void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd);
|
||||
/* Transitional bridge for helpers whose signatures still carry only an fd. */
|
||||
void protocol_session_bind(ProtocolSession* session);
|
||||
@@ -170,15 +228,20 @@ void protocol_session_unbind(void);
|
||||
void protocol_session_set_ssl(ProtocolSession* session, SSL* ssl);
|
||||
void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec);
|
||||
void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc);
|
||||
/* Override the per-message send/receive deadline for this session.
|
||||
* `sec` <= 0 restores the built-in 60 s default (used for --timeout=0/unset).
|
||||
* An explicit long deadline (e.g. the delete-ack wait) is applied per-call by
|
||||
* protocol_receive_status_timed and is unaffected by this setter. */
|
||||
/* Override the per-message send/receive deadline for this session. The value
|
||||
* is stored verbatim: a positive value sets the deadline, `sec` <= 0 disables
|
||||
* it (rsync's --timeout=0). An explicit long deadline (e.g. the delete-ack
|
||||
* wait) is applied per-call by protocol_receive_status_timed and is unaffected
|
||||
* by this setter. */
|
||||
void protocol_session_set_io_timeout(ProtocolSession* session, int sec);
|
||||
/* Effective per-message I/O deadline (seconds) for the currently-bound session,
|
||||
* falling back to the built-in default. Used by the plaintext sendfile path
|
||||
* which bypasses the protocol send primitive. */
|
||||
/* Effective per-message I/O deadline (seconds) for the currently-bound session.
|
||||
* Zero means the deadline is disabled (rsync's --timeout=0). Used by the
|
||||
* plaintext sendfile path which bypasses the protocol send primitive. */
|
||||
int protocol_get_io_timeout_sec(void);
|
||||
/* The server-side effective deadline for a client-requested timeout: a positive
|
||||
* client value is honored, otherwise the SERVER_IO_TIMEOUT_SEC floor applies so
|
||||
* a silent peer can never hold a session open forever. */
|
||||
int protocol_server_io_timeout_sec(int client_timeout);
|
||||
void* protocol_alloc(size_t size);
|
||||
void* protocol_realloc(void* ptr, size_t size);
|
||||
void protocol_session_set_8_bit_output(ProtocolSession* session, bool enabled);
|
||||
|
||||
@@ -139,6 +139,23 @@ void* queue_dequeue(Queue* queue) {
|
||||
return item;
|
||||
}
|
||||
|
||||
bool queue_push(Queue* queue, void* item) {
|
||||
return queue_enqueue(queue, item);
|
||||
}
|
||||
|
||||
void* queue_pop(Queue* queue) {
|
||||
if (queue == NULL || queue_is_empty(queue)) {
|
||||
log_perror("ERROR: Could not pop from null or empty queue.");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
queue->rear = (queue->rear - 1 + queue->capacity) % queue->capacity;
|
||||
void* item = queue->items[queue->rear];
|
||||
queue->items[queue->rear] = NULL;
|
||||
queue->size--;
|
||||
return item;
|
||||
}
|
||||
|
||||
void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty,
|
||||
cnd_t* condition_not_full, const bool* other_thread_done) {
|
||||
mtx_lock(mutex);
|
||||
|
||||
@@ -28,4 +28,11 @@ void* queue_dequeue(Queue* queue);
|
||||
void* queue_dequeue_multithreaded(Queue* queue, mtx_t* mutex, cnd_t* condition_not_empty,
|
||||
cnd_t* condition_not_full, const bool* other_thread_done);
|
||||
|
||||
/* LIFO stack operations over the same ring buffer. queue_push() is the enqueue
|
||||
primitive; queue_pop() removes from the rear, so a sequence of pushes is
|
||||
returned in reverse order. Used by the sequential scanner's depth-first
|
||||
traversal. */
|
||||
bool queue_push(Queue* queue, void* item);
|
||||
void* queue_pop(Queue* queue);
|
||||
|
||||
#endif
|
||||
|
||||
+181
-21
@@ -44,6 +44,181 @@ static bool parse_two_digits(const char* s, int* out) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* True when the current character of the cursor is a decimal digit. */
|
||||
static bool is_digit(const char* cp) {
|
||||
return *cp >= '0' && *cp <= '9';
|
||||
}
|
||||
|
||||
/* rsync 3.4.1's flexible --stop-at date parser (ported from
|
||||
* options.c:parse_time). Returns a time_t, or (time_t)-1 on a malformed value.
|
||||
* Accepted forms include Y-M-DTh:m, Y/M/DTh:m, Y-M-D, M-D, D, h:m, :m and
|
||||
* "T h:m"; a 1- or 2-digit year and omitted fields are resolved to the next
|
||||
* matching point in time in the local timezone. Seconds are NOT accepted
|
||||
* (rsync rejects them too); FastSync keeps its own HH:MM:SS spelling as an
|
||||
* extension handled by the caller. `now` is passed in so tests are
|
||||
* deterministic; production passes time(NULL). */
|
||||
static time_t parse_time_rsync(const char* value, time_t now) {
|
||||
const char* cp;
|
||||
time_t val;
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return (time_t)-1;
|
||||
struct tm t;
|
||||
int in_date, old_mday, n;
|
||||
|
||||
memset(&t, 0, sizeof t);
|
||||
t.tm_year = t.tm_mon = t.tm_mday = -1;
|
||||
t.tm_hour = t.tm_min = t.tm_isdst = -1;
|
||||
cp = value;
|
||||
if (*cp == 'T' || *cp == 't' || *cp == ':') {
|
||||
in_date = *cp == ':' ? 0 : -1;
|
||||
cp++;
|
||||
} else
|
||||
in_date = 1;
|
||||
for (;; cp++) {
|
||||
if (!is_digit(cp))
|
||||
return (time_t)-1;
|
||||
n = 0;
|
||||
do {
|
||||
n = n * 10 + *cp++ - '0';
|
||||
} while (is_digit(cp));
|
||||
if (*cp == ':')
|
||||
in_date = 0;
|
||||
if (in_date > 0) {
|
||||
if (t.tm_year != -1)
|
||||
return (time_t)-1;
|
||||
t.tm_year = t.tm_mon;
|
||||
t.tm_mon = t.tm_mday;
|
||||
t.tm_mday = n;
|
||||
if (!*cp)
|
||||
break;
|
||||
if (*cp == 'T' || *cp == 't') {
|
||||
if (!cp[1])
|
||||
break;
|
||||
in_date = -1;
|
||||
} else if (*cp != '-' && *cp != '/')
|
||||
return (time_t)-1;
|
||||
continue;
|
||||
}
|
||||
if (t.tm_hour != -1)
|
||||
return (time_t)-1;
|
||||
t.tm_hour = t.tm_min;
|
||||
t.tm_min = n;
|
||||
if (!*cp) {
|
||||
if (in_date < 0)
|
||||
return (time_t)-1;
|
||||
break;
|
||||
}
|
||||
if (*cp != ':')
|
||||
return (time_t)-1;
|
||||
in_date = 0;
|
||||
}
|
||||
|
||||
in_date = 0;
|
||||
if (t.tm_year < 0) {
|
||||
t.tm_year = today.tm_year;
|
||||
in_date = 1;
|
||||
} else if (t.tm_year < 100) {
|
||||
while (t.tm_year < today.tm_year)
|
||||
t.tm_year += 100;
|
||||
} else
|
||||
t.tm_year -= 1900;
|
||||
if (t.tm_mon < 0) {
|
||||
t.tm_mon = today.tm_mon;
|
||||
in_date = 2;
|
||||
} else
|
||||
t.tm_mon--;
|
||||
if (t.tm_mday < 0) {
|
||||
t.tm_mday = today.tm_mday;
|
||||
in_date = 3;
|
||||
}
|
||||
|
||||
n = 0;
|
||||
if (t.tm_min < 0) {
|
||||
t.tm_hour = t.tm_min = 0;
|
||||
} else if (t.tm_hour < 0) {
|
||||
if (in_date != 3)
|
||||
return (time_t)-1;
|
||||
in_date = 0;
|
||||
t.tm_hour = today.tm_hour;
|
||||
n = 60 * 60;
|
||||
}
|
||||
|
||||
/* mktime() may roll a too-large tm_mday into the following month; undo that
|
||||
* in the "next match" loop below. */
|
||||
old_mday = t.tm_mday;
|
||||
if (t.tm_hour > 23 || t.tm_min > 59 || t.tm_mon < 0 || t.tm_mon >= 12 || t.tm_mday < 1 ||
|
||||
t.tm_mday > 31 || (val = mktime(&t)) == (time_t)-1)
|
||||
return (time_t)-1;
|
||||
|
||||
while (in_date && (val <= now || t.tm_mday < old_mday)) {
|
||||
switch (in_date) {
|
||||
case 3:
|
||||
old_mday = ++t.tm_mday;
|
||||
break;
|
||||
case 2:
|
||||
if (t.tm_mday < old_mday)
|
||||
t.tm_mday = old_mday; /* the month already got bumped forward */
|
||||
else if (++t.tm_mon == 12) {
|
||||
t.tm_mon = 0;
|
||||
t.tm_year++;
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
if (t.tm_mday < old_mday) {
|
||||
/* mon==1 mday==29 got bumped to mon==2 */
|
||||
if (t.tm_mon != 2 || old_mday != 29)
|
||||
return (time_t)-1;
|
||||
t.tm_mon = 1;
|
||||
t.tm_mday = 29;
|
||||
}
|
||||
t.tm_year++;
|
||||
break;
|
||||
}
|
||||
if ((val = mktime(&t)) == (time_t)-1) {
|
||||
if (in_date != 3 || t.tm_mday <= 28)
|
||||
return (time_t)-1;
|
||||
t.tm_mday = old_mday = 1;
|
||||
in_date = 2;
|
||||
}
|
||||
}
|
||||
if (n) {
|
||||
while (val <= now)
|
||||
val += n;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
/* FastSync's HH:MM or HH:MM:SS spelling on the current local day. rsync's own
|
||||
* --stop-at accepts only HH:MM, so this is a strict superset extension. */
|
||||
static bool parse_clock_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
size_t len = strlen(value);
|
||||
if (len != 5 && len != 8)
|
||||
return false;
|
||||
if (value[2] != ':' || (len == 8 && value[5] != ':'))
|
||||
return false;
|
||||
int hh, mm, ss = 0;
|
||||
if (!parse_two_digits(value, &hh) || !parse_two_digits(value + 3, &mm))
|
||||
return false;
|
||||
if (len == 8 && !parse_two_digits(value + 6, &ss))
|
||||
return false;
|
||||
if (hh > 23 || mm > 59 || ss > 59)
|
||||
return false;
|
||||
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return false;
|
||||
today.tm_hour = hh;
|
||||
today.tm_min = mm;
|
||||
today.tm_sec = ss;
|
||||
today.tm_isdst = -1;
|
||||
time_t deadline = mktime(&today);
|
||||
if (deadline == (time_t)-1)
|
||||
return false;
|
||||
*out_deadline = deadline;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
if (!value || !out_deadline)
|
||||
return false;
|
||||
@@ -91,28 +266,13 @@ bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* HH:MM or HH:MM:SS on the current local day. */
|
||||
size_t len = strlen(value);
|
||||
if (len != 5 && len != 8)
|
||||
return false;
|
||||
if (value[2] != ':' || (len == 8 && value[5] != ':'))
|
||||
return false;
|
||||
int hh, mm, ss = 0;
|
||||
if (!parse_two_digits(value, &hh) || !parse_two_digits(value + 3, &mm))
|
||||
return false;
|
||||
if (len == 8 && !parse_two_digits(value + 6, &ss))
|
||||
return false;
|
||||
if (hh > 23 || mm > 59 || ss > 59)
|
||||
return false;
|
||||
/* HH:MM or HH:MM:SS on the current local day (FastSync extension). */
|
||||
if (parse_clock_time(value, now, out_deadline))
|
||||
return true;
|
||||
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return false;
|
||||
today.tm_hour = hh;
|
||||
today.tm_min = mm;
|
||||
today.tm_sec = ss;
|
||||
today.tm_isdst = -1;
|
||||
time_t deadline = mktime(&today);
|
||||
/* rsync's full/partial date-and-time form (e.g. 2000-12-31T23:59, 12-31,
|
||||
* 14:00, :59, 1, 1-30). */
|
||||
time_t deadline = parse_time_rsync(value, now);
|
||||
if (deadline == (time_t)-1)
|
||||
return false;
|
||||
*out_deadline = deadline;
|
||||
|
||||
+19
-11
@@ -289,14 +289,14 @@ void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
accept_loop(server, child_fn, child_ctx, log_fmt);
|
||||
}
|
||||
|
||||
static int g_timeout_sec = 30;
|
||||
static int g_contimeout_sec = 10;
|
||||
/* rsync defaults: --timeout=0 (disabled) and --contimeout=60. A non-positive
|
||||
* value means "no timeout" rather than "leave the built-in value in place". */
|
||||
static int g_timeout_sec = 0;
|
||||
static int g_contimeout_sec = 60;
|
||||
|
||||
void tcp_set_timeouts(int timeout_sec, int contimeout_sec) {
|
||||
if (timeout_sec > 0)
|
||||
g_timeout_sec = timeout_sec;
|
||||
if (contimeout_sec > 0)
|
||||
g_contimeout_sec = contimeout_sec;
|
||||
g_timeout_sec = timeout_sec > 0 ? timeout_sec : 0;
|
||||
g_contimeout_sec = contimeout_sec > 0 ? contimeout_sec : 0;
|
||||
}
|
||||
|
||||
int tcp_get_contimeout_sec(void) {
|
||||
@@ -308,6 +308,10 @@ int tcp_get_timeout_sec(void) {
|
||||
}
|
||||
|
||||
static void tcp_apply_socket_timeout(int fd) {
|
||||
/* timeout 0 means no timeout: leave the socket in its default (blocking)
|
||||
* mode instead of installing a zero SO_RCVTIMEO/SO_SNDTIMEO. */
|
||||
if (g_timeout_sec <= 0)
|
||||
return;
|
||||
struct timeval tv;
|
||||
tv.tv_sec = g_timeout_sec;
|
||||
tv.tv_usec = 0;
|
||||
@@ -483,11 +487,15 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
break;
|
||||
}
|
||||
|
||||
struct timeval ct;
|
||||
ct.tv_sec = g_contimeout_sec;
|
||||
ct.tv_usec = 0;
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &ct, sizeof(ct));
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &ct, sizeof(ct));
|
||||
/* --contimeout=0 disables the connect timeout: skip the pre-connect socket
|
||||
* timeouts entirely. */
|
||||
if (g_contimeout_sec > 0) {
|
||||
struct timeval ct;
|
||||
ct.tv_sec = g_contimeout_sec;
|
||||
ct.tv_usec = 0;
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &ct, sizeof(ct));
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &ct, sizeof(ct));
|
||||
}
|
||||
|
||||
if (bind_addr_family != 0) {
|
||||
if (rp->ai_family != bind_addr_family) {
|
||||
|
||||
+597
-192
@@ -2,6 +2,7 @@
|
||||
#include "array_list.h"
|
||||
#include "log.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <ctype.h>
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
@@ -52,6 +53,31 @@ bool path_is_within_root(const char* root, const char* path) {
|
||||
return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/');
|
||||
}
|
||||
|
||||
/* Borrowed transfer-relative view of `path`: strip any leading '/' and then a
|
||||
* `root` prefix (its own leading/trailing slashes tolerated), returning a
|
||||
* pointer into `path`. Non-allocating, so it is safe on the hot scan/print
|
||||
* paths. A NULL/empty root, or a path not under `root`, leaves only the
|
||||
* leading-slash strip. `path` must be NUL-terminated and live in the caller. */
|
||||
const char* utils_strip_transfer_root(const char* path, const char* root) {
|
||||
if (path == NULL)
|
||||
return NULL;
|
||||
const char* rel = path;
|
||||
while (*rel == '/')
|
||||
rel++;
|
||||
if (root == NULL)
|
||||
return rel;
|
||||
while (*root == '/')
|
||||
root++;
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 0 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (root_len == 0)
|
||||
return rel;
|
||||
if (strncmp(rel, root, root_len) == 0 && (rel[root_len] == '/' || rel[root_len] == '\0'))
|
||||
return rel + root_len + (rel[root_len] == '/' ? 1 : 0);
|
||||
return rel;
|
||||
}
|
||||
|
||||
/* Open the destination root directory itself, confined to the authorized root.
|
||||
* NOTE (do not merge with file_open_secure_parent): this walk opens dest_root
|
||||
* (a directory that must already exist) and returns its fd, whereas
|
||||
@@ -60,7 +86,7 @@ bool path_is_within_root(const char* root, const char* path) {
|
||||
* two differ in create-vs-no-create, in what path component they stop at, and
|
||||
* in the extra receiver policies they apply, so they are intentionally kept
|
||||
* separate. Both rely on the shared lexical path_is_within_root check. */
|
||||
static int open_authorized_destination(const char* dest_root) {
|
||||
int utils_open_authorized_destination(const char* dest_root) {
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
if (root_fd < 0 || !root_path || !dest_root || !path_is_within_root(root_path, dest_root))
|
||||
@@ -117,6 +143,47 @@ char* str_dup(const char* string) {
|
||||
return new_string;
|
||||
}
|
||||
|
||||
int env_choice_first(const char* env_name, int (*resolve)(const char*), bool* specified) {
|
||||
if (specified)
|
||||
*specified = false;
|
||||
if (!env_name || !resolve)
|
||||
return -1;
|
||||
const char* env = getenv(env_name);
|
||||
if (!env)
|
||||
return -1;
|
||||
|
||||
bool saw_nonblank = false;
|
||||
const char* p = env;
|
||||
while (*p) {
|
||||
if (*p == '&')
|
||||
break;
|
||||
if (isspace((unsigned char)*p)) {
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
saw_nonblank = true;
|
||||
char token[64];
|
||||
size_t len = 0;
|
||||
while (*p && *p != '&' && !isspace((unsigned char)*p)) {
|
||||
if (len < sizeof(token) - 1)
|
||||
token[len++] = *p;
|
||||
p++;
|
||||
}
|
||||
token[len] = '\0';
|
||||
if (len > 0) {
|
||||
int id = resolve(token);
|
||||
if (id >= 0) {
|
||||
if (specified)
|
||||
*specified = true;
|
||||
return id;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (specified)
|
||||
*specified = saw_nonblank;
|
||||
return -1;
|
||||
}
|
||||
|
||||
#define STR_HASH_SET_MIN_CAPACITY 16
|
||||
|
||||
static size_t str_hash_set_hash(const char* key, size_t len) {
|
||||
@@ -584,110 +651,45 @@ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSki
|
||||
return false;
|
||||
}
|
||||
|
||||
/* All-or-nothing max-delete needs to know BEFORE any unlink whether the run
|
||||
would delete more than max_delete entries. This rehearsal pass walks the
|
||||
destination with the same decisions as the delete pass but never touches the
|
||||
filesystem: it counts every regular file the delete pass would unlink and
|
||||
every directory it would rmdir (a directory is removed only once every entry
|
||||
below it has been removed and nothing the walker leaves in place survives).
|
||||
Entries the walker never removes (symlinks, manifest-listed files, protected
|
||||
prefixes) mark the enclosing directory as surviving, exactly as they would
|
||||
make a real rmdir fail with ENOTEMPTY. Stops early once *count reaches the
|
||||
cap (sets *exceeds). Returns false on a traversal error. */
|
||||
static bool count_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep, size_t cap,
|
||||
size_t* count, bool* exceeds, const DeleteSkipEntry* skips,
|
||||
int skip_count, bool* survives) {
|
||||
/* openat(dirfd, ".") opens an independent file description: a dup() would
|
||||
share dirfd's file offset, and a prior rehearsal pass must not have drained
|
||||
this directory's stream before the delete pass reads it again. */
|
||||
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
DIR* dir = fdopendir(scanfd);
|
||||
if (!dir) {
|
||||
close(scanfd);
|
||||
return false;
|
||||
}
|
||||
bool operation_ok = true;
|
||||
bool local_survives = false;
|
||||
bool at_root = rel_path[0] == '\0';
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
if (*exceeds)
|
||||
break;
|
||||
char* child_rel = path_cat((char*)rel_path, entry->d_name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) {
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT)
|
||||
operation_ok = false;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (S_ISLNK(st.st_mode)) {
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (S_ISDIR(st.st_mode)) {
|
||||
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_ok = true;
|
||||
bool child_survives = true;
|
||||
if (childfd >= 0) {
|
||||
child_ok = count_extras_fd(childfd, child_rel, keep, cap, count, exceeds, skips, skip_count,
|
||||
&child_survives);
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
if (!child_ok)
|
||||
operation_ok = false;
|
||||
if (keep_is_dir(keep, child_rel)) {
|
||||
/* A directory with kept content below it is never removed. */
|
||||
local_survives = true;
|
||||
} else if (child_survives) {
|
||||
/* The directory still holds entries the walker leaves in place, so an
|
||||
rmdir would fail with ENOTEMPTY; the delete pass leaves it behind
|
||||
rather than reporting an error (matching rsync). */
|
||||
local_survives = true;
|
||||
} else {
|
||||
if (*count >= cap) {
|
||||
*exceeds = true;
|
||||
} else {
|
||||
(*count)++;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
bool found = keep_is_file(keep, child_rel);
|
||||
if (!found) {
|
||||
if (*count >= cap) {
|
||||
*exceeds = true;
|
||||
} else {
|
||||
(*count)++;
|
||||
}
|
||||
}
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
closedir(dir);
|
||||
*survives = local_survives;
|
||||
return operation_ok;
|
||||
/* Per-run deletion budget and tallies. `max_delete` is the cap on the number
|
||||
of entries the walker may remove (SIZE_MAX = unlimited); once it is reached
|
||||
the remaining extras are counted in `skipped` and left in place, matching
|
||||
rsync's partial --max-delete behavior. */
|
||||
typedef struct {
|
||||
size_t max_delete;
|
||||
size_t deleted;
|
||||
size_t skipped;
|
||||
bool limit_hit;
|
||||
} DeleteBudget;
|
||||
|
||||
/* True when direct children of the directory named by `rel` may be removed.
|
||||
With no synchronization info (dirs == NULL) the whole tree is deletable; when
|
||||
a dirs index is supplied only its exact entries are (the receive root is the
|
||||
"." sentinel). */
|
||||
static bool is_synced_dir(const PathIndex* dirs, const char* rel) {
|
||||
if (!dirs)
|
||||
return true;
|
||||
return path_index_contains(dirs, rel[0] == '\0' ? "." : rel);
|
||||
}
|
||||
|
||||
static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep,
|
||||
size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips,
|
||||
int skip_count) {
|
||||
/* Independent file description (see count_extras_fd). */
|
||||
/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed
|
||||
strcmp would order bytes >= 0x80 differently). */
|
||||
static int delete_name_cmp(const char* a, const char* b) {
|
||||
const unsigned char* pa = (const unsigned char*)a;
|
||||
const unsigned char* pb = (const unsigned char*)b;
|
||||
while (*pa != '\0' && *pa == *pb) {
|
||||
pa++;
|
||||
pb++;
|
||||
}
|
||||
return (int)*pa - (int)*pb;
|
||||
}
|
||||
|
||||
bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count,
|
||||
bool* operation_ok) {
|
||||
*out = NULL;
|
||||
*count = 0;
|
||||
if (operation_ok)
|
||||
*operation_ok = true;
|
||||
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
@@ -696,12 +698,127 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k
|
||||
close(scanfd);
|
||||
return false;
|
||||
}
|
||||
bool operation_ok = true;
|
||||
DeleteDirEntry* entries = NULL;
|
||||
size_t used = 0;
|
||||
size_t capacity = 0;
|
||||
bool ok = true;
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entry->d_name);
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT && operation_ok)
|
||||
*operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (used == capacity) {
|
||||
size_t next = capacity == 0 ? 16 : capacity * 2;
|
||||
DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown));
|
||||
if (!grown) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
entries = grown;
|
||||
capacity = next;
|
||||
}
|
||||
entries[used].name = str_dup(entry->d_name);
|
||||
if (!entries[used].name) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
entries[used].is_dir = S_ISDIR(st.st_mode);
|
||||
used++;
|
||||
}
|
||||
closedir(dir);
|
||||
if (!ok) {
|
||||
delete_dir_entries_free(entries, used);
|
||||
return false;
|
||||
}
|
||||
*out = entries;
|
||||
*count = used;
|
||||
return true;
|
||||
}
|
||||
|
||||
void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) {
|
||||
if (!entries)
|
||||
return;
|
||||
for (size_t i = 0; i < count; i++)
|
||||
free(entries[i].name);
|
||||
free(entries);
|
||||
}
|
||||
|
||||
/* rsync's extraneous-entry order: subdirectories before files, each group in
|
||||
descending name order. */
|
||||
int delete_dir_entry_cmp_desc(const void* a, const void* b) {
|
||||
const DeleteDirEntry* ea = a;
|
||||
const DeleteDirEntry* eb = b;
|
||||
if (ea->is_dir != eb->is_dir)
|
||||
return ea->is_dir ? -1 : 1;
|
||||
return -delete_name_cmp(ea->name, eb->name);
|
||||
}
|
||||
|
||||
/* rsync's kept-subdirectory order: plain ascending name. */
|
||||
int delete_dir_entry_cmp_asc(const void* a, const void* b) {
|
||||
const DeleteDirEntry* ea = a;
|
||||
const DeleteDirEntry* eb = b;
|
||||
return delete_name_cmp(ea->name, eb->name);
|
||||
}
|
||||
|
||||
/* Remove the extras directly inside the directory open on `dirfd`, recursing
|
||||
into every child directory so kept content below a synchronized prefix is
|
||||
reached. `all_removed` reports whether every child entry was removed (so the
|
||||
caller may rmdir this directory). A child directory is never removed when it
|
||||
is itself a synchronized directory or holds kept content; with a dirs index
|
||||
supplied, direct children of a non-synchronized directory are never extras at
|
||||
all (they are left in place but still descended into). Symlinks are unlinked
|
||||
like any other non-directory extra (never followed).
|
||||
|
||||
Entries are processed in rsync's order (extraneous subdirectories in
|
||||
descending name order, then extraneous files, then kept subdirectories in
|
||||
ascending order) rather than readdir() order, so `--max-delete` leaves the
|
||||
same survivors and the `--info=del`/dry-run line order matches rsync. */
|
||||
static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep,
|
||||
const PathIndex* dirs, DeleteBudget* budget,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules, bool parent_deletable,
|
||||
bool* all_removed, DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
DeleteDirEntry* entries = NULL;
|
||||
size_t count = 0;
|
||||
bool collect_ok = true;
|
||||
if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok))
|
||||
return false;
|
||||
bool operation_ok = collect_ok;
|
||||
bool local_survives = false;
|
||||
bool* shielded = calloc(count ? count : 1, sizeof(bool));
|
||||
bool* is_extra = calloc(count ? count : 1, sizeof(bool));
|
||||
if (!shielded || !is_extra) {
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
return false;
|
||||
}
|
||||
/* A directory is deletable when it or ANY ancestor is synchronized; the
|
||||
`parent_deletable` flag carries that down the recursion so dest-only
|
||||
directories below a synchronized root are removed wholesale. */
|
||||
bool deletable = parent_deletable || is_synced_dir(dirs, rel_path);
|
||||
bool at_root = rel_path[0] == '\0';
|
||||
|
||||
/* Reproduce rsync's traversal order: extraneous subdirectories in descending
|
||||
name order, then extraneous files in descending name order, and kept
|
||||
subdirectories only afterwards (ascending). Sorting up front also fixes the
|
||||
identity of the survivors under a partial --max-delete. */
|
||||
if (count > 1)
|
||||
qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc);
|
||||
size_t dir_count = 0;
|
||||
while (dir_count < count && entries[dir_count].is_dir)
|
||||
dir_count++;
|
||||
|
||||
/* Classify every entry up front (the verdict does not depend on processing
|
||||
order) so the ordered passes below can act on it. */
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
@@ -713,94 +830,321 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* k
|
||||
top-level-only prefix) and the basis prefixes are protected: a nested
|
||||
destination directory that happens to be called .fastsync-stage is
|
||||
ordinary content. */
|
||||
if (path_under_skip_prefix(child_rel, rel_path[0] == '\0', skips, skip_count)) {
|
||||
free(child_rel);
|
||||
if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) {
|
||||
shielded[i] = true;
|
||||
local_survives = true;
|
||||
} else if (protect_rules &&
|
||||
filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir,
|
||||
FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) {
|
||||
/* A first-match protect rule shields the extra; for a directory the whole
|
||||
subtree is shielded (rsync prunes an excluded directory), so do not
|
||||
descend. */
|
||||
shielded[i] = true;
|
||||
local_survives = true;
|
||||
} else if (entries[i].is_dir) {
|
||||
bool child_synced = dirs && path_index_contains(dirs, child_rel);
|
||||
is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
local_survives = true;
|
||||
} else {
|
||||
is_extra[i] = deletable && !keep_is_file(keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
|
||||
/* Pass 1: extraneous subdirectories, descending. */
|
||||
for (size_t i = 0; i < dir_count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT)
|
||||
int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_all_removed = false;
|
||||
if (childfd >= 0) {
|
||||
if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count,
|
||||
protect_rules, deletable, &child_all_removed, observer,
|
||||
observer_context))
|
||||
operation_ok = false;
|
||||
free(child_rel);
|
||||
continue;
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
// Skip symlinks to prevent following them outside the destination tree
|
||||
if (S_ISLNK(st.st_mode)) {
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (S_ISDIR(st.st_mode)) {
|
||||
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_removed = false;
|
||||
if (childfd >= 0) {
|
||||
child_removed = delete_extras_fd(childfd, child_rel, keep, max_delete, deleted_count, skips,
|
||||
skip_count);
|
||||
if (!child_removed)
|
||||
if (child_all_removed && deletable) {
|
||||
if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
local_survives = true;
|
||||
} else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) {
|
||||
/* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still
|
||||
holds entries the walker leaves in place (a protected excluded
|
||||
prefix, a kept file the manifest protects, a symlink); rsync leaves
|
||||
such a directory behind, so this is not an error. Only genuine I/O
|
||||
failures abort the deletion. */
|
||||
if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST)
|
||||
operation_ok = false;
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
if (child_removed && !keep_is_dir(keep, child_rel)) {
|
||||
if (*deleted_count >= max_delete) {
|
||||
operation_ok = false;
|
||||
} else {
|
||||
if (unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0) {
|
||||
/* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory
|
||||
still holds entries the walker leaves in place (a protected
|
||||
excluded prefix, a kept file the manifest protects, a symlink);
|
||||
rsync leaves such a directory behind, so this is not an error.
|
||||
Only genuine I/O failures abort the deletion. */
|
||||
if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST)
|
||||
operation_ok = false;
|
||||
local_survives = true;
|
||||
} else {
|
||||
budget->deleted++;
|
||||
/* rsync reports a removed directory with a trailing slash. */
|
||||
if (observer) {
|
||||
size_t len = strlen(child_rel);
|
||||
char* with_slash = malloc(len + 2);
|
||||
if (with_slash) {
|
||||
memcpy(with_slash, child_rel, len);
|
||||
with_slash[len] = '/';
|
||||
with_slash[len + 1] = '\0';
|
||||
observer(observer_context, with_slash);
|
||||
free(with_slash);
|
||||
} else {
|
||||
(*deleted_count)++;
|
||||
observer(observer_context, child_rel);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// Check if relative path is in manifest
|
||||
bool found = keep_is_file(keep, child_rel);
|
||||
if (!found) {
|
||||
if (*deleted_count >= max_delete) {
|
||||
operation_ok = false;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (unlinkat(dirfd, entry->d_name, 0) != 0) {
|
||||
if (errno != ENOENT)
|
||||
operation_ok = false;
|
||||
} else {
|
||||
(*deleted_count)++;
|
||||
}
|
||||
local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
|
||||
/* Pass 2: extraneous files, descending. */
|
||||
for (size_t i = dir_count; i < count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
local_survives = true;
|
||||
} else if (unlinkat(dirfd, entries[i].name, 0) != 0) {
|
||||
if (errno != ENOENT)
|
||||
operation_ok = false;
|
||||
local_survives = true;
|
||||
} else {
|
||||
budget->deleted++;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (child_rel) {
|
||||
if (observer)
|
||||
observer(observer_context, child_rel);
|
||||
char* escaped_path = output_escape(child_rel, log_get_8_bit_output());
|
||||
fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : "<allocation failed>");
|
||||
free(escaped_path);
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
}
|
||||
|
||||
/* Pass 3: kept subdirectories, ascending (rsync descends into these only
|
||||
after the parent's own extras have been handled). */
|
||||
for (size_t i = dir_count; i-- > 0;) {
|
||||
if (is_extra[i] || shielded[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_all_removed = false;
|
||||
if (childfd >= 0) {
|
||||
if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count,
|
||||
protect_rules, deletable, &child_all_removed, observer,
|
||||
observer_context))
|
||||
operation_ok = false;
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
/* A kept/synchronized directory is never removed. */
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
}
|
||||
closedir(dir);
|
||||
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
*all_removed = !local_survives;
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
size_t max_delete, const DeleteSkipEntry* skips,
|
||||
int skip_count, size_t* deleted_out) {
|
||||
if (deleted_out)
|
||||
*deleted_out = 0;
|
||||
if (!manifest)
|
||||
return DELETE_WALK_ERROR;
|
||||
/* Index the keep-set once so both passes answer membership in O(path length)
|
||||
instead of scanning every manifest entry for every destination entry. */
|
||||
/* Read-only mirror of delete_extras_fd: records the paths that WOULD be removed
|
||||
without unlinking anything. A child directory is reported after its own
|
||||
reportable children (depth-first), matching the delete pass's ordering. */
|
||||
static bool list_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep,
|
||||
const PathIndex* dirs, ArrayList* out, size_t* recorded,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules, bool parent_deletable,
|
||||
bool* all_removed) {
|
||||
DeleteDirEntry* entries = NULL;
|
||||
size_t count = 0;
|
||||
bool collect_ok = true;
|
||||
if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok))
|
||||
return false;
|
||||
bool operation_ok = collect_ok;
|
||||
bool local_survives = false;
|
||||
bool* shielded = calloc(count ? count : 1, sizeof(bool));
|
||||
bool* is_extra = calloc(count ? count : 1, sizeof(bool));
|
||||
if (!shielded || !is_extra) {
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
return false;
|
||||
}
|
||||
bool deletable = parent_deletable || is_synced_dir(dirs, rel_path);
|
||||
bool at_root = rel_path[0] == '\0';
|
||||
|
||||
/* Mirror the delete walk's rsync order (extraneous subdirectories descending,
|
||||
then extraneous files descending, then kept subdirectories ascending). */
|
||||
if (count > 1)
|
||||
qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc);
|
||||
size_t dir_count = 0;
|
||||
while (dir_count < count && entries[dir_count].is_dir)
|
||||
dir_count++;
|
||||
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) {
|
||||
shielded[i] = true;
|
||||
local_survives = true;
|
||||
} else if (protect_rules &&
|
||||
filter_rules_apply_side(protect_rules, child_rel, entries[i].name, entries[i].is_dir,
|
||||
FILTER_SIDE_RECEIVER) == FILTER_ACTION_PROTECT) {
|
||||
/* Mirror the delete walk: a protected entry is never reported as a
|
||||
would-delete and a protected directory's subtree is not enumerated. */
|
||||
shielded[i] = true;
|
||||
local_survives = true;
|
||||
} else if (entries[i].is_dir) {
|
||||
bool child_synced = dirs && path_index_contains(dirs, child_rel);
|
||||
is_extra[i] = deletable && !child_synced && !keep_is_dir(keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
local_survives = true;
|
||||
} else {
|
||||
is_extra[i] = deletable && !keep_is_file(keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
|
||||
/* Pass 1: extraneous subdirectories, descending (recorded after contents). */
|
||||
for (size_t i = 0; i < dir_count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_all_removed = false;
|
||||
if (childfd >= 0) {
|
||||
if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count,
|
||||
protect_rules, deletable, &child_all_removed))
|
||||
operation_ok = false;
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
if (child_all_removed && deletable) {
|
||||
size_t len = strlen(child_rel);
|
||||
char* copy = malloc(len + 2);
|
||||
if (!copy) {
|
||||
operation_ok = false;
|
||||
} else {
|
||||
memcpy(copy, child_rel, len);
|
||||
copy[len] = '/';
|
||||
copy[len + 1] = '\0';
|
||||
if (!array_list_add(out, copy)) {
|
||||
free(copy);
|
||||
operation_ok = false;
|
||||
} else {
|
||||
(*recorded)++;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
|
||||
/* Pass 2: extraneous files, descending. */
|
||||
for (size_t i = dir_count; i < count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
char* copy = str_dup(child_rel);
|
||||
if (!copy || !array_list_add(out, copy)) {
|
||||
free(copy);
|
||||
operation_ok = false;
|
||||
} else {
|
||||
(*recorded)++;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
|
||||
/* Pass 3: kept subdirectories, ascending. */
|
||||
for (size_t i = dir_count; i-- > 0;) {
|
||||
if (is_extra[i] || shielded[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
int childfd = openat(dirfd, entries[i].name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_all_removed = false;
|
||||
if (childfd >= 0) {
|
||||
if (!list_extras_fd(childfd, child_rel, keep, dirs, out, recorded, skips, skip_count,
|
||||
protect_rules, deletable, &child_all_removed))
|
||||
operation_ok = false;
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
}
|
||||
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
*all_removed = !local_survives;
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out) {
|
||||
if (count_out)
|
||||
*count_out = 0;
|
||||
if (!manifest || !out)
|
||||
return false;
|
||||
PathIndex keep;
|
||||
if (!build_keep_index(manifest, &keep))
|
||||
return DELETE_WALK_ERROR;
|
||||
return false;
|
||||
PathIndex dirs;
|
||||
bool have_dirs = synced_dirs != NULL;
|
||||
if (have_dirs &&
|
||||
!path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) {
|
||||
path_index_free(&keep);
|
||||
return false;
|
||||
}
|
||||
int rootfd;
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
if (root_fd >= 0) {
|
||||
if (utils_get_authorized_root_path())
|
||||
rootfd = open_authorized_destination(dest_root);
|
||||
rootfd = utils_open_authorized_destination(dest_root);
|
||||
else if (dest_root == NULL)
|
||||
rootfd = dup(root_fd);
|
||||
else
|
||||
@@ -810,39 +1154,100 @@ DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* m
|
||||
}
|
||||
if (rootfd < 0) {
|
||||
path_index_free(&keep);
|
||||
return DELETE_WALK_ERROR;
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
return false;
|
||||
}
|
||||
if (max_delete != SIZE_MAX) {
|
||||
/* Rehearse the deletion first so a run that would exceed the cap removes
|
||||
nothing (rsync's all-or-nothing --max-delete contract). */
|
||||
size_t count = 0;
|
||||
bool exceeds = false;
|
||||
bool survives = false;
|
||||
bool counted_ok = count_extras_fd(rootfd, "", &keep, max_delete, &count, &exceeds, skips,
|
||||
skip_count, &survives);
|
||||
if (!counted_ok) {
|
||||
close(rootfd);
|
||||
path_index_free(&keep);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
if (exceeds) {
|
||||
close(rootfd);
|
||||
path_index_free(&keep);
|
||||
return DELETE_WALK_LIMIT_EXCEEDED;
|
||||
}
|
||||
}
|
||||
size_t deleted_count = 0;
|
||||
bool ok = delete_extras_fd(rootfd, "", &keep, max_delete, &deleted_count, skips, skip_count);
|
||||
bool all_removed = false;
|
||||
size_t recorded = 0;
|
||||
bool ok = list_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, out, &recorded, skips,
|
||||
skip_count, protect_rules, false, &all_removed);
|
||||
if (close(rootfd) != 0)
|
||||
ok = false;
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
if (count_out)
|
||||
*count_out = recorded;
|
||||
return ok;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules,
|
||||
size_t* deleted_out, size_t* skipped_out,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (deleted_out)
|
||||
*deleted_out = deleted_count;
|
||||
return ok ? DELETE_WALK_OK : DELETE_WALK_ERROR;
|
||||
*deleted_out = 0;
|
||||
if (skipped_out)
|
||||
*skipped_out = 0;
|
||||
if (!manifest)
|
||||
return DELETE_WALK_ERROR;
|
||||
/* Index the keep-set (and the synchronized-dir set, when supplied) once so
|
||||
membership is answered in O(path length) instead of scanning every entry
|
||||
for every destination entry. */
|
||||
PathIndex keep;
|
||||
if (!build_keep_index(manifest, &keep))
|
||||
return DELETE_WALK_ERROR;
|
||||
PathIndex dirs;
|
||||
bool have_dirs = synced_dirs != NULL;
|
||||
if (have_dirs &&
|
||||
!path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) {
|
||||
path_index_free(&keep);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
int rootfd;
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
if (root_fd >= 0) {
|
||||
if (utils_get_authorized_root_path())
|
||||
rootfd = utils_open_authorized_destination(dest_root);
|
||||
else if (dest_root == NULL)
|
||||
rootfd = dup(root_fd);
|
||||
else
|
||||
rootfd = -1;
|
||||
} else {
|
||||
rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
}
|
||||
if (rootfd < 0) {
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
bool all_removed = false;
|
||||
bool ok =
|
||||
delete_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &budget, skips, skip_count,
|
||||
protect_rules, false, &all_removed, observer, observer_context);
|
||||
if (close(rootfd) != 0)
|
||||
ok = false;
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
if (deleted_out)
|
||||
*deleted_out = budget.deleted;
|
||||
if (skipped_out)
|
||||
*skipped_out = budget.skipped;
|
||||
if (!ok)
|
||||
return DELETE_WALK_ERROR;
|
||||
return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules, size_t* deleted_out,
|
||||
size_t* skipped_out) {
|
||||
return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips,
|
||||
skip_count, protect_rules, deleted_out, skipped_out, NULL,
|
||||
NULL);
|
||||
}
|
||||
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest) {
|
||||
return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL, 0, NULL) == DELETE_WALK_OK;
|
||||
return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) ==
|
||||
DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
bool has_path_traversal(const char* path) {
|
||||
|
||||
+91
-16
@@ -2,6 +2,7 @@
|
||||
#define UTILS_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "filter.h"
|
||||
#include <stddef.h>
|
||||
#include <stdbool.h>
|
||||
#include <stdio.h>
|
||||
@@ -81,6 +82,17 @@ bool path_index_has_descendant(const PathIndex* index, const char* path);
|
||||
|
||||
char* str_dup(const char* string);
|
||||
char* output_escape(const char* string, bool eight_bit_output);
|
||||
|
||||
/* Resolve the first supported name from a rsync algorithm-preference
|
||||
* environment variable (RSYNC_COMPRESS_LIST / RSYNC_CHECKSUM_LIST). `resolve`
|
||||
* maps a case-insensitive name to an algorithm id (>= 0) or -1 for an unknown
|
||||
* name. rsync's syntax is a whitespace-separated list (comma/colon are NOT
|
||||
* separators); the client-side half ends at '&'. Unknown entries are skipped
|
||||
* and the first resolvable one wins. *specified is set true when the variable
|
||||
* holds at least one non-blank character. Returns the first resolvable id, or
|
||||
* -1 when the variable is unset/blank or names no supported algorithm. */
|
||||
int env_choice_first(const char* env_name, int (*resolve)(const char*), bool* specified);
|
||||
|
||||
/* Upper bound on one line/token read from a local list file (--files-from,
|
||||
* --exclude-from/--include-from, .rsync-filter). Mirrors MAX_STRING_SIZE and
|
||||
* stops a hostile multi-gigabyte line from forcing unbounded allocation. */
|
||||
@@ -98,10 +110,10 @@ bool glob_match(const char* pattern, const char* str);
|
||||
typedef enum {
|
||||
/* Every extra entry was removed (or there were none). */
|
||||
DELETE_WALK_OK = 0,
|
||||
/* The destination holds more extras than the numeric cap for this run. With
|
||||
the all-or-nothing max-delete semantics NOTHING was removed (the walker
|
||||
counts first and refuses to start when the run would exceed the limit). */
|
||||
DELETE_WALK_LIMIT_EXCEEDED,
|
||||
/* The numeric cap for this run was reached before every extra was removed.
|
||||
The walker removed exactly the entries the cap allowed and skipped (without
|
||||
removing) the rest, matching rsync's partial --max-delete behavior. */
|
||||
DELETE_WALK_LIMIT_REACHED,
|
||||
/* A traversal or unlink failure aborted the deletion (partial removal is
|
||||
possible, mirroring the delete pass). */
|
||||
DELETE_WALK_ERROR
|
||||
@@ -122,20 +134,78 @@ typedef struct {
|
||||
only DIRECT children of the destination root, i.e. child_rel has no '/'). */
|
||||
bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips,
|
||||
int skip_count);
|
||||
/* Remove files/dirs under dest_root that are not listed in manifest without
|
||||
ever descending into a protected prefix (see DeleteSkipEntry). When
|
||||
max_delete is not SIZE_MAX the run is all-or-nothing: extras are counted
|
||||
first and DELETE_WALK_LIMIT_EXCEEDED is returned (with nothing removed) when
|
||||
the count would exceed the cap. `deleted_out` optionally receives the number
|
||||
of entries actually removed. The all-or-nothing guarantee holds only while
|
||||
the destination tree is not being concurrently modified: the rehearsal pass
|
||||
and the delete pass are two separate walks, so a concurrent change between
|
||||
them (another process adding/removing entries) can make the second pass
|
||||
delete a different set than the first one counted. */
|
||||
/* One destination-directory entry collected up front so the delete walkers can
|
||||
reproduce rsync's traversal order instead of readdir() order. rsync processes
|
||||
a directory's extraneous subdirectories first (descending name, depth-first),
|
||||
then its extraneous files (descending name), and only afterwards descends into
|
||||
its kept subdirectories (ascending name). */
|
||||
typedef struct {
|
||||
char* name;
|
||||
bool is_dir;
|
||||
} DeleteDirEntry;
|
||||
/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."),
|
||||
stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of
|
||||
*count entries whose names the caller frees with delete_dir_entries_free().
|
||||
Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is
|
||||
skipped, any other stat failure is reported through *operation_ok while the
|
||||
walk continues. */
|
||||
bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok);
|
||||
void delete_dir_entries_free(DeleteDirEntry* entries, size_t count);
|
||||
/* Sort comparators: `_desc` orders subdirectories before files and each group by
|
||||
descending name (rsync's extraneous-entry order); `_asc` orders plain ascending
|
||||
name (rsync's kept-subdirectory order). */
|
||||
int delete_dir_entry_cmp_desc(const void* a, const void* b);
|
||||
int delete_dir_entry_cmp_asc(const void* a, const void* b);
|
||||
/* Remove files/dirs/symlinks under dest_root that are not listed in manifest
|
||||
without ever descending into a protected prefix (see DeleteSkipEntry). When
|
||||
`synced_dirs` is non-NULL, extras are only removed directly inside a directory
|
||||
whose destination-relative path is an exact entry in that list (the receive
|
||||
root is the "." sentinel); directories outside the synchronized set are still
|
||||
descended into so kept content below a listed directory is preserved, but
|
||||
nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of
|
||||
treating the whole destination tree as deletable. `max_delete` caps the
|
||||
number of removed entries (SIZE_MAX = unlimited): the walker removes up to the
|
||||
cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained.
|
||||
`deleted_out`/`skipped_out` optionally receive the number of entries removed
|
||||
and the number skipped because of the cap. */
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
size_t max_delete, const DeleteSkipEntry* skips,
|
||||
int skip_count, size_t* deleted_out);
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules, size_t* deleted_out,
|
||||
size_t* skipped_out);
|
||||
|
||||
/* Optional per-deletion observer: called for each destination-relative path
|
||||
actually removed (a file, symlink, or directory), in removal order, so the
|
||||
receiver can stream rsync's `--info=del`/`--info=remove` lines. */
|
||||
typedef void (*DeletePathObserver)(void* context, const char* rel_path);
|
||||
|
||||
/* `delete_extras_limited_observed` is delete_extras_limited with an optional
|
||||
* observer; the observer is invoked only for entries truly removed. When
|
||||
* `protect_rules` is non-NULL its receiver-side verdict is evaluated for every
|
||||
* candidate extra: a first-match PROTECT leaves the entry (and, for a
|
||||
* directory, its whole subtree) in place, while RISK/NONE fall through to the
|
||||
* ordinary skip-prefix/keep-set logic. */
|
||||
DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules,
|
||||
size_t* deleted_out, size_t* skipped_out,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context);
|
||||
/* Read-only companion to delete_extras_limited: walk the destination exactly as
|
||||
the delete pass would and APPEND (strdup'd) destination-relative paths that
|
||||
WOULD be removed, without touching disk. Used for -n/--dry-run --delete
|
||||
would-delete reporting. Returns true on a clean walk; the caller owns the
|
||||
strings appended to `out` and receives their count in *count_out. */
|
||||
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
|
||||
const FilterRuleList* protect_rules, ArrayList* out, size_t* count_out);
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest);
|
||||
/* Open the existing destination directory at `dest_root`, confined to the
|
||||
authorized root with an O_NOFOLLOW component walk (the same confinement the
|
||||
deletion walker uses for its root). Returns a new fd the caller owns, or -1
|
||||
on error (including a destination that does not exist). */
|
||||
int utils_open_authorized_destination(const char* dest_root);
|
||||
bool utils_set_authorized_root(int fd, const char* canonical_path);
|
||||
/* The fd-only compatibility form is fail-closed for path-based operations;
|
||||
* callers should use utils_set_authorized_root with the canonical identity. */
|
||||
@@ -161,6 +231,11 @@ const char* utils_get_authorized_root_path(void);
|
||||
* callers guarantee this); this is containment by string, not by resolved
|
||||
* symlinks. Shared by the utils and file secure-walk root confinement. */
|
||||
bool path_is_within_root(const char* root, const char* path);
|
||||
/* Non-allocating transfer-relative view of `path`: strip any leading '/' and
|
||||
* then a `root` prefix (leading/trailing slashes tolerated), returning a
|
||||
* borrowed pointer into `path`. A NULL/empty root, or a path not under
|
||||
* `root`, yields just the leading-slash strip. `path`/`root` must stay alive. */
|
||||
const char* utils_strip_transfer_root(const char* path, const char* root);
|
||||
/* True when `path` contains a ".." component. This is a purely lexical
|
||||
* dot-dot check: an absolute path is NOT rejected here, because default
|
||||
* (non-relative) transfers legitimately put the sender's absolute source path
|
||||
|
||||
+55
-43
@@ -2,6 +2,7 @@
|
||||
#include "xattr.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include "metadata.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include "file_types.h"
|
||||
@@ -37,6 +38,22 @@ void xattr_list_free(FileXattrList* list) {
|
||||
free(list);
|
||||
}
|
||||
|
||||
FileXattrList* xattr_list_clone(const FileXattrList* list) {
|
||||
if (!list)
|
||||
return NULL;
|
||||
FileXattrList* clone = xattr_list_new();
|
||||
if (!clone)
|
||||
return NULL;
|
||||
for (int i = 0; i < list->count; i++) {
|
||||
if (!xattr_list_append(clone, list->items[i].name, list->items[i].value,
|
||||
list->items[i].value_len)) {
|
||||
xattr_list_free(clone);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
return clone;
|
||||
}
|
||||
|
||||
bool xattr_list_append(FileXattrList* list, const char* name, const void* value, size_t value_len) {
|
||||
if (!list || !name || (!value && value_len != 0))
|
||||
return false;
|
||||
@@ -364,26 +381,12 @@ void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, int6
|
||||
}
|
||||
}
|
||||
|
||||
/* --fake-super replay: read the freshly-stored record and re-apply the source
|
||||
* stat fd-relative. A privileged (root) run can actually change the owner;
|
||||
* a non-root run silently skips the fchown on EPERM/EACCES (never fatal,
|
||||
* mirroring the normal metadata identity path; other errors are logged) and
|
||||
* still applies mode/mtime where permitted.
|
||||
*
|
||||
* The OWNER leg additionally honors three policies:
|
||||
* - an explicit ownership identity policy must be active (numeric-ids /
|
||||
* chown / usermap / groupmap / copy-as). --fake-super on its own only
|
||||
* RECORDS the source owner; replaying that owner as a live chown without an
|
||||
* explicit ownership opt-in would be an un-gated client-chosen-ownership
|
||||
* primitive.
|
||||
* - --no-super (privilege_super_permitted() false) suppresses it even for a
|
||||
* root receiver, exactly like the normal metadata identity path.
|
||||
* - an active --copy-as is AUTHORITATIVE: the identity path already forced the
|
||||
* target owner, so replaying the recorded source owner here would silently
|
||||
* override it. The xattr record is still stored/replayed for a later
|
||||
* privileged restore; only the live chown is skipped. Mode/mtime remain
|
||||
* applied either way so unprivileged --fake-super still works. */
|
||||
bool fake_super_restore_fd(int fd) {
|
||||
/* --fake-super replay: read the freshly-stored record and re-apply mode/mtime
|
||||
* fd-relative. The recorded uid/gid are retained for a later privileged
|
||||
* restore but are NEVER chowned here: --fake-super only RECORDS ownership, it
|
||||
* must not real-chown the recorded (resolved) owner. Mode/mtime still apply so
|
||||
* unprivileged --fake-super keeps working. */
|
||||
bool fake_super_restore_fd(int fd, FileAttrPolicy policy) {
|
||||
if (fd < 0)
|
||||
return false;
|
||||
char record[128];
|
||||
@@ -398,28 +401,37 @@ bool fake_super_restore_fd(int fd) {
|
||||
5)
|
||||
return false; /* malformed record: skip, never fatal */
|
||||
|
||||
/* Owner is applied best-effort only: a non-root process cannot chown and
|
||||
must not abort the transfer for that reason (FastSync identity philosophy).
|
||||
EPERM/EACCES (expected for a non-root receiver) are skipped silently; a
|
||||
genuine EINVAL (an impossible stored id) is logged so the corruption is
|
||||
not hidden. --no-super suppresses the owner leg even for root, and an
|
||||
active --copy-as is authoritative so its forced owner must not be
|
||||
overwritten by the recorded source owner. */
|
||||
if (identity_active_enabled() && privilege_super_permitted() && !identity_copy_as_active() &&
|
||||
fchown(fd, (uid_t)ul_uid, (gid_t)ul_gid) != 0 && errno != EPERM && errno != EACCES)
|
||||
log_message(LOG_LEVEL_WARNING, "--fake-super: could not restore owner on destination file: %s",
|
||||
strerror(errno));
|
||||
/* Mode is applied through the same sanitization the normal metadata path
|
||||
uses (metadata_mode): group/other write bits are never granted, so a
|
||||
recorded source mode of 0666 restores as 0644 — identical to a non-fake-
|
||||
super --preserve run, never a privilege-granting regression. */
|
||||
if (fchmod(fd, (mode_t)(ul_mode & 0777U & ~(S_IWGRP | S_IWOTH))) != 0)
|
||||
log_message(LOG_LEVEL_WARNING, "--fake-super: could not restore mode on destination file: %s",
|
||||
strerror(errno));
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = (time_t)mtime_sec, .tv_nsec = mtime_nsec}};
|
||||
if (futimens(fd, times) != 0)
|
||||
log_message(LOG_LEVEL_WARNING, "--fake-super: could not restore mtime on destination file: %s",
|
||||
strerror(errno));
|
||||
/* --fake-super NEVER performs a real chown: that would defeat the whole
|
||||
point of the flag (record privileged ownership on an unprivileged receiver
|
||||
for a later privileged restore). The uid/gid parsed above are retained in
|
||||
the record for that later restore, but no ownership change happens here. */
|
||||
(void)ul_uid;
|
||||
(void)ul_gid;
|
||||
/* Mode is applied only when the per-attribute policy asks for it, through the
|
||||
SAME shared helper the normal metadata path uses (metadata_mode_for_policy):
|
||||
under --perms the recorded source mode is copied exactly, including
|
||||
group/other write and setuid/setgid/sticky bits (rsync parity), and the -E
|
||||
rule derives exec bits from the destination's read bits exactly like
|
||||
file_restore_metadata_fd. */
|
||||
if (policy.perms || policy.executability) {
|
||||
struct stat cur;
|
||||
mode_t want = 0;
|
||||
if (fstat(fd, &cur) != 0) {
|
||||
log_message(LOG_LEVEL_WARNING, "--fake-super: could not read destination mode: %s",
|
||||
strerror(errno));
|
||||
} else if (metadata_mode_for_policy((mode_t)ul_mode, cur.st_mode, policy, &want)) {
|
||||
if (fchmod(fd, want) != 0)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--fake-super: could not restore mode on destination file: %s",
|
||||
strerror(errno));
|
||||
}
|
||||
}
|
||||
if (policy.times) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = (time_t)mtime_sec, .tv_nsec = mtime_nsec}};
|
||||
if (futimens(fd, times) != 0)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--fake-super: could not restore mtime on destination file: %s", strerror(errno));
|
||||
}
|
||||
return true;
|
||||
}
|
||||
+15
-10
@@ -1,6 +1,7 @@
|
||||
#ifndef XATTR_H
|
||||
#define XATTR_H
|
||||
|
||||
#include "file_attr.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
@@ -55,6 +56,8 @@ typedef struct {
|
||||
|
||||
FileXattrList* xattr_list_new(void);
|
||||
void xattr_list_free(FileXattrList* list);
|
||||
/* Deep-copy `list` (NULL in, NULL out). Returns NULL on allocation failure. */
|
||||
FileXattrList* xattr_list_clone(const FileXattrList* list);
|
||||
/* Append one entry (deep copy). Returns false on allocation failure. */
|
||||
bool xattr_list_append(FileXattrList* list, const char* name, const void* value, size_t value_len);
|
||||
|
||||
@@ -95,15 +98,17 @@ void fake_super_store_fd(int fd, uint32_t uid, uint32_t gid, uint32_t mode, int6
|
||||
int64_t mtime_nsec);
|
||||
|
||||
/* --fake-super replay: parse the FAKESUPER_XATTR record previously written on
|
||||
* `fd` by fake_super_store_fd and re-apply uid/gid/mode/mtime fd-relative.
|
||||
* Best-effort: absence of the xattr or a malformed record is a silent no-op
|
||||
* that never fails the transfer. The OWNER leg is applied only when an explicit
|
||||
* ownership identity policy is active (numeric-ids/chown/usermap/groupmap/
|
||||
* copy-as), when super-user activities are permitted, and when --copy-as is not
|
||||
* authoritative; a non-root EPERM/EACCES is skipped silently, matching
|
||||
* FastSync's identity philosophy. The mode is sanitized exactly like the normal
|
||||
* metadata path (group/other write bits never granted). Returns true when the
|
||||
* xattr was present and parsed. */
|
||||
bool fake_super_restore_fd(int fd);
|
||||
* `fd` by fake_super_store_fd and re-apply mode/mtime fd-relative. The
|
||||
* recorded uid/gid are deliberately NOT chowned for real: --fake-super only
|
||||
* RECORDS ownership (the caller stores the resolved mapping via
|
||||
* identity_resolve_storage_ids), it never performs a real chown. Best-effort:
|
||||
* absence of the xattr or a malformed record is a silent no-op that never fails
|
||||
* the transfer. The MODE leg is applied only when policy.perms||policy.
|
||||
* executability and the MTIME leg only when policy.times, so the fake-super
|
||||
* replay cannot bypass the per-attribute split; the mode follows the normal
|
||||
* metadata path exactly (under --perms the source mode is copied verbatim,
|
||||
* special and group/other write bits included).
|
||||
* Returns true when the xattr was present and parsed. */
|
||||
bool fake_super_restore_fd(int fd, FileAttrPolicy policy);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1 @@
|
||||
OLDDEST
|
||||
@@ -0,0 +1 @@
|
||||
NEWCONTENT
|
||||
@@ -0,0 +1 @@
|
||||
NEWCONTENT
|
||||
@@ -115,7 +115,16 @@ static void build_canonical_frame(void) {
|
||||
if (cfg->usermap) {
|
||||
cfg->usermap_count = 1;
|
||||
cfg->usermap[0].from = MAP_FROM;
|
||||
cfg->usermap[0].from_hi = MAP_FROM;
|
||||
cfg->usermap[0].to = MAP_TO;
|
||||
cfg->usermap[0].to_name = NULL;
|
||||
}
|
||||
/* Force a non-empty receiver delete-protection block so the fuzzer mutates
|
||||
* its rule count, action/sides codes and pattern strings. */
|
||||
cfg->filters = array_list_create(free);
|
||||
if (cfg->filters) {
|
||||
array_list_add(cfg->filters, str_dup("P *.log"));
|
||||
array_list_add(cfg->filters, str_dup("+r keep/**"));
|
||||
}
|
||||
if (!cfg->send_directory || !cfg->receive_root_directory || !cfg->usermap) {
|
||||
config_delete(cfg);
|
||||
|
||||
@@ -0,0 +1,80 @@
|
||||
# Integration tests
|
||||
|
||||
The integration suite drives the built `build/server` and `build/client`
|
||||
against local corpora. Unit tests live in `tests/` (the custom C framework);
|
||||
the Python suite here covers the full transfer pipeline, transports, features,
|
||||
and rsync parity.
|
||||
|
||||
## Running
|
||||
|
||||
```bash
|
||||
# Full suite (excludes privilege-dependent tests on CI runners)
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
|
||||
# Fast PR subset only
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m ci
|
||||
```
|
||||
|
||||
The tests expect `build/server` and `build/client` (configure/build with CMake
|
||||
first); `common.py` derives `BUILD_DIR` from the repository root.
|
||||
|
||||
## Differential rsync-parity gate
|
||||
|
||||
`test_differential_parity.py` runs the **same** transfer with real
|
||||
`rsync 3.4.1` and with FastSync over separate destinations, then compares:
|
||||
|
||||
- the destination trees — relative paths, file content hashes, symlink
|
||||
targets, modes (where the case is about perms), and hard-link grouping;
|
||||
- the normalized stdout for output-oriented flags (`-i`,
|
||||
`--out-format=...`, `--stats`), after stripping volatile fields
|
||||
(timings, rates, wire byte counts) and directory-only itemize lines that
|
||||
FastSync's recursive scanner documents as absent.
|
||||
|
||||
FastSync mirrors the absolute source path under its receive root (see
|
||||
`get_dest_received_dir`); the harness normalizes that layout (and the
|
||||
`-R`/`--files-from` layouts) before comparing.
|
||||
|
||||
```bash
|
||||
# Fast subset that guards the ✅ surface on pull requests
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci
|
||||
|
||||
# Full differential case table (`_CASES`): every row is marked `parity`, and a
|
||||
# case with an allowlisted residual in `parity_caveats.py` is included too.
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity
|
||||
```
|
||||
|
||||
`-m parity` selects only the `_CASES` table in this module. Differential
|
||||
coverage for options outside that table (`--temp-dir`, `--delay-updates`,
|
||||
`--dry-run`, `--fuzzy`, the basis-dir options, `-M` over daemon/TCP, and
|
||||
receiver filter-protect) lives in dedicated modules (`test_option_parity.py`,
|
||||
`test_parity_blockers.py`, `test_parity_quickwins.py`, ...) and is not part of
|
||||
this gate. The suite skips cleanly when `rsync` is not installed.
|
||||
|
||||
## Allowlist (`parity_caveats.py`)
|
||||
|
||||
`parity_caveats.py` is the single data-driven allowlist of known differences.
|
||||
Each entry maps a case id to the aspects that may differ (`tree`, `stdout`,
|
||||
`extra`, `rc`) and cites the governing row in `RSYNC_COMPAT.md`:
|
||||
|
||||
```python
|
||||
CAVEATS = {
|
||||
# no known residuals at present -- the burn-down reached zero
|
||||
# "some_case_id": {"tree": "documented residual ... ref: RSYNC_COMPAT.md ..."},
|
||||
}
|
||||
```
|
||||
|
||||
A differential mismatch in an aspect that is **not** listed fails the gate with
|
||||
a readable tree/stdout diff.
|
||||
|
||||
If a case is allowlisted but now matches rsync, the gate emits a loud warning
|
||||
naming the stale entry — that is the parity burn-down signal. Run with
|
||||
`FASTSYNC_PARITY_STRICT=1` to make stale entries fail instead (the full CI
|
||||
parity job sets this). To add a residual:
|
||||
|
||||
1. Reproduce it with `-m parity` and read the failure's tree/stdout diff.
|
||||
2. Confirm it is a documented `⚠️`/`❌` residual (or get the `✅` row
|
||||
reclassified) and cite the row.
|
||||
3. Add the case id and aspect(s) to `CAVEATS`, keeping the reason concise.
|
||||
|
||||
Do not allowlist an undocumented divergence from a `✅` row — fix it or get the
|
||||
row reclassified first.
|
||||
@@ -101,9 +101,15 @@ class CountingProxy:
|
||||
return
|
||||
counter[0] += len(data)
|
||||
|
||||
def run(self, cmd):
|
||||
def run(self, cmd, join_timeout=20):
|
||||
"""Forward one client run (the full command list) to the real server and
|
||||
return the CompletedProcess after the counts have settled."""
|
||||
return the CompletedProcess after the counts have settled.
|
||||
|
||||
``join_timeout`` bounds how long to wait for the forwarding threads. The
|
||||
client->server count is published as soon as the client side reaches EOF
|
||||
(i.e. once the client process has exited), so callers that only need that
|
||||
count can pass a small value instead of waiting for the server to close
|
||||
its idle socket."""
|
||||
|
||||
def serve():
|
||||
try:
|
||||
@@ -119,15 +125,15 @@ class CountingProxy:
|
||||
a.start()
|
||||
b.start()
|
||||
a.join()
|
||||
b.join()
|
||||
self.client_to_server = c2s[0]
|
||||
b.join()
|
||||
self.server_to_client = s2c[0]
|
||||
self._listener.close()
|
||||
|
||||
thread = threading.Thread(target=serve)
|
||||
thread = threading.Thread(target=serve, daemon=True)
|
||||
thread.start()
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, timeout=180)
|
||||
thread.join(20)
|
||||
thread.join(join_timeout)
|
||||
return result
|
||||
|
||||
|
||||
@@ -254,6 +260,13 @@ def _wait_for_port(port, timeout=5):
|
||||
|
||||
|
||||
def _wait_proc(proc, timeout=5):
|
||||
"""Stop a long-lived subprocess promptly. The server installs a SIGTERM
|
||||
handler, so signal first and only escalate to SIGKILL if it does not exit;
|
||||
waiting without signalling would burn the full timeout on every stop."""
|
||||
if proc.poll() is not None:
|
||||
proc.wait()
|
||||
return
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=timeout)
|
||||
except subprocess.TimeoutExpired:
|
||||
|
||||
@@ -0,0 +1,34 @@
|
||||
"""Data-driven allowlist for the differential rsync-parity gate.
|
||||
|
||||
Every entry maps a case id (see ``test_differential_parity.py``) to the aspects
|
||||
that are *known* to differ from ``rsync 3.4.1`` and the documented reason. A
|
||||
differential mismatch in an aspect that is **not** listed here fails the gate.
|
||||
|
||||
Aspect keys
|
||||
-----------
|
||||
``tree`` destination tree differs (paths, file hashes, symlink targets,
|
||||
modes, hardlink grouping)
|
||||
``stdout`` normalized output for ``-i`` / ``--stats`` / ``--out-format``
|
||||
``extra`` a case-specific assertion differs (basis/inode checks, ...)
|
||||
``rc`` exit status differs
|
||||
|
||||
Burn-down
|
||||
---------
|
||||
If a case is listed here but now matches rsync, the gate emits a loud
|
||||
``pytest`` warning naming the stale entry: delete the entry (and, when the
|
||||
underlying row in ``RSYNC_COMPAT.md`` is now parity, update that row). Set
|
||||
``FASTSYNC_PARITY_STRICT=1`` to turn stale entries into failures in CI.
|
||||
|
||||
Keep the values concise but cite the governing row so the entry can be
|
||||
re-triaged when the row moves.
|
||||
"""
|
||||
|
||||
# case id -> {aspect: "reason (ref: RSYNC_COMPAT.md ...)"}
|
||||
CAVEATS = {}
|
||||
|
||||
# Accepted aspect names (guards against typos in this file).
|
||||
ASPECTS = ("tree", "stdout", "extra", "rc")
|
||||
|
||||
|
||||
def caveat_for(case_id: str) -> dict:
|
||||
return CAVEATS.get(case_id, {})
|
||||
@@ -0,0 +1,584 @@
|
||||
"""Differential rsync-parity harness.
|
||||
|
||||
Runs the SAME transfer with real ``rsync`` and with FastSync over separate
|
||||
destinations and compares the resulting trees and (optionally) normalized
|
||||
stdout. ``test_differential_parity.py`` drives this module with a table of
|
||||
cases; ``parity_caveats.py`` is the data-driven allowlist of documented
|
||||
residuals.
|
||||
|
||||
Design notes
|
||||
------------
|
||||
FastSync mirrors the *absolute* source path below its receive root, while
|
||||
rsync copies the source contents directly into the destination. ``Case.layout``
|
||||
tells the harness which pair of directory roots to compare:
|
||||
|
||||
* ``MIRROR`` -- rsync ``DEST/`` vs FastSync ``DEST/<abs-src>/`` (the common
|
||||
case; matches ``common.get_dest_received_dir``).
|
||||
* ``MIRROR_ABS`` -- ``rsync -R`` without a cut lays the full absolute path
|
||||
under the destination, so rsync ``DEST/<abs-src>/`` is compared against the
|
||||
same FastSync mirror path.
|
||||
* ``RELATIVE`` -- ``rsync -R --files-from`` lays bare relative paths under the
|
||||
destination and FastSync does the same, so both destination roots compare
|
||||
directly.
|
||||
|
||||
Only ``tests/integration/common.py`` is used to reach the build products and the
|
||||
server manager; the harness never duplicates that plumbing.
|
||||
"""
|
||||
import difflib
|
||||
import hashlib
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
from dataclasses import dataclass
|
||||
from typing import Callable, Dict, List, Optional, Tuple
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402 (path bootstrap above)
|
||||
TEST_DATA_DIR,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
|
||||
# Comparison layouts (see module docstring).
|
||||
MIRROR = "mirror"
|
||||
MIRROR_ABS = "mirror_abs"
|
||||
RELATIVE = "relative"
|
||||
|
||||
# stdout comparators.
|
||||
STDOUT_NONE = None
|
||||
STDOUT_ITEMIZE = "itemize"
|
||||
STDOUT_OUTFMT = "outfmt"
|
||||
STDOUT_STATS = "stats"
|
||||
STDOUT_PROGRESS = "progress"
|
||||
|
||||
# rsync --stats lines that are protocol-independent and must match exactly.
|
||||
# `Number of files` and `Number of created files` carry rsync's per-type
|
||||
# breakdown; protocol 2.28.0 reports the receiver-created split over
|
||||
# STATUS_STATS. Deliberately excluded: Total bytes sent/received (protocol
|
||||
# framing differs, see the `--stats` row in RSYNC_COMPAT.md).
|
||||
STATS_KEYS = (
|
||||
"Number of files",
|
||||
"Number of created files",
|
||||
"Number of deleted files",
|
||||
"Number of regular files transferred",
|
||||
"Total file size",
|
||||
"Total transferred file size",
|
||||
"Literal data",
|
||||
"Matched data",
|
||||
"File list size",
|
||||
)
|
||||
|
||||
_ITEMIZE_RE = re.compile(r"^(<|>|c|h|\.|\*)[fdLDS][.+\-][.+\-][.+\-][.+\-]")
|
||||
_PROGRESS_TOTAL_RE = re.compile(r"to-chk=\d+/(\d+)")
|
||||
_PROGRESS_XFR_RE = re.compile(r"xfr#(\d+)")
|
||||
|
||||
|
||||
@dataclass
|
||||
class Case:
|
||||
"""One differential scenario: a corpus, a flag set, and how to compare."""
|
||||
|
||||
id: str
|
||||
corpus: str
|
||||
flags: List[str]
|
||||
fastsync_flags: Optional[List[str]] = None
|
||||
layout: str = MIRROR
|
||||
server_args: Tuple[str, ...] = ("--allow-super",)
|
||||
seed: Optional[Callable] = None
|
||||
stdout: Optional[str] = STDOUT_NONE
|
||||
compare_modes: bool = False
|
||||
compare_hardlinks: bool = False
|
||||
ignore_paths: Tuple[str, ...] = ()
|
||||
extra_check: Optional[Callable] = None
|
||||
files_from: Optional[Tuple[str, ...]] = None
|
||||
# rsync receives ``src + "/"``; FastSync mirrors the path it is given, so a
|
||||
# trailing-slash-sensitive case must hand FastSync the same form.
|
||||
fs_src_suffix: str = ""
|
||||
# Some cases have an unspecified result (e.g. which extras survive a
|
||||
# partial --max-delete abort): assert the case-specific invariants via
|
||||
# extra_check and skip the exact-tree comparison.
|
||||
compare_tree: bool = True
|
||||
ci: bool = False
|
||||
ref: str = ""
|
||||
|
||||
def fs_flags(self) -> List[str]:
|
||||
return list(self.flags if self.fastsync_flags is None else self.fastsync_flags)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Corpora
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
# Deterministic mtimes so quick-check decisions are reproducible.
|
||||
_SRC_MTIME = 1_600_000_000
|
||||
|
||||
|
||||
def _write(path: str, data: bytes, mode: Optional[int] = None) -> None:
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(data)
|
||||
os.utime(path, (_SRC_MTIME, _SRC_MTIME))
|
||||
if mode is not None:
|
||||
os.chmod(path, mode)
|
||||
|
||||
|
||||
def _set_mode(path: str, mode: int) -> None:
|
||||
os.chmod(path, mode)
|
||||
|
||||
|
||||
def corpus_basic(root: str) -> None:
|
||||
"""Regular files + nested dirs (dirs are implied by their files)."""
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "a.txt"), b"hello world\n")
|
||||
_write(os.path.join(root, "sub", "b.bin"),
|
||||
bytes((i * 7) & 0xFF for i in range(5000)))
|
||||
_write(os.path.join(root, "sub", "deep", "c.txt"), "w\u00f6rld\n".encode())
|
||||
|
||||
|
||||
def corpus_unicode(root: str) -> None:
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "uni \u00f1\u6587.txt"), b"unicode\n")
|
||||
_write(os.path.join(root, "sub", "sp ace \u00e9.dat"), b"spaced\n")
|
||||
_set_mode(os.path.join(root, "sub"), 0o750)
|
||||
|
||||
|
||||
def corpus_links(root: str) -> None:
|
||||
corpus_basic(root)
|
||||
os.symlink("a.txt", os.path.join(root, "rel_link"))
|
||||
os.symlink("/etc/hostname", os.path.join(root, "abs_link"))
|
||||
os.symlink("nowhere/target", os.path.join(root, "broken_link"))
|
||||
|
||||
|
||||
def corpus_hardlinks(root: str) -> None:
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "h1.txt"), b"hardlinked payload\n")
|
||||
os.link(os.path.join(root, "h1.txt"), os.path.join(root, "h2.txt"))
|
||||
_write(os.path.join(root, "other.txt"), b"other\n")
|
||||
|
||||
|
||||
def corpus_sparse(root: str) -> None:
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "small.txt"), b"small\n")
|
||||
sparse = os.path.join(root, "sparse.bin")
|
||||
with open(sparse, "wb") as fh:
|
||||
fh.seek(1024 * 1024 - 1)
|
||||
fh.write(b"\0")
|
||||
os.utime(sparse, (_SRC_MTIME, _SRC_MTIME))
|
||||
|
||||
|
||||
def corpus_filters(root: str) -> None:
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "keep.txt"), b"keep\n")
|
||||
_write(os.path.join(root, "drop.log"), b"log\n")
|
||||
_write(os.path.join(root, "sub", "keep2.txt"), b"keep2\n")
|
||||
_write(os.path.join(root, "sub", "drop2.log"), b"log2\n")
|
||||
_write(os.path.join(root, "sub", "data.bin"), b"bin\n")
|
||||
|
||||
|
||||
def corpus_empty_dir(root: str) -> None:
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "keep.txt"), b"keep\n")
|
||||
os.makedirs(os.path.join(root, "emptydir"), exist_ok=True)
|
||||
os.utime(os.path.join(root, "emptydir"), (_SRC_MTIME, _SRC_MTIME))
|
||||
_write(os.path.join(root, "nonempty", "f.txt"), b"f\n")
|
||||
|
||||
|
||||
def corpus_multidir(root: str) -> None:
|
||||
"""Multi-directory tree for the --progress file-list naming/denominator.
|
||||
|
||||
Nested files, a directory-only branch, an empty directory and a symlink
|
||||
exercise every file-list entry type rsync counts in `to-chk` but FastSync's
|
||||
streaming scanner never emits as a transfer entry.
|
||||
"""
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "a.txt"), b"alpha\n")
|
||||
_write(os.path.join(root, "b.txt"), b"bravo\n")
|
||||
_write(os.path.join(root, "sub1", "c.txt"), b"charlie\n")
|
||||
_write(os.path.join(root, "sub1", "deep", "d.txt"), b"delta\n")
|
||||
_write(os.path.join(root, "sub2", "e.txt"), b"echo\n")
|
||||
os.symlink("a.txt", os.path.join(root, "link1"))
|
||||
os.makedirs(os.path.join(root, "emptydir"), exist_ok=True)
|
||||
os.utime(os.path.join(root, "emptydir"), (_SRC_MTIME, _SRC_MTIME))
|
||||
|
||||
|
||||
def corpus_relative(root: str) -> None:
|
||||
"""Tree for the -R/--files-from cases."""
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "a.txt"), b"a\n")
|
||||
_write(os.path.join(root, "b.txt"), b"b\n")
|
||||
_write(os.path.join(root, "sub", "x.txt"), b"x\n")
|
||||
_write(os.path.join(root, "sub", "y.txt"), b"y\n")
|
||||
os.makedirs(os.path.join(root, "dir1"), exist_ok=True)
|
||||
os.utime(os.path.join(root, "dir1"), (_SRC_MTIME, _SRC_MTIME))
|
||||
_write(os.path.join(root, "dir1", "keep.txt"), b"keep\n")
|
||||
|
||||
|
||||
def corpus_iconv(root: str) -> None:
|
||||
"""Latin-1 (ISO-8859-1) encoded filenames, matching the --iconv direction."""
|
||||
clean_dir(root)
|
||||
for rel, data in ((b"caf\xe9.txt", b"caf\xe9\n"),
|
||||
(os.path.join(b"sub", b"\xfcber.txt"), b"\xfcber\n")):
|
||||
full = os.path.join(os.fsencode(root), rel)
|
||||
os.makedirs(os.path.dirname(full), exist_ok=True)
|
||||
with open(full, "wb") as fh:
|
||||
fh.write(data)
|
||||
os.utime(full, (_SRC_MTIME, _SRC_MTIME))
|
||||
|
||||
|
||||
# Payload for the --fuzzy basis corpus: large enough for the delta engine's
|
||||
# 16 KiB minimum and with repeated content so a coinciding basis yields a
|
||||
# non-zero (and identical) Matched data count in both tools.
|
||||
FUZZY_PAYLOAD = (b"the quick brown fox jumps over the lazy dog\n" * 2000)[:65536]
|
||||
|
||||
|
||||
def corpus_fuzzy(root: str) -> None:
|
||||
"""A named regular file; the `fuzzy` seed adds the similar-suffix sibling."""
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "report_v2.txt"), FUZZY_PAYLOAD)
|
||||
|
||||
|
||||
CORPORA: Dict[str, Callable[[str], None]] = {
|
||||
"basic": corpus_basic,
|
||||
"unicode": corpus_unicode,
|
||||
"links": corpus_links,
|
||||
"hardlinks": corpus_hardlinks,
|
||||
"sparse": corpus_sparse,
|
||||
"filters": corpus_filters,
|
||||
"empty_dir": corpus_empty_dir,
|
||||
"multidir": corpus_multidir,
|
||||
"relative": corpus_relative,
|
||||
"iconv": corpus_iconv,
|
||||
"fuzzy": corpus_fuzzy,
|
||||
}
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Tree snapshotting / comparison
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def snapshot(root: str, compare_modes: bool = False) -> Dict[str, tuple]:
|
||||
"""Map relative path -> descriptor for every entry below ``root``.
|
||||
|
||||
Files hash their contents with SHA-256 (structural comparison, so differing
|
||||
quick-check metadata cannot mask a payload difference). Symlinks record
|
||||
their target. Empty directories are included (as ``("dir", ...)``) so the
|
||||
recursive-empty-directory residual is observable.
|
||||
"""
|
||||
out: Dict[str, tuple] = {}
|
||||
if not os.path.isdir(root):
|
||||
return out
|
||||
|
||||
def describe(path: str) -> Optional[tuple]:
|
||||
st = os.lstat(path)
|
||||
if os.path.islink(path):
|
||||
return ("link", os.readlink(path))
|
||||
if os.path.isdir(path):
|
||||
mode = oct(st.st_mode & 0o7777) if compare_modes else None
|
||||
return ("dir", mode)
|
||||
h = hashlib.sha256()
|
||||
with open(path, "rb") as fh:
|
||||
for chunk in iter(lambda: fh.read(65536), b""):
|
||||
h.update(chunk)
|
||||
mode = oct(st.st_mode & 0o7777) if compare_modes else None
|
||||
return ("file", h.hexdigest()[:16], mode)
|
||||
|
||||
# The comparison root itself is not part of the tree diff: a no-transfer
|
||||
# result legitimately leaves FastSync's mirror directory absent while rsync
|
||||
# leaves an existing (empty) destination root.
|
||||
for dirpath, dirnames, filenames in os.walk(root, followlinks=False):
|
||||
dirnames.sort()
|
||||
for name in sorted(dirnames):
|
||||
p = os.path.join(dirpath, name)
|
||||
rel = os.path.relpath(p, root)
|
||||
if os.path.islink(p):
|
||||
out[rel] = ("link", os.readlink(p))
|
||||
dirnames.remove(name)
|
||||
else:
|
||||
out[rel] = describe(p)
|
||||
for name in sorted(filenames):
|
||||
p = os.path.join(dirpath, name)
|
||||
out[os.path.relpath(p, root)] = describe(p)
|
||||
return out
|
||||
|
||||
|
||||
def _hardlink_groups(root: str) -> Dict[str, str]:
|
||||
"""Assign a stable group letter to each inode shared by >1 regular file."""
|
||||
inodes: Dict[tuple, List[str]] = {}
|
||||
for dirpath, _dirs, filenames in os.walk(root, followlinks=False):
|
||||
for name in filenames:
|
||||
p = os.path.join(dirpath, name)
|
||||
if os.path.islink(p):
|
||||
continue
|
||||
st = os.lstat(p)
|
||||
if st.st_nlink > 1:
|
||||
inodes.setdefault((st.st_dev, st.st_ino), []).append(
|
||||
os.path.relpath(p, root))
|
||||
groups: Dict[str, str] = {}
|
||||
for i, (_key, members) in enumerate(sorted(inodes.items())):
|
||||
for rel in members:
|
||||
groups[rel] = chr(ord("A") + i)
|
||||
return groups
|
||||
|
||||
|
||||
def _drop_ignored(tree: Dict[str, tuple], ignore_paths) -> Dict[str, tuple]:
|
||||
if not ignore_paths:
|
||||
return tree
|
||||
out = {}
|
||||
for rel, desc in tree.items():
|
||||
if any(rel == ig or rel.startswith(ig.rstrip("/") + "/") for ig in ignore_paths):
|
||||
continue
|
||||
out[rel] = desc
|
||||
return out
|
||||
|
||||
|
||||
def tree_diff(rsync_root: str, fs_root: str, case: Case) -> List[str]:
|
||||
"""Return a list of human-readable differences (empty when identical)."""
|
||||
rtree = _drop_ignored(snapshot(rsync_root, case.compare_modes), case.ignore_paths)
|
||||
ftree = _drop_ignored(snapshot(fs_root, case.compare_modes), case.ignore_paths)
|
||||
if case.compare_hardlinks:
|
||||
rgroups = _hardlink_groups(rsync_root)
|
||||
fgroups = _hardlink_groups(fs_root)
|
||||
else:
|
||||
rgroups = fgroups = {}
|
||||
diffs: List[str] = []
|
||||
for rel in sorted(set(rtree) | set(ftree)):
|
||||
r = rtree.get(rel)
|
||||
f = ftree.get(rel)
|
||||
if r == f:
|
||||
continue
|
||||
if r is None:
|
||||
diffs.append(f"+ fastsync-only: {rel!r} {f}")
|
||||
elif f is None:
|
||||
diffs.append(f"- rsync-only: {rel!r} {r}")
|
||||
else:
|
||||
diffs.append(f"~ differs: {rel!r} rsync={r} fastsync={f}")
|
||||
if case.compare_hardlinks:
|
||||
for rel in sorted(set(rgroups) | set(fgroups)):
|
||||
if rgroups.get(rel) != fgroups.get(rel):
|
||||
diffs.append(
|
||||
f"~ hardlink group: {rel!r} rsync={rgroups.get(rel)} "
|
||||
f"fastsync={fgroups.get(rel)}")
|
||||
return diffs
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# stdout normalization
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _parse_bytes(text: str) -> str:
|
||||
m = re.match(r"([\d,]+)", text.strip())
|
||||
return m.group(1).replace(",", "") if m else text.strip()
|
||||
|
||||
|
||||
def normalize_stdout(text: str, mode: Optional[str]) -> object:
|
||||
if mode == STDOUT_ITEMIZE:
|
||||
lines = []
|
||||
for line in (text or "").splitlines():
|
||||
line = line.rstrip()
|
||||
if not line:
|
||||
continue
|
||||
if line.startswith("*deleting"):
|
||||
lines.append(line)
|
||||
continue
|
||||
if not _ITEMIZE_RE.match(line):
|
||||
continue
|
||||
# Directories are not transfer entries in FastSync's recursive
|
||||
# scanner, so rsync's `cd+++++++++ name/` lines have no counterpart
|
||||
# (documented recursive-empty-dir residual). Compare file/link
|
||||
# itemization only.
|
||||
if line.rsplit(" ", 1)[-1].endswith("/"):
|
||||
continue
|
||||
lines.append(line)
|
||||
return sorted(lines)
|
||||
if mode == STDOUT_OUTFMT:
|
||||
lines = []
|
||||
for line in (text or "").splitlines():
|
||||
line = line.rstrip()
|
||||
if not line:
|
||||
continue
|
||||
# Directory entries are emitted by rsync but not by FastSync's
|
||||
# recursive scanner (documented residual). Tokens are either
|
||||
# `%n %l` (path first) or `%i %n` (path last); drop a line when
|
||||
# either end-token is a directory path.
|
||||
first = line.split(" ", 1)[0]
|
||||
last = line.rsplit(" ", 1)[-1]
|
||||
if first.endswith("/") or last.endswith("/"):
|
||||
continue
|
||||
lines.append(line)
|
||||
return sorted(lines)
|
||||
if mode == STDOUT_STATS:
|
||||
found = {}
|
||||
for line in (text or "").splitlines():
|
||||
for key in STATS_KEYS:
|
||||
if line.startswith(key + ":"):
|
||||
found[key] = _parse_bytes(line.split(":", 1)[1])
|
||||
return found
|
||||
if mode == STDOUT_PROGRESS:
|
||||
# rsync prints the file-list entries in sorted depth-first order while
|
||||
# FastSync's streaming scan emits them in readdir/BFS order; only the
|
||||
# entry set and deterministic fields are compared. The transfer-root
|
||||
# `./` line's trigger condition is a separate documented residual, and
|
||||
# the per-frame rate/elapsed/xfr#/to-chk numerator are wall-clock- or
|
||||
# order-dependent, so only the `to-chk` denominator and the name set are
|
||||
# asserted.
|
||||
names = []
|
||||
totals = set()
|
||||
max_xfr = 0
|
||||
for line in (text or "").splitlines():
|
||||
line = line.rstrip()
|
||||
if not line:
|
||||
continue
|
||||
if "%" in line:
|
||||
m = _PROGRESS_TOTAL_RE.search(line)
|
||||
if m:
|
||||
totals.add(int(m.group(1)))
|
||||
mx = _PROGRESS_XFR_RE.search(line)
|
||||
if mx:
|
||||
max_xfr = max(max_xfr, int(mx.group(1)))
|
||||
continue
|
||||
if line == "sending incremental file list":
|
||||
continue
|
||||
if line.startswith("created directory "):
|
||||
continue
|
||||
if line == "./":
|
||||
continue
|
||||
names.append(line)
|
||||
return {"names": sorted(names), "total": sorted(totals), "xfr": max_xfr}
|
||||
# raw
|
||||
return sorted(l.rstrip() for l in (text or "").splitlines() if l.strip())
|
||||
|
||||
|
||||
def stdout_diff(rsync_out: str, fs_out: str, mode: Optional[str]) -> List[str]:
|
||||
r = normalize_stdout(rsync_out, mode)
|
||||
f = normalize_stdout(fs_out, mode)
|
||||
if r == f:
|
||||
return []
|
||||
if mode == STDOUT_STATS:
|
||||
return [f"stats rsync={r}", f"stats fastsync={f}"]
|
||||
if mode == STDOUT_PROGRESS:
|
||||
return [f"progress rsync={r}", f"progress fastsync={f}"]
|
||||
return list(difflib.unified_diff(
|
||||
[str(x) for x in r], [str(x) for x in f],
|
||||
fromfile="rsync", tofile="fastsync", lineterm=""))
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Running one case
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def run_rsync(src: str, rdst: str, flags: List[str]) -> subprocess.CompletedProcess:
|
||||
args = [RSYNC] + list(flags) + [src + "/", rdst + "/"]
|
||||
return subprocess.run(
|
||||
args, capture_output=True, text=True,
|
||||
env=dict(os.environ, LC_ALL="C"), timeout=180)
|
||||
|
||||
|
||||
def run_fastsync(src: str, fdst: str, flags: List[str], port: int):
|
||||
return run_client(src, fdst, flags=list(flags), port=port)
|
||||
|
||||
|
||||
def run_differential( # noqa: PLR0913 (explicit scenario parameters)
|
||||
src: str,
|
||||
rdst: str,
|
||||
fdst: str,
|
||||
rs_flags: List[str],
|
||||
fs_flags: List[str],
|
||||
server,
|
||||
layout: str = MIRROR,
|
||||
seed: Optional[Callable] = None,
|
||||
stdout: Optional[str] = STDOUT_NONE,
|
||||
compare_modes: bool = False,
|
||||
compare_hardlinks: bool = False,
|
||||
ignore_paths: Tuple[str, ...] = (),
|
||||
extra_check: Optional[Callable] = None,
|
||||
files_from: Optional[Tuple[str, ...]] = None,
|
||||
fs_src_suffix: str = "",
|
||||
compare_tree: bool = True,
|
||||
) -> Dict[str, object]:
|
||||
"""Run one rsync/FastSync pair and return the diff aspects.
|
||||
|
||||
Returned dict keys: ``rsync_rc``, ``fastsync_rc``, ``rsync_stderr``,
|
||||
``fastsync_stderr``, ``tree``, ``stdout``, ``extra``.
|
||||
"""
|
||||
clean_dir(rdst)
|
||||
clean_dir(fdst)
|
||||
abs_src = os.path.abspath(src)
|
||||
rel = abs_src.lstrip(os.sep)
|
||||
if layout == RELATIVE:
|
||||
rroot, froot = rdst, fdst
|
||||
elif layout == MIRROR_ABS:
|
||||
rroot, froot = os.path.join(rdst, rel), get_dest_received_dir(fdst, src)
|
||||
else:
|
||||
rroot, froot = rdst, get_dest_received_dir(fdst, src)
|
||||
# rsync's destination root always exists (clean_dir created it). FastSync's
|
||||
# logical transfer root is the mirror path below the destination argument,
|
||||
# so pre-create it too: `Number of created files` counts the root only when
|
||||
# it is genuinely absent, and the two tools must start from the same state.
|
||||
os.makedirs(froot, exist_ok=True)
|
||||
if seed:
|
||||
seed(src, rroot, froot)
|
||||
|
||||
rs_flags = list(rs_flags)
|
||||
fs_flags = list(fs_flags)
|
||||
if files_from is not None:
|
||||
list_path = os.path.join(TEST_DATA_DIR, "parity_" +
|
||||
os.path.basename(src) + ".list")
|
||||
write_list(list_path, files_from)
|
||||
rs_flags.append(f"--files-from={list_path}")
|
||||
fs_flags.append(f"--files-from={list_path}")
|
||||
|
||||
rs = run_rsync(src, rdst, rs_flags)
|
||||
fs_result, _ = run_fastsync(src + fs_src_suffix, fdst, fs_flags, server.port)
|
||||
|
||||
class _View:
|
||||
"""Adapter so tree_diff/extra_check keep the Case-shaped interface."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
self.compare_modes = compare_modes
|
||||
self.compare_hardlinks = compare_hardlinks
|
||||
self.ignore_paths = ignore_paths
|
||||
|
||||
result = {
|
||||
"rsync_rc": rs.returncode,
|
||||
"fastsync_rc": fs_result.returncode,
|
||||
"rsync_stderr": rs.stderr,
|
||||
"fastsync_stderr": fs_result.stderr or fs_result.stdout,
|
||||
"tree": tree_diff(rroot, froot, _View()) if compare_tree else [],
|
||||
"stdout": [],
|
||||
"extra": [],
|
||||
}
|
||||
if stdout is not None:
|
||||
result["stdout"] = stdout_diff(rs.stdout, fs_result.stdout, stdout)
|
||||
if extra_check:
|
||||
result["extra"] = list(extra_check(src, rroot, froot, rs, fs_result) or [])
|
||||
return result
|
||||
|
||||
|
||||
def execute_case(case: Case, server) -> Dict[str, object]:
|
||||
"""Run a table-driven case and return the diff aspects."""
|
||||
tag = case.id
|
||||
src = os.path.join(TEST_DATA_DIR, f"parity_{tag}_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"parity_{tag}_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, f"parity_{tag}_fdst")
|
||||
CORPORA[case.corpus](src)
|
||||
return run_differential(
|
||||
src, rdst, fdst,
|
||||
case.flags, case.fs_flags(), server,
|
||||
layout=case.layout, seed=case.seed, stdout=case.stdout,
|
||||
compare_modes=case.compare_modes, compare_hardlinks=case.compare_hardlinks,
|
||||
ignore_paths=case.ignore_paths, extra_check=case.extra_check,
|
||||
files_from=case.files_from, fs_src_suffix=case.fs_src_suffix,
|
||||
compare_tree=case.compare_tree,
|
||||
)
|
||||
|
||||
|
||||
def write_list(path: str, entries) -> str:
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "w", encoding="utf-8") as fh:
|
||||
for e in entries:
|
||||
fh.write(e + "\n")
|
||||
return path
|
||||
@@ -0,0 +1,213 @@
|
||||
"""Differential tests for the client CLI's codec defaults and env lists.
|
||||
|
||||
Track 3a of the rsync-parity plan pins two rsync 3.4.1 behaviors that are
|
||||
resolved entirely on the client:
|
||||
|
||||
* the per-codec default ``--compress-level`` (zstd 3, zlib/zlibx 6, lz4
|
||||
ignored) applied when the user omits ``--compress-level``/``--zl``, with an
|
||||
explicit level clamped to the codec's range; and
|
||||
* the ``RSYNC_COMPRESS_LIST`` / ``RSYNC_CHECKSUM_LIST`` preference lists that
|
||||
rsync's ``auto`` consults before its compiled-in order (whitespace-separated,
|
||||
unknown names skipped, first supported wins, all-unknown is exit 4).
|
||||
|
||||
The rsync side is observed through ``--debug=NSTR1``; FastSync publishes its
|
||||
resolved codec/level through ``--debug=util``. The checksum side is confirmed
|
||||
byte-for-byte through ``--out-format %C``. The rsync-based tests skip cleanly
|
||||
when rsync is not installed.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import (
|
||||
TEST_DATA_DIR,
|
||||
run_client,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
CODEC_ROOT = os.path.join(TEST_DATA_DIR, "cli_differential")
|
||||
|
||||
_COMPRESS_RE = re.compile(r"compress(?:ion)?: (\w+) \(level (-?\d+)\)")
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
def _scratch(tag):
|
||||
path = os.path.join(CODEC_ROOT, tag)
|
||||
clean_dir(path)
|
||||
os.makedirs(path, exist_ok=True)
|
||||
return path
|
||||
|
||||
|
||||
def _make_corpus(root):
|
||||
clean_dir(root)
|
||||
os.makedirs(root, exist_ok=True)
|
||||
with open(os.path.join(root, "big.bin"), "wb") as fh:
|
||||
fh.write(b"FastSync codec payload " * 4096)
|
||||
with open(os.path.join(root, "small.txt"), "wb") as fh:
|
||||
fh.write(b"hello codec world\n" * 32)
|
||||
return root
|
||||
|
||||
|
||||
def _rsync_compress_level(choice, level):
|
||||
src = _make_corpus(_scratch(f"lvl_src_{choice}_{level}"))
|
||||
dst = _scratch(f"lvl_rsync_{choice}_{level}")
|
||||
args = ["-a", "-z", f"--zc={choice}"]
|
||||
if level is not None:
|
||||
args.append(f"--zl={level}")
|
||||
args += ["--debug=NSTR1", src + "/", dst + "/"]
|
||||
result = _rsync(args)
|
||||
assert result.returncode == 0, result.stderr
|
||||
match = _COMPRESS_RE.search(result.stdout + result.stderr)
|
||||
assert match, (result.stdout, result.stderr)
|
||||
return match.group(1), int(match.group(2))
|
||||
|
||||
|
||||
def _fastsync_compress_level(choice, level, shared_server):
|
||||
src = _make_corpus(_scratch(f"lvl_src_fs_{choice}_{level}"))
|
||||
dst = _scratch(f"lvl_fs_{choice}_{level}")
|
||||
args = ["-a", "-z", f"--zc={choice}"]
|
||||
if level is not None:
|
||||
args.append(f"--zl={level}")
|
||||
args += ["-v", "--debug=util"]
|
||||
result, _ = run_client(src, dst, flags=args, port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
match = _COMPRESS_RE.search(result.stdout)
|
||||
assert match, result.stdout[:500]
|
||||
return match.group(1), int(match.group(2))
|
||||
|
||||
|
||||
class TestPerCodecCompressionLevelDefaults:
|
||||
"""``--compress-level`` defaults and clamping match rsync per codec."""
|
||||
|
||||
# FastSync uses a positive lz4 placeholder because its "level > 0" gate
|
||||
# enables compression; lz4_compress ignores the value, so rsync's level 0
|
||||
# and FastSync's level 1 produce the same bytes.
|
||||
CASES = [
|
||||
("zstd", None, 3),
|
||||
("zlib", None, 6),
|
||||
("zlibx", None, 6),
|
||||
("lz4", None, 1),
|
||||
("zstd", 10, 10),
|
||||
("zlib", 15, 9),
|
||||
("zlib", 3, 3),
|
||||
("lz4", 15, 1),
|
||||
]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("choice,level,fs_level", CASES)
|
||||
def test_level_matches_rsync(self, choice, level, fs_level, shared_server):
|
||||
rsync_algo, rsync_level = _rsync_compress_level(choice, level)
|
||||
fs_algo, fs_level_actual = _fastsync_compress_level(choice, level, shared_server)
|
||||
assert rsync_algo == choice
|
||||
assert fs_algo == choice
|
||||
if choice == "lz4":
|
||||
assert rsync_level == 0 and fs_level_actual > 0
|
||||
else:
|
||||
assert rsync_level == fs_level
|
||||
assert fs_level_actual == fs_level
|
||||
|
||||
|
||||
class TestEnvPreferenceLists:
|
||||
"""``RSYNC_COMPRESS_LIST`` / ``RSYNC_CHECKSUM_LIST`` drive auto like rsync."""
|
||||
|
||||
# (env value, expected codec, rsync level, FastSync level)
|
||||
COMPRESS_CASES = [
|
||||
("zlib lz4", "zlib", 6, 6),
|
||||
("lz4 zstd", "lz4", 0, 1),
|
||||
("bogus zstd zlib", "zstd", 3, 3),
|
||||
(" ", "zstd", 3, 3),
|
||||
]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("env,algo,rsync_level,fs_level", COMPRESS_CASES)
|
||||
def test_compress_list_matches_rsync(self, env, algo, rsync_level, fs_level, shared_server,
|
||||
monkeypatch):
|
||||
monkeypatch.setenv("RSYNC_COMPRESS_LIST", env)
|
||||
src = _make_corpus(_scratch(f"envc_src_{algo}"))
|
||||
rdst = _scratch(f"envc_rsync_{algo}")
|
||||
rs = _rsync(["-a", "-z", "--debug=NSTR1", src + "/", rdst + "/"])
|
||||
assert rs.returncode == 0, rs.stderr
|
||||
rm = _COMPRESS_RE.search(rs.stdout + rs.stderr)
|
||||
assert rm, (rs.stdout, rs.stderr)
|
||||
assert rm.group(1) == algo
|
||||
assert int(rm.group(2)) == rsync_level
|
||||
|
||||
fdst = _scratch(f"envc_fs_{algo}")
|
||||
result, _ = run_client(src, fdst, flags=["-a", "-z", "-v", "--debug=util"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
fm = _COMPRESS_RE.search(result.stdout)
|
||||
assert fm, result.stdout[:500]
|
||||
assert fm.group(1) == algo
|
||||
assert int(fm.group(2)) == fs_level
|
||||
received = get_dest_received_dir(fdst, src)
|
||||
assert _tree_bytes(received) == _tree_bytes(src)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("env,algo", [("md5", "md5"), ("sha1", "sha1"), ("xxh3 md5", "xxh3")])
|
||||
def test_checksum_list_matches_rsync(self, env, algo, shared_server, monkeypatch):
|
||||
monkeypatch.setenv("RSYNC_CHECKSUM_LIST", env)
|
||||
src = _make_corpus(_scratch(f"envcc_src_{algo}"))
|
||||
rdst = _scratch(f"envcc_rsync_{algo}")
|
||||
rs = _rsync(["-a", "--checksum", "--out-format=%C %n", src + "/", rdst + "/"])
|
||||
assert rs.returncode == 0, rs.stderr
|
||||
rs_digests = _digests(rs.stdout)
|
||||
|
||||
fdst = _scratch(f"envcc_fs_{algo}")
|
||||
result, _ = run_client(src, fdst, flags=["-a", "--checksum", "--out-format=%C %n"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert _digests(result.stdout) == rs_digests
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_all_unknown_lists_fail_like_rsync(self, shared_server, monkeypatch):
|
||||
src = _make_corpus(_scratch("envbad_src"))
|
||||
monkeypatch.setenv("RSYNC_COMPRESS_LIST", "bogus")
|
||||
rs = _rsync(["-a", "-z", src + "/", _scratch("envbad_rsync_c") + "/"])
|
||||
assert rs.returncode == 4, rs.stderr
|
||||
result, _ = run_client(src, _scratch("envbad_fs_c"), flags=["-a", "-z"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 4, (result.stderr or result.stdout)[:200]
|
||||
|
||||
monkeypatch.setenv("RSYNC_CHECKSUM_LIST", "bogus")
|
||||
rs = _rsync(["-a", src + "/", _scratch("envbad_rsync_s") + "/"])
|
||||
assert rs.returncode == 4, rs.stderr
|
||||
result, _ = run_client(src, _scratch("envbad_fs_s"), flags=["-a"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 4, (result.stderr or result.stdout)[:200]
|
||||
|
||||
|
||||
def _tree_bytes(root):
|
||||
out = {}
|
||||
for dirpath, _dirs, files in os.walk(root):
|
||||
for name in files:
|
||||
path = os.path.join(dirpath, name)
|
||||
with open(path, "rb") as fh:
|
||||
out[os.path.relpath(path, root)] = fh.read()
|
||||
return out
|
||||
|
||||
|
||||
def _digests(output):
|
||||
out = {}
|
||||
for line in output.splitlines():
|
||||
parts = line.split()
|
||||
if len(parts) == 2 and parts[0]:
|
||||
out[parts[1]] = parts[0]
|
||||
return out
|
||||
@@ -0,0 +1,367 @@
|
||||
"""Differential tests for --checksum-choice / --compress-choice against rsync 3.4.1.
|
||||
|
||||
These pin the accepted/rejected algorithm matrix and exit codes to real rsync,
|
||||
and verify that every codec FastSync now offers still transfers byte-exactly.
|
||||
The rsync-based tests skip cleanly when rsync is not installed.
|
||||
|
||||
The FastSync server confines transfers to its authorized root (the project
|
||||
directory when the shared test server is launched), so every scratch tree lives
|
||||
under ``TEST_DATA_DIR`` rather than pytest's ``tmp_path``.
|
||||
"""
|
||||
import os
|
||||
import random
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import (
|
||||
TEST_DATA_DIR,
|
||||
run_client,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
CHECKSUM_NAMES = ["xxh128", "xxh3", "xxh64", "md5", "md4", "sha1"]
|
||||
COMPRESS_NAMES = ["zstd", "lz4", "zlib", "zlibx"]
|
||||
|
||||
CODEC_ROOT = os.path.join(TEST_DATA_DIR, "codec_differential")
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
def _scratch(tag):
|
||||
"""A confined, uniquely named scratch directory under the project tree."""
|
||||
path = os.path.join(CODEC_ROOT, tag)
|
||||
clean_dir(path)
|
||||
os.makedirs(path, exist_ok=True)
|
||||
return path
|
||||
|
||||
|
||||
def _make_corpus(root):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "sub"), exist_ok=True)
|
||||
# Highly compressible payload so each codec is actually exercised.
|
||||
with open(os.path.join(root, "big.bin"), "wb") as fh:
|
||||
fh.write(b"FastSync codec payload " * 4096)
|
||||
with open(os.path.join(root, "sub", "text.txt"), "wb") as fh:
|
||||
fh.write(b"hello codec world\n" * 128)
|
||||
with open(os.path.join(root, "empty"), "wb"):
|
||||
pass
|
||||
return root
|
||||
|
||||
|
||||
def _tree_bytes(root):
|
||||
out = {}
|
||||
for dirpath, _dirs, files in os.walk(root):
|
||||
for name in files:
|
||||
path = os.path.join(dirpath, name)
|
||||
with open(path, "rb") as fh:
|
||||
out[os.path.relpath(path, root)] = fh.read()
|
||||
return out
|
||||
|
||||
|
||||
_CC_DELTA_T0 = 1_600_000_000
|
||||
_CC_DELTA_T1 = 1_600_000_100
|
||||
|
||||
_DELTA_STATS_KEYS = (
|
||||
"Number of created files",
|
||||
"Number of regular files transferred",
|
||||
"Total transferred file size",
|
||||
"Literal data",
|
||||
"Matched data",
|
||||
)
|
||||
|
||||
|
||||
def _pin_tree(root, mtime):
|
||||
for dirpath, dirnames, filenames in os.walk(root):
|
||||
for name in dirnames + filenames:
|
||||
path = os.path.join(dirpath, name)
|
||||
if not os.path.islink(path):
|
||||
os.utime(path, (mtime, mtime))
|
||||
os.utime(root, (mtime, mtime))
|
||||
|
||||
|
||||
def _make_delta_basis(src, size=512 * 1024):
|
||||
"""Build a source and a matching pre-modification basis tree.
|
||||
|
||||
The source's ``big.bin`` is then modified in a few disjoint places and given
|
||||
a newer mtime so both tools take the delta path. Returns the basis dir.
|
||||
"""
|
||||
clean_dir(src)
|
||||
original = random.Random(20240101).randbytes(size)
|
||||
with open(os.path.join(src, "big.bin"), "wb") as fh:
|
||||
fh.write(original)
|
||||
with open(os.path.join(src, "small.txt"), "wb") as fh:
|
||||
fh.write(b"hello world\n")
|
||||
_pin_tree(src, _CC_DELTA_T0)
|
||||
|
||||
basis = src.rstrip("/") + "_basis"
|
||||
clean_dir(basis)
|
||||
shutil.copy2(os.path.join(src, "big.bin"), os.path.join(basis, "big.bin"))
|
||||
shutil.copy2(os.path.join(src, "small.txt"), os.path.join(basis, "small.txt"))
|
||||
_pin_tree(basis, _CC_DELTA_T0)
|
||||
|
||||
modified = bytearray(original)
|
||||
for off in (0, size // 3, 2 * size // 3, size - 64):
|
||||
for i in range(32):
|
||||
modified[off + i] ^= 0x5A
|
||||
with open(os.path.join(src, "big.bin"), "wb") as fh:
|
||||
fh.write(bytes(modified))
|
||||
os.utime(os.path.join(src, "big.bin"), (_CC_DELTA_T1, _CC_DELTA_T1))
|
||||
return basis
|
||||
|
||||
|
||||
def _seed_from_basis(basis, target):
|
||||
clean_dir(target)
|
||||
for name in os.listdir(basis):
|
||||
shutil.copy2(os.path.join(basis, name), os.path.join(target, name))
|
||||
|
||||
|
||||
def _delta_stats(text):
|
||||
found = {}
|
||||
for line in text.splitlines():
|
||||
for key in _DELTA_STATS_KEYS:
|
||||
if line.startswith(key + ":"):
|
||||
found[key] = line.split(":", 1)[1].strip()
|
||||
return found
|
||||
|
||||
|
||||
def _big_bin_outfmt(text):
|
||||
"""The ``(c, C)`` pair from the ``big.bin`` out-format line (`%c|%C %n`)."""
|
||||
for line in text.splitlines():
|
||||
stripped = line.strip()
|
||||
if "|" not in stripped or not stripped.endswith("big.bin"):
|
||||
continue
|
||||
c_field, rest = stripped.split("|", 1)
|
||||
fields = rest.split()
|
||||
return c_field.strip(), (fields[0] if fields else "")
|
||||
return None, None
|
||||
|
||||
|
||||
class TestCodecChoiceMatrix:
|
||||
"""The CLI accept/reject set and exit codes must match rsync 3.4.1."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("name", CHECKSUM_NAMES)
|
||||
def test_checksum_names_accepted_by_both(self, name, shared_server):
|
||||
src = _make_corpus(_scratch(f"cc_src_{name}"))
|
||||
rdst = _scratch(f"cc_rsync_{name}")
|
||||
rsync_result = _rsync(["-a", f"--cc={name}", src + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
fdst = _scratch(f"cc_fs_{name}")
|
||||
result, _ = run_client(src, fdst, flags=[f"--cc={name}"], port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("name", COMPRESS_NAMES)
|
||||
def test_compress_names_accepted_by_both(self, name, shared_server):
|
||||
src = _make_corpus(_scratch(f"zc_src_{name}"))
|
||||
rdst = _scratch(f"zc_rsync_{name}")
|
||||
rsync_result = _rsync(["-az", f"--zc={name}", src + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
fdst = _scratch(f"zc_fs_{name}")
|
||||
result, _ = run_client(src, fdst, flags=["-z", f"--zc={name}"], port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("choice", ["md4,sha1", "sha1,md4", "auto,md5", "none,md5"])
|
||||
def test_checksum_two_name_accepted_by_both(self, choice, shared_server):
|
||||
tag = choice.replace(",", "_")
|
||||
src = _make_corpus(_scratch(f"two_src_{tag}"))
|
||||
rdst = _scratch(f"two_rsync_{tag}")
|
||||
rsync_result = _rsync(["-a", "--checksum", f"--cc={choice}", src + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
fdst = _scratch(f"two_fs_{tag}")
|
||||
result, _ = run_client(src, fdst, flags=["--checksum", f"--cc={choice}"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("name", ["sha256", "crc32", "md5,", "md4,md5,sha1"])
|
||||
def test_unknown_checksum_rejected_exit_4_both(self, name, shared_server):
|
||||
src = _make_corpus(_scratch(f"badcc_src_{name.replace(',', '_').replace(':', '_')}"))
|
||||
rdst = _scratch(f"badcc_rsync_{name.replace(',', '_').replace(':', '_')}")
|
||||
rsync_result = _rsync(["-a", f"--cc={name}", src + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 4, rsync_result.stderr
|
||||
|
||||
fdst = _scratch(f"badcc_fs_{name.replace(',', '_').replace(':', '_')}")
|
||||
result, _ = run_client(src, fdst, flags=[f"--cc={name}"], port=shared_server.port)
|
||||
assert result.returncode == 4, (result.stderr or result.stdout)[:200]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("choice", ["none", "md5,none"])
|
||||
def test_checksum_none_with_checksum_rejected_exit_4_both(self, choice, shared_server):
|
||||
tag = choice.replace(",", "_")
|
||||
src = _make_corpus(_scratch(f"nonecc_src_{tag}"))
|
||||
rdst = _scratch(f"nonecc_rsync_{tag}")
|
||||
rsync_result = _rsync(["-a", "--checksum", f"--cc={choice}", src + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 4, rsync_result.stderr
|
||||
|
||||
fdst = _scratch(f"nonecc_fs_{tag}")
|
||||
result, _ = run_client(src, fdst, flags=["--checksum", f"--cc={choice}"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 4, (result.stderr or result.stdout)[:200]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("name", ["bogus", "zstd,lz4"])
|
||||
def test_unknown_compress_rejected_exit_4_both(self, name, shared_server):
|
||||
tag = name.replace(",", "_")
|
||||
src = _make_corpus(_scratch(f"badzc_src_{tag}"))
|
||||
rdst = _scratch(f"badzc_rsync_{tag}")
|
||||
rsync_result = _rsync(["-az", f"--zc={name}", src + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 4, rsync_result.stderr
|
||||
|
||||
fdst = _scratch(f"badzc_fs_{tag}")
|
||||
result, _ = run_client(src, fdst, flags=["-z", f"--zc={name}"], port=shared_server.port)
|
||||
assert result.returncode == 4, (result.stderr or result.stdout)[:200]
|
||||
|
||||
|
||||
class TestCodecTransferDifferential:
|
||||
"""Each codec lands the same bytes rsync lands."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("name", COMPRESS_NAMES + ["none"])
|
||||
def test_compress_codec_matches_rsync_bytes(self, name, shared_server):
|
||||
src = _make_corpus(_scratch(f"byteszc_src_{name}"))
|
||||
rsync_dst = _scratch(f"byteszc_rsync_{name}")
|
||||
rsync_result = _rsync(["-a", "-z", f"--zc={name}", src + "/", rsync_dst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
fs_dst = _scratch(f"byteszc_fs_{name}")
|
||||
result, _ = run_client(src, fs_dst, flags=["-a", "-z", f"--zc={name}"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
received = get_dest_received_dir(fs_dst, src)
|
||||
assert _tree_bytes(received) == _tree_bytes(rsync_dst)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("name", CHECKSUM_NAMES)
|
||||
def test_checksum_codec_matches_rsync_bytes(self, name, shared_server):
|
||||
src = _make_corpus(_scratch(f"bytescc_src_{name}"))
|
||||
rsync_dst = _scratch(f"bytescc_rsync_{name}")
|
||||
rsync_result = _rsync(["-a", "--checksum", f"--cc={name}", src + "/", rsync_dst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
fs_dst = _scratch(f"bytescc_fs_{name}")
|
||||
result, _ = run_client(src, fs_dst, flags=["-a", "--checksum", f"--cc={name}"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
received = get_dest_received_dir(fs_dst, src)
|
||||
assert _tree_bytes(received) == _tree_bytes(rsync_dst)
|
||||
|
||||
|
||||
class TestChecksumChoiceDeltaSurface:
|
||||
"""--checksum-choice does not move the delta-transfer parity surface.
|
||||
|
||||
FastSync's delta BLOCK strong checksum is a fixed xxHash32, so the
|
||||
negotiated algorithm only selects the whole-file comparison digest (and the
|
||||
``%C`` transfer digest). A pre-seeded delta transfer must therefore land
|
||||
byte-identical bytes and report the same counters for every choice, while
|
||||
``%C`` -- the one token that tracks the choice -- stays byte-identical to
|
||||
rsync. This pins the Track-3b reclassification in RSYNC_COMPAT.md.
|
||||
"""
|
||||
|
||||
CHOICES = ["xxh64", "xxh128", "xxh3", "md5", "md4", "sha1",
|
||||
"xxh64,sha1", "sha1,xxh64"]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_delta_surface_invariant_to_checksum_choice(self, shared_server):
|
||||
src = _scratch("ccdelta_src")
|
||||
basis = _make_delta_basis(src)
|
||||
source_bytes = _tree_bytes(src)
|
||||
|
||||
rsync_c, fastsync_c, digests = {}, {}, {}
|
||||
rsync_stats, fastsync_stats = {}, {}
|
||||
for choice in self.CHOICES:
|
||||
tag = choice.replace(",", "_")
|
||||
rsync_dst = _scratch(f"ccdelta_rs_{tag}")
|
||||
fs_dst = _scratch(f"ccdelta_fs_{tag}")
|
||||
fs_root = get_dest_received_dir(fs_dst, src)
|
||||
_seed_from_basis(basis, rsync_dst)
|
||||
_seed_from_basis(basis, fs_root)
|
||||
|
||||
# Pin the block size on both ends so the literal/matched split is
|
||||
# comparable (rsync's adaptive default would otherwise differ from
|
||||
# FastSync's 8192-byte default).
|
||||
rsync_result = _rsync(["-a", "--no-whole-file", "-B8192", "--stats",
|
||||
"--out-format=%c|%C %n", f"--cc={choice}",
|
||||
src + "/", rsync_dst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(
|
||||
src, fs_dst,
|
||||
flags=["-a", "--incremental", "--delta", "-B8192", "--stats",
|
||||
"--out-format=%c|%C %n", f"--cc={choice}"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
|
||||
assert _tree_bytes(rsync_dst) == source_bytes, choice
|
||||
assert _tree_bytes(fs_root) == source_bytes, choice
|
||||
assert _delta_stats(rsync_result.stdout) == _delta_stats(result.stdout), choice
|
||||
|
||||
rs_c, rs_C = _big_bin_outfmt(rsync_result.stdout)
|
||||
fs_c, fs_C = _big_bin_outfmt(result.stdout)
|
||||
assert rs_C == fs_C, f"{choice}: %C rsync={rs_C!r} fastsync={fs_C!r}"
|
||||
rsync_c[choice] = rs_c
|
||||
fastsync_c[choice] = fs_c
|
||||
digests[choice] = fs_C
|
||||
rsync_stats[choice] = _delta_stats(rsync_result.stdout)
|
||||
fastsync_stats[choice] = _delta_stats(result.stdout)
|
||||
|
||||
# The choice is only observable in %C, and it is effective (the digests
|
||||
# are not all the same algorithm's output).
|
||||
assert len(set(digests.values())) > 1, digests
|
||||
# The compared --stats counters are invariant across choices in each tool
|
||||
# (and were asserted equal cross-tool inside the loop).
|
||||
assert len({tuple(sorted(s.items())) for s in rsync_stats.values()}) == 1, rsync_stats
|
||||
assert len({tuple(sorted(s.items())) for s in fastsync_stats.values()}) == 1, fastsync_stats
|
||||
# The block-checksum token (%c) is invariant across choices in each tool.
|
||||
assert len(set(rsync_c.values())) == 1, rsync_c
|
||||
assert len(set(fastsync_c.values())) == 1, fastsync_c
|
||||
|
||||
|
||||
class TestCodecNegotiationFallback:
|
||||
"""FastSync's auto negotiation and deterministic fallback order."""
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_default_checksum_and_compression_agree(self, shared_server):
|
||||
"""A default transfer (auto on both peers) succeeds; the negotiated
|
||||
default is xxh128 + zstd."""
|
||||
src = _make_corpus(_scratch("auto_src"))
|
||||
fdst = _scratch("auto_fs")
|
||||
result, _ = run_client(src, fdst, flags=["-a", "-z"], port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
received = get_dest_received_dir(fdst, src)
|
||||
assert _tree_bytes(received) == _tree_bytes(src)
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_explicit_choice_overrides_auto(self, shared_server):
|
||||
"""An explicit --zc/--cc wins over the negotiated default on both ends,
|
||||
so the receiver decodes with the sender's codec."""
|
||||
src = _make_corpus(_scratch("explicit_src"))
|
||||
fdst = _scratch("explicit_fs")
|
||||
result, _ = run_client(src, fdst, flags=["-a", "-z", "--zc=lz4", "--cc=sha1"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
received = get_dest_received_dir(fdst, src)
|
||||
assert _tree_bytes(received) == _tree_bytes(src)
|
||||
@@ -0,0 +1,291 @@
|
||||
"""Differential coverage for the delete-timing ABORT BOUNDARY (A9/A10).
|
||||
|
||||
rsync's generator runs ahead of its throttled sender, so on a mid-transfer abort
|
||||
it has already removed every extra it planned. FastSync now transmits the
|
||||
COMPLETE per-directory plan set before the first data frame, so an abort has the
|
||||
same effect. Before that change FastSync only removed the extras of the
|
||||
directories its (slower) data stream had reached, and ``-d/--dirs`` used an
|
||||
end-of-transfer commit that removed nothing on abort.
|
||||
|
||||
These tests abort both tools mid-transfer and assert the destination extras
|
||||
removed match real ``rsync 3.4.1``. The rsync side is driven locally with
|
||||
``--bwlimit`` and a small timing window (its generator's delete list is computed
|
||||
long before the throttled payload finishes); the FastSync side uses the
|
||||
byte-deterministic slicing proxy from ``test_delete_timing_parity``.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
TEST_DATA_DIR,
|
||||
ServerManager,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
from test_delete_timing_parity import _SlicingProxy # noqa: E402
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
# Exceeds the 10 MiB scanner chunk, so the next directory lands in a later chunk
|
||||
# (still unreached when the proxy cuts the stream).
|
||||
BIG_BYTES = 16 * 1024 * 1024
|
||||
# Cut well past the (small) config + delete-plan frames and into the big payload,
|
||||
# so the receiver has provably processed every plan before the abort.
|
||||
MID_TRANSFER_BYTES = 256 * 1024
|
||||
PROXY_THROTTLE = 0.001
|
||||
# Throttle rsync's sender so the generator has deleted long before the payload
|
||||
# finishes, then interrupt it mid-transfer.
|
||||
RSYNC_BWLIMIT = 512 # KiB/s -> ~32 s for 16 MiB
|
||||
RSYNC_ABORT_DELAY = 1.5
|
||||
|
||||
|
||||
def _write(path, content):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(content)
|
||||
|
||||
|
||||
def _rsync_aborted(args, delay=RSYNC_ABORT_DELAY):
|
||||
"""Start rsync, let its generator run, then interrupt it mid-transfer."""
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
proc = subprocess.Popen([RSYNC] + args, stdout=subprocess.PIPE, stderr=subprocess.PIPE,
|
||||
text=True, env=env)
|
||||
time.sleep(delay)
|
||||
proc.terminate()
|
||||
try:
|
||||
proc.wait(timeout=10)
|
||||
except subprocess.TimeoutExpired:
|
||||
proc.kill()
|
||||
proc.wait(timeout=5)
|
||||
return proc
|
||||
|
||||
|
||||
class TestDeleteDuringAbortBoundary:
|
||||
"""A9: on an abort, every planned removal has already been applied."""
|
||||
|
||||
def _seed_recursive(self, tag):
|
||||
source = os.path.join(TEST_DATA_DIR, f"dab_{tag}_src")
|
||||
clean_dir(source)
|
||||
# ``a/keep.bin`` sorts first, so the client streams it (and the proxy
|
||||
# cuts) before the data pass ever reaches ``z/deep``.
|
||||
_write(os.path.join(source, "a", "keep.bin"), b"B" * BIG_BYTES)
|
||||
_write(os.path.join(source, "z", "deep", "keep.txt"), b"keep\n")
|
||||
return source
|
||||
|
||||
@requires_rsync
|
||||
def test_recursive_abort_removes_all_planned_extras(self):
|
||||
# ---- FastSync: abort mid ``a/keep.bin``; ``z/deep`` is never reached.
|
||||
source = self._seed_recursive("rec_fs")
|
||||
dest = os.path.join(TEST_DATA_DIR, "dab_rec_fs_dst")
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
os.makedirs(os.path.join(received, "a"), exist_ok=True)
|
||||
_write(os.path.join(received, "a", "a_extra"), b"stale\n")
|
||||
os.makedirs(os.path.join(received, "z", "deep"), exist_ok=True)
|
||||
_write(os.path.join(received, "z", "deep", "old_extra"), b"stale\n")
|
||||
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
proxy = _SlicingProxy(server.port, forward_limit=MID_TRANSFER_BYTES,
|
||||
throttle=PROXY_THROTTLE)
|
||||
result, _ = run_client(source, dest, flags=["--delete-during"], port=proxy.port)
|
||||
proxy.finish()
|
||||
assert result.returncode != 0, "truncated transfer reported success"
|
||||
assert not os.path.exists(os.path.join(received, "a", "a_extra"))
|
||||
assert not os.path.exists(os.path.join(received, "z", "deep", "old_extra")), (
|
||||
"FastSync left an extra in a directory it never reached before the abort"
|
||||
)
|
||||
|
||||
# ---- rsync 3.4.1: same tree, same abort, same delete outcome.
|
||||
source = self._seed_recursive("rec_rs")
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, "dab_rec_rs_dst")
|
||||
clean_dir(rsync_dst)
|
||||
os.makedirs(os.path.join(rsync_dst, "a"), exist_ok=True)
|
||||
_write(os.path.join(rsync_dst, "a", "a_extra"), b"stale\n")
|
||||
os.makedirs(os.path.join(rsync_dst, "z", "deep"), exist_ok=True)
|
||||
_write(os.path.join(rsync_dst, "z", "deep", "old_extra"), b"stale\n")
|
||||
|
||||
proc = _rsync_aborted(["-a", "--delete-during", f"--bwlimit={RSYNC_BWLIMIT}",
|
||||
source + "/", rsync_dst + "/"])
|
||||
assert proc.returncode != 0, "rsync was not actually interrupted"
|
||||
assert not os.path.exists(os.path.join(rsync_dst, "a", "a_extra"))
|
||||
assert not os.path.exists(os.path.join(rsync_dst, "z", "deep", "old_extra")), (
|
||||
"rsync's generator did not delete ahead of its sender"
|
||||
)
|
||||
|
||||
|
||||
class TestDirsDeleteAbortBoundary:
|
||||
"""A10: ``-d/--dirs`` uses per-directory plans like rsync.
|
||||
|
||||
The listed directory's direct extras are removed by the up-front plan while
|
||||
a kept but untraversed subdirectory (and its destination content) is
|
||||
shielded.
|
||||
"""
|
||||
|
||||
def _seed_dirs(self, tag):
|
||||
source = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "big.bin"), b"B" * BIG_BYTES)
|
||||
_write(os.path.join(source, "subdir", "keep.txt"), b"inner\n")
|
||||
return source
|
||||
|
||||
@pytest.mark.parametrize("fs_flag,rs_flag", [("--delete-during", "--delete-during"),
|
||||
("--delete", "--delete")])
|
||||
@requires_rsync
|
||||
def test_dirs_abort_removes_direct_extras_only(self, fs_flag, rs_flag):
|
||||
label = f"{fs_flag.lstrip('-')}_{rs_flag.lstrip('-')}"
|
||||
# ---- FastSync: ``-d`` lists the immediate children; big.bin streams and
|
||||
# the abort lands mid-payload.
|
||||
source = self._seed_dirs(f"dirs_{label}_fs")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"ddb_{label}_fs_dst")
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
_write(os.path.join(received, "old_extra"), b"stale\n")
|
||||
os.makedirs(os.path.join(received, "subdir"), exist_ok=True)
|
||||
_write(os.path.join(received, "subdir", "stale.txt"), b"stale inner\n")
|
||||
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
proxy = _SlicingProxy(server.port, forward_limit=MID_TRANSFER_BYTES,
|
||||
throttle=PROXY_THROTTLE)
|
||||
result, _ = run_client(source + "/", dest, flags=["-d", fs_flag], port=proxy.port)
|
||||
proxy.finish()
|
||||
assert result.returncode != 0, f"{fs_flag}: truncated transfer reported success"
|
||||
assert not os.path.exists(os.path.join(received, "old_extra")), (
|
||||
f"{fs_flag}: the listed directory's direct extra survived the abort"
|
||||
)
|
||||
assert os.path.exists(os.path.join(received, "subdir", "stale.txt")), (
|
||||
f"{fs_flag}: descended into a kept, untraversed subdirectory"
|
||||
)
|
||||
|
||||
# ---- rsync 3.4.1: same shape and same abort.
|
||||
source = self._seed_dirs(f"dirs_{label}_rs")
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, f"ddb_{label}_rs_dst")
|
||||
clean_dir(rsync_dst)
|
||||
_write(os.path.join(rsync_dst, "old_extra"), b"stale\n")
|
||||
os.makedirs(os.path.join(rsync_dst, "subdir"), exist_ok=True)
|
||||
_write(os.path.join(rsync_dst, "subdir", "stale.txt"), b"stale inner\n")
|
||||
|
||||
proc = _rsync_aborted(["-d", rs_flag, f"--bwlimit={RSYNC_BWLIMIT}",
|
||||
source + "/", rsync_dst + "/"])
|
||||
assert proc.returncode != 0, "rsync was not actually interrupted"
|
||||
assert not os.path.exists(os.path.join(rsync_dst, "old_extra")), (
|
||||
f"rsync {rs_flag}: the listed directory's direct extra survived the abort"
|
||||
)
|
||||
assert os.path.exists(os.path.join(rsync_dst, "subdir", "stale.txt")), (
|
||||
f"rsync {rs_flag}: descended into a kept, untraversed subdirectory"
|
||||
)
|
||||
|
||||
|
||||
def _tree(root):
|
||||
out = []
|
||||
for dirpath, dirs, files in os.walk(root):
|
||||
for name in dirs:
|
||||
out.append(os.path.relpath(os.path.join(dirpath, name), root))
|
||||
for name in files:
|
||||
out.append(os.path.relpath(os.path.join(dirpath, name), root))
|
||||
return sorted(out)
|
||||
|
||||
|
||||
class TestDirsDeleteFinalStateParity:
|
||||
"""A10 completed run: ``-d DIR/ --delete`` (during default) and
|
||||
``--delete-during`` match rsync's final tree, including a kept but
|
||||
untraversed subdirectory whose destination content survives."""
|
||||
|
||||
@pytest.mark.parametrize("flag", ["--delete", "--delete-during"])
|
||||
@requires_rsync
|
||||
def test_dirs_final_state_matches_rsync(self, flag):
|
||||
source = os.path.join(TEST_DATA_DIR, f"ddf_{flag.lstrip('-')}_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"new keep\n")
|
||||
_write(os.path.join(source, "subdir", "inner.txt"), b"inner\n")
|
||||
|
||||
def seed_dest(root):
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "keep.txt"), b"old keep\n")
|
||||
_write(os.path.join(root, "extra.txt"), b"extra\n")
|
||||
_write(os.path.join(root, "extrasub", "ex.txt"), b"extra sub\n")
|
||||
_write(os.path.join(root, "subdir", "stale.txt"), b"stale inner\n")
|
||||
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, f"ddf_{flag.lstrip('-')}_rs_dst")
|
||||
seed_dest(rsync_dst)
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
rsync_result = subprocess.run(
|
||||
[RSYNC, "-d", flag, source + "/", rsync_dst + "/"],
|
||||
capture_output=True, text=True, env=env, timeout=120)
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_tree = _tree(rsync_dst)
|
||||
|
||||
dest = os.path.join(TEST_DATA_DIR, f"ddf_{flag.lstrip('-')}_fs_dst")
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
seed_dest(received)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source + "/", dest, flags=["-d", flag], port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
fastsync_tree = _tree(received)
|
||||
assert fastsync_tree == rsync_tree, (
|
||||
f"-d {flag}: fastsync tree {fastsync_tree} != rsync tree {rsync_tree}")
|
||||
|
||||
|
||||
class TestOneFileSystemDeleteParity:
|
||||
"""A9 side effect: the per-directory plan is now emitted only for directories
|
||||
whose children were enumerated, so a ``-x`` mount-point directory that is
|
||||
emitted but never traversed is shielded -- its destination content survives,
|
||||
exactly as rsync keeps a non-descended mount point under ``--delete``."""
|
||||
|
||||
@requires_rsync
|
||||
def test_mountpoint_content_survives_delete(self):
|
||||
local = os.stat(".")
|
||||
shm = "/dev/shm"
|
||||
if not os.path.isdir(shm) or os.stat(shm).st_dev == local.st_dev:
|
||||
pytest.skip("no cross-device filesystem available")
|
||||
probe = os.path.join(shm, f"fastsync_dofs_{os.getpid()}")
|
||||
clean_dir(probe)
|
||||
_write(os.path.join(probe, "inside.txt"), b"cross\n")
|
||||
try:
|
||||
source = os.path.join(TEST_DATA_DIR, "dofs_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"keep\n")
|
||||
os.symlink(probe, os.path.join(source, "nested_link"))
|
||||
|
||||
def seed_dest(root):
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "keep.txt"), b"old\n")
|
||||
_write(os.path.join(root, "nested_link", "stale.txt"), b"stale\n")
|
||||
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, "dofs_rs_dst")
|
||||
seed_dest(rsync_dst)
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
rsync_result = subprocess.run(
|
||||
[RSYNC, "-a", "--copy-links", "-x", "--delete-during",
|
||||
source + "/", rsync_dst + "/"],
|
||||
capture_output=True, text=True, env=env, timeout=120)
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
assert os.path.exists(os.path.join(rsync_dst, "nested_link", "stale.txt")), (
|
||||
"rsync unexpectedly descended into the mount point")
|
||||
|
||||
dest = os.path.join(TEST_DATA_DIR, "dofs_fs_dst")
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
seed_dest(received)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--copy-links", "-x", "--delete-during"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert os.path.exists(os.path.join(received, "nested_link", "stale.txt")), (
|
||||
"FastSync descended into a non-traversed mount point under --delete")
|
||||
finally:
|
||||
clean_dir(probe)
|
||||
|
||||
@@ -0,0 +1,185 @@
|
||||
"""Differential coverage for ``--delete-delay`` + ``--max-delete`` with a
|
||||
refilled deferred directory.
|
||||
|
||||
FastSync snapshots a directory's extras at plan time (``defer_add``) but charges
|
||||
``--max-delete`` only when a path is actually removed, and its deferred commit
|
||||
re-scans a queued directory and removes content created after the plan -- the
|
||||
same rules as rsync. These tests run both tools on the same fixture and assert
|
||||
both sides remove the late content (recursively) and bound the deletion with
|
||||
``--max-delete`` identically.
|
||||
|
||||
They are not part of the fast PR gate because the rsync side needs a wide
|
||||
real-time injection window (a throttled transfer), while the FastSync side uses
|
||||
the existing byte-deterministic slicing proxy.
|
||||
|
||||
The refilled directory sits at the transfer ROOT, whose delete plan is always
|
||||
processed before any subdirectory's, so the budget is deterministically charged
|
||||
to the refilled entry; the second extra lives under ``b`` and is skipped.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
TEST_DATA_DIR,
|
||||
ServerManager,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
from test_delete_timing_parity import _SlicingProxy # noqa: E402
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
BIG_BYTES = 8 * 1024 * 1024
|
||||
MID_TRANSFER_BYTES = 256 * 1024
|
||||
PROXY_THROTTLE = 0.001
|
||||
# rsync is driven locally, so the refill is injected on a wall-clock delay while
|
||||
# a throttled ~8 s transfer is in flight. 1.5 s is safely after rsync's plan
|
||||
# scan (t=0) and well before the deferred commit at the end.
|
||||
RSYNC_BWLIMIT = 1024 # 1 MiB/s
|
||||
RSYNC_INJECT_DELAY = 1.5
|
||||
|
||||
|
||||
def _write(path, content):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(content)
|
||||
|
||||
|
||||
def _seed_source(tag):
|
||||
source = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "a", "keep.bin"), b"B" * BIG_BYTES)
|
||||
_write(os.path.join(source, "b", "keep.txt"), b"keep\n")
|
||||
return source
|
||||
|
||||
|
||||
def _seed_fastsync(tag):
|
||||
"""FastSync mirrors the absolute source path under its receive root, so the
|
||||
extras live below ``received``."""
|
||||
source = _seed_source(tag)
|
||||
dest = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_dst")
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
os.makedirs(os.path.join(received, "xdir"), exist_ok=True)
|
||||
os.makedirs(os.path.join(received, "b", "ydir"), exist_ok=True)
|
||||
return source, dest, received
|
||||
|
||||
|
||||
def _seed_rsync(tag):
|
||||
"""rsync mirrors the source contents directly into the destination, so the
|
||||
extras are flat under ``rsync_dst``."""
|
||||
source = _seed_source(tag)
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, f"ddb_{tag}_dst")
|
||||
clean_dir(rsync_dst)
|
||||
os.makedirs(os.path.join(rsync_dst, "xdir"), exist_ok=True)
|
||||
os.makedirs(os.path.join(rsync_dst, "b", "ydir"), exist_ok=True)
|
||||
return source, rsync_dst
|
||||
|
||||
|
||||
def _deleted_count(text):
|
||||
for line in text.splitlines():
|
||||
if line.startswith("Number of deleted files:"):
|
||||
return int(line.split(":", 1)[1].split()[0])
|
||||
return None
|
||||
|
||||
|
||||
def _rsync(args, timeout=120):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=timeout)
|
||||
|
||||
|
||||
class TestDeleteDelayRefilledDirVsRsync:
|
||||
"""Both tools charge --max-delete on actual removals and recurse."""
|
||||
|
||||
def _fastsync_refilled(self, tag, max_delete=None):
|
||||
"""Run FastSync with the refill injected deterministically by the proxy
|
||||
hook (fired once the receiver has processed the plan frames)."""
|
||||
source, dest, received = _seed_fastsync(tag)
|
||||
late = os.path.join(received, "xdir", "new.txt")
|
||||
|
||||
def hook():
|
||||
_write(late, b"created mid-transfer\n")
|
||||
|
||||
flags = ["--delete-delay", "--incremental", "--ignore-times", "--stats"]
|
||||
if max_delete is not None:
|
||||
flags.append(f"--max-delete={max_delete}")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
proxy = _SlicingProxy(server.port, hook=hook, hook_after=MID_TRANSFER_BYTES,
|
||||
throttle=PROXY_THROTTLE, wait_for_reply=True)
|
||||
result, _ = run_client(source, dest, flags=flags, port=proxy.port)
|
||||
proxy.finish()
|
||||
assert proxy.hook_called.is_set(), "refill hook never fired"
|
||||
return result, received, late
|
||||
|
||||
@requires_rsync
|
||||
def test_max_delete_budget_bound_matches(self):
|
||||
# --- FastSync: the one actual removal is the late file; dirs survive ---
|
||||
result, received, late = self._fastsync_refilled("budget_fs", max_delete=1)
|
||||
assert result.returncode == 25, (result.stderr or result.stdout)[:300]
|
||||
assert _deleted_count(result.stdout) == 1, result.stdout
|
||||
assert not os.path.exists(late), "FastSync kept the late content of a queued dir"
|
||||
assert os.path.isdir(os.path.join(received, "xdir"))
|
||||
assert os.path.isdir(os.path.join(received, "b", "ydir")), (
|
||||
"FastSync did not bound the deletion with --max-delete=1"
|
||||
)
|
||||
|
||||
# --- rsync: same budget rule and recursive removal ---
|
||||
source, rsync_dst = _seed_rsync("budget_rsync")
|
||||
|
||||
def inject():
|
||||
time.sleep(RSYNC_INJECT_DELAY)
|
||||
_write(os.path.join(rsync_dst, "xdir", "new.txt"), b"created mid-transfer\n")
|
||||
|
||||
t = threading.Thread(target=inject)
|
||||
t.start()
|
||||
rsync_result = _rsync(
|
||||
["-a", "--delete-delay", "--max-delete=1", "--stats",
|
||||
f"--bwlimit={RSYNC_BWLIMIT}", source + "/", rsync_dst + "/"]
|
||||
)
|
||||
t.join()
|
||||
assert rsync_result.returncode == 25, rsync_result.stderr
|
||||
assert _deleted_count(rsync_result.stdout) == _deleted_count(result.stdout)
|
||||
assert not os.path.exists(os.path.join(rsync_dst, "xdir", "new.txt")), (
|
||||
"rsync kept late content inside a queued directory"
|
||||
)
|
||||
assert os.path.isdir(os.path.join(rsync_dst, "b", "ydir")), (
|
||||
"rsync did not bound the deletion with --max-delete=1"
|
||||
)
|
||||
assert os.path.isdir(os.path.join(received, "b", "ydir"))
|
||||
|
||||
@requires_rsync
|
||||
def test_refilled_extra_dir_recursive_removal_matches(self):
|
||||
"""Without --max-delete both tools remove the refilled extra directory
|
||||
(and its late content)."""
|
||||
result, received, late = self._fastsync_refilled("recur_fs")
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert not os.path.exists(late), "FastSync kept the refilled directory's late content"
|
||||
assert not os.path.isdir(os.path.join(received, "xdir"))
|
||||
|
||||
source, rsync_dst = _seed_rsync("recur_rsync")
|
||||
|
||||
def inject():
|
||||
time.sleep(RSYNC_INJECT_DELAY)
|
||||
_write(os.path.join(rsync_dst, "xdir", "new.txt"), b"created mid-transfer\n")
|
||||
|
||||
t = threading.Thread(target=inject)
|
||||
t.start()
|
||||
rsync_result = _rsync(
|
||||
["-a", "--delete-delay", "--stats", f"--bwlimit={RSYNC_BWLIMIT}",
|
||||
source + "/", rsync_dst + "/"]
|
||||
)
|
||||
t.join()
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
assert not os.path.exists(os.path.join(rsync_dst, "xdir")), (
|
||||
"rsync did not recursively remove the refilled extra directory"
|
||||
)
|
||||
@@ -0,0 +1,539 @@
|
||||
"""Differential + regression coverage for rsync's delete timing.
|
||||
|
||||
``--delete-during``/``--delete-delay`` stream a per-directory delete plan instead
|
||||
of one whole-tree manifest, so the timing is observable:
|
||||
|
||||
* ``--delete-during`` removes a directory's extras as it processes that
|
||||
directory (so an interrupted transfer has already removed the extras of the
|
||||
directories it reached);
|
||||
* ``--delete-delay`` snapshots those extras while scanning and commits the
|
||||
removals only after a fully-successful transfer (so an extra created in the
|
||||
destination after its directory's plan survives, and a failed transfer
|
||||
removes nothing);
|
||||
* ``--delete-after`` re-scans the destination at the end (so that same
|
||||
late-created extra is removed).
|
||||
|
||||
The final-state tests compare against real ``rsync 3.4.1`` where a deterministic
|
||||
comparison exists; the timing tests use a byte-slicing proxy to force a
|
||||
mid-transfer failure or to create a destination entry while the transfer is in
|
||||
flight.
|
||||
"""
|
||||
import os
|
||||
import select
|
||||
import shutil
|
||||
import socket
|
||||
import struct
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
BUILD_DIR,
|
||||
TEST_DATA_DIR,
|
||||
ServerManager,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
|
||||
# Every test here is deterministic (the proxy throttles until the delete-plan
|
||||
# frames are processed), so the PR gate runs the whole module.
|
||||
pytestmark = pytest.mark.ci
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
BIG_BYTES = 8 * 1024 * 1024
|
||||
# Forward/cut this far into the stream: past the (small) delete-plan frames and
|
||||
# well into the big payload, so the receiver has already processed the plan.
|
||||
MID_TRANSFER_BYTES = 256 * 1024
|
||||
# Throttle the proxy so the receiver keeps up with the (fast) client and the
|
||||
# plan frames are provably processed before the hook/cut offset is reached.
|
||||
PROXY_THROTTLE = 0.001
|
||||
|
||||
|
||||
def _write(path, content):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(content)
|
||||
|
||||
|
||||
def _seed_pair(tag, big=False):
|
||||
"""Create a source tree and a destination mirror seeded with extras.
|
||||
|
||||
The tree is a single directory ``d`` containing the transferred files plus,
|
||||
in the destination, an extra ``d/old_extra``.
|
||||
"""
|
||||
source = os.path.join(TEST_DATA_DIR, f"dtp_{tag}_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"dtp_{tag}_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
_write(os.path.join(source, "d", "keep.txt"), b"kept payload\n")
|
||||
if big:
|
||||
_write(os.path.join(source, "d", "big.bin"), b"B" * BIG_BYTES)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
os.makedirs(os.path.join(received, "d"), exist_ok=True)
|
||||
_write(os.path.join(received, "d", "old_extra"), b"stale extra\n")
|
||||
return source, dest, received
|
||||
|
||||
|
||||
def _tree(root):
|
||||
"""Sorted relative paths of every entry below root (files and dirs)."""
|
||||
out = []
|
||||
for dirpath, dirs, files in os.walk(root):
|
||||
for name in dirs:
|
||||
out.append(os.path.relpath(os.path.join(dirpath, name), root))
|
||||
for name in files:
|
||||
out.append(os.path.relpath(os.path.join(dirpath, name), root))
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
class _SlicingProxy:
|
||||
"""Forward the client stream to a server, optionally cutting it or invoking a
|
||||
hook after a byte threshold. ``forward_limit`` mode resets both ends after
|
||||
that many client bytes (a mid-transfer failure). ``hook`` mode calls the
|
||||
hook once and keeps forwarding to completion.
|
||||
|
||||
With ``wait_for_reply`` the hook is a real barrier, not a timing guess: it
|
||||
fires only after the server has sent *any* reply, which the receiver does
|
||||
only after it has consumed the frames that precede the payload (the
|
||||
per-directory delete plan for ``--delete-delay``). The caller pairs it with
|
||||
``--incremental`` so a per-file handshake reply is guaranteed mid-transfer.
|
||||
"""
|
||||
|
||||
def __init__(self, target_port, forward_limit=None, hook=None, hook_after=0,
|
||||
throttle=0.0, wait_for_reply=False):
|
||||
self.target = ("127.0.0.1", target_port)
|
||||
self.forward_limit = forward_limit
|
||||
self.hook = hook
|
||||
self.hook_after = hook_after
|
||||
self.throttle = throttle
|
||||
self.wait_for_reply = wait_for_reply
|
||||
self.server_replied = threading.Event()
|
||||
self.hook_called = threading.Event()
|
||||
self.listener = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
self.listener.setsockopt(socket.SOL_SOCKET, socket.SO_REUSEADDR, 1)
|
||||
self.listener.bind(("127.0.0.1", 0))
|
||||
self.listener.listen(1)
|
||||
self.listener.settimeout(20)
|
||||
self.port = self.listener.getsockname()[1]
|
||||
self._thread = threading.Thread(target=self._serve, daemon=True)
|
||||
self._thread.start()
|
||||
|
||||
def _serve(self):
|
||||
try:
|
||||
client, _ = self.listener.accept()
|
||||
except OSError:
|
||||
return
|
||||
try:
|
||||
backend = socket.create_connection(self.target, timeout=10)
|
||||
except OSError:
|
||||
client.close()
|
||||
return
|
||||
client.settimeout(20)
|
||||
backend.settimeout(20)
|
||||
forwarded = 0
|
||||
socks = [client, backend]
|
||||
try:
|
||||
while socks:
|
||||
ready, _, _ = select.select(socks, [], [], 20)
|
||||
if not ready:
|
||||
break
|
||||
for sock in ready:
|
||||
data = sock.recv(65536)
|
||||
if not data:
|
||||
socks.remove(sock)
|
||||
peer = backend if sock is client else client
|
||||
try:
|
||||
peer.shutdown(socket.SHUT_WR)
|
||||
except OSError:
|
||||
pass
|
||||
continue
|
||||
if sock is client:
|
||||
if self.forward_limit is not None:
|
||||
room = self.forward_limit - forwarded
|
||||
if room <= 0:
|
||||
socks = []
|
||||
break
|
||||
data = data[:room]
|
||||
backend.sendall(data)
|
||||
forwarded += len(data)
|
||||
self._maybe_hook(forwarded)
|
||||
if self.forward_limit is not None and forwarded >= self.forward_limit:
|
||||
socks = []
|
||||
break
|
||||
if self.throttle > 0:
|
||||
time.sleep(self.throttle)
|
||||
else:
|
||||
client.sendall(data)
|
||||
# Any server reply proves the receiver consumed the
|
||||
# frames that precede it, so the hook barrier is met.
|
||||
self.server_replied.set()
|
||||
self._maybe_hook(forwarded)
|
||||
except OSError:
|
||||
pass
|
||||
for sock in (client, backend):
|
||||
try:
|
||||
sock.setsockopt(socket.SOL_SOCKET, socket.SO_LINGER, struct.pack("ii", 1, 0))
|
||||
except OSError:
|
||||
pass
|
||||
try:
|
||||
sock.close()
|
||||
except OSError:
|
||||
pass
|
||||
try:
|
||||
self.listener.close()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
def _maybe_hook(self, forwarded):
|
||||
"""Fire the one-shot hook once its barrier is satisfied: enough client
|
||||
bytes have been forwarded and, when ``wait_for_reply`` is set, the
|
||||
server has sent a reply proving it processed the preceding frames."""
|
||||
if self.hook is None or self.hook_called.is_set():
|
||||
return
|
||||
if forwarded < self.hook_after:
|
||||
return
|
||||
if self.wait_for_reply and not self.server_replied.is_set():
|
||||
return
|
||||
self.hook()
|
||||
self.hook_called.set()
|
||||
|
||||
def finish(self):
|
||||
self._thread.join(30)
|
||||
try:
|
||||
self.listener.close()
|
||||
except OSError:
|
||||
pass
|
||||
|
||||
|
||||
class TestDeleteTimingFinalStateParity:
|
||||
"""On a successful transfer the per-directory timings match rsync's result.
|
||||
|
||||
Plain ``--delete`` has no rsync-incompatible spelling: it defaults to
|
||||
delete-during on both tools, so it is compared against rsync's own default.
|
||||
``--delete-commit`` is FastSync-only and selects the late whole-tree commit,
|
||||
which is rsync's ``--delete-after`` timing.
|
||||
"""
|
||||
|
||||
# (fastsync flag, rsync flag)
|
||||
PAIRS = [
|
||||
("--delete", "--delete"),
|
||||
("--delete-during", "--delete-during"),
|
||||
("--delete-delay", "--delete-delay"),
|
||||
("--delete-commit", "--delete-after"),
|
||||
]
|
||||
|
||||
def _run_fastsync(self, tag, timing):
|
||||
source, dest, received = _seed_pair(tag)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=[timing], port=server.port)
|
||||
return result, received
|
||||
|
||||
@pytest.mark.parametrize("fs_timing,rs_timing", PAIRS)
|
||||
@requires_rsync
|
||||
def test_success_final_state_matches_rsync(self, fs_timing, rs_timing):
|
||||
# Worker-safe names: xdist may run the parametrizations concurrently, so
|
||||
# the flags are part of every fixture path.
|
||||
label = f"{fs_timing.lstrip('-')}_vs_{rs_timing.lstrip('-')}"
|
||||
# Build the rsync fixture from the same seed so both sides start equal.
|
||||
source, dest, received = _seed_pair(f"parity_rsync_{label}")
|
||||
source2 = source
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, f"dtp_rsync_{label}_dst")
|
||||
clean_dir(rsync_dst)
|
||||
# rsync mirrors src/ into dst/; seed the same extra.
|
||||
_write(os.path.join(rsync_dst, "d", "old_extra"), b"stale extra\n")
|
||||
|
||||
rsync_result = _rsync(["-a", rs_timing, source2 + "/", rsync_dst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_tree = _tree(rsync_dst)
|
||||
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=[fs_timing], port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
fastsync_tree = _tree(received)
|
||||
assert fastsync_tree == rsync_tree, (
|
||||
f"{fs_timing} vs rsync {rs_timing}: fastsync tree {fastsync_tree} != "
|
||||
f"rsync tree {rsync_tree}"
|
||||
)
|
||||
|
||||
|
||||
class TestDeleteTimingTypeConflictParity:
|
||||
"""A destination entry whose type differs from the source is replaced, in
|
||||
both per-directory timings and in both directions, exactly like rsync."""
|
||||
|
||||
@pytest.mark.parametrize("timing", ["--delete-during", "--delete-delay"])
|
||||
@requires_rsync
|
||||
def test_type_conflicts_match_rsync(self, timing):
|
||||
label = timing.lstrip("-")
|
||||
source = os.path.join(TEST_DATA_DIR, f"dtc_{label}_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "foo"), b"now a file\n")
|
||||
_write(os.path.join(source, "bar", "inner.txt"), b"now a dir\n")
|
||||
|
||||
def seed_dest(root):
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "foo", "inner.txt"), b"was a dir\n")
|
||||
_write(os.path.join(root, "bar"), b"was a file\n")
|
||||
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, f"dtc_{label}_rsync_dst")
|
||||
seed_dest(rsync_dst)
|
||||
rsync_result = _rsync(["-a", timing, source + "/", rsync_dst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_tree = _tree(rsync_dst)
|
||||
|
||||
dest = os.path.join(TEST_DATA_DIR, f"dtc_{label}_dst")
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
seed_dest(received)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=[timing], port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert _tree(received) == rsync_tree, (
|
||||
f"{timing}: fastsync tree {_tree(received)} != rsync tree {rsync_tree}"
|
||||
)
|
||||
|
||||
|
||||
class TestDeleteTimingFailure:
|
||||
"""A mid-transfer failure distinguishes the during timings from the late
|
||||
commit timings.
|
||||
|
||||
Plain ``--delete`` must behave like ``--delete-during`` (the rsync default),
|
||||
removing the extras of the directories already reached; ``--delete-commit``
|
||||
must behave like ``--delete-after`` and remove nothing until the transfer
|
||||
has fully succeeded.
|
||||
"""
|
||||
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_during_removes_delay_preserves_on_failure(self, mt):
|
||||
source, dest, received = _seed_pair(f"failure_mt{int(mt)}", big=True)
|
||||
extra = os.path.join(received, "d", "old_extra")
|
||||
assert os.path.exists(extra)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
for timing, expect_removed in (
|
||||
("--delete-during", True),
|
||||
("--delete", True),
|
||||
("--delete-delay", False),
|
||||
("--delete-commit", False),
|
||||
("--delete-after", False)):
|
||||
# Re-seed the extra before each run.
|
||||
_write(extra, b"stale extra\n")
|
||||
proxy = _SlicingProxy(server.port, forward_limit=MID_TRANSFER_BYTES, throttle=PROXY_THROTTLE)
|
||||
flags = [timing] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=flags, port=proxy.port)
|
||||
proxy.finish()
|
||||
assert result.returncode != 0, f"{timing}: truncated transfer succeeded"
|
||||
present = os.path.exists(extra)
|
||||
assert present != expect_removed, (
|
||||
f"{timing} (mt={mt}): extra present={present}, expected "
|
||||
f"removed={expect_removed}"
|
||||
)
|
||||
|
||||
|
||||
class TestDeleteDelayDeletedCount:
|
||||
"""The reported deleted count must reflect entries actually removed."""
|
||||
|
||||
def test_refilled_deferred_dir_is_recursively_removed_and_counted(self):
|
||||
"""A directory snapshotted into a --delete-delay plan that is refilled
|
||||
before the commit is re-scanned and removed recursively (rsync parity):
|
||||
the late file and the directory are both counted as deleted."""
|
||||
source = os.path.join(TEST_DATA_DIR, "ddc_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "ddc_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
_write(os.path.join(source, "d", "keep.txt"), b"kept payload\n")
|
||||
_write(os.path.join(source, "d", "big.bin"), b"B" * BIG_BYTES)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
extra_dir = os.path.join(received, "d", "extradir")
|
||||
os.makedirs(extra_dir, exist_ok=True)
|
||||
|
||||
def hook():
|
||||
# Runs while big.bin is in flight, after d's delete plan was processed.
|
||||
_write(os.path.join(extra_dir, "new.txt"), b"created mid-transfer\n")
|
||||
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
proxy = _SlicingProxy(server.port, hook=hook, hook_after=MID_TRANSFER_BYTES,
|
||||
throttle=PROXY_THROTTLE, wait_for_reply=True)
|
||||
flags = ["--delete-delay", "--incremental", "--ignore-times", "--stats"]
|
||||
result, _ = run_client(source, dest, flags=flags, port=proxy.port)
|
||||
proxy.finish()
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
assert proxy.hook_called.is_set(), "hook never fired"
|
||||
assert not os.path.exists(os.path.join(extra_dir, "new.txt")), "late file survived"
|
||||
assert not os.path.isdir(extra_dir), "refilled extra dir survived"
|
||||
deleted = None
|
||||
for line in result.stdout.splitlines():
|
||||
if line.startswith("Number of deleted files:"):
|
||||
deleted = int(line.split(":", 1)[1].split()[0])
|
||||
assert deleted == 2, (deleted, result.stdout)
|
||||
|
||||
|
||||
class TestDeleteDelayMaxDeleteParity:
|
||||
"""--max-delete with --delete-delay: a partial deletion still reports the
|
||||
number of entries actually removed, matching rsync (the exact surviving set
|
||||
can differ; only the count is compared)."""
|
||||
|
||||
@requires_rsync
|
||||
def test_max_delete_count_matches_rsync(self):
|
||||
source = os.path.join(TEST_DATA_DIR, "ddm_src")
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, "ddm_rsync_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(rsync_dst)
|
||||
_write(os.path.join(source, "d", "keep.txt"), b"keep\n")
|
||||
for i in range(1, 6):
|
||||
_write(os.path.join(rsync_dst, "d", f"e{i}.txt"), f"extra{i}\n".encode())
|
||||
|
||||
rsync_result = _rsync(["-a", "--delete-delay", "--max-delete=2", "--stats",
|
||||
source + "/", rsync_dst + "/"])
|
||||
# rsync exits 25 ("the --max-delete limit stopped deletions").
|
||||
assert rsync_result.returncode == 25, rsync_result.stderr
|
||||
rsync_count = _deleted_count(rsync_result.stdout)
|
||||
assert rsync_count == 2, rsync_result.stdout
|
||||
|
||||
dest = os.path.join(TEST_DATA_DIR, "ddm_dst")
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
for i in range(1, 6):
|
||||
_write(os.path.join(received, "d", f"e{i}.txt"), f"extra{i}\n".encode())
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(
|
||||
source, dest,
|
||||
flags=["--delete-delay", "--max-delete=2", "--stats"],
|
||||
port=server.port,
|
||||
)
|
||||
# A capped --max-delete commit is a successful transfer that both tools
|
||||
# report with exit 25.
|
||||
assert result.returncode == 25, (result.stderr or result.stdout)[:300]
|
||||
assert _deleted_count(result.stdout) == rsync_count, result.stdout
|
||||
|
||||
|
||||
def _deleted_count(text):
|
||||
for line in text.splitlines():
|
||||
if line.startswith("Number of deleted files:"):
|
||||
return int(line.split(":", 1)[1].split()[0])
|
||||
return None
|
||||
|
||||
|
||||
class TestDeleteDelayVsAfterSnapshot:
|
||||
"""A destination entry created after its directory's scan survives under
|
||||
--delete-delay but is removed by --delete-after's fresh end scan."""
|
||||
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_late_created_extra_survives_delay_not_after(self, mt):
|
||||
source, dest, received = _seed_pair(f"latecreate_mt{int(mt)}", big=True)
|
||||
old_extra = os.path.join(received, "d", "old_extra")
|
||||
new_extra = os.path.join(received, "d", "new_extra")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
for timing, new_survives in (("--delete-delay", True),
|
||||
("--delete-after", False)):
|
||||
_write(old_extra, b"stale extra\n")
|
||||
if os.path.exists(new_extra):
|
||||
os.unlink(new_extra)
|
||||
|
||||
def hook():
|
||||
# Runs on the proxy thread while the big file is in flight,
|
||||
# after the directory's plan (delay) has been processed.
|
||||
_write(new_extra, b"created mid-transfer\n")
|
||||
|
||||
# --incremental gives the receiver a mid-transfer handshake
|
||||
# reply; the proxy waits for it (wait_for_reply) so the hook is
|
||||
# causally after the plan frame, never a timing guess.
|
||||
# --ignore-times forces the big file to transfer on the second
|
||||
# timing too (the first run already installed it), keeping the
|
||||
# mid-transfer reply present in both iterations.
|
||||
proxy = _SlicingProxy(server.port, hook=hook,
|
||||
hook_after=MID_TRANSFER_BYTES,
|
||||
throttle=PROXY_THROTTLE, wait_for_reply=True)
|
||||
flags = [timing, "--incremental", "--ignore-times"] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=flags, port=proxy.port)
|
||||
proxy.finish()
|
||||
assert result.returncode == 0, (
|
||||
f"{timing}: {(result.stderr or result.stdout)[:300]}"
|
||||
)
|
||||
assert proxy.hook_called.is_set(), f"{timing}: hook never fired"
|
||||
assert not os.path.exists(old_extra), f"{timing}: old extra survived"
|
||||
assert os.path.exists(new_extra) == new_survives, (
|
||||
f"{timing} (mt={mt}): new_extra present="
|
||||
f"{os.path.exists(new_extra)}, expected survives={new_survives}"
|
||||
)
|
||||
|
||||
|
||||
class TestDeleteAfterThreadsKeepSet:
|
||||
"""Regression: -j/--threads must still transmit the delete keep-set in every
|
||||
timing. PipelineContextSender.delete_suppressed was left uninitialized, so a
|
||||
garbage true silently skipped the late keep-set manifest under --threads.
|
||||
Plain --delete now uses the per-directory plans, while --delete-commit /
|
||||
--delete-after keep exercising the late whole-tree manifest."""
|
||||
|
||||
@pytest.mark.parametrize("delete_flag", ["--delete", "--delete-commit", "--delete-after"])
|
||||
def test_threads_delete_after_sends_keep_set(self, delete_flag):
|
||||
source, dest, received = _seed_pair("mtkeep")
|
||||
extra = os.path.join(received, "d", "old_extra")
|
||||
assert os.path.exists(extra)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=["--threads", delete_flag],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert not os.path.exists(extra), (
|
||||
f"{delete_flag} --threads did not remove an extra: delete keep-set was suppressed"
|
||||
)
|
||||
|
||||
class TestDeleteDelayMaxDeleteRefilledDir:
|
||||
"""--delete-delay charges the --max-delete budget on ACTUAL removals: the
|
||||
refilled directory's late content is removed first (consuming the one slot),
|
||||
so the directory itself and a later extra are skipped, matching rsync.
|
||||
|
||||
The refilled directory is at the destination ROOT (its plan is always sent
|
||||
first) and the skipped extra is under a separate source directory, so the
|
||||
ordering that decides the budget charge is deterministic -- not readdir
|
||||
order. The refill is injected through the byte-barrier proxy so it is
|
||||
causally after the plan frame."""
|
||||
|
||||
def test_budget_charged_on_actual_removal(self):
|
||||
source = os.path.join(TEST_DATA_DIR, "ddmb_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "ddmb_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
_write(os.path.join(source, "a", "keep.bin"), b"B" * BIG_BYTES)
|
||||
_write(os.path.join(source, "b", "keep.txt"), b"keep\n")
|
||||
received = get_dest_received_dir(dest, source)
|
||||
refilled_dir = os.path.join(received, "xdir")
|
||||
os.makedirs(refilled_dir, exist_ok=True)
|
||||
later_dir = os.path.join(received, "b", "ydir")
|
||||
os.makedirs(later_dir, exist_ok=True)
|
||||
|
||||
def hook():
|
||||
_write(os.path.join(refilled_dir, "new.txt"), b"created mid-transfer\n")
|
||||
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
proxy = _SlicingProxy(server.port, hook=hook, hook_after=MID_TRANSFER_BYTES,
|
||||
throttle=PROXY_THROTTLE, wait_for_reply=True)
|
||||
flags = ["--delete-delay", "--max-delete=1", "--incremental", "--ignore-times", "--stats"]
|
||||
result, _ = run_client(source, dest, flags=flags, port=proxy.port)
|
||||
proxy.finish()
|
||||
assert result.returncode == 25, (result.stderr or result.stdout)[:400]
|
||||
assert proxy.hook_called.is_set(), "hook never fired"
|
||||
# The late content consumes the single budget slot; the refilled
|
||||
# directory itself and the later extra are skipped.
|
||||
assert not os.path.exists(os.path.join(refilled_dir, "new.txt")), "late file survived"
|
||||
assert os.path.isdir(later_dir), "later extra was not skipped by the budget"
|
||||
# The one actual removal is reported.
|
||||
assert _deleted_count(result.stdout) == 1, result.stdout
|
||||
@@ -0,0 +1,719 @@
|
||||
"""Differential rsync-parity gate.
|
||||
|
||||
Runs real ``rsync 3.4.1`` and FastSync over the same corpora and flags, then
|
||||
compares the destination trees and the normalized output of the
|
||||
output-oriented flags. This is the executable counterpart of
|
||||
``RSYNC_COMPAT.md``: the fast subset (``-m parity_ci``) guards the ✅ surface on
|
||||
every pull request, and the full set (``-m parity``) burns the documented
|
||||
⚠️/❌ residuals down.
|
||||
|
||||
Known, documented differences live in ``parity_caveats.py``; anything else
|
||||
fails with a readable tree/stdout diff. A stale allowlist entry is reported
|
||||
loudly (and fails when ``FASTSYNC_PARITY_STRICT=1``).
|
||||
|
||||
Run locally::
|
||||
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import sys
|
||||
import warnings
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
ServerManager,
|
||||
TEST_DATA_DIR,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
)
|
||||
from parity_caveats import ASPECTS, caveat_for # noqa: E402
|
||||
import parity_harness as H # noqa: E402
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
# `--allow-super` matches the rest of the integration suite; `--allow-delete`
|
||||
# is needed only by the delete cases.
|
||||
SUPER = ("--allow-super",)
|
||||
DELETE = ("--allow-super", "--allow-delete")
|
||||
_OLD_MTIME = 1_500_000_000
|
||||
|
||||
parity = pytest.mark.parity
|
||||
parity_ci = pytest.mark.parity_ci
|
||||
|
||||
|
||||
@pytest.fixture(scope="session")
|
||||
def parity_server_factory():
|
||||
"""Lazily start one server per distinct extra-argument set, per xdist worker."""
|
||||
servers = {}
|
||||
|
||||
def get(extra):
|
||||
key = tuple(extra)
|
||||
if key not in servers:
|
||||
s = ServerManager()
|
||||
s.start(extra_args=list(extra))
|
||||
servers[key] = s
|
||||
return servers[key]
|
||||
|
||||
yield get
|
||||
for s in servers.values():
|
||||
s.stop()
|
||||
|
||||
|
||||
def _pin(path, mtime):
|
||||
os.utime(path, (mtime, mtime))
|
||||
|
||||
|
||||
def _mk(path, data, mtime=None):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(data)
|
||||
if mtime is not None:
|
||||
_pin(path, mtime)
|
||||
|
||||
|
||||
# --- destination seeds ------------------------------------------------------
|
||||
|
||||
def seed_extras(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "extra.txt"), b"extra\n")
|
||||
_mk(os.path.join(root, "extradir", "z.txt"), b"z\n")
|
||||
|
||||
|
||||
def seed_update(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
p = os.path.join(root, "a.txt")
|
||||
_mk(p, b"destination is newer and longer\n", 2_000_000_000)
|
||||
|
||||
|
||||
def seed_ignore_existing(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "a.txt"), b"destination-kept\n", _OLD_MTIME)
|
||||
|
||||
|
||||
def seed_append(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "a.txt"), b"hello ", _OLD_MTIME)
|
||||
|
||||
|
||||
def seed_backup(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "a.txt"), b"OLD-CONTENT\n", _OLD_MTIME)
|
||||
|
||||
|
||||
def seed_size_only(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "a.txt"), b"XXXXXXXXXXX\n", _OLD_MTIME)
|
||||
|
||||
|
||||
def seed_delete_excluded(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "drop.log"), b"stale log\n", _OLD_MTIME)
|
||||
_mk(os.path.join(root, "extra.txt"), b"extra\n", _OLD_MTIME)
|
||||
_mk(os.path.join(root, "keep.txt"), b"keep\n", _OLD_MTIME)
|
||||
|
||||
|
||||
def seed_filter_protect(_src, rroot, froot):
|
||||
"""Destination-only entries, including nested ones, for the receiver-side
|
||||
`protect` rule: the `.log` extras must survive --delete, the rest go."""
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "extra.log"), b"dest-only log\n", _OLD_MTIME)
|
||||
_mk(os.path.join(root, "other.txt"), b"dest-only other\n", _OLD_MTIME)
|
||||
_mk(os.path.join(root, "sub", "extra2.log"), b"nested dest-only log\n", _OLD_MTIME)
|
||||
_mk(os.path.join(root, "sub", "other2.txt"), b"nested dest-only other\n", _OLD_MTIME)
|
||||
|
||||
|
||||
def seed_max_delete(_src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "extra1.txt"), b"e1\n", _OLD_MTIME)
|
||||
_mk(os.path.join(root, "extra2.txt"), b"e2\n", _OLD_MTIME)
|
||||
|
||||
|
||||
def fuzzy_basis_seed(_src, rroot, froot):
|
||||
"""Seed a same-suffix sibling whose name is one edit from the source and
|
||||
whose content matches it, with a DIFFERENT mtime so rsync's exact
|
||||
size+mtime pass cannot fire: both tools must select it via the
|
||||
name-distance pass. Where the two tools' basis choices coincide the
|
||||
block-level results are identical when the block size is pinned."""
|
||||
for root in (rroot, froot):
|
||||
_mk(os.path.join(root, "report_v1.txt"), H.FUZZY_PAYLOAD, _OLD_MTIME)
|
||||
|
||||
|
||||
def max_delete_count_check(_src, rroot, froot, _rs, _fs):
|
||||
"""The exact survivor set is order-dependent; the count must still match."""
|
||||
r = H.snapshot(rroot)
|
||||
f = H.snapshot(froot)
|
||||
if len(r) != len(f):
|
||||
return [f"survivor count differs: rsync={len(r)} fastsync={len(f)}"]
|
||||
return []
|
||||
|
||||
|
||||
# --- case table -------------------------------------------------------------
|
||||
|
||||
_CASES = [
|
||||
# --- core archive / recursion -----------------------------------------
|
||||
H.Case("archive", "basic", ["-a"], ci=True, ref="-a/--archive"),
|
||||
H.Case("recursive", "basic", ["-r"], ci=True, ref="-r/--recursive"),
|
||||
H.Case("unicode_names", "unicode", ["-a"], ci=True, ref="-a unicode names"),
|
||||
H.Case("links_archive", "links", ["-a"], ci=True, ref="-l/--links"),
|
||||
H.Case("copy_links", "links", ["-aL"], ref="-L/--copy-links"),
|
||||
H.Case("hardlinks", "hardlinks", ["-a", "-H"], compare_hardlinks=True,
|
||||
ci=True, ref="-H/--hard-links"),
|
||||
H.Case("hardlinks_without_H", "hardlinks", ["-a"], compare_hardlinks=True,
|
||||
ref="hardlinks without -H"),
|
||||
H.Case("sparse", "sparse", ["-a", "-S"], ref="-S/--sparse"),
|
||||
|
||||
# --- compression / checksums ------------------------------------------
|
||||
H.Case("compress_zstd", "basic", ["-a", "-z"], ci=True, ref="-z/--compress"),
|
||||
H.Case("checksum", "basic", ["-a", "-c"], ref="-c/--checksum"),
|
||||
H.Case("checksum_choice_xxh64", "basic",
|
||||
["-a", "-c", "--checksum-choice=xxh64"], ref="--checksum-choice"),
|
||||
|
||||
# --- selection --------------------------------------------------------
|
||||
H.Case("exclude", "filters", ["-a", "--exclude=*.log"], ci=True,
|
||||
ref="--exclude"),
|
||||
H.Case("include_exclude", "filters",
|
||||
["-a", "--include=*.txt", "--exclude=*"], ci=True,
|
||||
ref="--include/--exclude ordering"),
|
||||
H.Case("filter_rules", "filters",
|
||||
["-a", "-f", "- *.log", "-f", "+ *.txt", "-f", "- *"],
|
||||
ref="--filter/-f grammar"),
|
||||
H.Case("max_size", "basic", ["-a", "--max-size=1000"], ref="--max-size"),
|
||||
H.Case("min_size", "basic", ["-a", "--min-size=1000"], ref="--min-size"),
|
||||
|
||||
# --- output-oriented --------------------------------------------------
|
||||
H.Case("stats", "basic", ["-a", "--stats"], stdout=H.STDOUT_STATS,
|
||||
ci=True, ref="--stats"),
|
||||
H.Case("itemize", "links", ["-a", "-i"], stdout=H.STDOUT_ITEMIZE,
|
||||
ci=True, ref="-i/--itemize-changes"),
|
||||
H.Case("out_format_n_l", "basic", ["-a", "--out-format=%n %l"],
|
||||
stdout=H.STDOUT_OUTFMT, ref="--out-format %n %l"),
|
||||
H.Case("out_format_i_n", "basic", ["-a", "--out-format=%i %n"],
|
||||
stdout=H.STDOUT_OUTFMT, ref="--out-format %i %n"),
|
||||
H.Case("progress", "multidir", ["-a", "--progress"], stdout=H.STDOUT_PROGRESS,
|
||||
ci=True, ref="--progress multi-directory file list"),
|
||||
H.Case("progress_threads", "multidir", ["-a", "--progress"],
|
||||
fastsync_flags=["-a", "--progress", "--threads"],
|
||||
stdout=H.STDOUT_PROGRESS, ci=True,
|
||||
ref="--progress multi-directory file list (--threads)"),
|
||||
|
||||
# --- transfer modifications -------------------------------------------
|
||||
H.Case("update", "basic", ["-a", "--update"], seed=seed_update,
|
||||
ref="-u/--update"),
|
||||
H.Case("ignore_existing", "basic", ["-a", "--ignore-existing"],
|
||||
seed=seed_ignore_existing, ci=True, ref="--ignore-existing"),
|
||||
H.Case("size_only", "basic",
|
||||
["-a", "--size-only"], fastsync_flags=["-a", "--incremental", "--size-only"],
|
||||
seed=seed_size_only, ref="--size-only"),
|
||||
H.Case("append", "basic", ["-a", "--append"], seed=seed_append,
|
||||
ref="--append"),
|
||||
H.Case("append_verify", "basic", ["-a", "--append-verify"], seed=seed_append,
|
||||
ref="--append-verify"),
|
||||
H.Case("backup", "basic", ["-a", "--backup"], seed=seed_backup,
|
||||
ref="--backup"),
|
||||
H.Case("chmod", "basic", ["-a", "--chmod=Fu+rwx"], compare_modes=True,
|
||||
ci=True, ref="--chmod"),
|
||||
|
||||
# --- delta / similar-file basis (--fuzzy) -----------------------------
|
||||
# Basis choices coincide here (same-suffix sibling, name distance one edit,
|
||||
# content identical); with the block size pinned both tools report the same
|
||||
# Matched/Literal/transferred counters. The residual (FastSync's narrower
|
||||
# delta size window) is covered by TestFuzzy in test_parity_quickwins.py.
|
||||
H.Case("fuzzy_basis", "fuzzy",
|
||||
["-a", "--no-whole-file", "--fuzzy", "--stats", "-B8192"],
|
||||
fastsync_flags=["-a", "--incremental", "--delta", "--fuzzy",
|
||||
"--stats", "--delta-block=8192"],
|
||||
seed=fuzzy_basis_seed, stdout=H.STDOUT_STATS, ci=True,
|
||||
ref="-y/--fuzzy similar-file basis"),
|
||||
|
||||
# --- deletion ---------------------------------------------------------
|
||||
H.Case("delete", "basic", ["-a", "--delete"], seed=seed_extras,
|
||||
server_args=DELETE, ci=True, ref="--delete"),
|
||||
H.Case("delete_before", "basic", ["-a", "--delete-before"], seed=seed_extras,
|
||||
server_args=DELETE, ref="--delete-before"),
|
||||
H.Case("delete_during", "basic", ["-a", "--delete-during"], seed=seed_extras,
|
||||
server_args=DELETE, ref="--delete-during"),
|
||||
H.Case("delete_delay", "basic", ["-a", "--delete-delay"], seed=seed_extras,
|
||||
server_args=DELETE, ref="--delete-delay"),
|
||||
H.Case("delete_after", "basic", ["-a", "--delete-after"], seed=seed_extras,
|
||||
server_args=DELETE, ref="--delete-after"),
|
||||
H.Case("delete_commit", "basic", ["-a", "--delete-after"], seed=seed_extras,
|
||||
fastsync_flags=["-a", "--delete-commit"], server_args=DELETE,
|
||||
ref="FastSync-only --delete-commit == rsync --delete-after"),
|
||||
H.Case("delete_excluded", "filters",
|
||||
["-a", "--delete", "--delete-excluded", "--exclude=*.log"],
|
||||
seed=seed_delete_excluded, server_args=DELETE, ref="--delete-excluded"),
|
||||
H.Case("exclude_protect_dest_only", "filters",
|
||||
["-a", "--delete", "--exclude=*.log"],
|
||||
seed=seed_delete_excluded, server_args=DELETE, ci=True,
|
||||
ref="--delete protects a destination-only excluded entry like rsync"),
|
||||
H.Case("max_delete", "basic", ["-a", "--delete", "--max-delete=1"],
|
||||
seed=seed_max_delete, server_args=DELETE,
|
||||
extra_check=max_delete_count_check, compare_tree=False,
|
||||
ref="--max-delete"),
|
||||
H.Case("filter_protect", "filters",
|
||||
["-a", "--delete", "--filter=P *.log"],
|
||||
seed=seed_filter_protect, server_args=DELETE, ci=True,
|
||||
ref="--filter P/--protect receiver-side delete protection (default during)"),
|
||||
H.Case("filter_protect_during", "filters",
|
||||
["-a", "--delete-during", "--filter=P *.log"],
|
||||
seed=seed_filter_protect, server_args=DELETE, ci=True,
|
||||
ref="--filter P/--protect under --delete-during"),
|
||||
H.Case("filter_protect_delay", "filters",
|
||||
["-a", "--delete-delay", "--filter=P *.log"],
|
||||
seed=seed_filter_protect, server_args=DELETE, ci=True,
|
||||
ref="--filter P/--protect under --delete-delay"),
|
||||
H.Case("filter_protect_after", "filters",
|
||||
["-a", "--delete-after", "--filter=P *.log"],
|
||||
seed=seed_filter_protect, server_args=DELETE, ci=True,
|
||||
ref="--filter P/--protect under the whole-tree --delete-after commit"),
|
||||
|
||||
# --- relative / dirs --------------------------------------------------
|
||||
H.Case("relative_general", "basic", ["-a", "-R"], layout=H.MIRROR_ABS,
|
||||
compare_modes=True, ref="-R/--relative"),
|
||||
H.Case("relative_no_implied_dirs", "basic",
|
||||
["-a", "-R", "--no-implied-dirs"], layout=H.MIRROR_ABS,
|
||||
compare_modes=True, ref="--no-implied-dirs"),
|
||||
H.Case("files_from", "relative", ["--dirs", "-R"],
|
||||
files_from=("dir1", "sub/x.txt"), layout=H.RELATIVE, ci=True,
|
||||
ref="-d/--dirs + --files-from"),
|
||||
H.Case("dirs_plain", "basic", ["-d"], fs_src_suffix="/",
|
||||
ref="-d/--dirs (plain)"),
|
||||
H.Case("empty_dirs_recursive", "empty_dir", ["-a"],
|
||||
ref="recursive empty-directory residual"),
|
||||
H.Case("empty_dirs_files_from", "empty_dir", ["--dirs", "-R"],
|
||||
files_from=("emptydir",), layout=H.RELATIVE, ci=True,
|
||||
ref="-d/--dirs explicit empty directory"),
|
||||
|
||||
# --- codecs -----------------------------------------------------------
|
||||
H.Case("iconv_identity", "basic", ["-a", "--iconv=UTF-8,UTF-8"],
|
||||
ref="--iconv identity"),
|
||||
H.Case("iconv_convert", "iconv",
|
||||
["-a", "--iconv=ISO-8859-1,UTF-8"],
|
||||
server_args=("--allow-super", "--iconv=UTF-8"),
|
||||
ref="--iconv conversion (receiver declares its own charset)"),
|
||||
# rsync's spec is LOCAL,REMOTE and the destination end's charset is REMOTE
|
||||
# on a push, so a default server writes the wire (UTF-8) names verbatim.
|
||||
H.Case("iconv_default_server", "iconv",
|
||||
["-a", "--iconv=ISO-8859-1,UTF-8"],
|
||||
ref="--iconv push direction (default receiver charset = REMOTE)"),
|
||||
|
||||
# --- partial ----------------------------------------------------------
|
||||
H.Case("partial_complete", "basic", ["-a", "--partial"], ref="--partial"),
|
||||
]
|
||||
|
||||
# Cases that must always be tolerated (documented ⚠️/❌ residuals) get an
|
||||
# allowlist entry; the table below stays the exact ✅ surface.
|
||||
ALL_CASES = _CASES
|
||||
|
||||
|
||||
def _params():
|
||||
out = []
|
||||
for case in ALL_CASES:
|
||||
marks = [parity]
|
||||
if case.ci:
|
||||
marks.append(parity_ci)
|
||||
out.append(pytest.param(case, id=case.id, marks=marks))
|
||||
return out
|
||||
|
||||
|
||||
def _aspects_to_check(result):
|
||||
return {
|
||||
"tree": result["tree"],
|
||||
"stdout": result["stdout"],
|
||||
"extra": result["extra"],
|
||||
}
|
||||
|
||||
|
||||
def _assert_no_unexpected(case_id, mismatches, caveat, ref=""):
|
||||
unexpected = {a: v for a, v in mismatches.items() if v and a not in caveat}
|
||||
if unexpected:
|
||||
lines = [f"differential parity mismatch for case {case_id!r}:"]
|
||||
lines.append(f" ref: {ref or 'see RSYNC_COMPAT.md'}")
|
||||
for aspect, detail in unexpected.items():
|
||||
lines.append(f" --- {aspect} ---")
|
||||
lines.extend(" " + str(d) for d in detail)
|
||||
lines.append("If this is a documented residual, add it to "
|
||||
"tests/integration/parity_caveats.py with a RSYNC_COMPAT.md "
|
||||
"reference. Do not allowlist an undocumented divergence.")
|
||||
pytest.fail("\n".join(lines))
|
||||
|
||||
stale = [a for a in ASPECTS
|
||||
if a in caveat and a != "rc" and not mismatches.get(a)]
|
||||
if stale:
|
||||
msg = (f"stale parity allowlist entry for case {case_id!r}, aspect(s) "
|
||||
f"{stale}: FastSync now matches rsync. Remove it from "
|
||||
f"parity_caveats.py (and update RSYNC_COMPAT.md if the row moved).")
|
||||
if os.environ.get("FASTSYNC_PARITY_STRICT") == "1":
|
||||
pytest.fail(msg)
|
||||
warnings.warn(msg, stacklevel=2)
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.parametrize("case", _params())
|
||||
def test_differential_case(case, parity_server_factory):
|
||||
server = parity_server_factory(case.server_args)
|
||||
result = H.execute_case(case, server)
|
||||
caveat = caveat_for(case.id)
|
||||
|
||||
mismatches = _aspects_to_check(result)
|
||||
if result["rsync_rc"] != result["fastsync_rc"]:
|
||||
mismatches["rc"] = [
|
||||
f"rsync rc={result['rsync_rc']} fastsync rc={result['fastsync_rc']} "
|
||||
f"(rsync stderr: {result['rsync_stderr'][:200]!r}, "
|
||||
f"fastsync stderr: {result['fastsync_stderr'][:200]!r})"]
|
||||
_assert_no_unexpected(case.id, mismatches, caveat, ref=case.ref)
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Multi-run and setup-heavy scenarios (kept as explicit tests)
|
||||
# ---------------------------------------------------------------------------
|
||||
|
||||
def _result_aspects(result):
|
||||
return _aspects_to_check(result)
|
||||
|
||||
|
||||
_STANDALONE_REFS = {
|
||||
"incremental_modified": "-i/--itemize-changes + incremental second run",
|
||||
"compare_dest": "--compare-dest",
|
||||
"copy_dest": "--copy-dest",
|
||||
"link_dest": "--link-dest",
|
||||
"link_dest_stats": "--link-dest + --stats",
|
||||
"verify_basis": "--verify-basis (FastSync-only)",
|
||||
"verify_basis_default": "--verify-basis (default quick-check vs rsync)",
|
||||
"added_and_deleted": "--delete across two runs",
|
||||
"added_and_deleted_seed": "--delete across two runs",
|
||||
"one_file_system": "-x/--one-file-system",
|
||||
}
|
||||
|
||||
|
||||
def _run_and_check(case_id, result, ref=""):
|
||||
mismatches = _result_aspects(result)
|
||||
if result["rsync_rc"] != result["fastsync_rc"]:
|
||||
mismatches["rc"] = [
|
||||
f"rsync rc={result['rsync_rc']} fastsync rc={result['fastsync_rc']} "
|
||||
f"(rsync stderr: {result['rsync_stderr'][:200]!r}, "
|
||||
f"fastsync stderr: {result['fastsync_stderr'][:200]!r})"]
|
||||
_assert_no_unexpected(case_id, mismatches, caveat_for(case_id),
|
||||
ref=ref or _STANDALONE_REFS.get(case_id, ""))
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_incremental_modified_file(parity_server_factory):
|
||||
"""A second run sends only the modified file; destinations stay identical."""
|
||||
case_id = "incremental_modified"
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_inc_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_inc_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_inc_fdst")
|
||||
H.CORPORA["basic"](src)
|
||||
server = parity_server_factory(SUPER)
|
||||
|
||||
# Seed both destinations with the initial content.
|
||||
H.run_differential(src, rdst, fdst, ["-a"], ["-a"], server,
|
||||
extra_check=lambda *a: [])
|
||||
with open(os.path.join(src, "a.txt"), "wb") as fh:
|
||||
fh.write(b"hello world, now modified and longer\n")
|
||||
_pin(os.path.join(src, "a.txt"), 1_650_000_000)
|
||||
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst, ["-a", "-i"], ["-a", "-i", "--incremental"], server,
|
||||
stdout=H.STDOUT_ITEMIZE)
|
||||
_run_and_check(case_id, result)
|
||||
|
||||
|
||||
def _seed_basis(rel_entries):
|
||||
def seed(src, rroot, froot):
|
||||
for root in (rroot, froot):
|
||||
os.makedirs(root, exist_ok=True)
|
||||
for rel, data in rel_entries.items():
|
||||
_mk(os.path.join(root, rel), data)
|
||||
return seed
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_compare_dest_skips_basis(parity_server_factory):
|
||||
"""--compare-dest: a file present in the basis is not copied."""
|
||||
case_id = "compare_dest"
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_cmpd_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_cmpd_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_cmpd_fdst")
|
||||
clean_dir(src)
|
||||
_mk(os.path.join(src, "f.txt"), b"basis-content\n")
|
||||
_pin(os.path.join(src, "f.txt"), _OLD_MTIME)
|
||||
server = parity_server_factory(SUPER)
|
||||
rel = os.path.abspath(src).lstrip(os.sep)
|
||||
|
||||
# rsync resolves --compare-dest relative to the destination dir; FastSync
|
||||
# resolves it under the receive root and appends the mirrored source path.
|
||||
# Both rely on rsync's size+mtime quick-check, so the basis mtime is pinned
|
||||
# to the source's to keep the match deterministic across a second boundary.
|
||||
def seed(_src, rroot, froot):
|
||||
_mk(os.path.join(rroot, "basis", "f.txt"), b"basis-content\n", _OLD_MTIME)
|
||||
_mk(os.path.join(fdst, "basis", rel, "f.txt"), b"basis-content\n", _OLD_MTIME)
|
||||
|
||||
def extra(_src, rroot, froot, _rs, _fs):
|
||||
out = []
|
||||
for label, root in (("rsync", rroot), ("fastsync", froot)):
|
||||
if os.path.exists(os.path.join(root, "f.txt")):
|
||||
out.append(f"{label} copied a file that is present in the "
|
||||
f"compare basis")
|
||||
return out
|
||||
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst,
|
||||
["-a", "--compare-dest=basis"],
|
||||
["-a", f"--compare-dest={os.path.join(fdst, 'basis')}", "--incremental"],
|
||||
server, seed=seed, ignore_paths=("basis",), extra_check=extra)
|
||||
_run_and_check(case_id, result)
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_link_dest_hardlinks_basis(parity_server_factory):
|
||||
"""--link-dest: an unchanged file is hard-linked to the basis, not copied."""
|
||||
case_id = "link_dest"
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_linkd_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_linkd_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_linkd_fdst")
|
||||
clean_dir(src)
|
||||
_mk(os.path.join(src, "f.txt"), b"link-basis-content\n")
|
||||
_pin(os.path.join(src, "f.txt"), _OLD_MTIME)
|
||||
server = parity_server_factory(SUPER)
|
||||
rel = os.path.abspath(src).lstrip(os.sep)
|
||||
|
||||
def seed(_src, rroot, froot):
|
||||
_mk(os.path.join(rroot, "basis", "f.txt"), b"link-basis-content\n", _OLD_MTIME)
|
||||
_mk(os.path.join(fdst, "basis", rel, "f.txt"), b"link-basis-content\n", _OLD_MTIME)
|
||||
|
||||
def extra(_src, rroot, froot, _rs, _fs):
|
||||
r_basis = os.stat(os.path.join(rroot, "basis", "f.txt")).st_ino
|
||||
f_basis = os.stat(os.path.join(fdst, "basis", rel, "f.txt")).st_ino
|
||||
out = []
|
||||
for label, root, basis in (("rsync", rroot, r_basis),
|
||||
("fastsync", froot, f_basis)):
|
||||
target = os.path.join(root, "f.txt")
|
||||
if not os.path.exists(target):
|
||||
out.append(f"{label}: f.txt missing")
|
||||
elif os.stat(target).st_ino != basis:
|
||||
out.append(f"{label}: f.txt is not hard-linked to the basis")
|
||||
return out
|
||||
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst,
|
||||
["-a", "--link-dest=basis"],
|
||||
["-a", f"--link-dest={os.path.join(fdst, 'basis')}", "--incremental"],
|
||||
server, seed=seed, ignore_paths=("basis",), extra_check=extra)
|
||||
_run_and_check(case_id, result)
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_link_dest_stats_matches_rsync(parity_server_factory):
|
||||
"""A basis hit must not be counted as created or literal data: rsync reports
|
||||
zero for both, so FastSync's receiver tallies must too (regression for the
|
||||
basis materialization over-report)."""
|
||||
case_id = "link_dest_stats"
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_linkds_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_linkds_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_linkds_fdst")
|
||||
clean_dir(src)
|
||||
_mk(os.path.join(src, "f.txt"), b"link-basis-content\n")
|
||||
_pin(os.path.join(src, "f.txt"), _OLD_MTIME)
|
||||
server = parity_server_factory(SUPER)
|
||||
rel = os.path.abspath(src).lstrip(os.sep)
|
||||
|
||||
def seed(_src, rroot, froot):
|
||||
_mk(os.path.join(rroot, "basis", "f.txt"), b"link-basis-content\n", _OLD_MTIME)
|
||||
_mk(os.path.join(fdst, "basis", rel, "f.txt"), b"link-basis-content\n", _OLD_MTIME)
|
||||
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst,
|
||||
["-a", "--link-dest=basis", "--stats"],
|
||||
["-a", f"--link-dest={os.path.join(fdst, 'basis')}", "--incremental", "--stats"],
|
||||
server, seed=seed, ignore_paths=("basis",), stdout=H.STDOUT_STATS)
|
||||
_run_and_check(case_id, result, ref="--link-dest + --stats")
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_copy_dest_copies_basis(parity_server_factory):
|
||||
"""--copy-dest: a basis match is materialized as an independent copy with the
|
||||
source's attributes, matching rsync (copy then fix attributes)."""
|
||||
case_id = "copy_dest"
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_copyd_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_copyd_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_copyd_fdst")
|
||||
clean_dir(src)
|
||||
_mk(os.path.join(src, "f.txt"), b"copy-basis-content\n")
|
||||
_pin(os.path.join(src, "f.txt"), 1_600_000_000)
|
||||
os.chmod(os.path.join(src, "f.txt"), 0o755)
|
||||
server = parity_server_factory(SUPER)
|
||||
rel = os.path.abspath(src).lstrip(os.sep)
|
||||
|
||||
def seed(_src, rroot, froot):
|
||||
# Basis content matches the source; give the basis a different mode so a
|
||||
# wrong "keep basis attributes" implementation is visible.
|
||||
_mk(os.path.join(rroot, "basis", "f.txt"), b"copy-basis-content\n",
|
||||
1_600_000_000)
|
||||
os.chmod(os.path.join(rroot, "basis", "f.txt"), 0o644)
|
||||
_mk(os.path.join(fdst, "basis", rel, "f.txt"), b"copy-basis-content\n",
|
||||
1_600_000_000)
|
||||
os.chmod(os.path.join(fdst, "basis", rel, "f.txt"), 0o644)
|
||||
|
||||
def extra(_src, rroot, froot, _rs, _fs):
|
||||
out = []
|
||||
bases = {"rsync": os.path.join(rroot, "basis", "f.txt"),
|
||||
"fastsync": os.path.join(fdst, "basis", rel, "f.txt")}
|
||||
for label, root in (("rsync", rroot), ("fastsync", froot)):
|
||||
target = os.path.join(root, "f.txt")
|
||||
if not os.path.exists(target):
|
||||
out.append(f"{label}: f.txt missing")
|
||||
continue
|
||||
if os.stat(target).st_ino == os.stat(bases[label]).st_ino:
|
||||
out.append(f"{label}: f.txt is hard-linked, not copied")
|
||||
if (os.stat(target).st_mode & 0o777) != 0o755:
|
||||
out.append(f"{label}: f.txt mode "
|
||||
f"{oct(os.stat(target).st_mode & 0o777)} != 0o755")
|
||||
return out
|
||||
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst,
|
||||
["-a", "--copy-dest=basis"],
|
||||
["-a", f"--copy-dest={os.path.join(fdst, 'basis')}", "--incremental"],
|
||||
server, seed=seed, ignore_paths=("basis",), extra_check=extra,
|
||||
compare_modes=True)
|
||||
_run_and_check(case_id, result)
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_verify_basis_restores_strict_content(parity_server_factory):
|
||||
"""Default matches rsync's metadata quick-check; FastSync-only
|
||||
`--verify-basis` restores strict content equality and transfers the source
|
||||
when a same-size/different-content basis would otherwise be trusted."""
|
||||
case_id = "verify_basis"
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_vbasis_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_vbasis_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_vbasis_fdst")
|
||||
clean_dir(src)
|
||||
_mk(os.path.join(src, "f.txt"), b"AAAA\n")
|
||||
_pin(os.path.join(src, "f.txt"), _OLD_MTIME)
|
||||
server = parity_server_factory(SUPER)
|
||||
rel = os.path.abspath(src).lstrip(os.sep)
|
||||
|
||||
def seed(_src, rroot, froot):
|
||||
# Same size and mtime as the source, different bytes: a metadata
|
||||
# quick-check trusts it; --verify-basis must not.
|
||||
for root, basis_rel in ((rroot, os.path.join("basis", "f.txt")),
|
||||
(fdst, os.path.join("basis", rel, "f.txt"))):
|
||||
_mk(os.path.join(root, basis_rel), b"BBBB\n", _OLD_MTIME)
|
||||
|
||||
# Default: both tools trust the basis (rsync's quick check), so the
|
||||
# destination carries the basis bytes and the trees match.
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst,
|
||||
["-a", "--link-dest=basis"],
|
||||
["-a", f"--link-dest={os.path.join(fdst, 'basis')}", "--incremental"],
|
||||
server, seed=seed, ignore_paths=("basis",))
|
||||
_run_and_check(case_id + "_default", result)
|
||||
|
||||
# --verify-basis (FastSync only): the digest mismatch rejects the basis and
|
||||
# the source is transferred, so the destination is the source bytes. rsync
|
||||
# has no such flag; assert the FastSync outcome directly against the source.
|
||||
fdst2 = os.path.join(TEST_DATA_DIR, "parity_vbasis_fdst2")
|
||||
clean_dir(fdst2)
|
||||
for root, basis_rel in ((fdst2, os.path.join("basis", rel, "f.txt")),):
|
||||
_mk(os.path.join(root, basis_rel), b"BBBB\n", _OLD_MTIME)
|
||||
result, _ = H.run_fastsync(src, fdst2,
|
||||
["-a", f"--link-dest={os.path.join(fdst2, 'basis')}",
|
||||
"--incremental", "--verify-basis"], server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
target = os.path.join(get_dest_received_dir(fdst2, src), "f.txt")
|
||||
with open(target, "rb") as fh:
|
||||
assert fh.read() == b"AAAA\n", \
|
||||
"--verify-basis must reject the same-size/different-content basis"
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_added_and_deleted_between_runs(parity_server_factory):
|
||||
"""A source deletion and addition sync correctly under --delete."""
|
||||
case_id = "added_and_deleted"
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_addel_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_addel_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_addel_fdst")
|
||||
server = parity_server_factory(DELETE)
|
||||
H.CORPORA["basic"](src)
|
||||
|
||||
seed = seed_extras
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst, ["-a", "--delete"], ["-a", "--delete"], server,
|
||||
seed=seed)
|
||||
_run_and_check(case_id + "_seed", result)
|
||||
|
||||
os.remove(os.path.join(src, "a.txt"))
|
||||
_mk(os.path.join(src, "added.txt"), b"added between runs\n")
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst, ["-a", "--delete", "-i"],
|
||||
["-a", "--delete", "-i", "--incremental"], server,
|
||||
stdout=H.STDOUT_ITEMIZE)
|
||||
_run_and_check(case_id, result)
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_one_file_system(parity_server_factory):
|
||||
"""-x emits the mount-point directory but not its contents."""
|
||||
case_id = "one_file_system"
|
||||
local = os.stat(".")
|
||||
shm = "/dev/shm"
|
||||
if not os.path.isdir(shm):
|
||||
pytest.skip("/dev/shm not available")
|
||||
if os.stat(shm).st_dev == local.st_dev:
|
||||
pytest.skip("no cross-device filesystem available")
|
||||
|
||||
src = os.path.join(TEST_DATA_DIR, "parity_ofs_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "parity_ofs_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "parity_ofs_fdst")
|
||||
clean_dir(src)
|
||||
_mk(os.path.join(src, "keep.txt"), b"keep\n")
|
||||
probe = os.path.join(shm, f"fastsync_ofs_{os.getpid()}")
|
||||
shutil.rmtree(probe, ignore_errors=True)
|
||||
os.makedirs(probe)
|
||||
_mk(os.path.join(probe, "inside.txt"), b"cross\n")
|
||||
try:
|
||||
os.symlink(probe, os.path.join(src, "nested_link"))
|
||||
server = parity_server_factory(SUPER)
|
||||
result = H.run_differential(
|
||||
src, rdst, fdst,
|
||||
["-a", "--copy-links", "-x"],
|
||||
["-a", "--copy-links", "-x"], server)
|
||||
_run_and_check(case_id, result)
|
||||
finally:
|
||||
shutil.rmtree(probe, ignore_errors=True)
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@parity
|
||||
def test_parity_caveats_reference_known_cases():
|
||||
"""Every allowlist entry must name a real case id and aspect."""
|
||||
from parity_caveats import CAVEATS
|
||||
known = {c.id for c in ALL_CASES} | {
|
||||
"incremental_modified", "compare_dest", "link_dest",
|
||||
"added_and_deleted", "added_and_deleted_seed", "one_file_system",
|
||||
}
|
||||
problems = []
|
||||
for case_id, entry in CAVEATS.items():
|
||||
if case_id not in known:
|
||||
problems.append(f"unknown case id in parity_caveats.py: {case_id!r}")
|
||||
for aspect in entry:
|
||||
if aspect not in ASPECTS:
|
||||
problems.append(
|
||||
f"{case_id!r}: unknown aspect {aspect!r} (expected {ASPECTS})")
|
||||
assert not problems, "\n".join(problems)
|
||||
@@ -36,7 +36,7 @@ from common import ( # noqa: E402
|
||||
verify_transfer,
|
||||
)
|
||||
|
||||
PROTOCOL_VERSION = b"2.21.0"
|
||||
PROTOCOL_VERSION = b"2.28.0"
|
||||
STATUS_MANIFEST = 5
|
||||
STATUS_OK = 0
|
||||
|
||||
@@ -59,6 +59,10 @@ def _seed_source():
|
||||
if os.path.exists(SOURCE_DIR):
|
||||
shutil.rmtree(SOURCE_DIR)
|
||||
os.makedirs(os.path.join(SOURCE_DIR, "nested"))
|
||||
# The receiver rejects a destination root that does not exist, and the
|
||||
# capture fixture can run before any test that creates it, so create it
|
||||
# here (test order/distribution must not matter).
|
||||
os.makedirs(DEST_DIR, exist_ok=True)
|
||||
with open(os.path.join(SOURCE_DIR, "hello.txt"), "wb") as fh:
|
||||
fh.write(b"fault injection payload\n" * 64)
|
||||
with open(os.path.join(SOURCE_DIR, "nested", "deep.bin"), "wb") as fh:
|
||||
|
||||
+1541
-255
File diff suppressed because it is too large
Load Diff
@@ -1,10 +1,12 @@
|
||||
"""--iconv=CONVERT_SPEC file-NAME charset conversion integration tests.
|
||||
|
||||
The client converts every source file name from LOCAL to REMOTE before it goes
|
||||
on the wire, and the receiver converts it back from REMOTE to LOCAL, so a
|
||||
source tree using one charset can be written into a destination tree using
|
||||
another (rsync compatibility; content bytes are never touched).
|
||||
rsync's spec is ``--iconv=LOCAL,REMOTE`` (the order is the same push or pull).
|
||||
The sender converts each source name from LOCAL to REMOTE for the wire, and on
|
||||
a PUSH the receiver's charset is the spec's REMOTE half, so it writes the wire
|
||||
bytes verbatim (only a server with its own ``--iconv`` declares a different
|
||||
destination charset and re-converts). Content bytes are never touched.
|
||||
"""
|
||||
import codecs
|
||||
import os
|
||||
import shutil
|
||||
|
||||
@@ -16,6 +18,11 @@ LATIN1_NAME = b"caf\xe9.txt"
|
||||
UTF8_NAME = "caf\u00e9.txt".encode("utf-8")
|
||||
|
||||
|
||||
def _to_utf8(name_bytes):
|
||||
"""The UTF-8 encoding of a name that is stored as ISO-8859-1 bytes."""
|
||||
return codecs.encode(codecs.decode(name_bytes, "iso-8859-1"), "utf-8")
|
||||
|
||||
|
||||
def _make(tag):
|
||||
source = os.path.join(TEST_DATA_DIR, f"iconv_{tag}_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"iconv_{tag}_dst")
|
||||
@@ -41,10 +48,10 @@ def _dest_file(source, dest, name):
|
||||
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_iconv_latin1_roundtrip(shared_server):
|
||||
"""A source file whose name is ISO-8859-1 bytes is transferred with
|
||||
--iconv=iso-8859-1,utf-8 and lands on the destination with the ORIGINAL
|
||||
latin1 name (the wire carried it as UTF-8)."""
|
||||
def test_iconv_latin1_to_utf8_dest(shared_server):
|
||||
"""rsync push parity: --iconv=iso-8859-1,utf-8 converts a latin1 source name
|
||||
to the spec's REMOTE (UTF-8) on the wire and the default receiver writes it
|
||||
verbatim, so the destination name is UTF-8 (not the source's latin1)."""
|
||||
source, dest = _make("latin1")
|
||||
_place_bytes(source, LATIN1_NAME)
|
||||
|
||||
@@ -53,8 +60,10 @@ def test_iconv_latin1_roundtrip(shared_server):
|
||||
)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
|
||||
dst = _dest_file(source, dest, LATIN1_NAME)
|
||||
assert os.path.exists(dst), f"dest latin1-named file not found under {dest}"
|
||||
dst = _dest_file(source, dest, UTF8_NAME)
|
||||
assert os.path.exists(dst), f"dest UTF-8-named file not found under {dest}"
|
||||
assert not os.path.exists(_dest_file(source, dest, LATIN1_NAME)), \
|
||||
"destination kept the latin1 name instead of the wire (UTF-8) charset"
|
||||
|
||||
|
||||
@pytest.mark.ci
|
||||
@@ -153,7 +162,7 @@ def test_iconv_expanding_name_growth(shared_server):
|
||||
)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
|
||||
assert os.path.exists(_dest_file(source, dest, name_bytes))
|
||||
assert os.path.exists(_dest_file(source, dest, _to_utf8(name_bytes)))
|
||||
|
||||
|
||||
def test_iconv_symlink_path_and_target(shared_server):
|
||||
@@ -170,11 +179,13 @@ def test_iconv_symlink_path_and_target(shared_server):
|
||||
)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
|
||||
dst_target = _dest_file(source, dest, target)
|
||||
dst_link = _dest_file(source, dest, b"link\xe9")
|
||||
assert os.path.exists(dst_target), "dest latin1 target file missing"
|
||||
assert os.path.islink(dst_link), "dest latin1 symlink missing"
|
||||
assert os.readlink(dst_link) == target, "symlink target not preserved/decoded"
|
||||
utf8_target = _to_utf8(target)
|
||||
utf8_link = _to_utf8(b"link\xe9")
|
||||
dst_target = _dest_file(source, dest, utf8_target)
|
||||
dst_link = _dest_file(source, dest, utf8_link)
|
||||
assert os.path.exists(dst_target), "dest UTF-8 target file missing"
|
||||
assert os.path.islink(dst_link), "dest UTF-8 symlink missing"
|
||||
assert os.readlink(dst_link) == utf8_target, "symlink target not wire-converted"
|
||||
with open(dst_link, "rb") as fh:
|
||||
assert fh.read() == b"t\n"
|
||||
|
||||
@@ -198,8 +209,8 @@ def test_iconv_hardlink_path_and_target(shared_server):
|
||||
)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
|
||||
dst_a = _dest_file(source, dest, a)
|
||||
dst_b = _dest_file(source, dest, b)
|
||||
dst_a = _dest_file(source, dest, _to_utf8(a))
|
||||
dst_b = _dest_file(source, dest, _to_utf8(b))
|
||||
assert os.path.exists(dst_a) and os.path.exists(dst_b)
|
||||
assert os.stat(dst_a).st_ino == os.stat(dst_b).st_ino, \
|
||||
"hard-link relationship not preserved across the transfer"
|
||||
@@ -221,16 +232,16 @@ def test_iconv_delete_manifest_consistent(shared_server):
|
||||
flags = ["--iconv=iso-8859-1,utf-8"]
|
||||
result, _ = run_client(source, dest, flags=flags, port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
assert os.path.exists(_dest_file(source, dest, keep))
|
||||
assert os.path.exists(_dest_file(source, dest, gone))
|
||||
assert os.path.exists(_dest_file(source, dest, _to_utf8(keep)))
|
||||
assert os.path.exists(_dest_file(source, dest, _to_utf8(gone)))
|
||||
|
||||
os.remove(os.path.join(os.fsencode(source), gone))
|
||||
result, _ = run_client(
|
||||
source, dest, flags=flags + ["--delete"], port=server.port
|
||||
)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
assert os.path.exists(_dest_file(source, dest, keep)), "kept file deleted"
|
||||
assert not os.path.exists(_dest_file(source, dest, gone)), \
|
||||
assert os.path.exists(_dest_file(source, dest, _to_utf8(keep))), "kept file deleted"
|
||||
assert not os.path.exists(_dest_file(source, dest, _to_utf8(gone))), \
|
||||
"missing file was not deleted"
|
||||
|
||||
|
||||
@@ -247,4 +258,4 @@ def test_iconv_chunk_serialization_blob(shared_server):
|
||||
)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:400]
|
||||
|
||||
assert os.path.exists(_dest_file(source, dest, name))
|
||||
assert os.path.exists(_dest_file(source, dest, _to_utf8(name)))
|
||||
@@ -0,0 +1,538 @@
|
||||
"""Differential parity tests for the option wave (bwlimit, --info=*, -M,
|
||||
--ignore-errors, --filter protect).
|
||||
|
||||
Every differential here runs the SAME scenario with real ``rsync 3.4.1`` and
|
||||
with fastsync and compares the observable result, so the modules are skipped
|
||||
when rsync is unavailable. The privilege-dependent --ignore-errors differential
|
||||
drops the client to an unprivileged uid so a mode-000 source directory is
|
||||
genuinely unreadable; it is marked ``setpriv`` (run as root locally, excluded
|
||||
from the root PR gate exactly like the other privilege tests).
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
CLIENT_CMD,
|
||||
TEST_DATA_DIR,
|
||||
ServerManager,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
|
||||
def _rsync(args, timeout=120, as_nobody=False):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
cmd = [RSYNC] + args
|
||||
if as_nobody:
|
||||
cmd = ["setpriv", "--reuid=65534", "--regid=65534", "--clear-groups"] + cmd
|
||||
return subprocess.run(cmd, capture_output=True, text=True, env=env, timeout=timeout)
|
||||
|
||||
|
||||
def _write(path, content):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(content)
|
||||
|
||||
|
||||
class TestBwlimitParity:
|
||||
"""--bwlimit must accept rsync 3.4.1's spellings and pace like it."""
|
||||
|
||||
ACCEPTED = ["100", "0", "1.5", "100K", "100KB", "100KiB", "1M", "1MB", "1m", "1G", "512"]
|
||||
REJECTED = ["-1", "abc", "1x", "1 000"]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_parse_acceptance_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "bwp_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "f.txt"), b"payload\n")
|
||||
|
||||
for value in self.ACCEPTED + self.REJECTED:
|
||||
rdst = os.path.join(TEST_DATA_DIR, "bwp_rdst")
|
||||
clean_dir(rdst)
|
||||
rsync_result = _rsync(["-a", "--bwlimit=" + value, source + "/", rdst + "/"])
|
||||
dest = os.path.join(TEST_DATA_DIR, "bwp_dst")
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest, flags=["-a", "--bwlimit=" + value],
|
||||
port=shared_server.port)
|
||||
assert (result.returncode == 0) == (rsync_result.returncode == 0), (
|
||||
f"--bwlimit={value}: fastsync rc={result.returncode} "
|
||||
f"({(result.stderr or result.stdout)[:120]!r}) "
|
||||
f"rsync rc={rsync_result.returncode} ({rsync_result.stderr[:120]!r})"
|
||||
)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_throttle_rate_matches_rsync(self, shared_server):
|
||||
"""A 4 MiB transfer at --bwlimit=2048 (2 MiB/s) must take about the same
|
||||
wall-clock time for both tools (~2 s with rsync's leaky bucket)."""
|
||||
source = os.path.join(TEST_DATA_DIR, "bwt_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "big.bin"), os.urandom(4 * 1024 * 1024))
|
||||
dest = os.path.join(TEST_DATA_DIR, "bwt_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "bwt_rdst")
|
||||
|
||||
clean_dir(rdst)
|
||||
start = time.monotonic()
|
||||
rsync_result = _rsync(["-a", "--bwlimit=2048", source + "/", rdst + "/"])
|
||||
rsync_secs = time.monotonic() - start
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
clean_dir(dest)
|
||||
result, fast_secs = run_client(source, dest, flags=["-a", "--bwlimit=2048"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
# 4 MiB at 2 MiB/s rendezvous near 2 s. Use a coarse band on each side
|
||||
# (an unthrottled transfer finishes well under 1.5 s) plus a generous
|
||||
# cross-tolerance so a loaded CI runner cannot flake the parity assert.
|
||||
lo, hi = 1.5, 4.5
|
||||
assert lo <= fast_secs <= hi, f"fastsync throttle out of band: {fast_secs:.2f}s"
|
||||
assert lo <= rsync_secs <= hi, f"rsync throttle out of band: {rsync_secs:.2f}s"
|
||||
assert abs(fast_secs - rsync_secs) < 2.0, (
|
||||
f"fastsync {fast_secs:.2f}s vs rsync {rsync_secs:.2f}s"
|
||||
)
|
||||
|
||||
|
||||
def _output_tree(root):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "sub"))
|
||||
_write(os.path.join(root, "a.txt"), b"top\n")
|
||||
_write(os.path.join(root, "sub", "b.txt"), b"nested\n")
|
||||
os.symlink("a.txt", os.path.join(root, "link"))
|
||||
|
||||
|
||||
class TestInfoParity:
|
||||
"""The --info categories that map to a FastSync event must print rsync's
|
||||
line format."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_flist_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_fl_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_fl_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_fl_rdst")
|
||||
_output_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
rsync_result = _rsync(["-a", "--info=flist", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--info=flist"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
assert "sending incremental file list" in result.stdout
|
||||
assert "sending incremental file list" in rsync_result.stdout
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_name_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_nm_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_nm_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_nm_rdst")
|
||||
_output_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
rsync_result = _rsync(["-a", "--info=name", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--info=name"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
|
||||
def entries(text):
|
||||
# Compare the transferred entries only: rsync also prints the
|
||||
# transfer-root `./` and every directory (FastSync records dirs),
|
||||
# which are a separate documented divergence.
|
||||
out = []
|
||||
for line in text.splitlines():
|
||||
if not line or line.startswith("sending ") or line.startswith("created "):
|
||||
continue
|
||||
if line == "./" or line.endswith("/"):
|
||||
continue
|
||||
out.append(line)
|
||||
return sorted(out)
|
||||
|
||||
assert entries(result.stdout) == entries(rsync_result.stdout), (
|
||||
f"rsync={entries(rsync_result.stdout)} fastsync={entries(result.stdout)}"
|
||||
)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_name_root_line_matches_rsync(self, shared_server):
|
||||
"""A fresh destination: rsync prints `created directory`, then the
|
||||
transfer-root `./` name line before the entries; FastSync must emit the
|
||||
same `./` line."""
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_root_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_root_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_root_rdst")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "f.bin"), b"payload\n")
|
||||
clean_dir(dest)
|
||||
shutil.rmtree(rdst, ignore_errors=True)
|
||||
rsync_result = _rsync(["-a", "--info=name", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--info=name"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
|
||||
def names(text):
|
||||
return [l for l in text.splitlines()
|
||||
if l and not l.startswith("created directory")
|
||||
and not (l.endswith("/") and l != "./")]
|
||||
|
||||
assert names(rsync_result.stdout) == ["./", "f.bin"], names(rsync_result.stdout)
|
||||
assert names(result.stdout) == ["./", "f.bin"], names(result.stdout)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_name2_uptodate_matches_rsync(self, shared_server):
|
||||
"""--info=name2 prints `NAME is uptodate` for entries the receiver
|
||||
already has, matching rsync byte-for-byte."""
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_up_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_up_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_up_rdst")
|
||||
clean_dir(source)
|
||||
os.makedirs(os.path.join(source, "sub"))
|
||||
_write(os.path.join(source, "a.txt"), b"a\n")
|
||||
_write(os.path.join(source, "sub", "b.txt"), b"b\n")
|
||||
clean_dir(rdst)
|
||||
assert _rsync(["-a", source + "/", rdst + "/"]).returncode == 0
|
||||
rsync_result = _rsync(["-a", "--info=name2", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
clean_dir(dest)
|
||||
seed, _ = run_client(source, dest, flags=["-a", "--incremental"],
|
||||
port=shared_server.port)
|
||||
assert seed.returncode == 0, (seed.stderr or seed.stdout)[:200]
|
||||
result, _ = run_client(source, dest, flags=["-a", "--incremental", "--info=name2"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
|
||||
if l.endswith("is uptodate"))
|
||||
fast_lines = sorted(l for l in result.stdout.splitlines()
|
||||
if l.endswith("is uptodate"))
|
||||
assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
|
||||
assert fast_lines == ["a.txt is uptodate", "sub/b.txt is uptodate"], fast_lines
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_nonreg_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_nr_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_nr_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_nr_rdst")
|
||||
clean_dir(source)
|
||||
os.mkfifo(os.path.join(source, "fifo"))
|
||||
_write(os.path.join(source, "a.txt"), b"a\n")
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
rsync_result = _rsync(["-rlt", "--info=nonreg", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-rlt", "--info=nonreg"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
|
||||
if l.startswith("skipping non-regular"))
|
||||
fast_lines = sorted(l for l in result.stdout.splitlines()
|
||||
if l.startswith("skipping non-regular"))
|
||||
assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
|
||||
assert fast_lines, "no non-regular skip line emitted"
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_del_real_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_dl_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_dl_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_dl_rdst")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"keep\n")
|
||||
clean_dir(rdst)
|
||||
_write(os.path.join(rdst, "extra.txt"), b"x\n")
|
||||
_write(os.path.join(rdst, "extra2.txt"), b"y\n")
|
||||
rsync_result = _rsync(["-a", "--delete", "--info=del", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
|
||||
if l.startswith("deleting "))
|
||||
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
_write(os.path.join(received, "extra.txt"), b"x\n")
|
||||
_write(os.path.join(received, "extra2.txt"), b"y\n")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=["-a", "--delete", "--info=del"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
fast_lines = sorted(l for l in result.stdout.splitlines()
|
||||
if l.startswith("deleting "))
|
||||
assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
|
||||
assert fast_lines, "no deletion lines emitted"
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_del_itemize_real_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_di_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_di_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_di_rdst")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"keep\n")
|
||||
clean_dir(rdst)
|
||||
_write(os.path.join(rdst, "extra.txt"), b"x\n")
|
||||
rsync_result = _rsync(["-a", "-i", "--delete", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
|
||||
if l.startswith("*deleting"))
|
||||
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
_write(os.path.join(received, "extra.txt"), b"x\n")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=["-a", "-i", "--delete"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
fast_lines = sorted(l for l in result.stdout.splitlines()
|
||||
if l.startswith("*deleting"))
|
||||
assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_del_dry_run_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "inf_dd_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "inf_dd_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "inf_dd_rdst")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"keep\n")
|
||||
clean_dir(rdst)
|
||||
_write(os.path.join(rdst, "extra.txt"), b"x\n")
|
||||
rsync_result = _rsync(["-a", "-n", "--delete", "--info=del", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
|
||||
if l.startswith("deleting "))
|
||||
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
_write(os.path.join(received, "extra.txt"), b"x\n")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=["-a", "-n", "--delete", "--info=del"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
fast_lines = sorted(l for l in result.stdout.splitlines()
|
||||
if l.startswith("deleting "))
|
||||
assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_remove_matches_rsync(self, shared_server):
|
||||
tag = "inf_rm"
|
||||
rsync_src = os.path.join(TEST_DATA_DIR, f"{tag}_rsrc")
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, f"{tag}_rdst")
|
||||
fast_src = os.path.join(TEST_DATA_DIR, f"{tag}_fsrc")
|
||||
fast_dst = os.path.join(TEST_DATA_DIR, f"{tag}_fdst")
|
||||
for root in (rsync_src, rsync_dst, fast_src, fast_dst):
|
||||
clean_dir(root)
|
||||
_write(os.path.join(rsync_src, "a.txt"), b"a\n")
|
||||
_write(os.path.join(rsync_src, "sub", "b.txt"), b"b\n")
|
||||
_write(os.path.join(fast_src, "a.txt"), b"a\n")
|
||||
_write(os.path.join(fast_src, "sub", "b.txt"), b"b\n")
|
||||
|
||||
rsync_result = _rsync(["-a", "--remove-source-files", "--info=remove",
|
||||
rsync_src + "/", rsync_dst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(l for l in rsync_result.stdout.splitlines()
|
||||
if l.startswith("sender removed "))
|
||||
|
||||
result, _ = run_client(fast_src, fast_dst,
|
||||
flags=["-a", "--remove-source-files", "--info=remove"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
fast_lines = sorted(l for l in result.stdout.splitlines()
|
||||
if l.startswith("sender removed "))
|
||||
assert fast_lines == rsync_lines, (rsync_lines, fast_lines)
|
||||
assert fast_lines, "no source-removal lines emitted"
|
||||
|
||||
|
||||
class TestIgnoreErrorsParity:
|
||||
"""--ignore-errors: a source I/O error skips deletion by default; the flag
|
||||
lets deletion proceed. Both exit 23. Run the client as an unprivileged user
|
||||
so the mode-000 directory is genuinely unreadable."""
|
||||
|
||||
@pytest.mark.setpriv
|
||||
def test_delete_after_io_error_matches_rsync(self):
|
||||
if os.geteuid() != 0 or shutil.which("setpriv") is None:
|
||||
pytest.skip("requires root + setpriv to drop privileges for the client")
|
||||
tag = f"ie_{os.getpid()}"
|
||||
source = os.path.join(TEST_DATA_DIR, f"{tag}_src")
|
||||
rsync_dst = os.path.join(TEST_DATA_DIR, f"{tag}_rdst")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"{tag}_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(rsync_dst)
|
||||
clean_dir(dest)
|
||||
_write(os.path.join(source, "top.txt"), b"top\n")
|
||||
_write(os.path.join(source, "locked", "blocked.txt"), b"blocked\n")
|
||||
os.chmod(os.path.join(source, "locked"), 0)
|
||||
os.chmod(TEST_DATA_DIR, 0o777)
|
||||
os.chmod(source, 0o755)
|
||||
os.chmod(rsync_dst, 0o777)
|
||||
os.chmod(dest, 0o777)
|
||||
try:
|
||||
for ignore in (False, True):
|
||||
flags = ["-a", "--delete-after"] + (["--ignore-errors"] if ignore else [])
|
||||
# rsync side
|
||||
_write(os.path.join(rsync_dst, "extra.txt"), b"x\n")
|
||||
os.chmod(os.path.join(rsync_dst, "extra.txt"), 0o666)
|
||||
rres = _rsync(flags + [source + "/", rsync_dst + "/"], as_nobody=True)
|
||||
rsync_extra = os.path.exists(os.path.join(rsync_dst, "extra.txt"))
|
||||
|
||||
# fastsync side
|
||||
received = get_dest_received_dir(dest, source)
|
||||
_write(os.path.join(received, "extra.txt"), b"x\n")
|
||||
os.chmod(os.path.join(received, "extra.txt"), 0o666)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
fflags = (["--delete", "--ignore-errors"] if ignore else ["--delete"])
|
||||
cmd = CLIENT_CMD + ["--source-dir", source, "--dest-dir", dest,
|
||||
"--save-to-disk", "--server-port", str(server.port)] + fflags
|
||||
fres = subprocess.run(
|
||||
["setpriv", "--reuid=65534", "--regid=65534", "--clear-groups"] + cmd,
|
||||
text=True, capture_output=True)
|
||||
fast_extra = os.path.exists(os.path.join(received, "extra.txt"))
|
||||
|
||||
assert rres.returncode == 23, (ignore, rres.returncode, rres.stderr[:200])
|
||||
assert fres.returncode == 23, (ignore, fres.returncode, fres.stderr[:200])
|
||||
assert rsync_extra == fast_extra, (
|
||||
f"ignore_errors={ignore}: rsync extra={rsync_extra} fastsync extra={fast_extra}"
|
||||
)
|
||||
assert fast_extra is (not ignore), (ignore, fast_extra)
|
||||
finally:
|
||||
os.chmod(os.path.join(source, "locked"), 0o755)
|
||||
|
||||
|
||||
class TestRemoteOptionDaemon:
|
||||
"""rsync forwards -M/--remote-option to its remote process over a daemon
|
||||
connection; FastSync's daemon has no per-connection argv channel and rejects
|
||||
it. This pins the documented divergence with evidence."""
|
||||
|
||||
@requires_rsync
|
||||
def test_rsync_forwards_M_over_daemon_and_fastsync_rejects(self, tmp_path):
|
||||
import socket
|
||||
|
||||
with socket.socket() as probe:
|
||||
probe.bind(("127.0.0.1", 0))
|
||||
port = probe.getsockname()[1]
|
||||
|
||||
module_root = tmp_path / "mod"
|
||||
module_root.mkdir()
|
||||
os.chmod(module_root, 0o777)
|
||||
source = tmp_path / "src"
|
||||
source.mkdir()
|
||||
(source / "a.txt").write_bytes(b"hello\n")
|
||||
conf = tmp_path / "rsyncd.conf"
|
||||
conf.write_text(
|
||||
f"port = {port}\nuse chroot = no\n[m]\npath = {module_root}\nread only = no\n"
|
||||
)
|
||||
daemon = subprocess.Popen(
|
||||
[RSYNC, "--daemon", "--no-detach", "--port", str(port), "--config", str(conf)],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL)
|
||||
try:
|
||||
deadline = time.monotonic() + 5
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
with socket.create_connection(("127.0.0.1", port), timeout=0.3):
|
||||
break
|
||||
except OSError:
|
||||
time.sleep(0.05)
|
||||
else:
|
||||
pytest.skip("rsync daemon did not start")
|
||||
|
||||
# A well-formed -M option is forwarded and accepted by the daemon...
|
||||
ok = _rsync(["-a", "-M--safe-links", source.as_posix() + "/",
|
||||
f"rsync://127.0.0.1:{port}/m/"])
|
||||
# ...and a bogus one is rejected ON THE REMOTE with "unknown option",
|
||||
# which proves the option reached the daemon's parser.
|
||||
bogus = _rsync(["-a", "-M--totally-bogus", source.as_posix() + "/",
|
||||
f"rsync://127.0.0.1:{port}/m/"])
|
||||
assert bogus.returncode != 0
|
||||
assert "unknown option" in (bogus.stderr + bogus.stdout), bogus.stderr
|
||||
del ok
|
||||
finally:
|
||||
daemon.terminate()
|
||||
try:
|
||||
daemon.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
daemon.kill()
|
||||
|
||||
# FastSync rejects -M for a non-SSH transport up front.
|
||||
dest = os.path.join(TEST_DATA_DIR, "ro_dst")
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source.as_posix(), dest, flags=["-a", "-M--safe-links"])
|
||||
assert result.returncode != 0
|
||||
assert "remote-option" in (result.stderr + result.stdout)
|
||||
|
||||
|
||||
class TestFilterProtect:
|
||||
"""Receiver-derived delete protection: a `protect`/`P` rule is compiled by
|
||||
the sender and sent on the config frame, so the receiver shields a
|
||||
destination-only entry that never appeared on the sender, matching rsync."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_protect_dest_only_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "fpd_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "fpd_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "fpd_rdst")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"keep\n")
|
||||
clean_dir(rdst)
|
||||
_write(os.path.join(rdst, "extra.log"), b"extra\n")
|
||||
_write(os.path.join(rdst, "other.txt"), b"other\n")
|
||||
|
||||
rsync_result = _rsync(["-a", "--delete", "--filter=P *.log", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
assert os.path.exists(os.path.join(rdst, "extra.log")), "rsync did not protect extra.log"
|
||||
assert not os.path.exists(os.path.join(rdst, "other.txt")), "rsync did not delete other.txt"
|
||||
|
||||
clean_dir(dest)
|
||||
received = get_dest_received_dir(dest, source)
|
||||
_write(os.path.join(received, "extra.log"), b"extra\n")
|
||||
_write(os.path.join(received, "other.txt"), b"other\n")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--delete", "--filter=P *.log"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:200]
|
||||
assert os.path.exists(os.path.join(received, "extra.log")), (
|
||||
"FastSync must protect a destination-only P match like rsync")
|
||||
assert not os.path.exists(os.path.join(received, "other.txt"))
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_protect_dest_only_dry_run_enumeration(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "fpd_nd_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "fpd_nd_dst")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"keep\n")
|
||||
received = get_dest_received_dir(dest, source)
|
||||
clean_dir(received)
|
||||
_write(os.path.join(received, "keep.txt"), b"keep\n")
|
||||
_write(os.path.join(received, "extra.log"), b"extra\n")
|
||||
_write(os.path.join(received, "other.txt"), b"other\n")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "-n", "--delete", "--out-format=%n",
|
||||
"--filter=P *.log"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert "other.txt" in result.stdout, result.stdout
|
||||
assert "extra.log" not in result.stdout, result.stdout
|
||||
assert os.path.exists(os.path.join(received, "extra.log"))
|
||||
assert os.path.exists(os.path.join(received, "other.txt"))
|
||||
@@ -0,0 +1,848 @@
|
||||
"""Output-parity tests (#291 selection/output, #292 output formatting).
|
||||
|
||||
These tests exercise rsync-style selection ordering and output formatting. The
|
||||
differential tests run the SAME transfer with real ``rsync 3.4.1`` and with
|
||||
fastsync and compare stdout, so they are skipped when rsync is unavailable.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import TEST_DATA_DIR, run_client, clean_dir, get_dest_received_dir, ServerManager
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run(
|
||||
[RSYNC] + args, capture_output=True, text=True, env=env, timeout=120
|
||||
)
|
||||
|
||||
|
||||
def _make_selection_tree(root):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "sub"))
|
||||
with open(os.path.join(root, "a.txt"), "wb") as fh:
|
||||
fh.write(b"top text\n")
|
||||
with open(os.path.join(root, "b.log"), "wb") as fh:
|
||||
fh.write(b"log data\n")
|
||||
with open(os.path.join(root, "sub", "c.txt"), "wb") as fh:
|
||||
fh.write(b"nested text\n")
|
||||
with open(os.path.join(root, "sub", "d.log"), "wb") as fh:
|
||||
fh.write(b"nested log\n")
|
||||
|
||||
|
||||
class TestSelectionOrdering:
|
||||
"""#291: --include/--exclude compile into one ordered rule list."""
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_include_then_exclude_keeps_only_matching(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_inc_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_inc_dst")
|
||||
_make_selection_tree(source)
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(
|
||||
source, dest,
|
||||
flags=["--preserve", "--include=*.txt", "--exclude=*"],
|
||||
port=shared_server.port,
|
||||
)
|
||||
assert result.returncode == 0, f"include/exclude failed: {result.stderr[:300]}"
|
||||
received = get_dest_received_dir(dest, source)
|
||||
assert os.path.exists(os.path.join(received, "a.txt"))
|
||||
# `*` also excludes the directory, so nothing below sub/ is sent.
|
||||
assert not os.path.exists(os.path.join(received, "b.log"))
|
||||
assert not os.path.exists(os.path.join(received, "sub", "c.txt"))
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_include_dirs_then_files_idiom(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_inc2_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_inc2_dst")
|
||||
_make_selection_tree(source)
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(
|
||||
source, dest,
|
||||
flags=["--preserve", "--include=*/", "--include=*.txt", "--exclude=*"],
|
||||
port=shared_server.port,
|
||||
)
|
||||
assert result.returncode == 0, f"include/exclude failed: {result.stderr[:300]}"
|
||||
received = get_dest_received_dir(dest, source)
|
||||
assert os.path.exists(os.path.join(received, "a.txt"))
|
||||
assert os.path.exists(os.path.join(received, "sub", "c.txt"))
|
||||
assert not os.path.exists(os.path.join(received, "b.log"))
|
||||
assert not os.path.exists(os.path.join(received, "sub", "d.log"))
|
||||
|
||||
@requires_rsync
|
||||
def test_include_idiom_matches_rsync_selection(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_inc3_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_inc3_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "out_inc3_rdst")
|
||||
_make_selection_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
flags = ["--include=*/", "--include=*.txt", "--exclude=*"]
|
||||
rsync_result = _rsync(["-a"] + flags + [source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["--preserve"] + flags,
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0
|
||||
received = get_dest_received_dir(dest, source)
|
||||
assert os.path.exists(os.path.join(received, "a.txt"))
|
||||
assert os.path.exists(os.path.join(received, "sub", "c.txt"))
|
||||
assert not os.path.exists(os.path.join(received, "b.log"))
|
||||
# rsync -a src/ dst/ writes directly into dst/
|
||||
assert os.path.exists(os.path.join(rdst, "a.txt"))
|
||||
assert os.path.exists(os.path.join(rdst, "sub", "c.txt"))
|
||||
assert not os.path.exists(os.path.join(rdst, "b.log"))
|
||||
|
||||
|
||||
class TestOneFileSystem:
|
||||
"""#291: -x emits the mount-point directory but not its contents."""
|
||||
|
||||
def test_one_file_system_emits_mount_point_dir(self, shared_server):
|
||||
local = os.stat(".")
|
||||
shm = "/dev/shm"
|
||||
try:
|
||||
shm_stat = os.stat(shm)
|
||||
except OSError:
|
||||
pytest.skip("/dev/shm not available")
|
||||
if shm_stat.st_dev == local.st_dev:
|
||||
pytest.skip("no cross-device filesystem available")
|
||||
|
||||
source = os.path.join(TEST_DATA_DIR, "out_ofs_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_ofs_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
os.makedirs(os.path.join(source, "nested"))
|
||||
os.makedirs(os.path.join(shm, "fastsync_ofs_probe"), exist_ok=True)
|
||||
with open(os.path.join(source, "keep.txt"), "wb") as fh:
|
||||
fh.write(b"keep\n")
|
||||
with open(os.path.join(shm, "fastsync_ofs_probe", "inside.txt"), "wb") as fh:
|
||||
fh.write(b"cross\n")
|
||||
link = os.path.join(source, "nested", "link")
|
||||
try:
|
||||
os.symlink(os.path.join(shm, "fastsync_ofs_probe"), link)
|
||||
except OSError:
|
||||
pytest.skip("cannot create symlink")
|
||||
|
||||
try:
|
||||
result, _ = run_client(
|
||||
source, dest,
|
||||
flags=["--preserve", "--copy-links", "-x"],
|
||||
port=shared_server.port,
|
||||
)
|
||||
assert result.returncode == 0, f"-x failed: {result.stderr[:300]}"
|
||||
received = get_dest_received_dir(dest, source)
|
||||
assert os.path.exists(os.path.join(received, "keep.txt"))
|
||||
# The mount-point directory entry is created but its contents are not.
|
||||
assert os.path.isdir(os.path.join(received, "nested", "link"))
|
||||
assert not os.path.exists(os.path.join(received, "nested", "link", "inside.txt"))
|
||||
finally:
|
||||
shutil.rmtree(os.path.join(shm, "fastsync_ofs_probe"), ignore_errors=True)
|
||||
|
||||
|
||||
def _make_output_tree(root):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "sub"))
|
||||
with open(os.path.join(root, "a.txt"), "wb") as fh:
|
||||
fh.write(b"hello\n")
|
||||
with open(os.path.join(root, "sub", "b.txt"), "wb") as fh:
|
||||
fh.write("wörld\n".encode("utf-8"))
|
||||
os.symlink("a.txt", os.path.join(root, "link"))
|
||||
|
||||
|
||||
class TestItemizeParity:
|
||||
"""#292: -i output matches rsync 3.4.1 for the cases fastsync can observe."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_itemize_first_transfer_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_item_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_item_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "out_item_rdst")
|
||||
_make_output_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
rsync_result = _rsync(["-a", "-i", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(
|
||||
line for line in rsync_result.stdout.splitlines()
|
||||
if line.startswith(">f") or line.startswith("cL")
|
||||
)
|
||||
result, _ = run_client(source, dest, flags=["-a", "-i"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
fast_lines = sorted(
|
||||
line for line in result.stdout.splitlines()
|
||||
if line.startswith(">f") or line.startswith("cL")
|
||||
)
|
||||
assert fast_lines == rsync_lines, f"rsync={rsync_lines} fastsync={fast_lines}"
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_itemize_modified_file_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_item2_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_item2_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "out_item2_rdst")
|
||||
_make_output_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
seed = run_client(source, dest, flags=["-a"], port=shared_server.port)
|
||||
assert seed[0].returncode == 0, seed[0].stderr[:300]
|
||||
assert _rsync(["-a", source + "/", rdst + "/"]).returncode == 0
|
||||
|
||||
with open(os.path.join(source, "a.txt"), "wb") as fh:
|
||||
fh.write(b"hello changed and longer\n")
|
||||
# Pin the source mtime so rsync's `t` column is deterministic (a write
|
||||
# that lands in the same whole second as the seed would not show `t`).
|
||||
os.utime(os.path.join(source, "a.txt"), (1000000000, 1000000000))
|
||||
|
||||
rsync_result = _rsync(["-a", "-i", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(
|
||||
line for line in rsync_result.stdout.splitlines() if line.startswith(">f")
|
||||
)
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "-i", "--incremental"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
fast_lines = sorted(
|
||||
line for line in result.stdout.splitlines() if line.startswith(">f")
|
||||
)
|
||||
assert fast_lines == rsync_lines, f"rsync={rsync_lines} fastsync={fast_lines}"
|
||||
|
||||
|
||||
class TestOutFormatParity:
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_out_format_n_l_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_fmt_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_fmt_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "out_fmt_rdst")
|
||||
_make_output_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
fmt = "%n %l"
|
||||
rsync_result = _rsync(["-a", "--out-format=" + fmt, source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(
|
||||
line for line in rsync_result.stdout.splitlines()
|
||||
if line and not line.split(" ", 1)[0].endswith("/")
|
||||
)
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--out-format=" + fmt],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
fast_lines = sorted(
|
||||
line for line in result.stdout.splitlines()
|
||||
if line and not line.split(" ", 1)[0].endswith("/")
|
||||
)
|
||||
assert fast_lines == rsync_lines, f"rsync={rsync_lines} fastsync={fast_lines}"
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_out_format_M_datetime_shape(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_M_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_M_dst")
|
||||
_make_output_tree(source)
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--out-format=%M %f"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
import re
|
||||
pattern = re.compile(r"^\d{4}/\d{2}/\d{2}-\d{2}:\d{2}:\d{2} ")
|
||||
for line in result.stdout.splitlines():
|
||||
if line:
|
||||
assert pattern.match(line), f"bad %M format: {line!r}"
|
||||
|
||||
|
||||
class TestListOnlyParity:
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_list_only_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "out_list_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "out_list_dst")
|
||||
_make_output_tree(source)
|
||||
clean_dir(dest)
|
||||
rsync_result = _rsync(["-r", "--list-only", source + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_lines = sorted(rsync_result.stdout.splitlines())
|
||||
result, _ = run_client(source, dest, flags=["--list-only", "-l"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
fast_lines = sorted(result.stdout.splitlines())
|
||||
assert fast_lines == rsync_lines, (
|
||||
f"rsync={rsync_lines}\nfastsync={fast_lines}"
|
||||
)
|
||||
|
||||
|
||||
def _make_one_file(root, name="f.bin", size=100):
|
||||
clean_dir(root)
|
||||
with open(os.path.join(root, name), "wb") as fh:
|
||||
fh.write(bytes((i * 7 + 3) & 0xFF for i in range(size)))
|
||||
|
||||
|
||||
def _make_multidir_tree(root):
|
||||
"""Multi-directory corpus for the --progress file-list tests: nested files,
|
||||
a directory-only branch, an empty directory and a symlink."""
|
||||
clean_dir(root)
|
||||
for rel, data in (("a.txt", b"alpha\n"), ("b.txt", b"bravo\n"),
|
||||
("sub1/c.txt", b"charlie\n"), ("sub1/deep/d.txt", b"delta\n"),
|
||||
("sub2/e.txt", b"echo\n")):
|
||||
path = os.path.join(root, rel)
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(data)
|
||||
os.symlink("a.txt", os.path.join(root, "link1"))
|
||||
os.makedirs(os.path.join(root, "emptydir"), exist_ok=True)
|
||||
|
||||
|
||||
def _parse_progress(text):
|
||||
"""Name lines and the `to-chk` denominators from a --progress run."""
|
||||
names = []
|
||||
totals = set()
|
||||
for line in text.splitlines():
|
||||
line = line.rstrip()
|
||||
if not line or line == "sending incremental file list":
|
||||
continue
|
||||
if "%" in line:
|
||||
match = re.search(r"to-chk=\d+/(\d+)", line)
|
||||
if match:
|
||||
totals.add(int(match.group(1)))
|
||||
continue
|
||||
if line == "./": # root-line trigger is a separate documented residual
|
||||
continue
|
||||
names.append(line)
|
||||
return sorted(names), totals
|
||||
|
||||
|
||||
def _pick_stats(text, keys):
|
||||
out = {}
|
||||
for line in text.splitlines():
|
||||
for key in keys:
|
||||
if line.startswith(key + ":"):
|
||||
out[key] = line
|
||||
return out
|
||||
|
||||
|
||||
class TestWireStatsParity:
|
||||
"""Wire-counter output parity: --out-format %b/%c/%C, --progress and
|
||||
--stats versus real rsync 3.4.1."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_out_format_checksum_matches_rsync(self, shared_server):
|
||||
"""%C (whole-file xxh128, seed 0) is protocol-independent, so the full
|
||||
`%C %l %n` line must be byte-identical to rsync."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_ck_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_ck_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_ck_rdst")
|
||||
_make_one_file(source, "f.bin", 200000)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
fmt = "%C %l %n"
|
||||
rsync_result = _rsync(["-a", "--out-format=" + fmt, source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--out-format=" + fmt],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
def file_lines(text):
|
||||
# Ignore the root directory entry: fastsync does not transfer the
|
||||
# source-root dir itself (a separate pre-existing divergence).
|
||||
return [
|
||||
line for line in text.splitlines() if not line.rsplit(" ", 1)[-1].endswith("/")
|
||||
]
|
||||
|
||||
assert file_lines(result.stdout) == file_lines(rsync_result.stdout), (
|
||||
f"rsync={rsync_result.stdout!r} fastsync={result.stdout!r}"
|
||||
)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_out_format_b_is_wire_bytes(self, shared_server):
|
||||
"""%b is the bytes actually transferred (wire), not the source length.
|
||||
|
||||
A differential run against rsync confirms both implementations report a
|
||||
framed value greater than %l. The exact numbers are not compared: each
|
||||
counts its own protocol framing and checksum trailer, so the two are
|
||||
protocol-specific and cannot be numerically equal (documented
|
||||
divergence)."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_b_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_b_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_b_rdst")
|
||||
_make_one_file(source, "f.bin", 5000)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
fmt = "%b %l"
|
||||
rsync_result = _rsync(["-a", "--out-format=" + fmt, source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--out-format=" + fmt],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
rb, rl = (int(x) for x in rsync_result.stdout.split()[:2])
|
||||
fb, fl = (int(x) for x in result.stdout.split()[:2])
|
||||
assert rl == fl == 5000, (rsync_result.stdout, result.stdout)
|
||||
assert rb > rl, f"rsync %b must include framing: {rsync_result.stdout!r}"
|
||||
assert fb > fl, f"fastsync %b must include framing: {result.stdout!r}"
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_out_format_c_whole_file_matches_rsync(self, shared_server):
|
||||
"""%c is the block-checksum bytes received. rsync reports its 16-byte
|
||||
sum header even for a whole-file transfer (no basis), so `%c` must match
|
||||
rsync exactly for the whole-file case."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_c_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_c_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_c_rdst")
|
||||
_make_one_file(source, "f.bin", 5000)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
fmt = "%c %l %n"
|
||||
rsync_result = _rsync(["-a", "--out-format=" + fmt, source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--out-format=" + fmt],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
def file_lines(text):
|
||||
return [
|
||||
line for line in text.splitlines()
|
||||
if line and not line.rsplit(" ", 1)[-1].endswith("/")
|
||||
]
|
||||
|
||||
assert file_lines(result.stdout) == file_lines(rsync_result.stdout), (
|
||||
f"rsync={rsync_result.stdout!r} fastsync={result.stdout!r}"
|
||||
)
|
||||
assert result.stdout.split()[0] == rsync_result.stdout.split()[0] == "16", (
|
||||
f"%c must be rsync's 16-byte sum header: {result.stdout!r}"
|
||||
)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_out_format_c_delta_mode_divergence(self, shared_server):
|
||||
"""Documented residual: with delta enabled, rsync's %c is its 16-byte sum
|
||||
header plus one checksum entry per block (protocol-specific, so it grows
|
||||
with the basis size), while FastSync's %c is the bytes of its own delta
|
||||
handshake. FastSync's delta %c therefore cannot match rsync numerically;
|
||||
only the whole-file case is aligned. Pinned here so a future change is
|
||||
noticed."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_cd_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_cd_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_cd_rdst")
|
||||
_make_one_file(source, "f.bin", 5000)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
fmt = "%c %l"
|
||||
# rsync local default is whole-file; force the block-delta path.
|
||||
rsync_result = _rsync(["-a", "--no-whole-file", "--out-format=" + fmt,
|
||||
source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--incremental", "--delta",
|
||||
"--out-format=" + fmt],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
rs_c = int(rsync_result.stdout.split()[0])
|
||||
fs_c = int(result.stdout.split()[0])
|
||||
# No basis exists, so rsync still reports only its sum header.
|
||||
assert rs_c == 16, rsync_result.stdout
|
||||
# FastSync reports its own handshake bytes and is not aligned.
|
||||
assert fs_c > 16, (
|
||||
f"FastSync delta %c changed to {fs_c}; the documented divergence "
|
||||
"may be closable now"
|
||||
)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
@pytest.mark.parametrize("progress_flag", ["--progress", "-P"])
|
||||
def test_progress_first_frame_matches_rsync(self, shared_server, progress_flag, mt):
|
||||
"""For a sub-32 KiB file the first --progress/-P frame is deterministic
|
||||
(0.00 kB/s, 0:00:00) and must be byte-identical to rsync's, in both the
|
||||
single-threaded and --threads send paths."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_pg_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_pg_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_pg_rdst")
|
||||
_make_one_file(source, "f.bin", 100)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
rsync_result = _rsync(["-a", progress_flag, source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
flags = ["-a", progress_flag] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
def frames(text):
|
||||
# subprocess text mode normalizes \r to \n (universal newlines).
|
||||
return [p for p in text.split("\n") if "%" in p]
|
||||
|
||||
rsync_frames = frames(rsync_result.stdout)
|
||||
fast_frames = frames(result.stdout)
|
||||
assert rsync_frames and fast_frames, (rsync_result.stdout, result.stdout)
|
||||
assert fast_frames[0] == rsync_frames[0], (rsync_frames[0], fast_frames[0])
|
||||
assert "(xfr#1," in fast_frames[-1], fast_frames[-1]
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_progress_leading_root_line_and_to_chk_match_rsync(self, shared_server):
|
||||
"""A single-file transfer: rsync emits the transfer-root `./` name line
|
||||
and a `to-chk=0/2` denominator that counts that root entry. Both must
|
||||
match FastSync byte-for-byte for the deterministic frames."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_pgroot_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_pgroot_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_pgroot_rdst")
|
||||
_make_one_file(source, "f.bin", 100)
|
||||
clean_dir(dest)
|
||||
# rsync prints the `./` root line only when the transfer root itself is
|
||||
# created, so make the rsync destination absent. The "created directory"
|
||||
# line it then emits has no FastSync counterpart (different mirror
|
||||
# layout), so only the name/frame lines are compared.
|
||||
shutil.rmtree(rdst, ignore_errors=True)
|
||||
rsync_result = _rsync(["-a", "--progress", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--progress"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
# subprocess text mode normalizes \r to \n (universal newlines).
|
||||
def lines_of(text):
|
||||
return [ln for ln in text.splitlines() if ln and not ln.startswith("created directory")]
|
||||
|
||||
rsync_lines = lines_of(rsync_result.stdout)
|
||||
fast_lines = lines_of(result.stdout)
|
||||
rsync_names = [ln for ln in rsync_lines if "%" not in ln]
|
||||
fast_names = [ln for ln in fast_lines if "%" not in ln]
|
||||
|
||||
assert rsync_names == ["sending incremental file list", "./", "f.bin"], rsync_names
|
||||
assert fast_names == rsync_names, (rsync_names, fast_names)
|
||||
# The final frame's to-chk denominator must include the source-root entry.
|
||||
assert "to-chk=0/2" in fast_lines[-1], fast_lines[-1]
|
||||
assert fast_lines[-1] == rsync_lines[-1], (rsync_lines[-1], fast_lines[-1])
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_progress_multidir_file_list_matches_rsync(self, shared_server, mt):
|
||||
"""A multi-directory tree: the paths-only pre-count must reproduce
|
||||
rsync's file-list set and `to-chk` denominator. Per-directory name
|
||||
lines are emitted for directories, symlinks and the empty directory; the
|
||||
name set and the denominator (every entry plus the transfer root) match
|
||||
rsync, while the emitted *order* remains a documented residual (rsync
|
||||
sorts depth-first, FastSync streams in readdir/BFS order)."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_pgmd_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_pgmd_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_pgmd_rdst")
|
||||
_make_multidir_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
|
||||
rsync_result = _rsync(["-a", "--progress", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
flags = ["-a", "--progress"] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
rsync_names, rsync_totals = _parse_progress(rsync_result.stdout)
|
||||
fast_names, fast_totals = _parse_progress(result.stdout)
|
||||
assert sorted(rsync_names) == [
|
||||
"a.txt", "b.txt", "emptydir/", "link1 -> a.txt", "sub1/",
|
||||
"sub1/c.txt", "sub1/deep/", "sub1/deep/d.txt", "sub2/", "sub2/e.txt",
|
||||
], rsync_names
|
||||
assert fast_names == rsync_names, (rsync_names, fast_names)
|
||||
# 10 entries + the transfer-root "." counted by rsync's file list.
|
||||
assert rsync_totals == {11}, rsync_totals
|
||||
assert fast_totals == rsync_totals, (rsync_totals, fast_totals)
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_progress_delete_during_reuses_pre_scan(self):
|
||||
"""--delete-during + --progress reuses the keep-set pre-scan instead of
|
||||
walking the tree a second time: the file-list total and directory name
|
||||
lines are identical to a plain --progress run."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_pgdel_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_pgdel_dst")
|
||||
_make_multidir_tree(source)
|
||||
clean_dir(dest)
|
||||
server = ServerManager()
|
||||
server.start(extra_args=["--allow-super", "--allow-delete"])
|
||||
try:
|
||||
result, _ = run_client(source, dest, flags=["-a", "--progress", "--delete-during"],
|
||||
port=server.port)
|
||||
finally:
|
||||
server.stop()
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
names, totals = _parse_progress(result.stdout)
|
||||
assert totals == {11}, totals
|
||||
assert "sub1/" in names and "sub1/deep/" in names and "emptydir/" in names, names
|
||||
assert "link1 -> a.txt" in names, names
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_stats_selected_lines_match_rsync(self, shared_server, mt):
|
||||
"""The protocol-independent --stats lines must match rsync exactly, in
|
||||
both the single-threaded and --threads (multithreaded) send paths."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_st_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_st_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_st_rdst")
|
||||
_make_one_file(source, "f.bin", 6000)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
# Start both tools from the same state: rsync's destination root exists,
|
||||
# so pre-create FastSync's mirrored logical root as well.
|
||||
os.makedirs(get_dest_received_dir(dest, source), exist_ok=True)
|
||||
rsync_result = _rsync(["-a", "--stats", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
flags = ["-a", "--stats"] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
keys = (
|
||||
"Number of regular files transferred",
|
||||
"Total file size",
|
||||
"Total transferred file size",
|
||||
"Literal data",
|
||||
"Matched data",
|
||||
"Number of deleted files",
|
||||
"File list size",
|
||||
)
|
||||
|
||||
def pick(text):
|
||||
out = {}
|
||||
for line in text.splitlines():
|
||||
for key in keys:
|
||||
if line.startswith(key + ":"):
|
||||
out[key] = line
|
||||
return out
|
||||
|
||||
assert pick(result.stdout) == pick(rsync_result.stdout), (
|
||||
f"rsync={pick(rsync_result.stdout)} fastsync={pick(result.stdout)}"
|
||||
)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_stats_file_count_breakdown_matches_rsync(self, shared_server):
|
||||
"""`Number of files` and `Number of created files` both carry rsync's
|
||||
per-type breakdown (protocol 2.28.0 reports the receiver-created
|
||||
reg/dir/link/special split over STATUS_STATS)."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_stc_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_stc_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_stc_rdst")
|
||||
_make_one_file(source, "f.bin", 6000)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
# Start both tools from the same state: rsync's destination root exists,
|
||||
# so pre-create FastSync's mirrored logical root as well.
|
||||
os.makedirs(get_dest_received_dir(dest, source), exist_ok=True)
|
||||
rsync_result = _rsync(["-a", "--stats", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=["-a", "--stats"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
def stats_line(text, key):
|
||||
for line in text.splitlines():
|
||||
if line.startswith(key + ":"):
|
||||
return line
|
||||
return None
|
||||
|
||||
r_files = stats_line(rsync_result.stdout, "Number of files")
|
||||
r_created = stats_line(rsync_result.stdout, "Number of created files")
|
||||
f_files = stats_line(result.stdout, "Number of files")
|
||||
f_created = stats_line(result.stdout, "Number of created files")
|
||||
|
||||
assert re.match(r"Number of files: 2 \(reg: 1, dir: 1\)$", r_files), r_files
|
||||
assert r_files == f_files, (r_files, f_files)
|
||||
assert re.match(r"Number of created files: 1 \(reg: 1\)$", r_created), r_created
|
||||
assert f_created == r_created, (r_created, f_created)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_stats_r_directory_breakdown_matches_rsync(self, shared_server, mt):
|
||||
"""A recursive `-r` scan (no -t/-p) exposes no directory metadata, but
|
||||
rsync still counts every directory in `Number of files`; the sender's
|
||||
lightweight directory counter must reproduce the `dir: N` category."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_stdir_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_stdir_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_stdir_rdst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
os.makedirs(os.path.join(source, "sub", "deep"))
|
||||
os.makedirs(os.path.join(source, "empty"))
|
||||
for rel in ("a.txt", os.path.join("sub", "b.txt"), os.path.join("sub", "deep", "c.txt")):
|
||||
with open(os.path.join(source, rel), "wb") as fh:
|
||||
fh.write(b"x\n")
|
||||
os.makedirs(get_dest_received_dir(dest, source), exist_ok=True)
|
||||
|
||||
rsync_result = _rsync(["-r", "--stats", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
flags = ["-r", "--stats"] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
def stats_line(text, key):
|
||||
for line in text.splitlines():
|
||||
if line.startswith(key + ":"):
|
||||
return line
|
||||
return None
|
||||
|
||||
r_files = stats_line(rsync_result.stdout, "Number of files")
|
||||
f_files = stats_line(result.stdout, "Number of files")
|
||||
# 3 regular files, 4 directories (root, sub, sub/deep, empty).
|
||||
assert re.match(r"Number of files: 7 \(reg: 3, dir: 4\)$", r_files), r_files
|
||||
assert f_files == r_files, (r_files, f_files)
|
||||
assert (stats_line(result.stdout, "Number of regular files transferred") ==
|
||||
stats_line(rsync_result.stdout, "Number of regular files transferred"))
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_stats_created_and_literal_fresh_update_delta(self, shared_server, mt):
|
||||
"""The receiver-observed counters must match rsync for the three
|
||||
transfer shapes: a fresh create (created breakdown + whole-file literal),
|
||||
an update (created == 0, whole-file literal), and a delta update (only
|
||||
the literal delta fragments are counted, not the whole file)."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_stcd_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_stcd_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_stcd_rdst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
os.makedirs(source, exist_ok=True)
|
||||
os.makedirs(get_dest_received_dir(dest, source), exist_ok=True)
|
||||
with open(os.path.join(source, "big.bin"), "wb") as fh:
|
||||
fh.write(bytes(range(256)) * 4096) # 1 MiB
|
||||
mt_flag = ["--threads"] if mt else []
|
||||
|
||||
def compare(tag):
|
||||
# Pin the delta block size on both ends: rsync's adaptive block size
|
||||
# would otherwise make the literal/matched split non-comparable.
|
||||
rsync_result = _rsync(["-a", "--stats", "--no-whole-file", "-B8192",
|
||||
source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(
|
||||
source, dest,
|
||||
flags=["-a", "--stats", "--incremental", "--delta", "-B8192"] + mt_flag,
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
keys = ("Number of created files", "Literal data", "Matched data",
|
||||
"Total transferred file size")
|
||||
r = _pick_stats(rsync_result.stdout, keys)
|
||||
f = _pick_stats(result.stdout, keys)
|
||||
assert r == f, f"{tag}: rsync={r} fastsync={f}"
|
||||
return r
|
||||
|
||||
fresh = compare("fresh")
|
||||
assert re.match(r"Number of created files: 1 \(reg: 1\)$",
|
||||
fresh["Number of created files"]), fresh
|
||||
|
||||
# Update the source and re-run: the destination already exists.
|
||||
sleep_mtime = os.path.getmtime(os.path.join(source, "big.bin")) + 2
|
||||
with open(os.path.join(source, "big.bin"), "r+b") as fh:
|
||||
fh.seek(100)
|
||||
fh.write(b"XXXXXXXXXX")
|
||||
os.utime(os.path.join(source, "big.bin"), (sleep_mtime, sleep_mtime))
|
||||
update = compare("update")
|
||||
assert update["Number of created files"] == "Number of created files: 0", update
|
||||
|
||||
# Second delta update: change bytes far apart, so rsync ships only the
|
||||
# literal fragments and FastSync must report the same Literal data.
|
||||
sleep_mtime = os.path.getmtime(os.path.join(source, "big.bin")) + 2
|
||||
with open(os.path.join(source, "big.bin"), "r+b") as fh:
|
||||
fh.seek(500000)
|
||||
fh.write(b"YYYYYYYYYY")
|
||||
os.utime(os.path.join(source, "big.bin"), (sleep_mtime, sleep_mtime))
|
||||
delta = compare("delta")
|
||||
assert delta["Number of created files"] == "Number of created files: 0", delta
|
||||
lit = int(delta["Literal data"].split(":", 1)[1].strip().split()[0].replace(",", ""))
|
||||
assert 0 < lit < 1024 * 1024, delta
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("choice", ["xxh128", "xxh64", "xxh3", "md5", "md4", "sha1", "none"])
|
||||
def test_out_format_C_selected_algorithm_matches_rsync(self, shared_server, choice):
|
||||
"""`%C` must use the algorithm selected by --checksum-choice, not always
|
||||
xxh128, and render it exactly like rsync (big-endian for the 64-bit
|
||||
hashes, high-then-low for xxh128, standard hex for md5/md4/sha1)."""
|
||||
source = os.path.join(TEST_DATA_DIR, f"wire_cc_{choice}_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"wire_cc_{choice}_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"wire_cc_{choice}_rdst")
|
||||
_make_one_file(source, "f.bin", 200000)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
fmt = "%C %l %n"
|
||||
rsync_result = _rsync(["-a", "--checksum-choice=" + choice,
|
||||
"--out-format=" + fmt, source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--checksum-choice=" + choice,
|
||||
"--out-format=" + fmt],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
|
||||
def file_lines(text):
|
||||
return [line for line in text.splitlines()
|
||||
if line and not line.rsplit(" ", 1)[-1].endswith("/")]
|
||||
|
||||
assert file_lines(result.stdout) == file_lines(rsync_result.stdout), (
|
||||
f"choice={choice}: rsync={rsync_result.stdout!r} fastsync={result.stdout!r}"
|
||||
)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_dry_run_delete_lines_match_rsync(self, mt):
|
||||
"""-n --delete emits transfer-relative `*deleting` lines like rsync
|
||||
(single-threaded and --threads)."""
|
||||
source = os.path.join(TEST_DATA_DIR, "wire_del_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "wire_del_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "wire_del_rdst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
with open(os.path.join(source, "a.txt"), "wb") as fh:
|
||||
fh.write(b"a\n")
|
||||
for root, entries in (
|
||||
(rdst, {"extra.txt": b"x\n"}),
|
||||
(rdst, {"sub/y.txt": b"y\n", "extradir/z.txt": b"z\n"}),
|
||||
):
|
||||
for rel, data in entries.items():
|
||||
full = os.path.join(root, rel)
|
||||
os.makedirs(os.path.dirname(full), exist_ok=True)
|
||||
with open(full, "wb") as fh:
|
||||
fh.write(data)
|
||||
# FastSync mirrors the source's absolute path under dest.
|
||||
received = get_dest_received_dir(dest, source)
|
||||
for rel, data in (
|
||||
("extra.txt", b"x\n"),
|
||||
("sub/y.txt", b"y\n"),
|
||||
("extradir/z.txt", b"z\n"),
|
||||
):
|
||||
full = os.path.join(received, rel)
|
||||
os.makedirs(os.path.dirname(full), exist_ok=True)
|
||||
with open(full, "wb") as fh:
|
||||
fh.write(data)
|
||||
|
||||
rsync_result = _rsync(["-a", "-n", "--delete", "-i", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_del = sorted(
|
||||
line for line in rsync_result.stdout.splitlines() if line.startswith("*deleting")
|
||||
)
|
||||
# The shared session server refuses deletion; start one that allows it.
|
||||
flags = ["-a", "-n", "--delete", "-i"] + (["--threads"] if mt else [])
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest, flags=flags, port=server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
fast_del = sorted(
|
||||
line for line in result.stdout.splitlines() if line.startswith("*deleting")
|
||||
)
|
||||
assert fast_del == rsync_del, f"rsync={rsync_del}\nfastsync={fast_del}"
|
||||
@@ -0,0 +1,198 @@
|
||||
"""Differential rsync-parity coverage for two residuals closed on this branch.
|
||||
|
||||
* A4 -- ``--compare-dest``/``--copy-dest``/``--link-dest`` relative-DIR
|
||||
resolution: rsync resolves a relative DIR against the destination directory
|
||||
and appends the file's TRANSFER-RELATIVE name. FastSync's default transfer
|
||||
mirrors the absolute source path below its receive root, so a naive relative
|
||||
DIR used to probe a different tree. These tests seed the basis at rsync's
|
||||
spelling and assert FastSync finds it (byte-exact / hard-linked / sparse),
|
||||
matching real rsync 3.4.1.
|
||||
|
||||
* A5 -- ``-y``/``--fuzzy`` candidate eligibility: rsync's ``find_fuzzy`` has no
|
||||
delta-size gate, so it reuses an oversized (>10x) or sub-16-KiB sibling;
|
||||
FastSync used to decline both. These tests assert FastSync now uses the same
|
||||
sibling as rsync (observable as ``Matched data``) with a byte-exact result.
|
||||
|
||||
Every test skips cleanly when rsync is absent.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
TEST_DATA_DIR,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
OLD_MTIME = 1_500_000_000
|
||||
|
||||
|
||||
def _write(path, content, mtime=None):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(content)
|
||||
if mtime is not None:
|
||||
os.utime(path, (mtime, mtime))
|
||||
|
||||
|
||||
def _read(path):
|
||||
with open(path, "rb") as fh:
|
||||
return fh.read()
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
def _stat_bytes(text, label):
|
||||
for line in text.splitlines():
|
||||
if line.startswith(label + ":"):
|
||||
return int(line.split(":", 1)[1].strip().split()[0].replace(",", ""))
|
||||
return None
|
||||
|
||||
|
||||
class TestRelativeBasisDirResolution:
|
||||
"""A4: a relative basis DIR must resolve to the same tree as rsync's."""
|
||||
|
||||
_FILES = {
|
||||
"root.txt": b"root-basis-content\n",
|
||||
"sub/nested.txt": b"nested-basis-content\n",
|
||||
}
|
||||
|
||||
def _seed_source(self, source):
|
||||
clean_dir(source)
|
||||
for rel, data in self._FILES.items():
|
||||
_write(os.path.join(source, rel), data, OLD_MTIME)
|
||||
return self._FILES
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.parametrize("flag", ["--compare-dest", "--link-dest"])
|
||||
def test_relative_dir_resolves_like_rsync(self, shared_server, flag):
|
||||
tag = flag.lstrip("-")
|
||||
source = os.path.join(TEST_DATA_DIR, f"relbasis_{tag}_src")
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"relbasis_{tag}_rdst")
|
||||
fdst = os.path.join(TEST_DATA_DIR, f"relbasis_{tag}_fdst")
|
||||
self._seed_source(source)
|
||||
|
||||
# rsync: relative DIR -> dest/basis/<transfer-relative name>.
|
||||
clean_dir(rdst)
|
||||
for rel, data in self._FILES.items():
|
||||
_write(os.path.join(rdst, "basis", rel), data, OLD_MTIME)
|
||||
rs = _rsync(["-a", f"{flag}=basis", source + "/", rdst + "/"])
|
||||
assert rs.returncode == 0, rs.stderr
|
||||
|
||||
# FastSync: the SAME relative spelling seeded at the SAME
|
||||
# transfer-relative location under its destination root.
|
||||
clean_dir(fdst)
|
||||
for rel, data in self._FILES.items():
|
||||
_write(os.path.join(fdst, "basis", rel), data, OLD_MTIME)
|
||||
result, _ = run_client(source, fdst,
|
||||
flags=["-a", f"{flag}=basis", "--incremental"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
received = get_dest_received_dir(fdst, source)
|
||||
|
||||
for rel, data in self._FILES.items():
|
||||
rfile = os.path.join(rdst, rel)
|
||||
ffile = os.path.join(received, rel)
|
||||
basis = os.path.join(fdst, "basis", rel)
|
||||
if flag == "--compare-dest":
|
||||
# compare-dest never copies: both destinations stay sparse.
|
||||
assert not os.path.exists(rfile), f"rsync copied {rel}"
|
||||
assert not os.path.exists(ffile), (
|
||||
f"FastSync did not resolve the relative basis DIR at {basis!r} "
|
||||
f"(expected {rel!r} to stay sparse like rsync)")
|
||||
else:
|
||||
# link-dest hard-links; a basis miss would transfer a new file.
|
||||
assert os.path.exists(ffile), f"FastSync lost {rel}"
|
||||
assert _read(ffile) == data
|
||||
assert os.stat(ffile).st_ino == os.stat(basis).st_ino, (
|
||||
f"FastSync did not hard-link {rel!r} to the relative basis at "
|
||||
f"{basis!r} (basis not resolved like rsync)")
|
||||
|
||||
@requires_rsync
|
||||
def test_relative_dir_copy_dest_content(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "relbasis_copy_src")
|
||||
fdst = os.path.join(TEST_DATA_DIR, "relbasis_copy_fdst")
|
||||
self._seed_source(source)
|
||||
clean_dir(fdst)
|
||||
for rel, data in self._FILES.items():
|
||||
_write(os.path.join(fdst, "basis", rel), data, OLD_MTIME)
|
||||
result, _ = run_client(source, fdst,
|
||||
flags=["-a", "--copy-dest=basis", "--incremental"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
received = get_dest_received_dir(fdst, source)
|
||||
for rel, data in self._FILES.items():
|
||||
ffile = os.path.join(received, rel)
|
||||
assert os.path.exists(ffile), f"copy-dest did not materialize {rel}"
|
||||
assert _read(ffile) == data
|
||||
assert os.stat(ffile).st_ino != os.stat(os.path.join(fdst, "basis", rel)).st_ino
|
||||
|
||||
|
||||
class TestFuzzyEligibilityWindow:
|
||||
"""A5: --fuzzy candidate eligibility must match rsync's uncapped window."""
|
||||
|
||||
BASE = b"the quick brown fox jumps over the lazy dog\n" * 4000
|
||||
|
||||
def _run_pair(self, shared_server, tag, payload, sibling):
|
||||
source = os.path.join(TEST_DATA_DIR, f"fzw_{tag}_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"fzw_{tag}_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"fzw_{tag}_rdst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
_write(os.path.join(source, "report_v2.txt"), payload)
|
||||
for root in (rdst, get_dest_received_dir(dest, source)):
|
||||
_write(os.path.join(root, "report_v1.txt"), sibling)
|
||||
|
||||
rs = _rsync(["-a", "--no-whole-file", "--fuzzy", "--stats",
|
||||
source + "/", rdst + "/"])
|
||||
assert rs.returncode == 0, rs.stderr
|
||||
result, _ = run_client(
|
||||
source, dest,
|
||||
flags=["-a", "--incremental", "--delta", "--fuzzy", "--stats"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
|
||||
# The reconstructed file is byte-exact in every case.
|
||||
assert _read(os.path.join(get_dest_received_dir(dest, source),
|
||||
"report_v2.txt")) == payload
|
||||
return rs, result
|
||||
|
||||
@requires_rsync
|
||||
def test_oversized_sibling_eligible_like_rsync(self, shared_server):
|
||||
"""A sibling 20x the source is used by rsync; FastSync must too (its old
|
||||
10x delta-size gate declined it)."""
|
||||
n = 65536
|
||||
payload = (self.BASE * ((n // len(self.BASE)) + 1))[:n]
|
||||
sibling = (self.BASE * 200)[: n * 20]
|
||||
rs, result = self._run_pair(shared_server, "big", payload, sibling)
|
||||
assert _stat_bytes(rs.stdout, "Matched data") > 0, \
|
||||
"rsync should use a >10x fuzzy basis"
|
||||
assert _stat_bytes(result.stdout, "Matched data") > 0, (
|
||||
"FastSync's fuzzy eligibility must accept a >10x sibling like rsync "
|
||||
f"(Matched data={_stat_bytes(result.stdout, 'Matched data')})")
|
||||
|
||||
@requires_rsync
|
||||
def test_small_source_sibling_eligible_like_rsync(self, shared_server):
|
||||
"""A sub-16-KiB source with an identical sibling is used by rsync;
|
||||
FastSync's old 16 KiB delta minimum declined it."""
|
||||
n = 8192
|
||||
payload = (self.BASE * ((n // len(self.BASE)) + 1))[:n]
|
||||
rs, result = self._run_pair(shared_server, "small", payload, payload)
|
||||
assert _stat_bytes(rs.stdout, "Matched data") > 0, \
|
||||
"rsync applies --fuzzy below 16 KiB"
|
||||
assert _stat_bytes(result.stdout, "Matched data") > 0, (
|
||||
"FastSync's fuzzy eligibility must accept a sub-16-KiB source like "
|
||||
f"rsync (Matched data={_stat_bytes(result.stdout, 'Matched data')})")
|
||||
@@ -0,0 +1,299 @@
|
||||
"""Differential/regression coverage for the parity-completion review blockers.
|
||||
|
||||
Each test pins a fix against real ``rsync 3.4.1`` where a deterministic
|
||||
comparison exists; the differential tests skip cleanly when rsync is absent.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
TEST_DATA_DIR,
|
||||
ServerManager,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
|
||||
def _write(path, content):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(content)
|
||||
|
||||
|
||||
def _tree(root):
|
||||
"""Sorted relative paths of every entry below root (files and dirs)."""
|
||||
out = []
|
||||
for dirpath, dirs, files in os.walk(root):
|
||||
for name in dirs:
|
||||
out.append(os.path.relpath(os.path.join(dirpath, name), root))
|
||||
for name in files:
|
||||
out.append(os.path.relpath(os.path.join(dirpath, name), root))
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
class TestRelativePerDirDeleteScope:
|
||||
"""Blocker #1: -R --delete-during/--delete-delay must not delete destination
|
||||
content outside the transferred prefix (rsync keeps sibling directories)."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
@pytest.mark.parametrize("timing", ["--delete-during", "--delete-delay"])
|
||||
def test_prefix_scoped_delete_matches_rsync(self, timing, mt):
|
||||
source = os.path.join(TEST_DATA_DIR, f"delblk_src{int(mt)}")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "foo", "a.txt"), b"payload\n")
|
||||
spec = source + "/./foo"
|
||||
|
||||
def seed(root):
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "foo", "extra.txt"), b"stale\n")
|
||||
_write(os.path.join(root, "unrelated", "keep.txt"), b"keep\n")
|
||||
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"delblk_rdst{int(mt)}")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"delblk_dst{int(mt)}")
|
||||
seed(rdst)
|
||||
seed(dest)
|
||||
r = _rsync(["-aR", timing, spec, rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
flags = ["-a", "-R", timing] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(spec, dest, flags=flags, port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
#The prefix's parent-directory sibling survives on both sides.
|
||||
assert os.path.isfile(os.path.join(dest, "unrelated", "keep.txt"))
|
||||
assert os.path.isfile(os.path.join(rdst, "unrelated", "keep.txt"))
|
||||
#The in - scope extra is removed on both sides.
|
||||
assert not os.path.exists(os.path.join(dest, "foo", "extra.txt"))
|
||||
assert not os.path.exists(os.path.join(rdst, "foo", "extra.txt"))
|
||||
assert _tree(dest) == _tree(rdst)
|
||||
|
||||
|
||||
def _stats_value(text, label):
|
||||
for line in text.splitlines():
|
||||
if line.startswith(label + ":"):
|
||||
return int(line.split(":", 1)[1].strip().split()[0].replace(",", ""))
|
||||
return None
|
||||
|
||||
|
||||
def _seed_delta_pair(tag):
|
||||
"""Source file plus a same-size/basis destination file whose mtime differs,
|
||||
and an extra destination file to be deleted."""
|
||||
source = os.path.join(TEST_DATA_DIR, f"stats_{tag}_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"stats_{tag}_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"stats_{tag}_rdst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
payload = (b"0123456789abcdef" * 16384)[:200000]
|
||||
_write(os.path.join(source, "f.bin"), payload)
|
||||
#Destination basis : same length, one byte changed, deliberately older.
|
||||
basis = bytearray(payload)
|
||||
basis[100000] ^= 0xFF
|
||||
received = get_dest_received_dir(dest, source)
|
||||
for root in (rdst, received):
|
||||
_write(os.path.join(root, "f.bin"), bytes(basis))
|
||||
_write(os.path.join(root, "extra.txt"), b"delete me\n")
|
||||
old = 1000000
|
||||
os.utime(os.path.join(root, "f.bin"), (old, old))
|
||||
return source, dest, rdst
|
||||
|
||||
|
||||
class TestReceiverWireStats:
|
||||
"""Blocker #3/#4: the receiver must populate the STATUS_STATS counters
|
||||
(matched data, deleted files) on both the single-threaded and -m paths."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("threads", [False, True])
|
||||
def test_stats_reports_matched_and_deleted(self, threads):
|
||||
source, dest, rdst = _seed_delta_pair(f"mt{int(threads)}")
|
||||
rsync_result = _rsync(["-a", "--stats", "--delete", "--no-whole-file", source + "/",
|
||||
rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
assert _stats_value(rsync_result.stdout, "Matched data") > 0
|
||||
assert _stats_value(rsync_result.stdout, "Number of deleted files") == 1
|
||||
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
flags = ["-a", "--stats", "--delete", "--delta", "--incremental"]
|
||||
if threads:
|
||||
flags.append("--threads")
|
||||
result, _ = run_client(source, dest, flags=flags, port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert _stats_value(result.stdout, "Matched data") > 0, result.stdout
|
||||
assert _stats_value(result.stdout, "Number of deleted files") == 1, result.stdout
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_threads_dry_run_delete_lines_match_rsync(self):
|
||||
"""-n --delete --threads must emit transfer-relative `*deleting` lines."""
|
||||
source = os.path.join(TEST_DATA_DIR, "stats_drydel_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "stats_drydel_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "stats_drydel_rdst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
_write(os.path.join(source, "a.txt"), b"a\n")
|
||||
for root in (rdst, get_dest_received_dir(dest, source)):
|
||||
_write(os.path.join(root, "extra.txt"), b"x\n")
|
||||
_write(os.path.join(root, "sub", "y.txt"), b"y\n")
|
||||
rsync_result = _rsync(["-a", "-n", "--delete", "-i", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
rsync_del = sorted(
|
||||
line for line in rsync_result.stdout.splitlines() if line.startswith("*deleting")
|
||||
)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "-n", "--delete", "-i", "--threads"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
fast_del = sorted(
|
||||
line for line in result.stdout.splitlines() if line.startswith("*deleting")
|
||||
)
|
||||
assert fast_del and fast_del == rsync_del, f"rsync={rsync_del}\nfastsync={fast_del}"
|
||||
|
||||
|
||||
class TestRelativeFilesFromProtect:
|
||||
"""Blocker #9: a -R + --files-from receiver-protect rule must record the bare
|
||||
relative wire path so the protected destination mirror survives --delete."""
|
||||
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_hidden_protected_mirror_survives_delete(self, mt):
|
||||
source = os.path.join(TEST_DATA_DIR, "rfprot_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "rfprot_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
#Root - level entry exercises the parallel root scanner; the nested one
|
||||
#exercises the sequential worker scanner.
|
||||
_write(os.path.join(source, "root_secret.tmp"), b"root\n")
|
||||
_write(os.path.join(source, "sub", "nested_secret.tmp"), b"nested\n")
|
||||
_write(os.path.join(source, "sub", "keep.txt"), b"keep\n")
|
||||
listfile = os.path.join(TEST_DATA_DIR, "rfprot.list")
|
||||
with open(listfile, "w") as fh:
|
||||
fh.write(".\n")
|
||||
#H hides from the sender, P protects the receiver mirror from-- delete.
|
||||
filters = ["--filter=H root_secret.tmp", "--filter=P root_secret.tmp",
|
||||
"--filter=H sub/nested_secret.tmp", "--filter=P sub/nested_secret.tmp"]
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
seed = ["--files-from", listfile, "-R"] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=seed, port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert os.path.isfile(os.path.join(dest, "root_secret.tmp"))
|
||||
assert os.path.isfile(os.path.join(dest, "sub", "nested_secret.tmp"))
|
||||
_write(os.path.join(dest, "extra.txt"), b"extra\n")
|
||||
_write(os.path.join(dest, "sub", "extra.txt"), b"extra\n")
|
||||
flags = seed + ["--delete"] + filters
|
||||
result, _ = run_client(source, dest, flags=flags, port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert os.path.isfile(os.path.join(dest, "root_secret.tmp")), \
|
||||
"root-level protected mirror was deleted"
|
||||
assert os.path.isfile(os.path.join(dest, "sub", "nested_secret.tmp")), \
|
||||
"nested protected mirror was deleted"
|
||||
assert not os.path.exists(os.path.join(dest, "extra.txt"))
|
||||
assert not os.path.exists(os.path.join(dest, "sub", "extra.txt"))
|
||||
|
||||
|
||||
class TestInvalidPerDirFilter:
|
||||
"""Blocker #8: a per-directory filter file that fails to parse must fail the
|
||||
scan even when an earlier merge file in the same directory existed."""
|
||||
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_invalid_dir_filter_fails_scan(self, mt):
|
||||
source = os.path.join(TEST_DATA_DIR, "badfilter_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "badfilter_dst")
|
||||
clean_dir(source)
|
||||
clean_dir(dest)
|
||||
#A valid.rsync - filter makes any_exists true for the directory; the
|
||||
#invalid.rules must not then be silently ignored.
|
||||
_write(os.path.join(source, ".rsync-filter"), b"- *.bak\n")
|
||||
_write(os.path.join(source, ".rules"), b"protect\n")
|
||||
_write(os.path.join(source, "a.txt"), b"a\n")
|
||||
flags = ["-a", "-F", "--filter=: .rules"]
|
||||
if mt:
|
||||
flags.append("--threads")
|
||||
with ServerManager() as server:
|
||||
result, _ = run_client(source, dest, flags=flags, port=server.port)
|
||||
assert result.returncode != 0, "invalid per-directory filter was silently ignored"
|
||||
assert "invalid per-directory filter" in (result.stderr + result.stdout)
|
||||
|
||||
|
||||
class TestWouldDeleteEscaping:
|
||||
"""Blocker #5: -n --delete --out-format must escape control bytes in a
|
||||
peer-supplied would-delete path so it cannot forge output lines."""
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_out_format_escapes_control_chars(self):
|
||||
source = os.path.join(TEST_DATA_DIR, "esc_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "esc_dst")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "a.txt"), b"a\n")
|
||||
received = get_dest_received_dir(dest, source)
|
||||
clean_dir(received)
|
||||
_write(os.path.join(received, "a.txt"), b"a\n")
|
||||
#A newline in a destination filename must not split the printed line.
|
||||
with open(os.path.join(received, "evil\nname.txt"), "wb") as fh:
|
||||
fh.write(b"x\n")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "-n", "--delete", "--out-format=%n"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert "\\#012" in result.stdout, result.stdout
|
||||
assert "evil\nname.txt" not in result.stdout, result.stdout
|
||||
|
||||
|
||||
class TestEmptySourceDirectoryDelete:
|
||||
"""Blocker #10: an empty in-scope source directory must survive
|
||||
--delete-during/--delete-delay (rsync keeps it) while its extras are still
|
||||
removed."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
@pytest.mark.parametrize("timing", ["--delete-during", "--delete-delay"])
|
||||
def test_empty_source_dir_survives_matches_rsync(self, timing, mt):
|
||||
source = os.path.join(TEST_DATA_DIR, f"emptydir_src{int(mt)}")
|
||||
dest = os.path.join(TEST_DATA_DIR, f"emptydir_dst{int(mt)}")
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"emptydir_rdst{int(mt)}")
|
||||
clean_dir(source)
|
||||
os.makedirs(os.path.join(source, "empty"))
|
||||
_write(os.path.join(source, "keep.txt"), b"keep\n")
|
||||
received = get_dest_received_dir(dest, source)
|
||||
for root in (rdst, received):
|
||||
clean_dir(root)
|
||||
_write(os.path.join(root, "keep.txt"), b"keep\n")
|
||||
_write(os.path.join(root, "empty", "extra.txt"), b"extra\n")
|
||||
rsync_result = _rsync(["-a", timing, source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
assert os.path.isdir(os.path.join(rdst, "empty"))
|
||||
assert not os.path.exists(os.path.join(rdst, "empty", "extra.txt"))
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
flags = ["-a", timing] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(source, dest, flags=flags, port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert os.path.isdir(os.path.join(received, "empty")), \
|
||||
"empty source directory was removed"
|
||||
assert not os.path.exists(os.path.join(received, "empty", "extra.txt"))
|
||||
assert _tree(received) == _tree(rdst)
|
||||
@@ -0,0 +1,105 @@
|
||||
"""`--debug=FLAGS` natural-event categories (no-wire).
|
||||
|
||||
FastSync maps the rsync `--debug` categories that correspond to a real event it
|
||||
already performs (``flist``, ``del``, ``hash``/``deltasum``, ``recv``,
|
||||
``filter`` and ``send``) onto debug output. A normal run prints none of it.
|
||||
"""
|
||||
import os
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import TEST_DATA_DIR, run_client, clean_dir, get_dest_received_dir, ServerManager
|
||||
|
||||
|
||||
def _make_tree(root):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "sub"))
|
||||
with open(os.path.join(root, "a.txt"), "wb") as fh:
|
||||
fh.write(b"alpha\n")
|
||||
with open(os.path.join(root, "keep.log"), "wb") as fh:
|
||||
fh.write(b"log\n")
|
||||
with open(os.path.join(root, "sub", "b.txt"), "wb") as fh:
|
||||
fh.write(b"beta\n")
|
||||
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_debug_flist_and_send_emit_output(shared_server):
|
||||
"""`--debug=flist,send` produces category-tagged debug output."""
|
||||
source = os.path.join(TEST_DATA_DIR, "dbg_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "dbg_dst")
|
||||
_make_tree(source)
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest, flags=["-a", "--debug=flist,send"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert "flist: scanning" in result.stdout, result.stdout
|
||||
assert "send: " in result.stdout, result.stdout
|
||||
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_debug_filter_emits_excluded_entry(shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "dbg_filter_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "dbg_filter_dst")
|
||||
_make_tree(source)
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--debug=filter", "--exclude=*.log"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert "filter: excluded keep.log" in result.stdout, result.stdout
|
||||
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_debug_hash_and_recv_emit_on_incremental(shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "dbg_hash_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "dbg_hash_dst")
|
||||
_make_tree(source)
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest,
|
||||
flags=["-a", "--incremental", "--checksum",
|
||||
"--debug=hash,recv"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert "hash: " in result.stdout, result.stdout
|
||||
assert "recv: " in result.stdout, result.stdout
|
||||
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_debug_del_emits_deleted_path():
|
||||
"""`--debug=del` reports the paths the receiver actually removed.
|
||||
|
||||
A deletion-capable server is required (the shared fixture refuses
|
||||
client-requested deletion)."""
|
||||
source = os.path.join(TEST_DATA_DIR, "dbg_del_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "dbg_del_dst")
|
||||
_make_tree(source)
|
||||
clean_dir(dest)
|
||||
seeded = get_dest_received_dir(dest, source)
|
||||
os.makedirs(seeded)
|
||||
with open(os.path.join(seeded, "extra.tmp"), "wb") as fh:
|
||||
fh.write(b"stale\n")
|
||||
server = ServerManager()
|
||||
server.start(extra_args=["--allow-super", "--allow-delete"])
|
||||
try:
|
||||
result, _ = run_client(source, dest, flags=["-a", "--delete", "--debug=del"],
|
||||
port=server.port)
|
||||
finally:
|
||||
server.stop()
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert "del: " in result.stdout and "extra.tmp" in result.stdout, result.stdout
|
||||
assert not os.path.exists(os.path.join(seeded, "extra.tmp"))
|
||||
|
||||
|
||||
@pytest.mark.ci
|
||||
def test_normal_run_has_no_debug_output(shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "dbg_quiet_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "dbg_quiet_dst")
|
||||
_make_tree(source)
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest, flags=["-a"], port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
assert "[DEBUG]" not in result.stdout
|
||||
assert "flist: scanning" not in result.stdout
|
||||
assert "send: " not in result.stdout
|
||||
@@ -0,0 +1,174 @@
|
||||
"""Differential parity for `--info=mount` and `--info=stats` (no-wire).
|
||||
|
||||
Both behaviours are compared against real rsync 3.4.1:
|
||||
|
||||
* `--info=mount` prints rsync's ``[sender] skipping mount-point dir NAME`` line
|
||||
when ``-xx`` drops a mount-point directory. Plain ``-x`` keeps the empty
|
||||
directory and stays silent, exactly like rsync.
|
||||
* `--info=stats` requests the same transfer-statistics block as `--stats`
|
||||
(rsync spells the full block ``--info=stats2``/``--stats``).
|
||||
|
||||
The tests are skipped when rsync is unavailable.
|
||||
"""
|
||||
import os
|
||||
import re
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import TEST_DATA_DIR, run_client, clean_dir, get_dest_received_dir
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
def _cross_device_mount_tree(source):
|
||||
"""Build a source whose ``nested_link`` is a symlink onto a tmpfs directory.
|
||||
|
||||
``--copy-links`` dereferences it so ``-x`` sees a mount-point directory on a
|
||||
different device. Returns the probe path to remove, or skips the test when
|
||||
no cross-device filesystem is available.
|
||||
"""
|
||||
local = os.stat(".")
|
||||
shm = "/dev/shm"
|
||||
try:
|
||||
shm_stat = os.stat(shm)
|
||||
except OSError:
|
||||
pytest.skip("/dev/shm not available")
|
||||
if shm_stat.st_dev == local.st_dev:
|
||||
pytest.skip("no cross-device filesystem available")
|
||||
|
||||
clean_dir(source)
|
||||
with open(os.path.join(source, "keep.txt"), "wb") as fh:
|
||||
fh.write(b"keep\n")
|
||||
probe = os.path.join(shm, f"fastsync_info_mount_{os.getpid()}")
|
||||
shutil.rmtree(probe, ignore_errors=True)
|
||||
os.makedirs(probe)
|
||||
with open(os.path.join(probe, "inside.txt"), "wb") as fh:
|
||||
fh.write(b"cross\n")
|
||||
try:
|
||||
os.symlink(probe, os.path.join(source, "nested_link"))
|
||||
except OSError:
|
||||
shutil.rmtree(probe, ignore_errors=True)
|
||||
pytest.skip("cannot create symlink")
|
||||
return probe
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_mount_xx_matches_rsync(shared_server):
|
||||
"""`-xx --info=mount` drops the mount-point dir and prints rsync's line."""
|
||||
source = os.path.join(TEST_DATA_DIR, "info_mount_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "info_mount_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "info_mount_rdst")
|
||||
probe = _cross_device_mount_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
flags = ["-a", "--copy-links", "-xx", "--info=mount"]
|
||||
try:
|
||||
rsync_result = _rsync(flags + [source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
|
||||
expected = "[sender] skipping mount-point dir nested_link"
|
||||
assert expected in rsync_result.stdout, rsync_result.stdout
|
||||
assert expected in result.stdout, (result.stdout, result.stderr)
|
||||
|
||||
received = get_dest_received_dir(dest, source)
|
||||
assert os.path.exists(os.path.join(received, "keep.txt"))
|
||||
# -xx omits the mount-point directory entirely.
|
||||
assert not os.path.exists(os.path.join(received, "nested_link"))
|
||||
assert not os.path.exists(os.path.join(rdst, "nested_link"))
|
||||
finally:
|
||||
shutil.rmtree(probe, ignore_errors=True)
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_mount_single_x_is_silent(shared_server):
|
||||
"""Plain `-x` keeps the empty mount-point directory and prints no line."""
|
||||
source = os.path.join(TEST_DATA_DIR, "info_mount1_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "info_mount1_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "info_mount1_rdst")
|
||||
probe = _cross_device_mount_tree(source)
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
flags = ["-a", "--copy-links", "-x", "--info=mount"]
|
||||
try:
|
||||
rsync_result = _rsync(flags + [source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
result, _ = run_client(source, dest, flags=flags, port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
|
||||
assert "skipping mount-point dir" not in rsync_result.stdout
|
||||
assert "skipping mount-point dir" not in result.stdout
|
||||
|
||||
received = get_dest_received_dir(dest, source)
|
||||
assert os.path.isdir(os.path.join(received, "nested_link"))
|
||||
assert not os.path.exists(os.path.join(received, "nested_link", "inside.txt"))
|
||||
assert os.path.isdir(os.path.join(rdst, "nested_link"))
|
||||
assert not os.path.exists(os.path.join(rdst, "nested_link", "inside.txt"))
|
||||
finally:
|
||||
shutil.rmtree(probe, ignore_errors=True)
|
||||
|
||||
|
||||
def _make_stats_tree(root):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "sub"))
|
||||
with open(os.path.join(root, "a.txt"), "wb") as fh:
|
||||
fh.write(b"alpha\n")
|
||||
with open(os.path.join(root, "sub", "b.txt"), "wb") as fh:
|
||||
fh.write(b"beta\n")
|
||||
|
||||
|
||||
def _pick_stats(text):
|
||||
keys = ("Number of files", "Number of regular files transferred", "Total file size",
|
||||
"Total transferred file size", "Literal data", "Matched data")
|
||||
out = {}
|
||||
for line in text.splitlines():
|
||||
for key in keys:
|
||||
if line.startswith(key + ":"):
|
||||
out[key] = line
|
||||
return out
|
||||
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_info_stats_emits_full_stats_block(shared_server):
|
||||
"""`--info=stats` is the same full block as `--stats` and matches rsync."""
|
||||
source = os.path.join(TEST_DATA_DIR, "info_stats_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "info_stats_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "info_stats_rdst")
|
||||
dest2 = os.path.join(TEST_DATA_DIR, "info_stats_dst2")
|
||||
_make_stats_tree(source)
|
||||
for path in (dest, rdst, dest2):
|
||||
clean_dir(path)
|
||||
os.makedirs(get_dest_received_dir(path, source), exist_ok=True)
|
||||
|
||||
rsync_result = _rsync(["-a", "--stats", source + "/", rdst + "/"])
|
||||
assert rsync_result.returncode == 0, rsync_result.stderr
|
||||
|
||||
info_result, _ = run_client(source, dest, flags=["-a", "--info=stats"],
|
||||
port=shared_server.port)
|
||||
assert info_result.returncode == 0, (info_result.stderr or info_result.stdout)[:300]
|
||||
stats_result, _ = run_client(source, dest2, flags=["-a", "--stats"],
|
||||
port=shared_server.port)
|
||||
assert stats_result.returncode == 0, (stats_result.stderr or stats_result.stdout)[:300]
|
||||
|
||||
# --info=stats must print the same block as --stats...
|
||||
assert _pick_stats(info_result.stdout) == _pick_stats(stats_result.stdout), (
|
||||
f"info={info_result.stdout} stats={stats_result.stdout}")
|
||||
# ...and the protocol-independent counters must match real rsync.
|
||||
assert _pick_stats(info_result.stdout) == _pick_stats(rsync_result.stdout), (
|
||||
f"rsync={_pick_stats(rsync_result.stdout)} fastsync={_pick_stats(info_result.stdout)}")
|
||||
assert re.search(r"^Number of files: \d+ \(reg: 2, dir: 2\)$", info_result.stdout,
|
||||
re.MULTILINE), info_result.stdout
|
||||
@@ -0,0 +1,185 @@
|
||||
"""Differential rsync-parity coverage for FastSync's transfer/delete ORDER.
|
||||
|
||||
rsync walks a source tree in its sorted flist order: within each directory the
|
||||
non-directories come first (ascending name), then the subdirectories (ascending
|
||||
name), each subdirectory immediately followed by its own subtree (depth-first).
|
||||
The sequential scanner now reproduces that order, which makes both the
|
||||
``--info=name`` stream and the ``--delete-during`` deletion sequence match real
|
||||
``rsync 3.4.1`` exactly. ``--threads`` has no rsync analogue and is unordered.
|
||||
|
||||
Every test skips cleanly when rsync is absent.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import ( # noqa: E402
|
||||
TEST_DATA_DIR,
|
||||
ServerManager,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
run_client,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
MTIME = 1_500_000_000
|
||||
|
||||
_TREE = {
|
||||
"a.txt": b"a\n",
|
||||
"b.txt": b"b\n",
|
||||
"z.txt": b"z\n",
|
||||
"a_dir/f.txt": b"f\n",
|
||||
"a_dir/deep/g.txt": b"g\n",
|
||||
"m_dir/h.txt": b"h\n",
|
||||
"Z_dir/i.txt": b"i\n",
|
||||
}
|
||||
|
||||
|
||||
def _write(path, data):
|
||||
os.makedirs(os.path.dirname(path), exist_ok=True)
|
||||
with open(path, "wb") as fh:
|
||||
fh.write(data)
|
||||
os.utime(path, (MTIME, MTIME))
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
def _deleting(text):
|
||||
out = []
|
||||
for line in text.splitlines():
|
||||
stripped = line.strip()
|
||||
if stripped.startswith("*deleting") or stripped.startswith("deleting"):
|
||||
out.append(stripped.split()[-1])
|
||||
return out
|
||||
|
||||
|
||||
class TestTransferOrderParity:
|
||||
@requires_rsync
|
||||
def test_info_name_file_order_matches_rsync(self, shared_server):
|
||||
source = os.path.join(TEST_DATA_DIR, "order_name_src")
|
||||
clean_dir(source)
|
||||
for rel, data in _TREE.items():
|
||||
_write(os.path.join(source, rel), data)
|
||||
|
||||
rdst = os.path.join(TEST_DATA_DIR, "order_name_rdst")
|
||||
clean_dir(rdst)
|
||||
rs = _rsync(["-a", "--info=name", source + "/", rdst + "/"])
|
||||
assert rs.returncode == 0, rs.stderr
|
||||
# rsync also names the directories (trailing '/'); FastSync names the
|
||||
# transferred entries. Compare the file/symlink sequence, which is what
|
||||
# the traversal order determines.
|
||||
rsync_files = [l for l in rs.stdout.splitlines() if l.strip() and not l.endswith("/")]
|
||||
|
||||
fdst = os.path.join(TEST_DATA_DIR, "order_name_fdst")
|
||||
clean_dir(fdst)
|
||||
result, _ = run_client(source, fdst, flags=["-a", "--info=name"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
fsync_files = [
|
||||
l for l in result.stdout.splitlines()
|
||||
if l.strip() and l.strip() != "./" and not l.startswith("sending")
|
||||
]
|
||||
assert fsync_files == rsync_files, (
|
||||
f"transfer order differs\nrsync={rsync_files}\nfastsync={fsync_files}")
|
||||
|
||||
|
||||
class TestDeleteOrderParity:
|
||||
_EXTRA = {
|
||||
"a_extra.txt": b"a\n",
|
||||
"z_extra.txt": b"z\n",
|
||||
"a_extra_dir/f": b"f\n",
|
||||
"z_extra_dir/f": b"f\n",
|
||||
"a_extra_dir/sub/g": b"g\n",
|
||||
}
|
||||
|
||||
def _seed_source(self):
|
||||
source = os.path.join(TEST_DATA_DIR, "order_del_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"k\n")
|
||||
_write(os.path.join(source, "keepdir", "x.txt"), b"x\n")
|
||||
_write(os.path.join(source, "keep2", "y.txt"), b"y\n")
|
||||
return source
|
||||
|
||||
def _assert_order(self, timing, dry_run=False):
|
||||
source = self._seed_source()
|
||||
rdst = os.path.join(TEST_DATA_DIR, f"order_{timing}_rdst")
|
||||
clean_dir(rdst)
|
||||
for rel, data in self._EXTRA.items():
|
||||
_write(os.path.join(rdst, rel), data)
|
||||
rs_flags = ["-a", "-n"] if dry_run else ["-a"]
|
||||
rs = _rsync(rs_flags + [timing, "--info=del", source + "/", rdst + "/"])
|
||||
assert rs.returncode == 0, rs.stderr
|
||||
|
||||
fdst = os.path.join(TEST_DATA_DIR, f"order_{timing}_fdst")
|
||||
clean_dir(fdst)
|
||||
received = get_dest_received_dir(fdst, source)
|
||||
for rel, data in self._EXTRA.items():
|
||||
_write(os.path.join(received, rel), data)
|
||||
fs_flags = ["-a", "-n"] if dry_run else ["-a"]
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, fdst, flags=fs_flags + [timing, "--info=del"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, (result.stderr or result.stdout)[:300]
|
||||
|
||||
rsync_order = _deleting(rs.stdout)
|
||||
fsync_order = _deleting(result.stdout)
|
||||
assert sorted(fsync_order) == sorted(rsync_order), (
|
||||
f"{timing} deleted set differs\nrsync={rsync_order}\nfastsync={fsync_order}")
|
||||
assert fsync_order == rsync_order, (
|
||||
f"{timing} deletion order differs\nrsync={rsync_order}\nfastsync={fsync_order}")
|
||||
|
||||
@requires_rsync
|
||||
def test_delete_during_deletion_order_matches_rsync(self):
|
||||
self._assert_order("--delete-during")
|
||||
|
||||
@requires_rsync
|
||||
def test_delete_delay_deletion_order_matches_rsync(self):
|
||||
self._assert_order("--delete-delay")
|
||||
|
||||
@requires_rsync
|
||||
def test_dry_run_delete_order_matches_rsync(self):
|
||||
self._assert_order("--delete", dry_run=True)
|
||||
|
||||
@requires_rsync
|
||||
def test_partial_max_delete_survivor_order_matches_rsync(self):
|
||||
"""With the exact removal order matching rsync, a --max-delete cap stops
|
||||
after the same entries, so the survivor set is identical too."""
|
||||
source = os.path.join(TEST_DATA_DIR, "order_maxdel_src")
|
||||
clean_dir(source)
|
||||
_write(os.path.join(source, "keep.txt"), b"k\n")
|
||||
extra = {f"e{i}.txt": b"x\n" for i in range(6)}
|
||||
extra["ed/f"] = b"f\n"
|
||||
extra["ed/g"] = b"g\n"
|
||||
|
||||
rdst = os.path.join(TEST_DATA_DIR, "order_maxdel_rdst")
|
||||
clean_dir(rdst)
|
||||
for rel, data in extra.items():
|
||||
_write(os.path.join(rdst, rel), data)
|
||||
rs = _rsync(["-a", "--delete-during", "--max-delete=3", "--info=del",
|
||||
source + "/", rdst + "/"])
|
||||
assert rs.returncode in (0, 25), (rs.returncode, rs.stderr)
|
||||
|
||||
fdst = os.path.join(TEST_DATA_DIR, "order_maxdel_fdst")
|
||||
clean_dir(fdst)
|
||||
received = get_dest_received_dir(fdst, source)
|
||||
for rel, data in extra.items():
|
||||
_write(os.path.join(received, rel), data)
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
result, _ = run_client(source, fdst,
|
||||
flags=["-a", "--delete-during", "--max-delete=3", "--info=del"],
|
||||
port=server.port)
|
||||
assert result.returncode in (0, 25), (result.returncode, result.stderr[:300])
|
||||
assert _deleting(result.stdout) == _deleting(rs.stdout), (
|
||||
f"partial --max-delete survivor order differs\n"
|
||||
f"rsync={_deleting(rs.stdout)}\nfastsync={_deleting(result.stdout)}")
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,326 @@
|
||||
"""rsync 3.4.1 parity for selection/path semantics and client option aliases.
|
||||
|
||||
Each test pins behaviour against real ``rsync 3.4.1``; the differential tests
|
||||
skip cleanly when rsync is not installed.
|
||||
"""
|
||||
import os
|
||||
import shutil
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
import pytest
|
||||
|
||||
sys.path.insert(0, os.path.dirname(__file__))
|
||||
from common import (
|
||||
TEST_DATA_DIR,
|
||||
CLIENT_CMD,
|
||||
ServerManager,
|
||||
run_client,
|
||||
clean_dir,
|
||||
get_dest_received_dir,
|
||||
)
|
||||
|
||||
RSYNC = shutil.which("rsync")
|
||||
requires_rsync = pytest.mark.skipif(RSYNC is None, reason="rsync 3.4.1 not installed")
|
||||
|
||||
|
||||
def _tree(root):
|
||||
"""Sorted relative paths of directories (``D ``) and files (``F ``)."""
|
||||
out = []
|
||||
for dirpath, dirs, files in os.walk(root):
|
||||
rel = os.path.relpath(dirpath, root)
|
||||
for d in dirs:
|
||||
out.append("D " + (d if rel == "." else os.path.join(rel, d)))
|
||||
for f in files:
|
||||
out.append("F " + (f if rel == "." else os.path.join(rel, f)))
|
||||
return sorted(out)
|
||||
|
||||
|
||||
def _rsync(args):
|
||||
env = dict(os.environ, LC_ALL="C")
|
||||
return subprocess.run([RSYNC] + args, capture_output=True, text=True, env=env, timeout=120)
|
||||
|
||||
|
||||
def _make_tree(root):
|
||||
clean_dir(root)
|
||||
for rel, content in {
|
||||
"top.txt": b"top\n",
|
||||
"foo/bar/baz/f.txt": b"deep\n",
|
||||
"sub/x.txt": b"x\n",
|
||||
}.items():
|
||||
full = os.path.join(root, rel)
|
||||
os.makedirs(os.path.dirname(full), exist_ok=True)
|
||||
with open(full, "wb") as fh:
|
||||
fh.write(content)
|
||||
return root
|
||||
|
||||
|
||||
class TestRelativeGeneral:
|
||||
"""#11: -R without --files-from uses rsync's '/./' cut and relative
|
||||
reconstruction instead of always mirroring the full source path."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("suffix", ["", "/./foo", "/./foo/bar", "/./"])
|
||||
def test_relative_cut_matches_rsync(self, shared_server, suffix):
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_rel_src"))
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_rel_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_rel_rdst")
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
spec = source + suffix
|
||||
r = _rsync(["-aR", spec, rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
result, _ = run_client(spec, dest, flags=["-R"], port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
assert _tree(rdst) == _tree(dest), f"layout mismatch for {spec!r}"
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_no_implied_dirs_matches_rsync(self, shared_server):
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_nid_src"))
|
||||
a = os.path.join(source, "foo")
|
||||
b = os.path.join(source, "foo", "bar")
|
||||
os.chmod(a, 0o700)
|
||||
os.chmod(b, 0o711)
|
||||
os.utime(a, (978307200, 978307200))
|
||||
os.utime(b, (978307200, 978307200))
|
||||
spec = source + "/./foo/bar"
|
||||
for extra in ([], ["--no-implied-dirs"]):
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_nid_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_nid_rdst")
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
r = _rsync(["-aR"] + extra + [spec, rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
result, _ = run_client(spec, dest, flags=["-a", "-R"] + extra,
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
for rel in ("foo", "foo/bar"):
|
||||
rs = os.stat(os.path.join(rdst, rel))
|
||||
fs = os.stat(os.path.join(dest, rel))
|
||||
assert (rs.st_mode & 0o7777) == (fs.st_mode & 0o7777), \
|
||||
f"mode mismatch for {rel} with {extra}"
|
||||
if extra == ["--no-implied-dirs"]:
|
||||
# The implied parent directory is created at run time (no
|
||||
# metadata applied), so rsync's and FastSync's separate runs
|
||||
# can differ by a second; compare with a tolerance.
|
||||
assert abs(rs.st_mtime - fs.st_mtime) <= 2, \
|
||||
f"mtime mismatch for {rel} with {extra}"
|
||||
else:
|
||||
assert int(rs.st_mtime) == int(fs.st_mtime), \
|
||||
f"mtime mismatch for {rel} with {extra}"
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_no_implied_dirs_files_from_matches_rsync(self, shared_server):
|
||||
"""-R --no-implied-dirs --files-from: a listed file whose parent is not
|
||||
itself listed still transfers; the implied parent is created with
|
||||
default attributes (rsync 3.4.1 parity)."""
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_nidff_src"))
|
||||
# Make the implied parent unmistakably non-default on the source so a
|
||||
# wrongly-applied attribute would be observable.
|
||||
os.chmod(os.path.join(source, "foo"), 0o700)
|
||||
os.chmod(os.path.join(source, "foo", "bar"), 0o711)
|
||||
os.utime(os.path.join(source, "foo"), (978307200, 978307200))
|
||||
os.utime(os.path.join(source, "foo", "bar"), (978307200, 978307200))
|
||||
lst = os.path.join(TEST_DATA_DIR, "sel_nidff_list")
|
||||
with open(lst, "w") as fh:
|
||||
fh.write("foo/bar/baz/f.txt\n")
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_nidff_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_nidff_rdst")
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
r = _rsync(["-rlpt", "-R", "--no-implied-dirs", "--files-from=" + lst,
|
||||
source + "/", rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
result, _ = run_client(source, dest, flags=[
|
||||
"-rlpt", "-R", "--no-implied-dirs", "--files-from", lst],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
assert _tree(rdst) == _tree(dest), "implied-parent layout mismatch"
|
||||
# The implied parents exist on both sides and carry the run-time default
|
||||
# attributes, not the source's (non-default) ones.
|
||||
for rel in ("foo", "foo/bar", "foo/bar/baz"):
|
||||
rs = os.stat(os.path.join(rdst, rel))
|
||||
fs = os.stat(os.path.join(dest, rel))
|
||||
assert (rs.st_mode & 0o7777) == (fs.st_mode & 0o7777), \
|
||||
f"mode mismatch for implied {rel}"
|
||||
# The listed file is transferred with its content.
|
||||
with open(os.path.join(dest, "foo", "bar", "baz", "f.txt"), "rb") as fh:
|
||||
assert fh.read() == b"deep\n"
|
||||
|
||||
|
||||
class TestDirsOneLevel:
|
||||
"""#13: -d with a trailing slash (or '.') lists the source's immediate
|
||||
contents; FastSync mirrors them below the source-root mirror, so compare
|
||||
rsync's destination tree against that mirror."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_dirs_trailing_slash_matches_rsync(self, shared_server):
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_dirs_src"))
|
||||
os.makedirs(os.path.join(source, "empty"), exist_ok=True)
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_dirs_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_dirs_rdst")
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
r = _rsync(["-d", source + "/", rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
result, _ = run_client(source + "/", dest, flags=["-d"], port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
mirror = get_dest_received_dir(dest, source)
|
||||
assert _tree(rdst) == _tree(mirror)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_dirs_relative_matches_rsync(self, shared_server):
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_dirsr_src"))
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_dirsr_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_dirsr_rdst")
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
spec = source + "/./foo"
|
||||
r = _rsync(["-d", "-R", spec, rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
result, _ = run_client(spec, dest, flags=["-d", "-R"], port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
assert _tree(rdst) == _tree(dest)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_relative_delete_scope_matches_rsync(self):
|
||||
"""-R --delete must be confined to the transferred prefix subtree so a
|
||||
sibling destination directory survives (rsync parity)."""
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_delscope_src"))
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_delscope_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_delscope_rdst")
|
||||
spec = source + "/./foo"
|
||||
for root in (dest, rdst):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "foo"))
|
||||
with open(os.path.join(root, "foo", "extra.txt"), "wb") as fh:
|
||||
fh.write(b"extra\n")
|
||||
os.makedirs(os.path.join(root, "unrelated"))
|
||||
with open(os.path.join(root, "unrelated", "keep.txt"), "wb") as fh:
|
||||
fh.write(b"keep\n")
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
r = _rsync(["-aR", "--delete", spec, rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
result, _ = run_client(spec, dest, flags=["-a", "-R", "--delete"],
|
||||
port=server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
assert (os.path.isfile(os.path.join(dest, "unrelated", "keep.txt"))
|
||||
== os.path.isfile(os.path.join(rdst, "unrelated", "keep.txt")))
|
||||
assert _tree(dest) == _tree(rdst)
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
@pytest.mark.parametrize("mt", [False, True])
|
||||
def test_relative_delete_protects_excluded_mirror(self, mt):
|
||||
"""-R --delete with --exclude must protect the destination mirror of an
|
||||
excluded source path (recorded as a prefix-relative wire path)."""
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_delexc_src"))
|
||||
with open(os.path.join(source, "foo", "secret.tmp"), "wb") as fh:
|
||||
fh.write(b"secret\n")
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_delexc_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_delexc_rdst")
|
||||
for root in (dest, rdst):
|
||||
clean_dir(root)
|
||||
os.makedirs(os.path.join(root, "foo"))
|
||||
with open(os.path.join(root, "foo", "secret.tmp"), "wb") as fh:
|
||||
fh.write(b"secret\n")
|
||||
with open(os.path.join(root, "foo", "extra.txt"), "wb") as fh:
|
||||
fh.write(b"extra\n")
|
||||
spec = source + "/./foo"
|
||||
with ServerManager() as server:
|
||||
server.start(extra_args=["--allow-delete"])
|
||||
r = _rsync(["-aR", "--delete", "--exclude=*.tmp", spec, rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
flags = ["-a", "-R", "--delete", "--exclude=*.tmp"] + (["--threads"] if mt else [])
|
||||
result, _ = run_client(spec, dest, flags=flags, port=server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
assert _tree(dest) == _tree(rdst)
|
||||
assert os.path.isfile(os.path.join(dest, "foo", "secret.tmp"))
|
||||
assert not os.path.exists(os.path.join(dest, "foo", "extra.txt"))
|
||||
|
||||
|
||||
class TestClientAliases:
|
||||
"""#5: safe rsync option aliases accepted client-side."""
|
||||
|
||||
def _seed(self):
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_alias_src"))
|
||||
return source
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"flag",
|
||||
[
|
||||
"--ignore-non-existing",
|
||||
"--protect-args",
|
||||
"--msgs2stderr",
|
||||
"--no-msgs2stderr",
|
||||
"--no-iconv",
|
||||
"--iconv=.",
|
||||
"--iconv=-",
|
||||
],
|
||||
)
|
||||
def test_alias_accepted(self, shared_server, flag):
|
||||
source = self._seed()
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_alias_dst")
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest, flags=[flag], port=shared_server.port)
|
||||
assert result.returncode == 0, f"{flag} rejected: {result.stderr[:300]}"
|
||||
|
||||
def test_lone_h_prints_help(self):
|
||||
result = subprocess.run([CLIENT_CMD[0], "-h"], capture_output=True, text=True,
|
||||
timeout=30)
|
||||
assert result.returncode == 0, result.stderr
|
||||
assert "Usage" in (result.stdout + result.stderr)
|
||||
|
||||
def test_h_with_args_still_human_readable(self, shared_server):
|
||||
source = self._seed()
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_h_dst")
|
||||
clean_dir(dest)
|
||||
result, _ = run_client(source, dest, flags=["-h"], port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
received = get_dest_received_dir(dest, source)
|
||||
assert os.path.isfile(os.path.join(received, "top.txt"))
|
||||
|
||||
|
||||
class TestFilesFromEdges:
|
||||
"""#8/#58: --files-from empty list succeeds; rsync 3.4.1 rejects the
|
||||
--no-ignore-missing-args negation, so FastSync must reject it too."""
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_empty_files_from_list_succeeds(self, shared_server):
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_ff_src"))
|
||||
dest = os.path.join(TEST_DATA_DIR, "sel_ff_dst")
|
||||
rdst = os.path.join(TEST_DATA_DIR, "sel_ff_rdst")
|
||||
clean_dir(dest)
|
||||
clean_dir(rdst)
|
||||
lst = os.path.join(TEST_DATA_DIR, "sel_ff_empty")
|
||||
with open(lst, "w") as fh:
|
||||
fh.write("")
|
||||
r = _rsync(["-a", "--files-from=" + lst, source + "/", rdst + "/"])
|
||||
assert r.returncode == 0, r.stderr
|
||||
result, _ = run_client(source, dest, flags=["--files-from", lst],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, result.stderr[:300]
|
||||
assert _tree(rdst) == []
|
||||
assert _tree(dest) == []
|
||||
|
||||
@requires_rsync
|
||||
@pytest.mark.ci
|
||||
def test_no_ignore_missing_args_rejected_like_rsync(self):
|
||||
source = _make_tree(os.path.join(TEST_DATA_DIR, "sel_nima_src"))
|
||||
r = _rsync(["-a", "--no-ignore-missing-args", source + "/",
|
||||
os.path.join(TEST_DATA_DIR, "sel_nima_rdst") + "/"])
|
||||
assert r.returncode != 0, "rsync unexpectedly accepted --no-ignore-missing-args"
|
||||
|
||||
cmd = CLIENT_CMD + ["--source-dir", source, "--dest-dir",
|
||||
os.path.join(TEST_DATA_DIR, "sel_nima_dst"), "--save-to-disk",
|
||||
"--no-ignore-missing-args"]
|
||||
result = subprocess.run(cmd, capture_output=True, text=True, timeout=30)
|
||||
assert result.returncode != 0, "FastSync unexpectedly accepted the negation"
|
||||
@@ -94,14 +94,14 @@ def _seed_protocol_source(source):
|
||||
class TestProtocol:
|
||||
@pytest.mark.ci
|
||||
def test_protocol_current_version_accepted(self, shared_server):
|
||||
"""--protocol=2.21.0 (the current PROTOCOL_VERSION) is accepted and the
|
||||
"""--protocol=2.28.0 (the current PROTOCOL_VERSION) is accepted and the
|
||||
transfer completes normally."""
|
||||
source = os.path.join(TEST_DATA_DIR, "proto_ok_src")
|
||||
dest = os.path.join(TEST_DATA_DIR, "proto_ok_dst")
|
||||
shutil.rmtree(dest, ignore_errors=True)
|
||||
os.makedirs(dest)
|
||||
_seed_protocol_source(source)
|
||||
result, _ = run_client(source, dest, flags=["--protocol=2.21.0"],
|
||||
result, _ = run_client(source, dest, flags=["--protocol=2.28.0"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode == 0, \
|
||||
f"--protocol current run failed: {(result.stderr or result.stdout)[:400]}"
|
||||
@@ -118,7 +118,8 @@ class TestProtocol:
|
||||
shutil.rmtree(dest, ignore_errors=True)
|
||||
os.makedirs(dest)
|
||||
_seed_protocol_source(source)
|
||||
for bad in ("2.20.0", "2.19.0", "2.18.0", "2.17.0", "2.15.0", "2.16.0", "216", "31"):
|
||||
for bad in ("2.27.0", "2.26.0", "2.25.0", "2.24.0", "2.23.0", "2.22.0", "2.21.0", "2.20.0",
|
||||
"2.19.0", "2.18.0", "2.17.0", "2.15.0", "2.16.0", "216", "31"):
|
||||
result, _ = run_client(source, dest, flags=[f"--protocol={bad}"],
|
||||
port=shared_server.port)
|
||||
assert result.returncode != 0, f"--protocol={bad} should be rejected"
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user