Compare commits
340
Commits
6968ff6734
..
dev
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b7e95b8d96 | ||
|
|
13627e82fc | ||
|
|
09ab81120e | ||
|
|
2a7afbda0a | ||
|
|
bab6fcc706 | ||
|
|
6a426cffa8 | ||
|
|
b74182c7c1 | ||
|
|
f2a8eeaea7 | ||
|
|
18d7d495cf | ||
|
|
59d20e867a | ||
|
|
cc931790e9 | ||
|
|
60776b78fc | ||
|
|
52ad0aa512 | ||
|
|
759ffea117 | ||
|
|
ef37c2b988 | ||
|
|
a14fbb2c10 | ||
|
|
7408686370 | ||
|
|
90f2fb613f | ||
|
|
8843f1e3cf | ||
|
|
c387beb182 | ||
|
|
96e02f52c0 | ||
|
|
f7d5dda93b | ||
|
|
6d9cdb83ba | ||
|
|
7c748f1f7a | ||
|
|
884530c9a2 | ||
|
|
729f3ef8e1 | ||
|
|
b414f197af | ||
|
|
29f8be161c | ||
|
|
cb2979fdf1 | ||
|
|
00197102bf | ||
|
|
13b257d1dd | ||
|
|
f3d7672694 | ||
|
|
18b821d32c | ||
|
|
6ad065925f | ||
|
|
46dcefe218 | ||
|
|
b88acdbd3c | ||
|
|
c29bce54fc | ||
|
|
d98e971fcc | ||
|
|
492ce0ce89 | ||
|
|
ec6692ac41 | ||
|
|
1f8d60e30d | ||
|
|
6db9827b87 | ||
|
|
a07cc00bb0 | ||
|
|
3ae7685655 | ||
|
|
4f945a8e39 | ||
|
|
15f38f5b76 | ||
|
|
787967d3ce | ||
|
|
06e7aef5c9 | ||
|
|
ee265d78ba | ||
|
|
47b1b9b915 | ||
|
|
bb6c788cf9 | ||
|
|
0b40c47d6a | ||
|
|
6f81005094 | ||
|
|
193d358c64 | ||
|
|
8792e3265a | ||
|
|
3ec0ffb644 | ||
|
|
80e8d7f450 | ||
|
|
86741725fd | ||
|
|
f96f1764af | ||
|
|
d780a4625e | ||
|
|
5d1303ffdf | ||
|
|
06b53c5b3d | ||
|
|
bf09e8893e | ||
|
|
034c27926f | ||
|
|
aa15099d22 | ||
|
|
ff4db23831 | ||
|
|
79e45441c0 | ||
|
|
6537227467 | ||
|
|
309c9aed98 | ||
|
|
84594197ee | ||
|
|
7bf25048f6 | ||
|
|
b6b5eee20e | ||
|
|
8dcd87609d | ||
|
|
19fe63bd59 | ||
|
|
1f5f8dc5a7 | ||
|
|
3787695ba6 | ||
|
|
b7cb213c8c | ||
|
|
8cc3dd993b | ||
|
|
1167e7970b | ||
|
|
e98729f00e | ||
|
|
c0020364b2 | ||
|
|
d7ac940a6b | ||
|
|
be20e836de | ||
|
|
b549887138 | ||
|
|
1ffd4744c6 | ||
|
|
494cef2a0b | ||
|
|
ed7527cc2c | ||
|
|
934defa965 | ||
|
|
ade9be8600 | ||
|
|
707bb659e8 | ||
|
|
33980bc4c8 | ||
|
|
6a9372f4f5 | ||
|
|
5e1d6b6e10 | ||
|
|
d2d1b63f44 | ||
|
|
a6659472fe | ||
|
|
4638030288 | ||
|
|
07dfec629f | ||
|
|
c062a0762e | ||
|
|
283f9f0823 | ||
|
|
5754b9a952 | ||
|
|
b8a0efef7b | ||
|
|
2e77c09447 | ||
|
|
06c4026b74 | ||
|
|
9f47b13712 | ||
|
|
b1eddf0133 | ||
|
|
338c27db73 | ||
|
|
8d46a26c04 | ||
|
|
707ba272df | ||
|
|
bc71a3c0a5 | ||
|
|
cee9b7647c | ||
|
|
3adb6dddb5 | ||
|
|
d77849774e | ||
|
|
b5c6f8b60f | ||
|
|
134dcd74cc | ||
|
|
42f2f845ac | ||
|
|
0669335ca5 | ||
|
|
391f76cd56 | ||
|
|
2021afe613 | ||
|
|
ba914e8ab3 | ||
|
|
df8fe1ae4e | ||
|
|
2c490d58b7 | ||
|
|
d55dabff2e | ||
|
|
b478a59a81 | ||
|
|
865941f850 | ||
|
|
221cefa7cc | ||
|
|
bb9b59024a | ||
|
|
5baf120243 | ||
|
|
84b7e1fd2f | ||
|
|
800a978e15 | ||
|
|
0f40e747f0 | ||
|
|
b07306d5bc | ||
|
|
f2c89b6e7c | ||
|
|
379f127859 | ||
|
|
3799200f71 | ||
|
|
91c4a967e6 | ||
|
|
88c8968b4c | ||
|
|
082b31886a | ||
|
|
423a62e691 | ||
|
|
bc18ae205b | ||
|
|
10a61c6101 | ||
|
|
bc1e1191af | ||
|
|
0fbb9de915 | ||
|
|
9b05972375 | ||
|
|
d119f35066 | ||
|
|
00829fd265 | ||
|
|
ff261bc38a | ||
|
|
b235721f8b | ||
|
|
79a28cdb96 | ||
|
|
402cae80ad | ||
|
|
9691dba6f0 | ||
|
|
558782d339 | ||
|
|
5597e74f6a | ||
|
|
ee6523afac | ||
|
|
4163caa1d3 | ||
|
|
10159dc120 | ||
|
|
cd7b96d0bb | ||
|
|
f6f49d536e | ||
|
|
38d304103c | ||
|
|
4b09213b88 | ||
|
|
36fd0774e8 | ||
|
|
711b7e50b3 | ||
|
|
67076bf218 | ||
|
|
8ec8cb7203 | ||
|
|
82959395fb | ||
|
|
c9f94ea46e | ||
|
|
eff9852038 | ||
|
|
c80098623f | ||
|
|
6119e1e75c | ||
|
|
25062352f6 | ||
|
|
00d628d4ba | ||
|
|
3545d88905 | ||
|
|
596a039454 | ||
|
|
ae037cc27b | ||
|
|
3d0672a721 | ||
|
|
b82aab72c5 | ||
|
|
8e7764007d | ||
|
|
1042d15db7 | ||
|
|
cbe37a77dd | ||
|
|
a5d45ef266 | ||
|
|
a960391b34 | ||
|
|
134f8b027a | ||
|
|
eb7e3fd2e0 | ||
|
|
6a129b54d4 | ||
|
|
c7b2c7eb2b | ||
|
|
e77dfbec70 | ||
|
|
f3ac4df4d0 | ||
|
|
91197fd7cf | ||
|
|
13708352ec | ||
|
|
d162d93570 | ||
|
|
9fa1696eff | ||
|
|
b705fb807f | ||
|
|
cee281ff55 | ||
|
|
5c509831b8 | ||
|
|
ee57aeea4a | ||
|
|
803c1d3385 | ||
|
|
b8ec62beef | ||
|
|
59bfd32b9e | ||
|
|
3432a33d9a | ||
|
|
a1b081d328 | ||
|
|
bbecff9c04 | ||
|
|
c1553bd5d6 | ||
|
|
1a550bda24 | ||
|
|
902f86192d | ||
|
|
6a40ac86e5 | ||
|
|
410ba6e992 | ||
|
|
6c6f02e5dd | ||
|
|
d9006d1fda | ||
|
|
36375010d3 | ||
|
|
5efa0dba7c | ||
|
|
e32733fbf6 | ||
|
|
b02799327d | ||
|
|
946aa934cc | ||
|
|
125921c11b | ||
|
|
9dd5288381 | ||
|
|
3f2c74dd9e | ||
|
|
51e41dee2a | ||
|
|
dbf1b39d47 | ||
|
|
845f20a28d | ||
|
|
a5083776da | ||
|
|
76a81f1684 | ||
|
|
24b81c7e5a | ||
|
|
0e33f84f38 | ||
|
|
c7ac039523 | ||
|
|
e771cc9da6 | ||
|
|
7f9f82a068 | ||
|
|
695b5c8c25 | ||
|
|
5b2188d909 | ||
|
|
e5da916d54 | ||
|
|
de640bba1b | ||
|
|
f0f5719be0 | ||
|
|
6200b298ac | ||
|
|
394a9aae22 | ||
|
|
12d4af1b89 | ||
|
|
9d7c55d3c0 | ||
|
|
5a104bfd88 | ||
|
|
2ada8f9ad5 | ||
|
|
1493f1806d | ||
|
|
ea28e25535 | ||
|
|
d6295d62ce | ||
|
|
a9f416ce44 | ||
|
|
448edc0432 | ||
|
|
4a7703b06a | ||
|
|
6fc297544e | ||
|
|
583d3c8edb | ||
|
|
9883757190 | ||
|
|
a690109975 | ||
|
|
36d4d0e43e | ||
|
|
a0b9d9794b | ||
|
|
3e9f70d9ba | ||
|
|
2b5aaef409 | ||
|
|
478f80be9f | ||
|
|
e674b25213 | ||
|
|
1116da9f64 | ||
|
|
684153350a | ||
|
|
ec206b02d0 | ||
|
|
88bdfeeb58 | ||
|
|
3f5b0250f4 | ||
|
|
c41bfb2cdb | ||
|
|
1b2632f968 | ||
|
|
7dbca70a4b | ||
|
|
1a053f06e5 | ||
|
|
58b3a33e82 | ||
|
|
17b0632098 | ||
|
|
376e6500ab | ||
|
|
82a1d5e240 | ||
|
|
3eec5a4cc3 | ||
|
|
6144c7fc7f | ||
|
|
ea4ab661b4 | ||
|
|
84827ca617 | ||
|
|
23552e823d | ||
|
|
d1a567f7e3 | ||
|
|
93c1fc3c1f | ||
|
|
4815b1b281 | ||
|
|
ad7bc3348b | ||
|
|
34970b961c | ||
|
|
b3f7cad4db | ||
|
|
cd8a84c0a2 | ||
|
|
09c384d7d0 | ||
|
|
81ad313ee5 | ||
|
|
919a729206 | ||
|
|
8cd2b550d9 | ||
|
|
99c0fd8016 | ||
|
|
a2200f039a | ||
|
|
1437c6dc6b | ||
|
|
38356ecc1e | ||
|
|
7badac7f97 | ||
|
|
6ccf16b650 | ||
|
|
1174993d6b | ||
|
|
d5fcfa2c5c | ||
|
|
e79d2b47b0 | ||
|
|
7c24a365cf | ||
|
|
5893de4a34 | ||
|
|
825ba69753 | ||
|
|
34abaadb9a | ||
|
|
10c4ffebdf | ||
|
|
dfa2a42028 | ||
|
|
5b0ec5880d | ||
|
|
1a26bde2d4 | ||
|
|
58a28334b8 | ||
|
|
e48f19ee2b | ||
|
|
a2370433b2 | ||
|
|
9da5a0a9ed | ||
|
|
80c1ff321c | ||
|
|
551c187005 | ||
|
|
0d6c1f784f | ||
|
|
f75a69f96a | ||
|
|
df887c73b1 | ||
|
|
5d39619a8a | ||
|
|
6269ae54e5 | ||
|
|
07f7555c1d | ||
|
|
5b8aca5799 | ||
|
|
a1eaa93357 | ||
|
|
99df0a8a6d | ||
|
|
334fc5b3e8 | ||
|
|
88aee6ce94 | ||
|
|
f6e8b6ddc4 | ||
|
|
6f974eff19 | ||
|
|
72cccaa256 | ||
|
|
eb71b29d1c | ||
|
|
4d5befedfe | ||
|
|
f00844cf9a | ||
|
|
2ec17e821c | ||
|
|
6d47d93fd7 | ||
|
|
6ea966781f | ||
|
|
0a7f5faea6 | ||
|
|
25909110ac | ||
|
|
bd43448af2 | ||
|
|
264964411c | ||
|
|
18d1b84246 | ||
|
|
0f95f48899 | ||
|
|
c78a21de57 | ||
|
|
5c8970c64f | ||
|
|
4e918a1b69 | ||
|
|
87f6cb0243 | ||
|
|
e1f8f75e7c | ||
|
|
4c17122b00 | ||
|
|
0abaa62193 | ||
|
|
5334397b81 | ||
|
|
3260a39ab4 | ||
|
|
5d3c43305e |
No files matched your search
@@ -9,7 +9,7 @@ on:
|
||||
jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
# suite) run on merge to dev/main, so PR CI stays well under ~3 minutes.
|
||||
build-and-test:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
steps:
|
||||
- name: Checkout
|
||||
@@ -49,9 +49,50 @@ jobs:
|
||||
if: github.event_name == 'push'
|
||||
run: python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" --durations=25 --tb=short -q
|
||||
|
||||
# Differential rsync-parity gate: runs real rsync 3.4.1 and FastSync over the
|
||||
# same corpora and compares destinations + normalized output. The fast subset
|
||||
# guards the ✅ surface on every PR; the full set (with FASTSYNC_PARITY_STRICT
|
||||
# so a fixed caveat must be removed from the allowlist) burns the documented
|
||||
# ⚠️/❌ residuals down on push. See tests/integration/README.md.
|
||||
parity-fast:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'pull_request'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build -j$(nproc)
|
||||
|
||||
- name: Differential parity (fast subset)
|
||||
run: python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci -q
|
||||
|
||||
parity-full:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build -j$(nproc)
|
||||
|
||||
- name: Differential parity (full set)
|
||||
run: FASTSYNC_PARITY_STRICT=1 python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity -q
|
||||
|
||||
sanitizers:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
strategy:
|
||||
@@ -72,7 +113,7 @@ jobs:
|
||||
|
||||
fuzz-build:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
@@ -94,7 +135,7 @@ jobs:
|
||||
|
||||
coverage:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
@@ -118,7 +159,7 @@ jobs:
|
||||
|
||||
valgrind:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
|
||||
+13
@@ -8,3 +8,16 @@ build-*/
|
||||
build2/
|
||||
build3/
|
||||
build_docker2/
|
||||
|
||||
# Test/run artifacts
|
||||
root/
|
||||
test_partial_install_tmp/
|
||||
|
||||
# Editor/tooling + test caches/artifacts
|
||||
.pytest_cache/
|
||||
*.gcda
|
||||
*.gcno
|
||||
*.gcov
|
||||
di/
|
||||
test_data-manual/
|
||||
*.log
|
||||
@@ -128,7 +128,7 @@ Do not wait for the user to tell you CI failed — check proactively. The user s
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -92,7 +92,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,16 +27,19 @@ FetchContent_Declare(xxhash GIT_REPOSITORY https://github.com/Cyan4973/xxHash GI
|
||||
FetchContent_MakeAvailable(xxhash)
|
||||
|
||||
# Sanitizer option
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread none)
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, undefined, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread undefined none)
|
||||
if(SANITIZER STREQUAL "address")
|
||||
add_compile_options(-fsanitize=address -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=address)
|
||||
elseif(SANITIZER STREQUAL "thread")
|
||||
add_compile_options(-fsanitize=thread -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=thread)
|
||||
elseif(SANITIZER STREQUAL "undefined")
|
||||
add_compile_options(-fsanitize=undefined -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=undefined)
|
||||
elseif(NOT SANITIZER STREQUAL "none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, undefined, none")
|
||||
endif()
|
||||
|
||||
option(STRICT_WARNINGS "Enable strict warnings" OFF)
|
||||
@@ -52,6 +55,16 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
file(GLOB SHARED_SRCS "src/shared/*.c")
|
||||
@@ -61,15 +74,15 @@ file(GLOB TEST_SRCS "tests/*.c")
|
||||
|
||||
add_executable(server ${SERVER_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(server PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(client PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} src/client/scanner.c)
|
||||
target_include_directories(tests PRIVATE tests src/shared src/server src/client)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
```
|
||||
|
||||
### Source Layout
|
||||
@@ -82,19 +95,25 @@ tests/integration/ — Python pytest integration tests
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)`
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)` (default compression codec)
|
||||
- **zlib** — found via `find_library(ZLIB_LIBRARY z)` (the `zlib`/`zlibx` codecs)
|
||||
- **lz4** — found via `find_library(LZ4_LIBRARY lz4)` (the `lz4` codec)
|
||||
- **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport)
|
||||
- **xxHash** — fetched via `FetchContent` from GitHub (delta transfer hashing, v0.8.3)
|
||||
- **xxHash** — fetched via `FetchContent` from the upstream repository (delta transfer hashing, v0.8.3)
|
||||
- **pthreads** — found via `find_package(Threads REQUIRED)`
|
||||
- **C11 standard** — required
|
||||
- **CMake 3.22+** — minimum version
|
||||
|
||||
The codec matrix (protocol 2.26.0) uses zstd/zlib/lz4 for compression and
|
||||
xxHash/OpenSSL for the `xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1` checksums
|
||||
(`none` needs no library); both codec families are negotiated per transfer.
|
||||
|
||||
## Conventions
|
||||
|
||||
- Use `file(GLOB ...)` for source collection (existing pattern).
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `${ZLIB_LIBRARY}`, `${LZ4_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- Include directories: `src/shared`, `src/server`, `src/client`, `tests` (for test target).
|
||||
- Sanitizer support: pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (live option in CMakeLists.txt).
|
||||
- Sanitizer support: pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (live option in CMakeLists.txt).
|
||||
- Build with `cmake -B build -S . && cmake --build build -j$(nproc)`.
|
||||
- For CI, dependencies are provided by the project's custom Docker image (repo-root `Dockerfile`, same image CI uses). For local development, use `nix-shell`. Never add `apt-get install` / `pip install` to CI workflows. See `AGENTS.md`.
|
||||
|
||||
@@ -105,7 +124,7 @@ tests/integration/ — Python pytest integration tests
|
||||
3. Add new dependencies with `find_package` or `find_library`.
|
||||
4. When adding a new executable target, follow the pattern of existing targets.
|
||||
5. When adding a new library (static/shared), use `add_library` and follow the project's naming.
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (matching CI's matrix strategy).
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (matching CI's matrix strategy).
|
||||
7. Always verify the build compiles after changes.
|
||||
|
||||
## Sanitizer Configurations
|
||||
@@ -119,11 +138,9 @@ cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (race conditions)
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
For UndefinedBehaviorSanitizer (no `-DSANITIZER=undefined` option in CMakeLists.txt yet), use the manual flag approach:
|
||||
UndefinedBehaviorSanitizer uses the same built-in option:
|
||||
```bash
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=undefined -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=undefined"
|
||||
cmake -B build -S . -DSANITIZER=undefined
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
@@ -159,7 +176,7 @@ cmake -B build -S . -DCMAKE_BUILD_TYPE=RelWithDebInfo
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/server
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
./build/client
|
||||
./build/tests
|
||||
```
|
||||
@@ -187,7 +204,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ FastSync is a file synchronization tool (like rsync, but faster). It transfers f
|
||||
cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Server (TCP mode)
|
||||
./build/server
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
|
||||
# Client (TCP mode)
|
||||
./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk
|
||||
@@ -37,13 +37,13 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Run tests
|
||||
./build/tests # unit tests
|
||||
python3 test.py # integration tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # integration tests
|
||||
```
|
||||
|
||||
## Code Walkthrough
|
||||
|
||||
### Client Entry Point (`src/client/client_cli.c`)
|
||||
- Parses CLI arguments using `getopt_long`
|
||||
- Parses CLI arguments using a custom option-table parser (`OPTION_TABLE` in `src/client/client_cli.c`); there is no `getopt*` usage
|
||||
- Creates `Config` struct with all options
|
||||
- Detects SSH destinations (contains `:`)
|
||||
- Calls into `client_send.c` for the actual transfer
|
||||
@@ -109,7 +109,7 @@ Collection of files for batch transfer. Serialized with file count, then per-fil
|
||||
zstd streaming compression via `ZSTD_compressStream2`/`ZSTD_decompressStream`. Compression happens per-chunk in the sender stage. Level 1-22 (default 5). Streaming means memory usage stays bounded regardless of file size.
|
||||
|
||||
### "How does sendfile() work?"
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `-f` flag. Only works with TCP (not SSH, not compression).
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `--sendfile` (long form only). Only works with TCP (not SSH, not compression).
|
||||
|
||||
### "How does incremental sync work?"
|
||||
Client sends file metadata (path, size, mtime) to server. Server checks if destination file has same size+mtime. If match, server responds `STATUS_OK` (skip). If mismatch, server responds `STATUS_NEXT` (send).
|
||||
@@ -138,7 +138,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -316,7 +316,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -14,10 +14,9 @@ Diagnose crashes, memory errors, hangs, and logic bugs. You use structured debug
|
||||
### Memory Errors
|
||||
```bash
|
||||
# AddressSanitizer (fast, recommended first)
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/client # or ./build/server
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/client # or ./build-asan/server -p 8080 --allow-unauthenticated
|
||||
|
||||
# Valgrind (slower, more thorough)
|
||||
valgrind --leak-check=full --show-leak-kinds=all --track-origins=yes \
|
||||
@@ -32,10 +31,9 @@ valgrind --tool=drd ./build/client ...
|
||||
|
||||
### Thread Sanitizer
|
||||
```bash
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=thread" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
|
||||
### GDB
|
||||
@@ -143,7 +141,7 @@ gprof ./build/client gmon.out
|
||||
|
||||
### Step 5: Verify
|
||||
- Run `./build/tests` (unit tests)
|
||||
- Run `python3 test.py` (integration tests)
|
||||
- Run `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` (integration tests)
|
||||
- Run under valgrind again to confirm clean
|
||||
- Test under ASan again
|
||||
|
||||
@@ -162,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -96,7 +96,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -16,12 +16,18 @@ Scan the codebase for patterns that suggest new feature opportunities. You ident
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_cli.c Entry point, OPTION_TABLE parser, config setup
|
||||
usage.c Usage/help text (authoritative CLI flag list)
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
change_list.c File change-list bookkeeping
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
server_cli.c Server option-table CLI parsing
|
||||
receiver.c Receiver-side file handling
|
||||
receiver_pipeline.c Receiver worker pipeline
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
@@ -32,40 +38,63 @@ src/shared/ Shared libraries (used by both client and server)
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_send.c/h Sender-side file transfer
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_list.c/h File list model
|
||||
file_store.c/h Destination file store
|
||||
array_list.c/h Dynamic array
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
batch.c/h Batch files (--write-batch/--read-batch)
|
||||
charset.c/h Filename charset conversion (--iconv)
|
||||
chmod.c/h Permission modification (--chmod)
|
||||
xattr.c/h Extended attributes
|
||||
hardlink.c/h Hard-link handling
|
||||
identity.c/h uid/gid mapping (--usermap/--groupmap/--chown)
|
||||
credentials.c/h Daemon credentials
|
||||
daemon_conf.c/h Daemon module configuration
|
||||
motd.c/h Daemon MOTD
|
||||
delay_updates.c/h Delayed update staging
|
||||
stop_condition.c/h Stop-after/stop-at handling
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
file_types.h Shared file type definitions
|
||||
```
|
||||
|
||||
### Existing CLI Flags (from client_cli.c)
|
||||
### Existing CLI Flags (authoritative source: `src/client/usage.c`)
|
||||
```
|
||||
--source-dir <dir> Source directory to sync (required)
|
||||
--dest-dir <dir> Destination directory on server (required)
|
||||
--host <host> Server hostname/IP (required)
|
||||
--port <port> Server TCP port
|
||||
--server-mode Listen as server
|
||||
--use-compression, -c Enable zstd compression
|
||||
--use-multithreading, -m Enable multithreaded transfer
|
||||
--use-sendfile, -s Use sendfile() zero-copy TCP
|
||||
--use-ssh, -S Use SSH transport
|
||||
--use-tls, -T Enable TLS encryption
|
||||
--cert <file> TLS certificate file
|
||||
--key <file> TLS key file
|
||||
--ca <file> TLS CA certificate file
|
||||
--insecure Skip TLS verification
|
||||
--bwlimit <bytes/s> Bandwidth limit
|
||||
--delete Delete files not in source
|
||||
--include <pattern> Include filter pattern
|
||||
--exclude <pattern> Exclude filter pattern
|
||||
--dry-run Print what would be transferred
|
||||
--save-to-disk Save transferred files to disk (for server tests)
|
||||
--source-dir <dir> Source directory
|
||||
--dest-dir <dir> Destination directory on server
|
||||
--server-host <ip> Server IP address (default: 127.0.0.1)
|
||||
--server-port <n> Server port (default: 8080); --port is an alias
|
||||
-c, --checksum Verify content by checksum instead of size+mtime
|
||||
-z, --compress [level] Enable compression (level 1-22, default 5)
|
||||
-j, --threads[=N] Enable multithreaded scanner/loader/sender pipeline
|
||||
--chunk-serialization Enable chunk serialization (long form only)
|
||||
--sendfile sendfile() zero-copy (TCP only; long form only)
|
||||
-s, --secluded-args Protect-args compatibility option (no effect)
|
||||
--tls Enable TLS encryption; --cert/--key/--ca give PEMs
|
||||
--bwlimit <KB/s> Bandwidth limit in kilobytes per second
|
||||
--delete Delete files on receiver not in source
|
||||
--incremental Skip files unchanged since last transfer
|
||||
--delta Delta transfer for changed files (needs --incremental)
|
||||
-f, --filter=RULE rsync-style filter rule (+/- include/exclude)
|
||||
--exclude <pattern> Exclude files matching pattern
|
||||
--include <pattern> Only include files matching pattern
|
||||
-m, --prune-empty-dirs Do not transfer empty directory entries
|
||||
-n, --dry-run Show what would be transferred
|
||||
--save-to-disk Write received files to disk
|
||||
--version Print version and exit
|
||||
--help Print help
|
||||
--help Show help
|
||||
```
|
||||
> Always confirm the current flags with `./build/client --help`; the table above
|
||||
> is a representative subset. `src/client/usage.c` is the authoritative list and
|
||||
> `OPTION_TABLE` in `src/client/client_cli.c` is the parser (there is no `getopt*`).
|
||||
|
||||
## Feature Scout Checklist
|
||||
|
||||
@@ -288,7 +317,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ Design integration tests that verify the full transfer pipeline works end-to-end
|
||||
- Multiple configurations (TCP, SSH, TLS, compression, multithreading)
|
||||
- Network shaping (LAN, WAN profiles)
|
||||
- Feature tests (dry run, archive, exclude, delete, incremental, bandwidth limit)
|
||||
- Run: `python3 -m pytest tests/ -v --tb=short`
|
||||
- Run: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
|
||||
### 3. New: Focused Integration Tests
|
||||
When adding new features or fixing bugs, write targeted integration tests.
|
||||
@@ -35,13 +35,14 @@ mkdir -p /tmp/fastsync_test/src
|
||||
echo "test content" > /tmp/fastsync_test/src/file.txt
|
||||
|
||||
# Start server
|
||||
./build/server &
|
||||
./build/server -p 8080 --allow-unauthenticated &
|
||||
SERVER_PID=$!
|
||||
sleep 0.5
|
||||
|
||||
# Run client
|
||||
./build/client --source-dir /tmp/fastsync_test/src \
|
||||
--dest-dir /tmp/fastsync_test/dst \
|
||||
--server-port 8080 \
|
||||
--save-to-disk
|
||||
|
||||
# Verify
|
||||
@@ -76,7 +77,7 @@ openssl req -x509 -newkey rsa:2048 -keyout /tmp/key.pem -out /tmp/cert.pem \
|
||||
### Pattern 4: Incremental Sync
|
||||
```bash
|
||||
# First sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
|
||||
# Modify source
|
||||
echo "updated" >> /tmp/src/file.txt
|
||||
@@ -89,14 +90,14 @@ echo "updated" >> /tmp/src/file.txt
|
||||
### Pattern 5: Delete Verification
|
||||
```bash
|
||||
# Initial sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
|
||||
# Add extra file to dest
|
||||
echo "extra" > /tmp/dst/.../extra.txt
|
||||
|
||||
# Sync with --delete
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst \
|
||||
--save-to-disk --delete -M
|
||||
--save-to-disk --delete
|
||||
|
||||
# Verify extra.txt is gone
|
||||
test ! -f /tmp/dst/.../extra.txt
|
||||
@@ -115,7 +116,7 @@ The project uses Gitea Actions. Key jobs:
|
||||
jobs:
|
||||
new-job:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v7
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Configure
|
||||
@@ -127,7 +128,7 @@ jobs:
|
||||
- name: Unit Tests
|
||||
run: ./build-${{ matrix.sanitizer }}/tests
|
||||
- name: Integration Tests
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/ -v --tb=short
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
The symlink step is required because `tests/conftest.py` expects `./build` to exist.
|
||||
|
||||
@@ -135,7 +136,7 @@ The symlink step is required because `tests/conftest.py` expects `./build` to ex
|
||||
|
||||
After any code change:
|
||||
- [ ] Unit tests pass: `./build/tests`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/ -v --tb=short`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
- [ ] Build clean: no warnings with `-Wall`
|
||||
- [ ] No memory errors: ASan clean
|
||||
- [ ] No thread errors: TSan clean (if threading involved)
|
||||
@@ -156,7 +157,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
---
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates GitHub issues from their findings.
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates Gitea issues from their findings.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
@@ -12,7 +12,7 @@ You are the primary orchestrator agent. Your job is to:
|
||||
2. Decide which specialized sub-agents to dispatch for analysis
|
||||
3. Delegate analysis work using the task tool
|
||||
4. Receive structured findings from sub-agents
|
||||
5. Create GitHub issues from those findings using `gh issue create`
|
||||
5. Create Gitea issues from those findings using `tea issues create`
|
||||
6. Coordinate the overall analysis workflow end-to-end
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
@@ -97,7 +97,7 @@ First, read the repository structure to understand what exists:
|
||||
### Phase 2: Determine Analysis Scope
|
||||
Based on what the user requests or what needs attention:
|
||||
- **New features wanted?** → Dispatch `feature-scout` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-screener` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-auditor` sub-agent
|
||||
- **Code quality review?** → Dispatch `code-quality-guardian` sub-agent
|
||||
- **All of the above?** → Run all three in parallel
|
||||
|
||||
@@ -110,7 +110,7 @@ Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
```
|
||||
Task: Ask the security-screener agent to analyze the codebase.
|
||||
Task: Ask the security-auditor agent to analyze the codebase.
|
||||
Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
@@ -138,14 +138,14 @@ Each sub-agent returns findings in this structured format:
|
||||
- **Labels**: comma-separated labels for the issue
|
||||
```
|
||||
|
||||
### Phase 5: Create GitHub Issues
|
||||
For each finding, create a GitHub issue:
|
||||
### Phase 5: Create Gitea Issues
|
||||
For each finding, create a Gitea issue:
|
||||
|
||||
```bash
|
||||
gh issue create \
|
||||
tea issues create --repo TapTap/FastSync \
|
||||
--title "<Finding Title>" \
|
||||
--label "<labels>" \
|
||||
--body "## Description
|
||||
--labels "<labels>" \
|
||||
--description "## Description
|
||||
<description>
|
||||
|
||||
## Location
|
||||
@@ -175,11 +175,13 @@ _This issue was automatically generated by the issue-creator agent._"
|
||||
|
||||
### Duplicate Detection
|
||||
Before creating an issue:
|
||||
1. Check existing open issues: `gh issue list --state open --label "<label>"`
|
||||
2. Search for similar titles using `gh issue list --search "<keywords>"`
|
||||
1. Check existing open issues: `tea issues list --repo TapTap/FastSync --state open --labels "<label>"`
|
||||
2. Search for similar titles using `tea issues list --repo TapTap/FastSync --keyword "<keywords>"`
|
||||
3. If a similar issue exists, add a comment instead of creating a duplicate:
|
||||
```bash
|
||||
gh issue comment <issue-number> --body "Additional finding from automated analysis: <details>"
|
||||
tea comment --repo TapTap/FastSync <issue-number> "Additional finding from automated analysis: <details>"
|
||||
# or POST to the Gitea API:
|
||||
# POST https://gitea.tap-tap.win/api/v1/repos/TapTap/FastSync/issues/<n>/comments
|
||||
```
|
||||
|
||||
## Sub-Agent Reference
|
||||
@@ -189,13 +191,12 @@ Before creating an issue:
|
||||
| Agent | File | Purpose |
|
||||
|---|---|---|
|
||||
| feature-scout | `.opencode/agents/feature-scout.md` | Scans for feature opportunities |
|
||||
| security-screener | `.opencode/agents/security-screener.md` | Scans for security vulnerabilities |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits and vulnerability scans |
|
||||
| code-quality-guardian | `.opencode/agents/code-quality-guardian.md` | Scans for code quality improvements |
|
||||
| architect | `.opencode/agents/architect.md` | Architecture reviews |
|
||||
| c-reviewer | `.opencode/agents/c-reviewer.md` | C code correctness reviews |
|
||||
| debugger | `.opencode/agents/debugger.md` | Bug diagnosis |
|
||||
| refactorer | `.opencode/agents/refactorer.md` | Code refactoring |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits |
|
||||
| test-writer | `.opencode/agents/test-writer.md` | Test development |
|
||||
| perf-analyst | `.opencode/agents/perf-analyst.md` | Performance analysis |
|
||||
| protocol-designer | `.opencode/agents/protocol-designer.md` | Protocol design |
|
||||
@@ -242,7 +243,7 @@ tests/test_file.c — File tests
|
||||
tests/test_transport_tcp.c — TCP transport tests
|
||||
tests/test_transport_tls.c — TLS transport tests
|
||||
tests/test_array_list.c — Array list tests
|
||||
tests/pytest/ — Python integration tests
|
||||
tests/integration/ — Python pytest integration tests
|
||||
```
|
||||
|
||||
### Build & Config Files
|
||||
@@ -259,7 +260,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -56,9 +56,11 @@ DirectoryScanner → Queue(Scanner→Loader) → ChunkBuilder → Queue(Loader
|
||||
|
||||
### Benchmark Context
|
||||
|
||||
From README benchmarks (25MB mixed files, localhost):
|
||||
- Best config: `-m -c` (multithread + compression) → 0.20s, 11.2× faster than rsync
|
||||
- `sendfile()` bypasses userspace → ~2× faster on localhost
|
||||
Use the maintained benchmark tool — do not cite stale README numbers:
|
||||
- `python3 benchmark/bench.py` runs the repeatable throughput benchmark.
|
||||
- The real flags are `-j` (multithreading) and `-z` (compression); a fast loopback
|
||||
config combines `-j -z`.
|
||||
- `sendfile()` (via `--sendfile`) bypasses userspace → ~2× faster on localhost
|
||||
- Compression reduces wire data enough that transfer becomes latency-bound on WAN
|
||||
|
||||
## Output Format
|
||||
@@ -120,6 +122,7 @@ time ./build/client [args...]
|
||||
|
||||
# High precision
|
||||
perf stat -e task-clock ./build/client [args...]
|
||||
```
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
@@ -127,9 +130,8 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
```
|
||||
@@ -91,7 +91,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -160,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -3,25 +3,64 @@ description: Audits FastSync for security vulnerabilities — TLS config, input
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are a security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
You are the security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport. This is the single canonical security agent.
|
||||
|
||||
## Your Role
|
||||
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices.
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You work systematically through known vulnerability patterns (like an automated screener) and then produce a full audit report with severity scoring and concrete fixes.
|
||||
|
||||
## Attack Surface
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
|
||||
### Network Input Points
|
||||
1. **TCP server** (`src/server/server.c`) — accepts connections from any client
|
||||
2. **SSH transport** (`src/shared/transport_ssh.c`) — receives data via stdio pipe
|
||||
3. **Protocol parsing** (`src/shared/protocol.c`) — deserializes all incoming data
|
||||
4. **Config deserialization** (`src/shared/config.c`) — receives remote config
|
||||
5. **Chunk deserialization** (`src/shared/chunk.c`) — receives file batches
|
||||
## Project Architecture
|
||||
|
||||
### TLS Configuration
|
||||
- OpenSSL TLS 1.2+ via `src/shared/transport_tls.c`
|
||||
- Certificate/key loading, CA verification
|
||||
- SSL context setup, cipher suite selection
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
receiver.c Receiver-side file handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_store.c/h Destination file store
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
xattr.c/h Extended attributes
|
||||
identity.c/h uid/gid mapping
|
||||
credentials.c/h Daemon credentials
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` / `receiver.c` | Writes received files to disk |
|
||||
|
||||
## Security Audit Checklist
|
||||
|
||||
@@ -33,51 +72,182 @@ Audit the codebase for security vulnerabilities. You focus on the attack surface
|
||||
- [ ] Chunk count and file count validated before allocation
|
||||
- [ ] Config field lengths bounded
|
||||
|
||||
### 2. Buffer Safety
|
||||
- [ ] No `strcpy` — use `snprintf` or `strncpy` with null termination
|
||||
- [ ] `malloc` size calculations don't overflow (e.g., `count * sizeof(...)`)
|
||||
- [ ] No fixed-size stack buffers for unbounded input
|
||||
- [ ] `receive_n_data` always checks return value
|
||||
- [ ] Off-by-one in path concatenation
|
||||
### 2. Buffer Overflow Risks
|
||||
|
||||
### 3. Memory Safety in Error Paths
|
||||
- [ ] All error paths free allocated resources
|
||||
- [ ] No use-after-free on error paths
|
||||
- [ ] No double-free on error paths
|
||||
- [ ] Partial reads handled (don't use incomplete data)
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
|
||||
### 4. TLS/SSL Security
|
||||
- [ ] TLS 1.2 minimum enforced (no SSLv3, TLS 1.0, TLS 1.1)
|
||||
- [ ] Certificate verification enabled when CA provided
|
||||
- [ ] Certificate verification disabled only with explicit warning
|
||||
- [ ] Private key file permissions checked
|
||||
- [ ] No hardcoded certificates or keys
|
||||
- [ ] Cipher suites restricted to strong algorithms
|
||||
- [ ] SSL error codes checked after `SSL_read`/`SSL_write`
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
|
||||
### 5. Authentication & Authorization
|
||||
### 3. Path Traversal in File Operations
|
||||
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 4. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 5. TLS / SSL Security
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--ca`/warning
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
- [ ] **No hardcoded certificates or keys**
|
||||
- [ ] **SSL error codes checked after `SSL_read`/`SSL_write`**
|
||||
|
||||
### 6. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
- [ ] **All error paths free allocated resources** (no leaks / UAF / double-free)
|
||||
- [ ] **Partial reads handled** (don't use incomplete data)
|
||||
|
||||
### 7. Integer Overflow in Allocation
|
||||
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 8. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 9. Authentication & Authorization
|
||||
- [ ] SSH transport relies on SSH authentication (not custom auth)
|
||||
- [ ] No password/credential storage in plaintext
|
||||
- [ ] Server doesn't trust client-supplied paths blindly
|
||||
- [ ] Destination directory validated before writing
|
||||
|
||||
### 6. Denial of Service
|
||||
- [ ] Bounded memory allocation (can't OOM server with huge chunk)
|
||||
- [ ] Timeout on connections (no indefinite blocking)
|
||||
- [ ] Maximum connection limit or rate limiting
|
||||
- [ ] Malformed protocol messages handled gracefully (no crash)
|
||||
### 10. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
|
||||
### 7. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
### 11. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
|
||||
### 8. File System Security
|
||||
### 12. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 13. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 14. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
### 15. File System Security
|
||||
- [ ] Received file permissions validated (no SUID/SGID injection)
|
||||
- [ ] Symlink attack prevention (don't follow symlinks in destination)
|
||||
- [ ] Race conditions in file creation (TOCTOU)
|
||||
- [ ] Temporary file security (if any)
|
||||
|
||||
### 16. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
|
||||
## Common Vulnerability Patterns
|
||||
|
||||
### Format String Bugs
|
||||
@@ -118,9 +288,59 @@ receive_n_data(fd, buffer, expected_size);
|
||||
if (!receive_n_data(fd, buffer, expected_size)) { /* handle error */ }
|
||||
```
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
For each vulnerability found:
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Detailed Finding Fields
|
||||
|
||||
For each vulnerability found, also be prepared to report:
|
||||
1. **Location** — file:line
|
||||
2. **Severity** — critical / high / medium / low / informational
|
||||
3. **Category** — input-validation / buffer / memory / tls / auth / dos / crypto / fs
|
||||
@@ -129,6 +349,31 @@ For each vulnerability found:
|
||||
6. **Fix** — concrete code change
|
||||
7. **CVSS estimate** — rough severity score if exploitable
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### Audit Summary
|
||||
|
||||
Also provide a summary:
|
||||
```
|
||||
=== SECURITY AUDIT SUMMARY ===
|
||||
@@ -140,13 +385,30 @@ Low: <count>
|
||||
Informational: <count>
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -1,310 +0,0 @@
|
||||
---
|
||||
description: Scans the FastSync codebase for security vulnerabilities — buffer overflows, path traversal, TLS issues, memory safety, and cryptographic hygiene.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are a security screener for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
|
||||
## Your Role
|
||||
|
||||
Scan the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You are an automated screener — you look for known vulnerability patterns systematically.
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
|
||||
## Project Architecture
|
||||
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
array_list.c/h Dynamic array
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` | Writes received files to disk |
|
||||
|
||||
## Security Screener Checklist
|
||||
|
||||
### 1. Buffer Overflow Risks
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
|
||||
### 2. Path Traversal in File Operations
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 3. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 4. TLS / SSL Misconfiguration
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--insecure` flag
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
|
||||
### 5. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
|
||||
### 6. Integer Overflow in Allocation
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 7. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 8. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
|
||||
### 9. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
|
||||
### 10. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 11. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 12. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
@@ -138,11 +138,9 @@ int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
|
||||
|
||||
Build for fuzzing:
|
||||
```bash
|
||||
cmake -B build-fuzz -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=fuzzer,address,undefined -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=fuzzer,address,undefined"
|
||||
CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
|
||||
cmake --build build-fuzz -j$(nproc)
|
||||
./build-fuzz/tests/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
./build-fuzz/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
```
|
||||
|
||||
### AFL++ Harness
|
||||
@@ -216,7 +214,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -32,23 +32,19 @@ Try to reproduce the issue with the exact command the user provides.
|
||||
|
||||
**Memory errors (first priority):**
|
||||
```bash
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
# or run the failing command
|
||||
```
|
||||
|
||||
**Thread errors:**
|
||||
```bash
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
|
||||
**Valgrind (if ASan doesn't find it):**
|
||||
@@ -108,13 +104,12 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
|
||||
# If integration test needed
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
|
||||
# Re-run under sanitizer to confirm fix
|
||||
rm -rf build
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
# reproduce the original failing command
|
||||
```
|
||||
|
||||
|
||||
@@ -19,7 +19,7 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log main..HEAD --oneline
|
||||
git log dev..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Clean build
|
||||
@@ -39,17 +39,13 @@ If the PR touches threading, memory management, or network code, also build with
|
||||
```bash
|
||||
# AddressSanitizer
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
|
||||
# ThreadSanitizer (if threading changes)
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
@@ -91,10 +87,10 @@ If tests fail:
|
||||
### Step 6: Run integration tests (optional)
|
||||
|
||||
```bash
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
This runs the integration + benchmark suite. It takes longer — only run if the user asks or if unit tests pass.
|
||||
This runs the integration suite (benchmarking is `benchmark/bench.py`). It takes longer — only run if the user asks or if unit tests pass.
|
||||
|
||||
### Step 7: Fix and commit
|
||||
|
||||
|
||||
@@ -19,13 +19,13 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log main..HEAD --oneline
|
||||
git log dev..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Get changed files
|
||||
|
||||
```bash
|
||||
git diff main --name-only -- '*.c' '*.h'
|
||||
git diff dev --name-only -- '*.c' '*.h'
|
||||
```
|
||||
|
||||
This gives the list of C source and header files changed in the PR.
|
||||
@@ -125,7 +125,7 @@ STYLE: <count>
|
||||
|
||||
If the user wants to post the review as a PR comment:
|
||||
```bash
|
||||
tea pr comment <number> --comment "<review report>"
|
||||
tea comment --repo TapTap/FastSync <number> "<review report>"
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -16,7 +16,7 @@ Ask the user or determine from context:
|
||||
- **Minor** (x.Y.0) — new features, backward compatible
|
||||
- **Patch** (x.y.Z) — bug fixes, no protocol changes
|
||||
|
||||
Current version: `PROTOCOL_VERSION "1.1.0"` in `src/shared/config.h`
|
||||
Current version: `PROTOCOL_VERSION "2.29.0"` in `src/shared/config.h`
|
||||
|
||||
### Step 2: Check Protocol Version
|
||||
|
||||
@@ -37,7 +37,7 @@ rm -rf build
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
ALL tests must pass before release.
|
||||
@@ -46,12 +46,10 @@ ALL tests must pass before release.
|
||||
|
||||
```bash
|
||||
# ASan
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
```
|
||||
|
||||
### Step 5: Update README (If Needed)
|
||||
@@ -79,12 +77,23 @@ git commit -m "Release vX.Y.Z
|
||||
git tag -a vX.Y.Z -m "Release vX.Y.Z"
|
||||
```
|
||||
|
||||
### Step 8: Push
|
||||
### Step 8: Push and Open dev → main PR
|
||||
|
||||
`main` is protected and only receives changes via `dev` → `main` PRs (see AGENTS.md). Never push directly to `main`.
|
||||
|
||||
```bash
|
||||
git push origin main --tags
|
||||
# Push the release commit and tag to dev
|
||||
git push origin dev
|
||||
git push origin vX.Y.Z
|
||||
|
||||
# Open the dev → main release PR for review + CI
|
||||
tea pr create --repo TapTap/FastSync --head dev --base main \
|
||||
--title "Release vX.Y.Z" \
|
||||
--description "Release vX.Y.Z"
|
||||
```
|
||||
|
||||
Then wait for the full CI to pass and request review before the PR is merged to `main`.
|
||||
|
||||
### Step 9: Report
|
||||
|
||||
```
|
||||
|
||||
@@ -102,9 +102,9 @@ Informational: <count>
|
||||
...
|
||||
|
||||
=== VERDICT ===
|
||||
[PASS] No critical/high issues found
|
||||
[PASS] No critical/high-severity issues found
|
||||
— or —
|
||||
[FAIL] <N> critical/high issues must be fixed
|
||||
[FAIL] <N> critical/high-severity issues must be fixed
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -4,18 +4,19 @@ FastSync is a high-performance file synchronization system written in C11. It su
|
||||
|
||||
## Dependency installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v10`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, and Node.js.
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v11`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, zlib1g-dev, liblz4-dev, libxxhash-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, Node.js, plus `rsync` 3.4.1 (with zstd/xxhash/lz4), `acl` and `attr` (setfacl/getfacl, setfattr/getfattr) for drop-in parity tests. (CMake hard-requires zstd, zlib, and lz4; xxHash is fetched via `FetchContent`.)
|
||||
|
||||
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
|
||||
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, zlib, lz4, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
|
||||
|
||||
```bash
|
||||
# Use the prebuilt CI image directly (faster, guaranteed CI parity)
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v10 fastsync-ci:local
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v11 fastsync-ci:local
|
||||
|
||||
# Or build the image from the repo-root Dockerfile
|
||||
# (Note: the prebuilt :v10 image reflects the previous Dockerfile state;
|
||||
# rebuild from source to pick up any newly added packages like lcov/valgrind.)
|
||||
# (Note: the prebuilt :v11 image is built from the current Dockerfile and
|
||||
# includes rsync 3.4.1 plus acl/attr; rebuild from source after changing
|
||||
# the Dockerfile.)
|
||||
docker build -t fastsync-ci:local .
|
||||
|
||||
# Build, run unit tests, and run integration tests inside the container
|
||||
@@ -28,7 +29,7 @@ docker run --rm --user "$(id -u):$(id -g)" -v "$PWD:/workspace" \
|
||||
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load'
|
||||
```
|
||||
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash from GitHub via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
|
||||
If a dependency is missing from the CI image, add it to the `Dockerfile` (and rebuild) rather than adding an install step to the CI workflow.
|
||||
|
||||
@@ -37,11 +38,12 @@ If a dependency is missing from the CI image, add it to the `Dockerfile` (and re
|
||||
When configuring for CI parity, use:
|
||||
```bash
|
||||
cmake -B build -S . -DSTRICT_WARNINGS=ON # -Wextra -Wpedantic -Werror
|
||||
cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan)
|
||||
cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan)
|
||||
cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan); in the CI matrix
|
||||
cmake -B build -S . -DSANITIZER=undefined # UndefinedBehaviorSanitizer (UBSan); in the CI matrix
|
||||
cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan); local-only, NOT in CI
|
||||
```
|
||||
|
||||
The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, sanitizer, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. The two `setpriv` privilege tests are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions.
|
||||
The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, the `address`+`undefined` sanitizer matrix, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. TSan is not part of the CI matrix and is a local-only configuration. The `setpriv`-marked privilege tests (four decorated functions, collecting to eight instances because two are parametrized) are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions.
|
||||
|
||||
## Build
|
||||
|
||||
@@ -55,32 +57,47 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
./build/tests # unit tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # full integration suite (CI excludes env-dependent privilege tests)
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset only
|
||||
|
||||
# Differential rsync-parity gate (real rsync 3.4.1 vs FastSync)
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci # fast PR subset
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity # full set
|
||||
```
|
||||
|
||||
See `tests/integration/README.md` for the differential parity gate and its
|
||||
`parity_caveats.py` allowlist (the residual burn-down mechanism).
|
||||
|
||||
Unit tests under valgrind must set `FASTSYNC_UNDER_VALGRIND=1` (CI does): the
|
||||
tests use it to skip fork-based tests, because valgrind 3.22 does not expose
|
||||
`vgpreload` in the guest's `/proc/self/maps`.
|
||||
|
||||
```bash
|
||||
FASTSYNC_UNDER_VALGRIND=1 valgrind --leak-check=full --show-leak-kinds=definite --error-exitcode=1 ./build/tests
|
||||
```
|
||||
|
||||
## CI Workflow — Waiting for Results
|
||||
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Use `gh run watch` or similar to monitor CI status, then inspect logs on failure.
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Monitor CI status via the Gitea API (see below) or `tea actions`, then inspect logs on failure.
|
||||
|
||||
## CI Troubleshooting
|
||||
|
||||
### If lint (clang-format) fails
|
||||
Run clang-format in the CI Docker image to match the exact CI version:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i'
|
||||
```
|
||||
|
||||
### If cppcheck fails
|
||||
Fix reported issues locally, then verify with:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/'
|
||||
```
|
||||
|
||||
### If integration tests fail
|
||||
Run locally before pushing:
|
||||
```bash
|
||||
python3 -m pytest tests/ -v --tb=short
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
## Branch Strategy
|
||||
@@ -89,7 +106,7 @@ Two main branches: `dev` (integration) and `main` (stable releases).
|
||||
|
||||
### Rules
|
||||
- **All PRs target `dev`** — never target `main` directly
|
||||
- **`dev` is the default branch** in Gitea repo settings
|
||||
- **`dev` is intended to be the default branch** in Gitea repo settings — verify in the repo settings, since this clone's `origin/HEAD` still points at `main`
|
||||
- **`main` is protected** — only merged from `dev` via PR with 2 approvals + full CI pass
|
||||
- **Feature/bug branches** branch from `dev`, PR back to `dev`
|
||||
- **`dev` → `main` merges** happen on-demand or weekly, requiring full CI + review
|
||||
@@ -165,7 +182,7 @@ This can be cron'd locally if desired (e.g., `crontab -e` with `opencode run`).
|
||||
## Is opencode a good option?
|
||||
|
||||
**Yes, for FastSync's needs.** The hybrid model works well:
|
||||
- opencode's 17 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- opencode's 16 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- The assistant orchestrates subagents, merges branches, and iterates on CI
|
||||
- You only review the final output
|
||||
|
||||
|
||||
+601
@@ -4,6 +4,607 @@ All notable changes to FastSync are documented here. Versions match
|
||||
`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must
|
||||
run the same version because the handshake is strict.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
Wire backlog cycle (protocol 2.29.0 → 2.30.0; config-frame layout unchanged).
|
||||
|
||||
- **`--stderr=client` client-message channel (#313):** the client now accepts
|
||||
`--stderr=client` (and maps the deprecated `--no-msgs2stderr` to it), routing
|
||||
its own diagnostics over the new bounded `STATUS_CLIENT_MSG` client->server
|
||||
frame instead of writing them locally; the server writes each received
|
||||
message to its stderr (respecting the server log destination). `errors`/`all`
|
||||
behavior is unchanged.
|
||||
- **Receiver partial failures exit 23 (#320):** a per-entry receiver failure
|
||||
that does not abort the stream (e.g. an unprivileged `--devices` mknod) now
|
||||
sends the terminal `STATUS_PARTIAL`; the client exits 23 like rsync and, under
|
||||
`--remove-source-files`, still removes the sources it successfully
|
||||
transferred. A clean run stays 0 and a fatal/connection error stays non-23.
|
||||
- **Directory/symlink destination-state itemize (#314):** when `report_dest_info`
|
||||
is negotiated (now also for `--progress`), the receiver answers `STATUS_MKDIR`
|
||||
and `STATUS_SYMLINK` with the entry's pre-transfer destination snapshot
|
||||
(existence, type, perms/owner/group/time, and whether an existing symlink's
|
||||
target already matches), and the sender probes every ancestor directory before
|
||||
the receiver creates it implicitly. A re-run over an unchanged tree no longer
|
||||
emits per-directory `cd+++++++++` or unchanged-symlink lines, a changed
|
||||
directory renders rsync's `.d..t......`, and a changed symlink renders
|
||||
`cLc........` / `.L..t......`. Directory/symlink time comparison uses whole
|
||||
seconds (rsync's `cmp_time`). The `STATUS_MKDIR` body gains a probe flag and
|
||||
the `STATUS_DEST_INFO` record gains a symlink-target-match field; the
|
||||
config-frame layout is unchanged. Differential-tested against rsync 3.4.1.
|
||||
- **`--stats` deleted per-type breakdown (#316):** `STATUS_STATS` gains
|
||||
`deleted_reg/dir/link/special`, tallied by the delete observers and rendered
|
||||
as rsync's `Number of deleted files: X (reg: A, dir: B, link: C, special: D)`.
|
||||
Differential-tested against rsync 3.4.1 for a mixed-type `--delete` tree.
|
||||
|
||||
## [2.29.0] - 2026-09-23
|
||||
|
||||
The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0).
|
||||
`RSYNC_COMPAT.md` moves from **116 ✅ / 14 ⚠️ / 27 ❌** to
|
||||
**120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows.
|
||||
|
||||
An audit cycle follows on the same wire version (`PROTOCOL_VERSION` stays
|
||||
2.28.0): a security-and-correctness pass over the parity-2.29 baseline, plus a
|
||||
set of audit follow-ups (filter merge modifiers, the `--inplace`/`--partial-dir`
|
||||
conflict, credential-file hardening, and small leak/log/test fixes). It fixes
|
||||
a `--temp-dir` symlink escape, gates client-controlled special permission bits,
|
||||
corrects `--partial-dir`/`--bwlimit`/`-z` behavior, handles unsupported filter
|
||||
modifiers, and tightens client and wire validation. The only parity
|
||||
reclassification is `--filter=RULE` moving ✅ → ⚠️, because its merge-only
|
||||
`e`/`n`/`w`/`-` modifiers are now accepted and consumed but their semantics
|
||||
remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ /
|
||||
11 ⚠️ / 27 ❌** of 157 rows. The affected rows' notes and the summary tally in
|
||||
`RSYNC_COMPAT.md` were updated. A following triage-fix cycle (see **Triage
|
||||
fixes** below) moves `-F` and `-i` to ⚠️, for a final **117 ✅ / 13 ⚠️ / 27 ❌**
|
||||
of 157 rows.
|
||||
|
||||
A no-wire parity burn-down cycle follows on 2.28.0: it accepts
|
||||
`--inc-recursive`/`--no-inc-recursive` as inert no-ops, accepts an absolute
|
||||
`--temp-dir` that canonicalizes inside the receive root, closes the
|
||||
`--delete-before` phase-0 divergence (both the single-threaded and `--threads`
|
||||
data passes replay the pre-scan list), makes `--fake-super` interoperable with
|
||||
rsync's `user.rsync.%stat` key/grammar (regular files and char/block devices
|
||||
faked as regular files), turns a failed device `mknod` into a continuing
|
||||
per-entry failure, and accepts a practical subset of rsync's `rsyncd.conf`
|
||||
grammar (modules are read-only by default, and accepted-but-unenforced
|
||||
access-control keys emit a startup warning). The matrix moves to **119 ✅ /
|
||||
14 ⚠️ / 24 ❌** of 157 rows.
|
||||
|
||||
A structural cycle then lands a transport I/O vtable over TCP/TLS (fixing the
|
||||
TLS-multithreaded sendfile path and making the per-thread SSL resolution
|
||||
explicit) and bumps the wire to **2.29.0**: the `STATUS_SYMLINK` frame grows an
|
||||
optional symlink-xattr block (captured no-follow with `llistxattr`/`lgetxattr`,
|
||||
applied no-follow with `lsetxattr`). Because the handshake is strict, 2.28.0 and
|
||||
2.29.0 peers are incompatible. Note: Linux refuses to associate xattrs with a
|
||||
symlink at all, so the symlink-xattr block is a no-op on Linux and is carried
|
||||
for correctness on platforms/filesystems that do support it; the config-frame
|
||||
layout is unchanged (golden length still 886).
|
||||
|
||||
### Changed
|
||||
|
||||
- **rsync-exact traversal order.** The sequential scanner now walks each
|
||||
directory's entries in rsync 3.4.1's flist order (non-directories ascending,
|
||||
then directories ascending, depth-first), so `--info=name`, the
|
||||
`--delete-during`/`--delete-delay`/`-n` would-delete order and the partial
|
||||
`--max-delete` survivor set match rsync byte-for-byte. `--threads` has no
|
||||
rsync analogue and stays unordered.
|
||||
- **Delete timing.** The complete `--delete-during`/`--delete-delay`
|
||||
per-directory plan set is transmitted before the first data frame, so a
|
||||
mid-transfer abort has already removed every planned extra like rsync's
|
||||
generator; `-d/--dirs` uses per-directory plans (shielded untraversed
|
||||
subdirectories) instead of the end-of-transfer commit. `-n`, `--delete`,
|
||||
`--del`/`--delete-during` and `--delete-delay` are now ✅ Parity.
|
||||
- **Basis directories.** A relative `--compare-dest`/`--copy-dest`/`--link-dest`
|
||||
DIR resolves against the destination directory with the transfer-relative
|
||||
name appended, exactly like rsync 3.4.1.
|
||||
- **`-y`/`--fuzzy`.** The candidate search no longer inherits the ordinary delta
|
||||
engine's 16 KiB minimum or 10× size-ratio bound, so an oversized or
|
||||
sub-16-KiB sibling is reused exactly as rsync reuses it.
|
||||
- `--info=mount` prints rsync's mount-point skip line (repeated `-xx` drops the
|
||||
mount-point directory); `--info=stats` enables the `--stats` block; `-x` is
|
||||
repeatable. `--stats` counts traversed directories for the `Number of files`
|
||||
breakdown under a plain `-r` scan. `--debug` emits real output for
|
||||
`flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send`.
|
||||
|
||||
### Known residuals
|
||||
|
||||
- `--progress` and `--info` still need a receiver→sender event channel for the
|
||||
root `./` line, ancestor-directory suppression, receiver-side `skip`/`backup`
|
||||
wording, and symlink/empty-directory quick-checks.
|
||||
- `--delete-before`'s phase-0 late-file divergence remains (rsync's pre-scan
|
||||
fixes the file list before the data pass).
|
||||
- A whole-file sender that cannot stream its codec (lz4's one-shot block
|
||||
format) or a `--append`/delta source above the bound still buffers; the
|
||||
default zstd/zlib and the uncompressed paths stream (see #318).
|
||||
- `--stats` byte totals and `--msgs2stderr` stay documented divergences.
|
||||
|
||||
### Security
|
||||
|
||||
- **`--temp-dir` symlink escape fixed.** The receiver's scratch directory was
|
||||
opened with a bare `open()`, so a symlink planted under the receive root could
|
||||
redirect receiver scratch files outside the authorized root. The opened
|
||||
directory is now judged by the real path of its fd (`/proc/self/fd` via
|
||||
`realpath`) and an escaping target is refused (`EACCES`, logged); an in-root
|
||||
link to another filesystem (the `EXDEV` fallback case) still works.
|
||||
- **Client-controlled special bits masked when super-user activities are not
|
||||
permitted.** Setuid/setgid/sticky bits (`--perms`, `--chmod`, the symlink and
|
||||
special-node paths, and deferred directory modes) are now stripped when the
|
||||
connection forbids super activities (`--no-super`, a non-opted daemon module,
|
||||
a privileged listener without `--allow-super`); exact rsync semantics are
|
||||
preserved wherever super activities are permitted.
|
||||
- **Daemon umask no longer forced to `0`.** `daemonize()` now sets the
|
||||
conventional `022`, so implied parent directories created without `-p` are no
|
||||
longer world-writable `0777`.
|
||||
- **Daemon modules are read-only by default.** A `--daemon` module is now
|
||||
served read-only unless it sets `read only = no` (or rsync's `write only =
|
||||
yes`), matching rsync: a real `rsyncd.conf` that omits `read only` is no
|
||||
longer silently writable. A global `read only` still sets the default for
|
||||
later modules, and an explicit module value wins. This is a behavior change
|
||||
for existing FastSync-native configs that relied on the old writable default;
|
||||
add `read only = no` to keep them writable. An rsync `write only = yes` is
|
||||
mapped to writability (FastSync is push-only, so a module can never be read
|
||||
from the network).
|
||||
- **Accepted-but-unenforced rsync security keys now warn at startup.** The
|
||||
rsync keys FastSync recognizes but does not implement — `secrets file`,
|
||||
`refuse options`, `exclude`/`include`/`filter`, `max size`/`min size`,
|
||||
`pre-xfer exec`/`post-xfer exec`, `incoming chmod`/`outgoing chmod`,
|
||||
`name converter`, `use chroot`, `uid`/`gid`, and the rest of the
|
||||
access-control set — load for migration compatibility but now emit a
|
||||
`WARN` naming the key (and module) so an operator does not believe the
|
||||
restriction is enforced. `auth users`/`secrets file` stay fail-closed: a
|
||||
module declaring `auth users` still requires a FastSync credential store.
|
||||
- **Credentials and signal handling hardened.** Secret files are opened with
|
||||
`O_NOFOLLOW|O_NONBLOCK` (while allowing fd-backed store paths and bound-waiting
|
||||
a FIFO read for ~3 s so a slow process substitution works but a connected-but-
|
||||
silent FIFO cannot hang), and signal handlers use `sigaction` with
|
||||
async-signal-safe bodies.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **`-z` on 100–256 MiB files.** The decompressor's internal ceiling was 100 MiB
|
||||
while the receiver advertises and the sender compresses whole files up to
|
||||
`MAX_RECEIVE_WHOLE_FILE_SIZE` (256 MiB), so `-z` on a 100–256 MiB regular file
|
||||
failed with `Declared decompressed size exceeds 104857600 bytes`. The ceiling
|
||||
is now defined in terms of the protocol whole-file bound (still an
|
||||
allocation-clamped bomb guard).
|
||||
- **`--bwlimit` now paces `--sendfile`.** The plaintext-TCP `--sendfile` fast
|
||||
path bypassed the protocol's token bucket, so the limit was ignored there. It
|
||||
now throttles through the same per-session leaky bucket as the TLS path.
|
||||
- **`--partial-dir` implies `--partial`.** Matching rsync 3.4.1 (which sets
|
||||
`keep_partial` after option parsing), `--partial-dir=DIR` alone retains an
|
||||
interrupted transfer's partial and wins over an explicit `--no-partial`;
|
||||
`--inplace` still bypasses the partial machinery, and combining `--inplace`
|
||||
with `--partial-dir` is now rejected up front with rsync's message
|
||||
(`--inplace cannot be used with --partial-dir`).
|
||||
- **Filter modifiers handled.** The `x` xattr-name modifier is rejected with a
|
||||
clear error everywhere. The merge-only `e`/`n`/`w` and `-` modifiers are now
|
||||
accepted and consumed on `merge`/`dir-merge` rules (so they no longer leak
|
||||
into the merge filename) while still being rejected on non-merge rules,
|
||||
matching rsync; their semantics remain unimplemented (accepted-but-ignored).
|
||||
Glued patterns (`-newfile`, `-e2e`) and mixed tokens (`H,!secret`) keep their
|
||||
historical parsing.
|
||||
- **Credential-file reads hardened.** Secret files (`--password-file`/
|
||||
`--early-input`/`--hash-credentials` input) are opened with `O_NOFOLLOW`, so a
|
||||
symlinked credential path now fails closed (`ELOOP`) instead of being followed
|
||||
before the owner/mode gate; literal fd-backed paths (`/dev/fd/<digits>`,
|
||||
`/proc/self/fd/<digits>`) are exempt so process substitution still works. A
|
||||
FIFO/process-substitution read now waits under a bounded ~3 s deadline for its
|
||||
writer, so a slow producer works while a connected-but-silent FIFO fails
|
||||
instead of hanging.
|
||||
- **Miscellaneous correctness fixes:** `--filter` rule count is checked
|
||||
client-side against `MAX_FILTER_RULES` before any network I/O (the receiver
|
||||
still re-checks the expanded count); unknown wire `Status` values are rejected
|
||||
as protocol errors; a mutex leak on an init-failure path, an `errno` read
|
||||
after `free()` in deferred delete application, `log_perror` misuse for
|
||||
non-`errno` conditions, and a `NULL` `server_host`/`ssh_destination`
|
||||
allocation path were fixed (the `config_create` failure now releases through
|
||||
`config_delete`); the decompression-limit log now prints the effective bound
|
||||
rather than the compile-time ceiling; the daemon umask and root test fixtures
|
||||
were hardened; `SSL_read` length is clamped and `sendfile` `poll()` retries on
|
||||
`EINTR`.
|
||||
|
||||
### Refactored / Docs
|
||||
|
||||
- Dropped dead `filter_rules_apply` and dead `--old-args` plumbing, unified
|
||||
`set_error`, deduplicated `path_is_within` and shared constants, and added
|
||||
printf format attributes (fixing format mismatches). `RSYNC_COMPAT.md`,
|
||||
`CHANGELOG.md` and `HANDOFF.md` were updated for the audit cycle; the
|
||||
`RSYNC_COMPAT.md` summary tally was corrected to match the rows.
|
||||
|
||||
### Triage fixes
|
||||
|
||||
- **`--dirs` directory xattrs applied inline.** A `-d/--dirs` transfer now
|
||||
applies captured directory `-X`/`-A` xattrs fd-relative on the directory entry
|
||||
instead of dropping them, so directory xattrs survive the non-recursive path
|
||||
(`src/shared/file_save.c`, `tests/test_xattr.c`).
|
||||
- **Directory/root itemize and `--out-format` lines.** `-i`/`--itemize-changes`
|
||||
and `--out-format` now emit the transfer-root `./` line and per-directory
|
||||
`cd...`/`.d..t...` lines, rendered by the shared itemize code. This matches
|
||||
rsync's fresh-transfer output; because the root line is unconditional and an
|
||||
incremental re-run may itemize directories/symlinks that rsync's quick-check
|
||||
leaves silent, `-i` is now a ⚠️ Caveat row.
|
||||
- **FROM name globs for identity maps.** `--usermap`/`--groupmap` `FROM` tokens
|
||||
now accept `*`/`?`/`[...]` globs, expanded sender-side against the passwd/group
|
||||
database and collapsed into bounded numeric ranges (`MAX_IDENTITY_MAP`),
|
||||
matching rsync.
|
||||
- **Transport fallback unit tests.** Added unit coverage for the TCP/TLS
|
||||
transport fallback paths (`tests/test_transport_tcp.c`,
|
||||
`tests/test_transport_tls.c`).
|
||||
- **Docs corrections.** `RSYNC_COMPAT.md`/`README.md` corrected stale parity
|
||||
claims for issues #286–#297: the `-F` and `-i` reclassifications, the
|
||||
`--munge-links` direction, the accepted checksum/compression name sets,
|
||||
`--bwlimit` parsing, `--stop-at` grammar, `--trust-sender`, symlink xattrs, and
|
||||
the native/non-interoperable batch and credential notes. The summary tally is
|
||||
now **117 ✅ / 13 ⚠️ / 27 ❌** of 157 rows.
|
||||
|
||||
## [2.28.0] - 2026-09-20
|
||||
|
||||
The rsync-parity cycle. `PROTOCOL_VERSION` moves `2.26.0 → 2.27.0 → 2.28.0`;
|
||||
client and server must run the same version (the handshake is strict). See
|
||||
`RSYNC_COMPAT.md` for the per-option matrix, now **116 ✅ / 14 ⚠️ / 27 ❌** of
|
||||
157 rows.
|
||||
|
||||
### Added
|
||||
|
||||
- **Differential rsync 3.4.1 parity gate** (`tests/integration/
|
||||
test_differential_parity.py`, `parity_harness.py`, `parity_caveats.py`): runs
|
||||
real `rsync` and FastSync over generated corpora and diffs the destination
|
||||
tree, normalized stdout and exit code. A fast subset runs on pull requests and
|
||||
the full strict set on push; the residual allowlist is empty.
|
||||
- FastSync-only long option **`--verify-basis`**: require a
|
||||
`--compare-dest`/`--copy-dest`/`--link-dest` hit to match the source by
|
||||
whole-file digest instead of trusting the size+mtime quick-check.
|
||||
- FastSync-only long option **`--delete-commit`** (implies `--delete`): the old
|
||||
atomic late whole-tree commit.
|
||||
- `--bwlimit` now parses rsync's units exactly and paces like rsync's leaky
|
||||
bucket; `--ignore-errors` reproduces rsync's skip-unreadable-subdir and
|
||||
IO-error-suppressed deletion (exit 23).
|
||||
- `--info=name/flist/del/remove/nonreg/progress` emit rsync's line format,
|
||||
including real-run `deleting`/`*deleting` lines carried by a new
|
||||
`report_deletes` wire bool.
|
||||
- Receiver-observed `--stats` counters: `Number of created files` now carries
|
||||
rsync's `(reg/dir/link/special)` breakdown and `Literal data` is exact for a
|
||||
delta transfer (extended `STATUS_STATS`).
|
||||
- `--progress` uses an opt-in paths-only pre-count so the `to-chk` denominator
|
||||
counts every entry like rsync, and emits per-directory/symlink/special names.
|
||||
- Receiver-side `protect`/`risk` filter engine (new bounded filter-rule wire
|
||||
block): `--filter='P ...'` now shields a destination-only entry like rsync.
|
||||
- `auto` for `--compress-choice`/`--checksum-choice` honors
|
||||
`RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST`, and per-codec compression-level
|
||||
defaults match rsync.
|
||||
- Empty source directories are recreated recursively; `-R --no-implied-dirs
|
||||
--files-from` places listed files under missing implied parents; `--iconv`
|
||||
matches rsync's push direction; `--delete-delay` reports actual removals and
|
||||
recursively removes a refilled deferred directory.
|
||||
|
||||
### Changed
|
||||
|
||||
- **`--delete` now defaults to delete-during (rsync `--del`) timing.** With no
|
||||
explicit timing flag, a plain `--delete` removes each directory's extras as
|
||||
that directory is processed instead of committing one whole-tree deletion only
|
||||
after the entire transfer succeeds. This matches rsync, frees destination
|
||||
space progressively, and avoids the whole-old+new-tree peak that could
|
||||
`ENOSPC` a tight destination. The client maps the default onto the existing
|
||||
`delete_during` wire boolean, so `PROTOCOL_VERSION` stays `2.28.0`.
|
||||
- Basis directories (`--compare-dest`/`--copy-dest`/`--link-dest`) now default
|
||||
to rsync's metadata quick-check (equal size and mtime; `--size-only` drops the
|
||||
mtime leg) instead of FastSync's historical always-verify content hash.
|
||||
`--copy-dest` re-applies the source attributes, and basis materialization is
|
||||
streamed so the 256 MiB whole-file cap no longer applies to a basis hit.
|
||||
- The per-directory `STATUS_DELETE_PLAN` frame gained a one-int `apply` flag:
|
||||
the one-shot per-run config block (protected prefixes, size-pruned mirrors,
|
||||
`--delete-missing-args` exact paths) is now always transmitted first on a
|
||||
config-only carrier (`apply=false`), fixing a latent bug where a
|
||||
`--delete-missing-args` run whose `--files-from` list synchronized no directory
|
||||
never sent its exact deletions.
|
||||
|
||||
### Notes
|
||||
|
||||
- `--delete`/`--delete-during` remain caveats for the mid-transfer abort
|
||||
boundary (rsync's generator removes all planned extras ahead of its throttled
|
||||
sender; FastSync removes only reached directories — final trees agree).
|
||||
`--delete-before`, `--progress`, `--stats`, `--fuzzy` and the basis rows keep
|
||||
their documented residuals in `RSYNC_COMPAT.md`; `--filter` and
|
||||
`--delete-excluded` are now parity, including protection of a destination-only
|
||||
excluded entry under default `--delete`.
|
||||
|
||||
### Migration
|
||||
|
||||
- Scripts that relied on plain `--delete` deleting nothing until the transfer
|
||||
fully succeeded must pass **`--delete-commit`** (or `--delete-after`) to keep
|
||||
that behavior. Plain `--delete` now removes reached directories' extras during
|
||||
the transfer, exactly like rsync's default; on a completed run the final tree
|
||||
is unchanged.
|
||||
- Deployments that relied on FastSync's stricter basis verification should pass
|
||||
**`--verify-basis`**; the default now trusts the size+mtime quick-check like
|
||||
rsync.
|
||||
|
||||
## [2.26.0] - 2026-09-17
|
||||
|
||||
|
||||
### Added
|
||||
|
||||
- **Parity-completion wave.** Closed the remaining rsync-parity gaps against
|
||||
rsync 3.4.1 and reclassified the inherently non-rsync rows. It moved the wire
|
||||
protocol three times (`2.23.0 → 2.24.0 → 2.25.0 → 2.26.0`).
|
||||
- **Delete timing (2.24.0):** per-directory delete plans
|
||||
(`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`. An interrupted
|
||||
during-transfer has already removed the reached directories' extras, while a
|
||||
delayed transfer commits per directory only after the whole transfer
|
||||
succeeds (a late-created extra survives `--delete-delay` but not
|
||||
`--delete-after`). `-R --delete` is scoped to the transferred prefix; empty
|
||||
in-scope source directories survive; dry-run never deletes.
|
||||
- **Wire stats (2.25.0):** `STATUS_STATS` carries the receiver counters
|
||||
(matched data, deleted files) and the dry-run would-delete list. `--stats`
|
||||
prints rsync's protocol-independent lines; `--progress`/`-P` print per-file
|
||||
blocks; `--out-format` gains `%b` (wire bytes), `%c` (block-sum bytes) and
|
||||
`%C` (whole-file digest); `-n --delete` prints escaped `*deleting` lines in
|
||||
the sequential and `--threads` paths.
|
||||
- **Codecs (2.26.0):** `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/
|
||||
`none` checksums, with rsync-style `auto` negotiation (default `xxh128` +
|
||||
`zstd`) and exit-4 rejection of unknown names; the resolved `compression_algo`
|
||||
crosses the wire.
|
||||
- General `-R`/`--relative` (including the `/./` cut) and `--no-implied-dirs`;
|
||||
one-level `-d`/`--dirs` listing for `dir`, `dir/` and `.`; the full filter
|
||||
grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` and
|
||||
modifiers) with `-f` bound to `--filter`; a single `-F` transfers
|
||||
`.rsync-filter` and `-FF` excludes it.
|
||||
- Receiver-side `--chown`/`--usermap`/`--groupmap` TO-name resolution; absolute
|
||||
basis directories and a `--link-dest` relink of an up-to-date destination;
|
||||
a receiver-side `--ignore-existing` short-circuit before any payload;
|
||||
`--preallocate` now wins over `--sparse` via `fallocate(2)`.
|
||||
- Client quick wins: `--iconv=.`/`-`/`--no-iconv`, a lone `-h` prints help, an
|
||||
empty `--files-from` succeeds (exit 0), a broken referent under
|
||||
`-L`/`--copy-unsafe-links` exits 23, the full `--info`/`--debug`
|
||||
vocabularies, and the aliases `--ignore-non-existing`, `--protect-args`,
|
||||
`--msgs2stderr`.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.23.0 → 2.24.0` (delete plans),
|
||||
`2.24.0 → 2.25.0` (`STATUS_STATS` + `report_stats`), and
|
||||
`2.25.0 → 2.26.0` (codec negotiation + `md4`/`sha1`/`none`).
|
||||
- `--checksum-choice`/`--cc` now accepts `md4`, `sha1`, `none` and the two-name
|
||||
form; the negotiated whole-file default is `xxh128`.
|
||||
- `--compress-choice`/`--zc` now accepts `lz4`, `zlib`, `zlibx`.
|
||||
- `RSYNC_COMPAT.md` reclassifies the matrix: 9 already-parity rows to ✅, 17
|
||||
inherently non-rsync rows to ❌ (native daemon config/auth, batch, privileged
|
||||
xattr namespaces, and the safe-subset device/privilege flags), and the genuine
|
||||
fixes to ✅; new rows cover `--bwlimit`, `--partial`, `--partial-dir`,
|
||||
`--no-whole-file`, `--inc-recursive`/`--no-inc-recursive`, `--protect-args`
|
||||
and `--msgs2stderr`.
|
||||
- The client `--help` `--max-delete` text now describes the implemented partial
|
||||
semantics (delete up to N, skip the rest, exit 25).
|
||||
|
||||
### Notes
|
||||
|
||||
- Remaining documented divergences include the `--stats` per-type file-count
|
||||
breakdown, `%b`/`%c` being FastSync wire counts, `-n --delete` line ordering,
|
||||
the default `--delete` timing (delete-after, not rsync's delete-during),
|
||||
destination-only exclude protection (still sender-derived), `--temp-dir`
|
||||
absolute paths, basis-dir attribute re-application and the 256 MiB whole-file
|
||||
cap, `--fuzzy` tie-breaking, `--bwlimit=0`/decimal rates, `zlibx`==`zlib`, and
|
||||
recursive empty-directory creation.
|
||||
- Build: adds zlib and lz4 as link dependencies.
|
||||
|
||||
## [2.23.0] - 2026-09-16
|
||||
|
||||
### Added
|
||||
|
||||
- **Rsync-parity wave.** Closed the remaining CLI, filesystem, ownership,
|
||||
deletion, and output gaps against rsync 3.4.1.
|
||||
- Short options `-r` (`--recursive`), `-b` (`--backup`), `-L`
|
||||
(`--copy-links`), and `-B` (`--block-size`/`--delta-block`); rsync
|
||||
short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached/inline values
|
||||
(`--opt=value`, `-B1000`, `-essh`, `-MOPT`). A value that starts with `-`
|
||||
is not mistaken for a cluster.
|
||||
- `-c`/`--checksum` now implies the incremental checksum quick-check (and,
|
||||
like rsync, does not imply `-t`).
|
||||
- `--checksum-choice`/`--cc` accepts `xxh64`/`xxhash`/`xxh3`/`xxh128`/`md5`/
|
||||
`auto` and rejects `md4`/`sha1`/`none` and the two-name form by name;
|
||||
`--checksum-seed=0` (the default) is randomized per transfer and the chosen
|
||||
seed is sent to the receiver.
|
||||
- `--compress-choice`/`--zc` accepts `zstd`/`none`/`auto` and rejects
|
||||
`lz4`/`zlib`/`zlibx` by name; `--skip-compress` defaults to rsync 3.4.1's
|
||||
built-in suffix list; `--no-whole-file` is accepted.
|
||||
- `--timeout` defaults to 0 (disabled) and `--contimeout` to 60 s (both `0`
|
||||
disables), matching rsync; `--max-alloc=0` means no local limit.
|
||||
- `--temp-dir` is confined to the receive root (absolute/`..` rejected by the
|
||||
receiver) and an `EXDEV` install falls back to a non-atomic copy.
|
||||
- `--numeric-ids` is documented as a mapping modifier only;
|
||||
`--usermap`/`--groupmap` support inclusive `LOW-HIGH` ranges, `*`,
|
||||
empty-`FROM` (unnamed ids), and receiver-resolved `TO` names; `--chown`
|
||||
conflicts with a map on the same side are rejected.
|
||||
- `--fake-super` records the *resolved* owner (never a real chown) and replays
|
||||
mode/time; directory ownership and directory xattrs/ACLs are preserved.
|
||||
- `-l`/`--links` stores symlink targets verbatim (absolute and `..`-bearing
|
||||
included), matching rsync; `--safe-links`/`--copy-unsafe-links` are applied
|
||||
sender-side and `--munge-links` uses rsync's `/rsyncd-munged/` marker;
|
||||
`--trust-sender` no longer affects symlink targets.
|
||||
- `--specials` recreates unix sockets with `mknod(S_IFSOCK)` (so `-D` covers
|
||||
the full rsync node set).
|
||||
- Deletion: the manifest carries a synchronized-directory section so
|
||||
`--files-from` subsets no longer delete untransmitted paths;
|
||||
`--delete-excluded` leaves size-pruned mirrors protected; extraneous
|
||||
destination symlinks are unlinked (never followed); `--max-delete=N` is
|
||||
partial (delete up to N, skip the rest, exit 25) and `--delete-missing-args`
|
||||
removals draw from the same budget; `--force` is honored during
|
||||
`--delay-updates` publication.
|
||||
- `-x`/`--one-file-system` emits the mount-point directory entry; the
|
||||
`--include`/`--exclude` layers are an ordered first-match rule list.
|
||||
- `--chmod` is a faithful port of rsync 3.4.1 (numeric/symbolic, `D`/`F`/`X`,
|
||||
`s`/`t`, append semantics, no `-p` implication, no sanitization).
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.22.0 → 2.23.0`: the delete manifest gains a
|
||||
synchronized-directory section and the terminal status gains
|
||||
`STATUS_DELETE_LIMIT` (client exit 25 on a `--max-delete`-capped commit).
|
||||
- **The 2.22.0 mode-masking divergence is removed.** Under `-p` the source mode
|
||||
is copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky;
|
||||
`--chmod` no longer implies `-p`. New files without `-p` still use
|
||||
`source_mode & ~umask` when metadata is present (else `0644`), and new
|
||||
directories without `-p` still use the `0755` creation default.
|
||||
- `--protocol=NUM` accepts only the current `2.23.0` version string.
|
||||
|
||||
### Notes
|
||||
|
||||
- The rsync-compatibility matrix (`RSYNC_COMPAT.md`) now classifies every row
|
||||
as **parity**, **caveat** (works with a documented divergence), or
|
||||
**divergent** (not supported/no-op/impossible), replacing the previous
|
||||
misleading "N implemented / 0 divergence" summary. Durable documented
|
||||
divergences remain: receiver-side symlink target containment is not enforced
|
||||
by default (verbatim storage is rsync parity; use `--safe-links`),
|
||||
`--temp-dir` rejects absolute/foreign-filesystem paths, `--copy-devices`
|
||||
reads a bounded `st_size`, a broken referent under `--copy-links` exits 0,
|
||||
new directories without `-p` use `0755`, `--stats` receiver-only counters are
|
||||
0, and `--password-file`/`--early-input`/`--hash-credentials`/`--iterations`
|
||||
and the batch format are FastSync-native.
|
||||
|
||||
## [2.22.0] - 2026-09-15
|
||||
|
||||
### Added
|
||||
|
||||
- **Per-attribute metadata preservation (protocol 2.22.0).** The former single
|
||||
metadata bundle is split into four independent, rsync-compatible flags:
|
||||
`-p/--perms`, `-t/--times`, `-o/--owner`, and `-g/--group`, each applied
|
||||
independently on the receiver, with negations `--no-perms`/`--no-times`/
|
||||
`--no-owner`/`--no-group` (short `--no-p`/`--no-t`/`--no-o`/`--no-g`) and
|
||||
`--no-preserve` clearing all four. `-a/--archive` is now full rsync
|
||||
`-rlptgoD` (owner and group included; their application stays
|
||||
privilege-gated). `-A/--acls` and `--chmod` imply `-p`, `-X/--xattrs` does
|
||||
not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply
|
||||
`-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the
|
||||
user explicitly negated them.
|
||||
- Receiver applies directory modes under `-p` (at the end of the transfer,
|
||||
alongside the deferred directory times) and symlink mode under `-p`; `-O`
|
||||
suppresses directory times only.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.21.0 → 2.22.0`: the binary config frame gains
|
||||
four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/
|
||||
`preserve_group`) after `omit_link_times`. The fixed-width `FileMetadata`
|
||||
layout is unchanged; the receiver derives the metadata-frame gate
|
||||
(`use_metadata`) from the four attributes.
|
||||
|
||||
### Notes
|
||||
|
||||
- Documented divergences from rsync: a client-supplied mode never grants
|
||||
group/other write (`S_IWGRP|S_IWOTH` are stripped for files, directories,
|
||||
symlinks, and specials; rsync's `-p` preserves them exactly); a brand-new file
|
||||
without `-p` gets `source_mode & ~umask` (sanitized) when metadata is present,
|
||||
else the historical fixed `0644`; `--chmod` implies `-p` (rsync does not);
|
||||
`-o`/`-g` map by name on the receiver with a raw-numeric fallback (only
|
||||
numeric ids cross the wire); and a daemon module without `client owner = yes`
|
||||
does not refuse a plain `-a`/`-o`/`-g` but forces super-user activities off,
|
||||
applies no ownership, and logs a warning (explicit `--chown`/`--usermap`/
|
||||
`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused).
|
||||
|
||||
## [2.21.0] - 2026-09-14
|
||||
|
||||
### Added
|
||||
|
||||
- Optional server→client rejection detail (protocol 2.21.0). A rejected
|
||||
operation may now carry a bounded human-readable reason via
|
||||
`STATUS_ERROR_DETAIL` instead of a bare `STATUS_ERROR`, so the client can
|
||||
report *why* the server refused (daemon module gate, config validation,
|
||||
receiver-side path/node validation). `receive_status()` transparently maps the
|
||||
new status back to `STATUS_ERROR` for every existing call site and captures
|
||||
the reason into a thread-local buffer exposed by `protocol_last_error()`. The
|
||||
detail body is always consumed, so the stream cannot desynchronize, and
|
||||
messages are sliced to `MAX_ERROR_DETAIL_BYTES` (4096) on send.
|
||||
- **Server-contacting `--dry-run` (protocol 2.21.0).** `--dry-run` now performs
|
||||
a real handshake with a remote/daemon receiver and reports exactly what WOULD
|
||||
change based on receiver state (existing destination files, mtimes, checksums,
|
||||
basis dirs). The wire config carries the dry-run intent (`Config.dry_run`) and
|
||||
the receiver answers each per-file check with `STATUS_DRY_RUN_TRANSFER` (would
|
||||
transfer) or `STATUS_OK` (already up to date); the sender prints the
|
||||
would-transfer set and its trailer without sending any file data. The receiver
|
||||
performs the normal read-only incremental decision but mutates nothing: no temp
|
||||
files, writes, renames, deletes, metadata/xattr/chown, or directory creation.
|
||||
A plain local destination (no explicit `--server-port`/remote) keeps the
|
||||
original client-side dry-run. Would-delete reporting for `--delete*` is
|
||||
deferred to a follow-up; dry-run never deletes.
|
||||
- Daemon `max connections per host` (per-source-IP concurrent cap, default 0 =
|
||||
unlimited), `auth lockout threshold` (default 10; 0 disables) and
|
||||
`auth lockout duration` (default 300 s) config keys.
|
||||
- `fastsync-server --allow-super` opt-in for a privileged standalone TCP server;
|
||||
without it a root standalone receiver forces super-user activities off (device
|
||||
nodes, `--write-devices`, ownership). The `--stdio` SSH argv is client-composed,
|
||||
so super activities always stay off there.
|
||||
|
||||
### Changed
|
||||
|
||||
- Config wire fields are now declared once in an X-macro table
|
||||
(`CONFIG_WIRE_FIELDS` in `src/shared/config.h`) that generates the struct
|
||||
members, defaults, and the send/receive sequence, removing the manual
|
||||
six-site field sync. Wire bytes and `PROTOCOL_VERSION` are unchanged.
|
||||
- `receive_incremental_check()` (the per-file `STATUS_CHECK` fast path) is split
|
||||
into small static helpers with a short linear orchestrator. Pure refactor: the
|
||||
wire byte stream and all cleanup are unchanged.
|
||||
- `authorized_root` state has a single owner (`utils.c`) with read accessors; the
|
||||
duplicated statics in `file.c` and the server were removed.
|
||||
- `Data` records its owning `ProtocolSession` so its memory charge is returned to
|
||||
the session that reserved it, regardless of the destroying thread.
|
||||
- The receiver pipeline moved out of `shared` into `server/receiver_pipeline.[ch]`;
|
||||
the build now uses explicit `fastsync_shared` / `fastsync_client_core` /
|
||||
`fastsync_server_core` targets instead of a GLOB, and the client no longer links
|
||||
server code.
|
||||
- The benchmark tool generates the requested random/compressible data mix
|
||||
accurately, verifies each transfer before recording it, computes correct
|
||||
percentiles, adds a MB/s column, handles `tc`/netem without requiring `sudo`
|
||||
when already root, builds into a dedicated `build-bench/` directory, and adds a
|
||||
`--warm` incremental-transfer mode.
|
||||
- The `nix-shell` dev environment provides the full toolchain (clang-format,
|
||||
cppcheck, pytest-xdist, OpenSSH, rsync, iproute2, valgrind, lcov) and no longer
|
||||
builds on entry.
|
||||
|
||||
### Security
|
||||
|
||||
- Enforce the daemon's per-module `max connections` cap (0 = unlimited) and add
|
||||
the shared per-source `max connections per host` cap plus a cross-process
|
||||
`auth lockout`. Because the listener forks one child per connection, the
|
||||
counters live in an anonymous shared mapping created before the accept loop and
|
||||
reclaimed by the parent's `SIGCHLD` handler, so the per-module, per-source and
|
||||
auth-failure state is shared across every child (including after `SIGKILL`). The
|
||||
per-source table has a bounded lifetime (expired/idle entries are reclaimed,
|
||||
with a rate-limited warning when genuinely full), and the occupancy counters are
|
||||
re-derived from the shared slot table on every child exit. Trusted loopback
|
||||
peers are exempt (they share one address); clients behind a shared NAT/proxy
|
||||
share a single per-host budget and lockout, which is documented.
|
||||
- Hardening from a full security audit:
|
||||
- Fail a truncated zstd frame instead of spinning forever (remote DoS).
|
||||
- Open receiver destination/basis/hard-link entries `O_NONBLOCK` so a
|
||||
client-planted FIFO cannot block a worker indefinitely.
|
||||
- Require a regular file before `--inplace` writes, closing a FIFO-hang and a
|
||||
raw-device write that bypassed the `--write-devices` gate.
|
||||
- Reject SSH destinations whose user/host begins with `-` and insert `--` before
|
||||
the host token, closing `-o ProxyCommand=…` argument injection (RCE).
|
||||
- Gate client `--force` recursive removal behind the server `--allow-delete`
|
||||
policy.
|
||||
- Reject empty `hosts allow`/`hosts deny`/`auth users` values instead of
|
||||
silently meaning "unrestricted".
|
||||
- Restrict TLS 1.2 to AEAD suites and set server cipher preference; load the
|
||||
private key TOCTOU-safely from an `O_NOFOLLOW` fd; verify IP literals against
|
||||
IP SANs; guard client-cert CN truncation.
|
||||
- Make `--dry-run` content-blind: it neither reads destination files nor
|
||||
hashes basis files, removing a 1-bit content oracle against `read only`
|
||||
modules.
|
||||
- Bound glob matching (iterative DP, no exponential backtracking) and bound
|
||||
line reads for filter/`--files-from`/pattern files.
|
||||
- Gate `system.posix_acl_*` xattrs on `--acls` and charge decompression/chunk
|
||||
allocations against the per-connection memory budget.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Pre-auth NULL dereference in `config_delete()` when an over-long
|
||||
`basis_count` (and the analogous count fields) was received and then failed
|
||||
validation; received counts are now validated before being published.
|
||||
- Leaked inherited `Data` in the forked compression-truncation unit test
|
||||
(valgrind definite leak).
|
||||
- `receive_status()` no longer loses a captured rejection reason when owed
|
||||
keepalives are drained.
|
||||
|
||||
## [2.20.0] - 2026-09-13
|
||||
|
||||
### Security
|
||||
|
||||
+34
-7
@@ -1,6 +1,6 @@
|
||||
cmake_minimum_required(VERSION 3.22)
|
||||
|
||||
project(FastFileTransfer VERSION 2.20.0)
|
||||
project(FastFileTransfer VERSION 2.30.0)
|
||||
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_C_STANDARD 11)
|
||||
@@ -26,9 +26,9 @@ elseif(NOT SANITIZER STREQUAL "none")
|
||||
endif()
|
||||
|
||||
# --- Strict warnings option ---
|
||||
option(STRICT_WARNINGS "Enable strict warnings (Wextra, Wpedantic, Werror)" OFF)
|
||||
option(STRICT_WARNINGS "Enable strict warnings (Wextra, Wpedantic, Wformat-signedness, Werror)" OFF)
|
||||
if(STRICT_WARNINGS)
|
||||
add_compile_options(-Wextra -Wpedantic -Werror)
|
||||
add_compile_options(-Wextra -Wpedantic -Wformat-signedness -Werror)
|
||||
endif()
|
||||
|
||||
# --- Coverage option ---
|
||||
@@ -68,6 +68,16 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
# --- Explicit source lists ---
|
||||
@@ -86,17 +96,24 @@ set(SHARED_SRCS
|
||||
src/shared/config.c
|
||||
src/shared/credentials.c
|
||||
src/shared/daemon_conf.c
|
||||
src/shared/daemon_limits.c
|
||||
src/shared/data.c
|
||||
src/shared/delay_updates.c
|
||||
src/shared/delete.c
|
||||
src/shared/delete_commit.c
|
||||
src/shared/delete_plan.c
|
||||
src/shared/delta.c
|
||||
src/shared/file.c
|
||||
src/shared/file_list.c
|
||||
src/shared/file_receive.c
|
||||
src/shared/file_save.c
|
||||
src/shared/file_send.c
|
||||
src/shared/file_store.c
|
||||
src/shared/filter.c
|
||||
src/shared/format.c
|
||||
src/shared/hardlink.c
|
||||
src/shared/identity.c
|
||||
src/shared/incremental_check.c
|
||||
src/shared/log.c
|
||||
src/shared/metadata.c
|
||||
src/shared/motd.c
|
||||
@@ -123,9 +140,14 @@ set(SERVER_MAIN_SRCS src/server/server.c)
|
||||
# Client implementation (no main): everything except the CLI entry point.
|
||||
set(CLIENT_CORE_SRCS
|
||||
src/client/change_list.c
|
||||
src/client/client_manifest.c
|
||||
src/client/client_report.c
|
||||
src/client/client_scan.c
|
||||
src/client/client_send.c
|
||||
src/client/client_validation.c
|
||||
src/client/scanner.c
|
||||
src/client/scanner_filter.c
|
||||
src/client/scanner_parallel.c
|
||||
src/client/usage.c
|
||||
)
|
||||
set(CLIENT_MAIN_SRCS src/client/client_cli.c)
|
||||
@@ -133,8 +155,8 @@ set(CLIENT_MAIN_SRCS src/client/client_cli.c)
|
||||
# --- Library targets ---
|
||||
add_library(fastsync_shared STATIC ${SHARED_SRCS})
|
||||
target_include_directories(fastsync_shared PUBLIC src/shared)
|
||||
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL
|
||||
OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_library(fastsync_client_core STATIC ${CLIENT_CORE_SRCS})
|
||||
target_include_directories(fastsync_client_core PUBLIC src/client)
|
||||
@@ -205,12 +227,16 @@ set(TEST_SRCS
|
||||
tests/test_config.c
|
||||
tests/test_credentials.c
|
||||
tests/test_daemon_conf.c
|
||||
tests/test_daemon_limits.c
|
||||
tests/test_data.c
|
||||
tests/test_delay_updates.c
|
||||
tests/test_delete_plan.c
|
||||
tests/test_delta.c
|
||||
tests/test_file.c
|
||||
tests/test_file_list.c
|
||||
tests/test_file_sendfile.c
|
||||
tests/test_filter.c
|
||||
tests/test_format.c
|
||||
tests/test_fuzz_smoke.c
|
||||
tests/test_glob.c
|
||||
tests/test_hardlink.c
|
||||
@@ -221,6 +247,7 @@ set(TEST_SRCS
|
||||
tests/test_multiprocessing.c
|
||||
tests/test_property.c
|
||||
tests/test_protocol.c
|
||||
tests/test_protocol_error.c
|
||||
tests/test_queue.c
|
||||
tests/test_receiver_timeout.c
|
||||
tests/test_robustness.c
|
||||
@@ -270,7 +297,7 @@ if(ENABLE_FUZZ)
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE tests src/shared src/server)
|
||||
target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer)
|
||||
target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL
|
||||
OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
endforeach()
|
||||
endif()
|
||||
+15
-1
@@ -2,8 +2,22 @@ FROM ubuntu:24.04
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \
|
||||
python3 python3-pip python3-venv openssl openssh-client \
|
||||
lcov valgrind clang libclang-rt-18-dev && \
|
||||
lcov valgrind clang libclang-rt-18-dev \
|
||||
acl attr zlib1g-dev liblz4-dev libxxhash-dev && \
|
||||
pip3 install --break-system-packages pytest pytest-xdist && \
|
||||
curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \
|
||||
apt-get install -y --no-install-recommends nodejs && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# rsync is used as the reference implementation for drop-in parity tests.
|
||||
# Ubuntu 24.04 ships 3.2.7, so build the pinned 3.4.1 reference from source.
|
||||
ARG RSYNC_VERSION=3.4.1
|
||||
ARG RSYNC_SHA256=2924bcb3a1ed8b551fc101f740b9f0fe0a202b115027647cf69850d65fd88c52
|
||||
RUN curl -fsSL "https://download.samba.org/pub/rsync/src/rsync-${RSYNC_VERSION}.tar.gz" -o /tmp/rsync.tar.gz && \
|
||||
echo "${RSYNC_SHA256} /tmp/rsync.tar.gz" | sha256sum -c - && \
|
||||
tar -xzf /tmp/rsync.tar.gz -C /tmp && \
|
||||
cd "/tmp/rsync-${RSYNC_VERSION}" && \
|
||||
./configure --enable-zstd --enable-xxhash --enable-lz4 && \
|
||||
make -j"$(nproc)" && \
|
||||
make install && \
|
||||
rm -rf "/tmp/rsync-${RSYNC_VERSION}" /tmp/rsync.tar.gz
|
||||
+262
@@ -0,0 +1,262 @@
|
||||
# FastSync — Session Handoff (2026-09-21)
|
||||
|
||||
## Current status
|
||||
- **Release `v2.28.0`** is tagged and merged to `main`: tag `v2.28.0` points at
|
||||
`ee6523a`, and the PR #304 merge commit `b4d54504` is on `main`.
|
||||
- **`dev` is at `0fbb9de`** — the merge of parity cycle 2.29 (PR #305). The old
|
||||
`558782d` (incremental-check flake fix) is an ancestor.
|
||||
- **`PROTOCOL_VERSION` = `"2.28.0"`** (`src/shared/config.h`); CMake
|
||||
`project(FastFileTransfer VERSION 2.28.0)`.
|
||||
- **Parity cycle 2.29 is merged to `dev`** (PR #305), no wire change. It closed
|
||||
the scanner-order, delete-timing, relative-basis and fuzzy-eligibility
|
||||
residuals and improved the `--info`/`--stats`/`--debug` partials. Parity
|
||||
matrix: **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. Remaining ⚠️ rows: `--info`,
|
||||
`--debug`, `--msgs2stderr`, `--stats`, `--progress`, `--delete-before`, the
|
||||
three basis-dir options, and `-y`/`--fuzzy`.
|
||||
- **Audit cycle complete on branch `fix/audit-cycle`** (branched from `dev` @
|
||||
`0fbb9de`), integration PR to `dev` pending. No wire change
|
||||
(`PROTOCOL_VERSION` stays 2.28.0). It lands the receiver/client security and
|
||||
correctness fixes — `--temp-dir` symlink-escape confinement, special-bit
|
||||
masking under a super-off policy, daemon `umask(022)`, the `-z` decompression
|
||||
ceiling raised to the 256 MiB whole-file bound, `--bwlimit` pacing the
|
||||
plaintext `--sendfile` path, `--partial-dir` implying `--partial`, rejection
|
||||
of unsupported filter modifiers (`x`/`e`/`n`/`w`), client-side
|
||||
`MAX_FILTER_RULES` enforcement, unknown wire `Status` rejection, and the
|
||||
accompanying refactors/docs. The parity matrix is unchanged at
|
||||
**120 ✅ / 10 ⚠️ / 27 ❌ = 157**; this docs pass (worktree `fix/audit-docs2`)
|
||||
corrects the `RSYNC_COMPAT.md` summary tally to match the rows.
|
||||
|
||||
|
||||
## What landed this session
|
||||
1. **Wave 8 (refactors):** Config X-macro wire table; single-owner `authorized_root`;
|
||||
daemon per-module/per-host caps + cross-process auth lockout (`daemon_limits.[ch]`);
|
||||
`Data` charge returns to its owning `ProtocolSession`.
|
||||
2. **Wave 9 (protocol 2.21.0):** optional `STATUS_ERROR_DETAIL` rejection reasons;
|
||||
server-contacting `--dry-run` (`STATUS_DRY_RUN_TRANSFER`, receiver mutates nothing).
|
||||
3. **Security wave:** ran 5 parallel audits (wire parsing; daemon/transport/TLS/auth;
|
||||
receiver confinement; client/CLI/SSH; crypto/memory/limits). Fixed all HIGH and the
|
||||
confirmed MEDIUMs:
|
||||
- SSH `-o ProxyCommand=…` argument injection (RCE) — reject leading `-`, insert `--`.
|
||||
- Truncated zstd frame infinite CPU loop (remote DoS).
|
||||
- FIFO receiver opens lacked `O_NONBLOCK` (indefinite hang).
|
||||
- `--inplace` could write a FIFO/device (bypass of `--write-devices` gate).
|
||||
- `--force` not gated by server `--allow-delete`.
|
||||
- Privileged standalone server defaulted super activities on; added `--allow-super`
|
||||
(never honored with `--stdio`).
|
||||
- `--dry-run` content/hash oracle on `read only`/basis files removed.
|
||||
- Empty `hosts allow`/`deny`/`auth users` now rejected.
|
||||
- TLS: AEAD-only 1.2 + server preference, TOCTOU-safe key load, IP-SAN verify,
|
||||
CN-truncation guard. Glob backtracking bounded; line reads bounded; ACL xattrs
|
||||
gated on `--acls`; decompression/chunk memory charged; pre-auth `basis_count`
|
||||
NULL-deref fixed.
|
||||
4. **Tooling:** benchmark accuracy (data mix, verification, percentiles, `tc`,
|
||||
`build-bench/`, `--warm` mode); `shell.nix` full toolchain and no build-on-entry;
|
||||
docs state push-only / remote-source unsupported.
|
||||
5. **Preserve-attribute split (protocol 2.22.0)** landed on `feat/preserve-attr-split`: per-attribute `-p/-t/-o/-g` + `--no-*` negations, `-a` = `-rlptgoD`, and the 2.21.0 → 2.22.0 wire bump.
|
||||
6. **Rsync-parity wave (protocol 2.23.0)** on `feat/rsync-parity`: rsync short options/clustering/attached values (`-r`/`-b`/`-L`/`-B`, `-av`, `-aAX`, `-B1000`, `-essh`, `-MOPT`), `-c` checksum quick-check, `--checksum-choice`/`--compress-choice` validation and seed randomization, rsync timeout/max-alloc defaults, temp-dir confinement + `EXDEV` fallback, ownership/mapping parity (numeric-ids modifier, map ranges/`*`/empty-FROM, `--chown`+map conflicts, fake-super resolved-owner record), verbatim symlink storage with rsync `--safe-links`/`--munge-links`, socket recreation under `--specials`, `--chmod` 3.4.1 semantics, and delete scoping + `--max-delete` partial/exit-25. Wire: appended delete-manifest synchronized-directory section and `STATUS_DELETE_LIMIT`.
|
||||
7. **Parity-completion wave (protocol 2.24.0 → 2.26.0)** on `feat/parity-completion`: per-directory delete plans (`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`; receiver `STATUS_STATS` counters feeding `--stats`/`--progress` and `--out-format %b/%c/%C`, plus `-n --delete` lines; `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/`none` checksums with `auto` negotiation (default `xxh128`/`zstd`); general `-R`/`--no-implied-dirs`/`-d`; the full filter grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` + modifiers) and corrected `-F`/`-FF`; receiver-side `--chown`/map TO-name resolution; absolute basis dirs + `--link-dest` relink; receiver-side `--ignore-existing` short-circuit; `--preallocate` over `--sparse` via `fallocate(2)`; `--iconv=.`/`-`/`--no-iconv`; lone `-h` help; aliases `--ignore-non-existing`/`--protect-args`/`--msgs2stderr`; and the full `--info`/`--debug` vocabulary. `RSYNC_COMPAT.md` reclassifies the matrix to 106 ✅ / 27 ⚠️ / 23 ❌; the later rsync-parity-stats pass (`fix/parity-stats`) moves it to 107 ✅ / 25 ⚠️ / 24 ❌ (see item 8).
|
||||
8. **rsync-parity-stats pass** on `fix/parity-stats` (no wire change, `PROTOCOL_VERSION` stays `2.26.0`): `--delete-delay` now reports only entries it actually removes, while the `--max-delete` budget is charged at plan/snapshot time (`planned`, via `defer_add`) to bound the deferred list (a refilled deferred directory that survives `ENOTEMPTY` is not reported but still consumes budget); `--stats` gained the `(reg/dir/link/special)` `Number of files` breakdown and now counts only regular files actually stored for `Number of regular files transferred`/transferred size/literal data (up-to-date re-runs report 0); `Total file size` includes symlink target lengths; `--progress` prints the leading `./` root line and counts it in `to-chk` so a single-file transfer matches rsync; and `%C` uses the selected transfer checksum with `checksum_digest_file` supporting md4/sha1/none, byte-identical to rsync for every algorithm. `--out-format` reclassified ❌ (`%b`/delta-`%c` are protocol-specific). Differential + regression tests added; full suite + ASan + clang-format + cppcheck clean.
|
||||
9. **Option-parity wave (protocol 2.26.0 → 2.27.0, on `fix/parity-options`):**
|
||||
`--bwlimit` now ports rsync 3.4.1's units/quantization and paces like its
|
||||
leaky bucket; `--ignore-errors` reproduces rsync's default (an I/O error
|
||||
skips deletion unless the flag is set; the readable tree still transfers and
|
||||
the run exits 23) across every delete timing; the `--info` categories with a
|
||||
FastSync event (`name`/`flist`/`del`/`remove`/`nonreg`/`progress`) emit
|
||||
rsync's line format, with real-run `deleting`/`*deleting` lines carried over
|
||||
the new trailing config bool `report_deletes` (golden wire updated by
|
||||
`tests/test_config.c`). Two residuals were reclassified **divergent**: `-M`
|
||||
over daemon/TCP (no argv channel in FastSync's binary config handshake;
|
||||
rsync-daemon differential pins the rsync behavior) and receiver-side
|
||||
`protect`/`risk` re-derivation for destination-only entries (would need a
|
||||
receiver filter engine; differential pins the divergence — **reversed by
|
||||
track 4a below**, which adds that engine). The options pass
|
||||
stands at **110 ✅ / 21 ⚠️ / 26 ❌**. New `tests/integration/test_option_parity.py`
|
||||
holds the rsync differentials (bwlimit parse+rate, info lines, real-setpriv
|
||||
`--ignore-errors`, rsync-daemon `-M`, filter-protect pin).
|
||||
|
||||
10. **rsync-parity-fs pass** on `fix/parity-fs` (no wire change of its own; integrated
|
||||
on top of the 2.27.0 options wave): recursive transfers now recreate empty source directories (and
|
||||
`-m/--prune-empty-dirs` still suppresses them), a directory entry replaces a
|
||||
blocking destination regular file, and `-R --no-implied-dirs --files-from`
|
||||
places a listed file under a missing implied parent with default attributes
|
||||
instead of refusing (real rsync 3.4.1 parity, differential-tested). `--iconv`
|
||||
now reproduces rsync's push direction (destination charset = the spec's REMOTE
|
||||
half; a server `--iconv` overrides), and `-T/--temp-dir` relative semantics are
|
||||
confirmed identical while the absolute-path confinement is a deliberate
|
||||
divergence. The basis-dir options, `--delay-updates` and `--dry-run` were
|
||||
reclassified to ❌ after a differential test reproduced each exact residual
|
||||
(basis content verification, fixed staging-name collision, and dry-run
|
||||
would-delete over-report). `--fuzzy` was also reclassified to ❌ (deterministic
|
||||
heuristic with a 10× size window, not rsync's matcher), but its residual is the
|
||||
candidate-selection heuristic itself: the final tree is byte-exact by design, so
|
||||
it is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync
|
||||
differential. (Track 5b later found the name heuristic is rsync's own and moved
|
||||
the row ❌ → ⚠️, leaving only the narrower delta size window; see entry 15.) The parity-review pass then moved `--delete-delay` to ⚠️ (the
|
||||
plan-time `--max-delete` charge and non-recursive deferred removal differ from
|
||||
rsync when a snapshotted entry fails removal). Differential-gate allowlist
|
||||
entries `min_size`/`empty_dirs_recursive`/`dirs_plain` were removed. The
|
||||
integrated stats+options+fs branch stands at **111 ✅ / 13 ⚠️ / 33 ❌ = 157**;
|
||||
full suite + ASan + clang-format + cppcheck clean.
|
||||
|
||||
11. **No-wire parity track 1** on `feat/parity-2.28` (no protocol change):
|
||||
`-n --delete` now sends the same filter-excluded + size-pruned protected
|
||||
prefixes and synchronized-directory scope as a real run (dry-run would-delete
|
||||
matches rsync for source-derived protections; the destination-only exclude
|
||||
residual was later closed by track 4a, readdir ordering remains);
|
||||
`--delete-delay` now charges
|
||||
`--max-delete` on actual removals and re-scans a queued directory at commit
|
||||
to remove content created after the plan, with an independent deferred-list
|
||||
cap (only partial-delete ordering remains); and `--info=name2` emits `NAME is
|
||||
uptodate` plus the leading `./` root name line for `--info=name` (only the
|
||||
root-line trigger condition and receiver-side `skip` wording remain). Matrix
|
||||
now **111 ✅ / 14 ⚠️ / 32 ❌ = 157**; differential + unit tests added in
|
||||
`test_features.py`, `test_option_parity.py`, the unit test
|
||||
`tests/test_delete_plan.c`,
|
||||
`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`.
|
||||
12. **No-wire parity track 2b** on `feat/parity-2.28` (no protocol change):
|
||||
`--progress`/`-P`/`--info=progress` (when not `--quiet`) now run an opt-in
|
||||
paths-only metadata pre-count (no file reads/hashing) that supplies rsync's
|
||||
full file-list total for the `to-chk` denominator and the directory names,
|
||||
and emits per-directory/symlink/special name lines, in both the sequential
|
||||
and `--threads` paths. `--delete-during`/`--delete-delay` reuse their
|
||||
keep-set pre-scan instead of a second walk; non-progress runs are
|
||||
unaffected. Differential tests (`progress`/`progress_threads` over a new
|
||||
`multidir` corpus) match rsync's name set and `to-chk` denominator on a
|
||||
fresh transfer, and the single-file byte-identical test still passes;
|
||||
emission order (rsync's sorted depth-first vs FastSync's readdir/BFS stream)
|
||||
plus re-run over-naming (unconditional `./`, ancestor dirs named with a
|
||||
transferred child, and no quick-check for symlinks/empty dirs) remain the
|
||||
caveats, so the row stays ⚠️ and the matrix is unchanged at
|
||||
**111 ✅ / 14 ⚠️ / 32 ❌ = 157**.
|
||||
|
||||
13. **Wire parity track 4a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0`): the receiver now has a delete-time filter engine. The sender
|
||||
compiles its root-level selection rules exactly as the scanner does
|
||||
(`filter_base_build`) and streams them as one bounded, self-describing
|
||||
config-frame block (action, sides, anchored, dir-only, negate, owner,
|
||||
pattern; bounded rule count and pattern bytes, unknown action/sides is a
|
||||
protocol error). The receiver reconstructs `protect_rules` and applies them
|
||||
first-match-wins to each extraneous destination path in every delete timing
|
||||
(the whole-tree commit walker, the `--delete-during`/`--delete-delay`
|
||||
per-directory plans, and the `-n` would-delete enumeration), so a
|
||||
`P *.log` rule protects a destination-only `extra.log` like rsync (with
|
||||
`risk` cancelling); the sender-derived protected-prefix behavior is
|
||||
preserved when no rules are sent and `--delete-excluded` semantics are
|
||||
unchanged. Per-directory merge (`:`/`.`) receiver re-derivation remains the
|
||||
residual. `TestFilterProtect` (real + dry-run) plus differential cases
|
||||
`filter_protect`, `filter_protect_during`, `filter_protect_delay` added and
|
||||
the `--filter=RULE` row moves ❌ → ✅: matrix now
|
||||
**115 ✅ / 11 ⚠️ / 31 ❌ = 157**; unit tests, the three named integration
|
||||
files, clang-format and cppcheck clean.
|
||||
|
||||
14. **Wire parity track 5a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0` by project decision): the three basis-dir options now default to
|
||||
rsync's metadata quick-check (equal size + equal mtime, or size alone under
|
||||
`--size-only`; `-I` disables matching) instead of FastSync's historical
|
||||
xxHash64 content equality, so a same-size/different-content basis is trusted
|
||||
exactly as rsync trusts it. A new FastSync-only, long-only `--verify-basis`
|
||||
flag restores the strict whole-file content equality; its bool is appended to
|
||||
the basis block of the config frame (golden wire frame 882 → 886 bytes).
|
||||
`--verify-basis` streams the confined basis descriptor to hash it, and a
|
||||
basis hit is no longer capped at the 256 MiB whole-file payload bound:
|
||||
`--copy-dest` streams the basis through a bounded buffer and `--link-dest`'s
|
||||
copy fallback streams from the basis, so an over-limit hit materializes (a
|
||||
basis MISS still falls back to the normal transfer and keeps its own bound).
|
||||
A `--copy-dest` hit re-applies the SOURCE attributes (the sender transmits
|
||||
the source metadata with the basis check frame), matching rsync's
|
||||
"copy then fix attributes"; a `--link-dest` success keeps the shared inode's
|
||||
attributes (writing through it would mutate the basis). Differential cases
|
||||
`copy_dest` and `verify_basis` added; `test_basis_dir_size_only_content_residual`
|
||||
converted to a passing parity assertion; `TestBasisDestDirs` updated for the
|
||||
new default + `--verify-basis`; unit tests cover the quick-check/verify
|
||||
decision and the same-size/different-content handshake. The
|
||||
`--compare-dest`/`--copy-dest`/`--link-dest` rows move ❌ → ⚠️ (relative-DIR
|
||||
resolution base and over-limit MISS refusal): matrix now
|
||||
**116 ✅ / 13 ⚠️ / 28 ❌ = 157**.
|
||||
|
||||
15. **No-wire parity track 5b** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0` by project decision): `-y`/`--fuzzy` reclassified ❌ → ⚠️. A probe
|
||||
against real rsync 3.4.1 (pinned `-B8192`, repeated-content 64 KiB corpus)
|
||||
showed the name heuristic is already rsync's (`util1.c fuzzy_distance` /
|
||||
`find_filename_suffix` + the exact size+mtime pass) and the output is always
|
||||
byte-exact; the only residual is candidate ELIGIBILITY, because FastSync's
|
||||
`delta_should_attempt` gate caps the size ratio at 10× and requires both
|
||||
files ≥ 16 KiB while rsync will reuse a basis from 0.25× to 10000× and below
|
||||
16 KiB. The choice is observable only as `--stats` bandwidth counters. Added
|
||||
differential case `fuzzy_basis` (same-suffix sibling, one name edit,
|
||||
identical content, block size pinned) asserting tree **and** normalized
|
||||
`--stats` parity where the choices coincide, plus `TestFuzzy` pinning the
|
||||
window boundary on both sides (>10× and <16 KiB siblings declined by
|
||||
FastSync while rsync uses them, both trees byte-identical). Matrix now
|
||||
**116 ✅ / 14 ⚠️ / 27 ❌ = 157**.
|
||||
|
||||
16. **Lockstep delete-default track 6** on `feat/parity-2.28` (`PROTOCOL_VERSION`
|
||||
stays `2.28.0`): plain `--delete` now defaults to rsync's delete-during
|
||||
(`--del`) timing, normalized on the client onto the existing `delete_during`
|
||||
wire bool. The old late whole-tree commit is opt-in via `--delete-after` or
|
||||
the FastSync-only long `--delete-commit` (identical `delete_after` timing).
|
||||
`-d/--dirs` still falls back to the end commit, `--delay-updates` still
|
||||
deletes before publication, and `--files-from`/`-R` scope is unchanged. The
|
||||
`STATUS_DELETE_PLAN` frame gained a one-int `apply` flag so the per-run
|
||||
config block (including `--delete-missing-args` exact paths) is always
|
||||
transmitted, on a config-only carrier when the scope allows no directory
|
||||
plan — fixing a latent bug with a file-only `--files-from` list. Differential
|
||||
cases `delete`/`delete_commit`/`filter_protect_after` plus the extended
|
||||
`test_delete_timing_parity.py` (plain `--delete` mid-abort removes reached
|
||||
extras, `--delete-commit` defers) pass; full `-m "not setpriv"` suite,
|
||||
clang-format and cppcheck clean. Matrix unchanged at
|
||||
**116 ✅ / 14 ⚠️ / 27 ❌ = 157** (the `--delete`/`--delete-during` rows stay
|
||||
⚠️ for the abort boundary; `--delete-after` stays ✅).
|
||||
|
||||
17. **Audit cycle** on `fix/audit-cycle` (from `dev` @ `0fbb9de`;
|
||||
`PROTOCOL_VERSION` stays `2.28.0`): a security/correctness pass over the
|
||||
parity-2.29 baseline. It raises the decompression ceiling to the 256 MiB
|
||||
protocol whole-file bound (`-z` on 100–256 MiB files now works), paces the
|
||||
plaintext-TCP `--sendfile` path with `--bwlimit`, confines the `--temp-dir`
|
||||
scratch dir by the fd's real path (symlink escape refused), masks
|
||||
client-controlled setuid/setgid/sticky bits when super activities are not
|
||||
permitted, sets the daemon umask to `022`, makes `--partial-dir` imply
|
||||
`--partial`, rejects the unsupported filter modifiers (`x`/`e`/`n`/`w`),
|
||||
enforces `MAX_FILTER_RULES` client-side, rejects unknown wire `Status`
|
||||
values, and hardens credentials/signal handling (with the accompanying
|
||||
refactors and docs). No row changes classification, so the matrix stays
|
||||
**120 ✅ / 10 ⚠️ / 27 ❌ = 157**. This docs pass is on `fix/audit-docs2`.
|
||||
|
||||
## Next steps
|
||||
1. **Open and merge the audit-cycle PR** (`fix/audit-cycle`, including this
|
||||
`fix/audit-docs2` docs pass) into `dev` once reviewed. `dev` is the default
|
||||
branch; all PRs target `dev`, never `main` directly.
|
||||
2. **Remaining deferred items:**
|
||||
- **Large structural refactors:** delete-engine consolidation
|
||||
(`delete_extras_fd`/`manifest_delete_extras`/the delete-plan path),
|
||||
god-function splits, and translation-unit splits.
|
||||
- **`--progress`/`--info` receiver→sender event channel:** the root `./`
|
||||
line, ancestor-directory suppression, receiver-side `skip`/`backup` echo,
|
||||
and symlink/empty-dir quick-check feedback.
|
||||
- **`--delete-before` phase-0 keep-set** (rsync fixes the file list before
|
||||
the data pass; FastSync keeps its pre-scan snapshot race).
|
||||
- ~~**>256 MiB single-file streaming** (B4, the general whole-file limit).~~
|
||||
Closed by #318: the whole-file payload, the basis read/verify and the fuzzy
|
||||
basis are streamed through bounded buffers (lz4/append remain buffered).
|
||||
- **Wire native-size framing:** lengths are native `size_t` and the protocol
|
||||
assumes homogeneous word size/endianness — document or move to fixed-width
|
||||
framing.
|
||||
- **SCRAM-like daemon auth channel binding:** no TLS channel binding today
|
||||
(and it is not RFC 5802).
|
||||
- Still-open security nits: the pre-auth config/daemon-auth handshake has no
|
||||
aggregate wall-clock deadline (per-message timeout only — slowloris holds
|
||||
connection slots); the per-source registry fails open when the shared table
|
||||
is full (per-module/global caps and host ACLs still apply).
|
||||
3. **Out of scope / intentional:** pull (remote source) mode is **not** planned —
|
||||
FastSync is push-only; see `RSYNC_COMPAT.md#direction`.
|
||||
|
||||
## Key facts / commands
|
||||
- CI image: `gitea.tap-tap.win/taptap/fastsync-ci:v11` (alias `fastsync-ci:local`).
|
||||
- Build/test: `cmake -B build -S . -DSTRICT_WARNINGS=ON && cmake --build build -j$(nproc) && ./build/tests`
|
||||
then `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`.
|
||||
- Dev shell: `nix-shell` (provides clang-format, cppcheck, pytest-xdist, openssh,
|
||||
rsync, iproute2, valgrind, lcov; does not build on entry).
|
||||
- Gitea API token: supplied out-of-band via the `TOKEN` environment variable; it is
|
||||
intentionally **not** recorded in this file.
|
||||
- CI polling: `GET /api/v1/repos/TapTap/FastSync/actions/runs?limit=N`, match `head_sha`,
|
||||
then `/actions/runs/<id>/jobs`.
|
||||
@@ -1,4 +1,4 @@
|
||||
#FastSync
|
||||
# FastSync
|
||||
|
||||
FastSync is a high-performance file synchronization tool designed to become a
|
||||
drop-in replacement for common `rsync` workflows. It keeps the familiar
|
||||
@@ -7,7 +7,7 @@ multithreading, streaming zstd compression, chunking, zero-copy TCP transfers,
|
||||
and native TCP/TLS transports.
|
||||
|
||||
The release version is FastSync's client/server protocol version (printed by
|
||||
`fastsync --version`); client and server must match. See
|
||||
`./build/client --version`); client and server must match. See
|
||||
[CHANGELOG.md](CHANGELOG.md) for the history.
|
||||
|
||||
The compatibility target is straightforward:
|
||||
@@ -28,7 +28,7 @@ FastSync uses a producer-consumer transfer pipeline and can combine several
|
||||
optimizations for large or high-latency transfers:
|
||||
|
||||
- Multithreaded scanning, loading, and sending.
|
||||
- Streaming zstd compression with levels 1 through 22.
|
||||
- Streaming compression (zstd by default, plus lz4/zlib/zlibx) with levels 1 through 22.
|
||||
- Configurable file chunking and compact chunk serialization.
|
||||
- `sendfile()` zero-copy transfers over TCP.
|
||||
- Batched incremental checks to reduce round trips.
|
||||
@@ -51,28 +51,52 @@ replacement for every rsync feature or protocol mode.
|
||||
- Rsync-style source and destination arguments.
|
||||
- SSH transport using `user@host:destination` paths below the remote authorized root.
|
||||
- TCP client/server transfers.
|
||||
- Dry runs, excludes, includes, size filters, backups, statistics, and
|
||||
bandwidth limiting.
|
||||
- Incremental size/mtime checks and optional xxHash64 content checks.
|
||||
- Dry runs (server-contacting since protocol 2.21.0 for server-routed targets),
|
||||
excludes, includes, size filters, backups, statistics, and bandwidth
|
||||
limiting.
|
||||
- Incremental size/mtime checks and optional content checks (`xxh128` by
|
||||
default, selectable with `--checksum-choice`).
|
||||
- FastSync-native delta transfer for changed files.
|
||||
- Optional mode and timestamp preservation.
|
||||
- Delete manifests with server-side delete authorization.
|
||||
- Temporary-file writes with atomic rename by default.
|
||||
- Path traversal checks and destination-root confinement.
|
||||
|
||||
### Not yet equivalent to rsync
|
||||
### Boundaries and documented divergences
|
||||
|
||||
The items below summarize FastSync's rsync compatibility status — recently
|
||||
closed gaps and the remaining known divergences. Each row of the detailed
|
||||
matrix is classified as parity, caveat, or divergent in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
- The FastSync wire protocol is not the rsync wire protocol.
|
||||
- SSH mode requires `fastsync-server` on the remote host.
|
||||
- Archive mode does not yet provide all of rsync's `-rlptgoD` behavior.
|
||||
- Symlink transfer is incomplete; link targets are not yet recreated in all
|
||||
modes.
|
||||
- Owner/group, ACL, xattr, and hard-link handling is incomplete or
|
||||
unavailable.
|
||||
- Archive mode covers rsync's `-rlptgoD` behavior — links, permissions, times,
|
||||
owner, group, devices, and special files — and does not imply compression or
|
||||
multithreading (see [Client](#client)). Ownership application is still
|
||||
privilege-gated: a receiver that cannot `chown` logs a warning and skips it.
|
||||
Under `-p` the source mode is copied exactly, including group/other-write
|
||||
bits; setuid/setgid/sticky bits are copied only when super-user activities are
|
||||
permitted, and are masked under `SUPER_MODE_OFF`/`--no-super` (see
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)).
|
||||
- Symlink transfer stores targets **verbatim** (`-l`/`--links`), including
|
||||
absolute and `..`-bearing targets, matching rsync. The receiver does not
|
||||
enforce a containment predicate by default; `--safe-links` drops unsafe
|
||||
targets on the sender, and `--munge-links` rewrites them with rsync's
|
||||
`/rsyncd-munged/` marker. `--trust-sender` does not affect symlink targets.
|
||||
A destination later consumed by a link-following tool can therefore follow a
|
||||
link outside the receive root — use `--safe-links` for untrusted sources.
|
||||
- Hard links (`-H`/`--hard-links`), extended attributes (`-X`/`--xattrs`), and
|
||||
POSIX ACLs (`-A`/`--acls`) are preserved; owner/group is applied through
|
||||
`-o`/`-g` (or an `-a`/`--archive` transfer), through the opt-in identity flags
|
||||
(`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`), and only when
|
||||
the receiver has permission. See
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md) for the exact semantics and documented
|
||||
divergences.
|
||||
- Device and special-file preservation is implemented with documented
|
||||
divergences: recreated device nodes require `CAP_MKNOD` on the receiver (a
|
||||
non-root receiver skips the entry), and sockets cannot be recreated (FIFOs
|
||||
are).
|
||||
non-root receiver skips the entry), while FIFOs **and unix sockets** are
|
||||
recreated (`--specials`).
|
||||
- Sparse-file hole preservation (`-S`, `--sparse`) is implemented receiver-side:
|
||||
long all-zero runs are written as holes (no wire change; the full file image
|
||||
is already in memory).
|
||||
@@ -80,76 +104,164 @@ replacement for every rsync feature or protocol mode.
|
||||
the write atomic (temp + rename). With `--partial`, a failed/interrupted write
|
||||
now retains the already-written temp at the destination path (best-effort) so
|
||||
a later `--append`/`--append-verify` run can resume it.
|
||||
- `--dirs` is not implemented. Its compatibility aliases `--old-dirs` and
|
||||
`--old-d` are recognized but rejected explicitly rather than silently using
|
||||
FastSync's recursive directory behavior.
|
||||
- `-d`/`--dirs` and its aliases `--old-dirs`/`--old-d` transfer the named
|
||||
directory entries without recursing into their contents.
|
||||
- Short-option names are now rsync-parity (Phase 7 Wave A): FastSync's former
|
||||
collisions were renamed (`-j`/`--threads`, `--preserve`, `--sendfile`,
|
||||
`--chunk-serialization`, `--timeout`, `--ssh-port`), so `-m`, `-M`, `-f`,
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync. See `RSYNC_COMPAT.md`.
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync.
|
||||
- Short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached values
|
||||
(`-B1000`, `-essh`, `-MOPT`, `--opt=value`) are accepted, matching rsync.
|
||||
- `-r`, `-b`, `-L`, and `-B` are parsed with the rsync short names.
|
||||
- `--stats` prints the counters FastSync can observe plus the receiver-only
|
||||
counters reported over the wire (`Matched data`, deleted files, and the
|
||||
created/literal counters); `Number of files` and `Number of created files`
|
||||
carry rsync's per-type breakdown. `--progress` prints rsync-style per-file
|
||||
blocks including the leading `./` line, and (when progress is requested) a
|
||||
paths-only pre-count supplies rsync's `to-chk` denominator.
|
||||
- Codecs match rsync 3.4.1: `zstd`/`lz4`/`zlib`/`zlibx` compression and
|
||||
`xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1`/`none` checksums. `auto` honors
|
||||
`RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST` and otherwise follows rsync's
|
||||
compiled-in order. An omitted `--compress-level` uses the codec's rsync
|
||||
default (zstd 3, zlib/zlibx 6, lz4 ignored); `zlib`/`zlibx` share the
|
||||
literal-only zlib path (rsync's zlibx semantics), and the transfer checksum is
|
||||
not separately selectable.
|
||||
|
||||
The detailed flag matrix is maintained in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented,
|
||||
partial, alternate, and planned behavior.
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It reports each row as **parity**,
|
||||
**caveat** (works with a documented divergence), or **divergent** (not
|
||||
supported), rather than treating "parsed" as parity.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Build
|
||||
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
This produces `./build/client` and `./build/server`. `compile_commands.json` is a symlink to `build/compile_commands.json` and is used by clangd/editor tooling; its target is generated by the build, so it dangles until the first build.
|
||||
|
||||
### Client
|
||||
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| Positional | `<source> <dest>` — automatic SSH detection if dest contains `:` |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime |
|
||||
| `-z, --compress [level]` | Enable streaming zstd compression (level 1–22, default 5) |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials (not compression/multithreading) |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime (implies the incremental checksum quick-check) |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh128` (default), `xxh3`, `xxh64`/`xxhash`, `md5`, `md4`, `sha1`, `none`, or `auto` (plus rsync's two-name `transfer,pre-transfer` form) |
|
||||
| `-z, --compress [level]` | Enable streaming compression (default `zstd`; level 1–22, default 5) |
|
||||
| `--compress-choice <alg>` | Compression algorithm: `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, or `auto` |
|
||||
| `--skip-compress <list>` | Skip compression for suffixes (`/`- or `,`-separated); defaults to rsync 3.4.1's built-in suffix list |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated (not compression/multithreading) |
|
||||
| `-j, --threads[=N]` | Multithreading mode; `N` (1–256) sets the parallel scanner worker count, bare `-j`/`--threads` uses the default |
|
||||
| `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) |
|
||||
| `-r, --recursive` | Recurse into directories (FastSync is always recursive; accepted for rsync compatibility) |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents; aliases `--old-dirs`/`--old-d` |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root |
|
||||
| `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) |
|
||||
| `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) |
|
||||
| `--sendfile` | Sendfile zero-copy. Incompatible with compression / chunk serialization. TCP only. Long form only. |
|
||||
| `--preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) |
|
||||
| `-n, --dry-run` | Scan and print what would be transferred |
|
||||
| `-p, --perms` | Preserve permission bits (part of the metadata bundle) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show real-time transfer speed |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded) |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk) |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`) |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch) |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing; `--no-whole-file` clears it |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`) |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime |
|
||||
| `--preserve` | Preserve mode and mtime (`-p` + `-t`; add `-o`/`-g` for owner/group or `-U`/`--atimes` for atime; `-N`/`--crtimes` captures birth time but cannot apply it) |
|
||||
| `-U, --atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N, --crtimes` | Capture birth time; cannot be applied (documented divergence) |
|
||||
| `-p, --perms` | Preserve permission bits. The source mode is copied exactly, including group/other-write bits; setuid/setgid/sticky are copied only when super-user activities are permitted (`SUPER_MODE_OFF`/`--no-super` masks them) |
|
||||
| `-t, --times` | Preserve modification times |
|
||||
| `-o, --owner` | Preserve the source owner (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `-g, --group` | Preserve the source group (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group`, `--no-preserve` | Negate the per-attribute flags (short `--no-p`/`--no-t`/`--no-o`/`--no-g`; `--no-preserve` clears all four) |
|
||||
| `-E, --executability` | Preserve executable permission bits |
|
||||
| `-X, --xattrs` | Preserve user `user.*` extended attributes |
|
||||
| `-A, --acls` | Preserve POSIX ACLs |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax) |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership |
|
||||
| `--numeric-ids` | Apply source numeric uid/gid directly instead of mapping by name |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP] (requires a privileged receiver) |
|
||||
| `--fake-super` | Record the resolved owner plus full mode/rdev in rsync's reserved `user.rsync.%stat` xattr (rsync 3.4.1 grammar) and replay the permission bits; never performs a real chown |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes) |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`) |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`) |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--include-from <file>` | Read include patterns from a file |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `-x, --one-file-system` | Do not cross filesystem boundaries; the mount-point directory entry is emitted (empty at the destination) without descending |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G; `0` = no local limit, matching rsync) |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--ignore-existing` | Skip files that already exist on the receiver; like rsync it does not apply to directories or symlinks. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`; a basis of any size is supported, streamed in bounded chunks — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination (the copy is streamed, so a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win; a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `--verify-basis` | FastSync-only: require a basis hit (`--compare-dest`/`--copy-dest`/`--link-dest`) to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync) |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-during, matching rsync, so destination space is freed progressively). Scoped to the synchronized directories, so `--files-from` subsets are safe |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) |
|
||||
| `--delete-after` | Explicit delete-after timing (implies `--delete`) |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G) |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | Positive I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`, built-in default 30 s) and the per-message protocol poll deadline (built-in default 60 s). Omit the option to keep both built-ins; `0` is rejected. The server side keeps the built-in 60 s protocol window (the value is not sent on the wire). |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 10) |
|
||||
| `--backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing) |
|
||||
| `-h, --human-readable` | Format transfer byte sizes with binary units |
|
||||
| `--delete-commit` | FastSync-only: keep the pre-2.28 atomic timing — delete only after the whole transfer succeeded (identical timing to `--delete-after`) |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication; the fixed `.fastsync-stage` staging name diverges from rsync — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install; confined to the receive root (a relative path resolves below it; an absolute path is accepted only when it canonicalizes inside it), with an `EXDEV` non-atomic copy fallback |
|
||||
| `-n, --dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created) |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing), including the receiver-only counters reported over the wire; `Number of files` and `Number of created files` carry rsync's per-type breakdown (deleted files are reported as a single total) |
|
||||
| `-i, --itemize-changes` | Print an rsync-style per-file change line |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %c %C %i %M %%`) |
|
||||
| `--list-only` | List source files instead of transferring |
|
||||
| `--fsync` | Fsync every written file before publication |
|
||||
| `-h, --human-readable` | Format transfer byte/rate counts with rsync's decimal (base-1000) units |
|
||||
| `--max-depth <n>` | Maximum directory depth to recurse (0 = unlimited, default: 0) |
|
||||
| `--log-file <path>` | Write log messages to file instead of stderr |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree (FastSync-native format, not rsync-interoperable) |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server); FastSync-native format, not rsync-interoperable |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server); FastSync-native format, not rsync-interoperable |
|
||||
| `--source-dir <path>` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) |
|
||||
| `--dest-dir <path>` | Server destination directory (overrides `FASTSYNC_DEST_DIR`) |
|
||||
| `--save-to-disk` | Write received files to disk |
|
||||
| `--server-host <ip>` | Server IP address (default: `127.0.0.1`) |
|
||||
| `--server-port <n>` | Server port (default: `8080`) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-e, --rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments, e.g. `-e "ssh -p 2222"`) |
|
||||
| `-M, --remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable) |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address |
|
||||
| `-4, --ipv4` | Force IPv4 for destination resolution |
|
||||
| `-6, --ipv6` | Force IPv6 for destination resolution |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect (`TCP_NODELAY`, `SO_KEEPALIVE`, `SO_RCVBUF`, `SO_SNDBUF`, `SO_REUSEADDR`) |
|
||||
| `--bwlimit <RATE>` | Bandwidth limit, using rsync's exact `parse_size_arg` grammar: a bare value is KiB/s; `K`/`M`/`G`/`T`/`P` are binary suffixes; `KB`/`MB` are decimal and `KiB`/`MiB` binary; decimals are accepted and quantized to whole KiB; `0` (or empty) means no limit. Also paces `--sendfile` transfers |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline. Default `0` = disabled (matching rsync); `0` disables it. `--no-timeout` is the negation. The value is not sent on the wire; the server side keeps its own safe floor. |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 60, matching rsync); `0` disables it (`--no-contimeout` is the negation) |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes (a positive integer); whatever was already transferred is kept |
|
||||
| `--stop-at=TIME` | Stop at an absolute time. Accepts rsync's `parse_time` forms (`Y-M-DTh:m`, `Y/M/DTh:m`, `Y-M-D`, `M-D`, `D`, `h:m`, `:m`, `T h:m`; omitted fields resolve to the next matching point in the local timezone), plus `now+N[smhd]` and FastSync's `HH:MM`/`HH:MM:SS` clock-time spelling. An early stop skips the late `--delete` keep-set |
|
||||
| `-b, --backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--tls` | Enable TLS encryption |
|
||||
| `--cert <path>` | TLS certificate file (PEM) |
|
||||
| `--key <path>` | TLS private key file (PEM) |
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--client-cn <name>` | TLS client certificate common name; mandatory with `--tls` (a TLS connection always verifies the client CN) |
|
||||
|
||||
The exhaustive rsync flag matrix is in [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
**Per-message vs. connection timeouts.** `--timeout` bounds each individual protocol
|
||||
send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It
|
||||
@@ -176,6 +288,7 @@ transfer is never aborted.
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--destination-root <path>` | Authorized destination root (default: `.`) |
|
||||
| `--allow-delete` | Permit manifest deletion |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. **Rejected with `--stdio`** (the SSH remote argv is client-composed, so a client could otherwise pass it and defeat the secure default; operators exposing `fastsync-server --stdio` over SSH must use a forced command if the default must hold). No effect when not root. |
|
||||
| `--allow-unauthenticated` | Permit plaintext TCP clients. For an `auth users` module this opts in **loopback plaintext only**; remote auth still requires verified TLS, so the flag never permits remote plaintext auth. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `--help` | Show help |
|
||||
@@ -187,60 +300,70 @@ transfer is never aborted.
|
||||
| `FASTSYNC_SOURCE_DIR` | — | Source directory fallback |
|
||||
| `FASTSYNC_DEST_DIR` | — | Destination directory fallback |
|
||||
| `FASTSYNC_SAVE_TO_DISK` | `false` | Disk persistence fallback |
|
||||
| `FASTSYNC_SSH_PORT` | `22` | Default SSH port |
|
||||
| `FASTSYNC_SERVER_HOST` | `127.0.0.1` | Default server host |
|
||||
| `FASTSYNC_SERVER_PORT` | `8080` | Default server port |
|
||||
| `FASTSYNC_TLS_CERT` | — | Default TLS certificate path |
|
||||
| `FASTSYNC_TLS_KEY` | — | Default TLS private key path |
|
||||
| `FASTSYNC_TLS_CA` | — | Default TLS CA certificate path |
|
||||
| `FASTSYNC_MAX_WHOLE_FILE_SIZE` | `268435456` | Receiver-only test hook: a byte count that lowers the whole-file streaming bound. Payloads above it are streamed through a bounded buffer. Values are clamped to the 256 MiB protocol ceiling, so it can only lower, never raise, the bound. |
|
||||
|
||||
## Implementation Details
|
||||
|
||||
### Data Structures
|
||||
1. **Chunk** — collection of files (~10 MB total by default)
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`;
|
||||
uid / gid are advisory wire fields and are never applied by the receiver;
|
||||
atime is unsupported
|
||||
4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement
|
||||
|
||||
1. **Chunk** — collection of files (~10 MB total by default).
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer.
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec` (plus
|
||||
atime/crtime fields). `uid`/`gid` are applied only through the opt-in
|
||||
identity path; atime is preserved with `-U`/`--atimes`; crtime is captured
|
||||
but cannot be set on the destination.
|
||||
4. **Config** — runtime parameters. Most cross the wire (TLS settings
|
||||
excluded); `backup` and `backup_dir` are in the serialized wire table, while
|
||||
`timeout`, `contimeout`, `quiet`, `stats`, `max_depth`, and `log_file` are
|
||||
client-only.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables.
|
||||
6. **DirectoryScanner** — recursive traversal that buffers and sorts each
|
||||
directory (non-directories ascending, then directories ascending) and walks
|
||||
depth-first in rsync flist order, with exclude and include pattern support and
|
||||
max-depth enforcement.
|
||||
|
||||
### Key Algorithms
|
||||
1. **File scanning** — BFS directory traversal;
|
||||
entries matched against exclude and include patterns,
|
||||
max - depth enforced 2. * *Chunking ** — files accumulated until `chunk_size` threshold,
|
||||
then flushed 3. *
|
||||
*Compression ** — streaming zstd
|
||||
via `ZSTD_compressStream2` / `ZSTD_decompressStream` 4. *
|
||||
*Network protocol ** — status -
|
||||
code - driven exchange with metadata packing,
|
||||
keep - alive,
|
||||
and abort support 5. * *Incremental check ** — client sends `STATUS_CHECK` + path + size +
|
||||
mtime and,
|
||||
with `--checksum`, XXH64 content checksum; server compares against destination. Can be batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with `nanosleep` throttling on 64 KB write chunks
|
||||
7. **Metadata restoration** — `chmod()`, `chown()`, `utimensat()` on the receiving side
|
||||
8. **`--delete`** — sender tracks all sent paths;
|
||||
receiver walks destination tree and removes unlisted files / directories 9. *
|
||||
*SSH transport *
|
||||
* — `socketpair()` + `fork()` + `execvp("ssh",
|
||||
...)` with `ControlMaster` and port support
|
||||
10. *
|
||||
*TLS transport ** — OpenSSL `SSL_CTX` with TLS
|
||||
1.2 minimum,
|
||||
mutual CA verification,
|
||||
transparent `SSL_read`/`SSL_write` via `io_set_ssl()` 11. *
|
||||
*Path traversal protection ** — `has_path_traversal()` rejects any file path
|
||||
containing `..` components,
|
||||
preventing directory escape attacks 12. *
|
||||
*Connection limiting ** — server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100)13. *
|
||||
*Keep
|
||||
- alive ** — idle connections receive periodic `STATUS_KEEPALIVE` to detect half
|
||||
- open TCP connections 14. * *Abort handling ** — `SIGINT` sets an abort flag; the next protocol operation sends `STATUS_ABORT` for clean server cleanup
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically renamed via `rename()`, preventing partial files
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir` (or same directory with `~` suffix) preserving the original
|
||||
|
||||
1. **File scanning** — sorted depth-first traversal in rsync flist order (each
|
||||
directory's non-directories ascending, then its directories ascending);
|
||||
entries matched against exclude and include patterns, with max-depth
|
||||
enforced. The `--threads` parallel scanner remains unordered.
|
||||
2. **Chunking** — files accumulated until the `chunk_size` threshold (default
|
||||
10 MiB) is reached, then flushed.
|
||||
3. **Compression** — streaming zstd via `ZSTD_compressStream2()` /
|
||||
`ZSTD_decompressStream()`, with lz4 and zlib/zlibx codecs also supported
|
||||
(selectable with `--compress-choice`).
|
||||
4. **Network protocol** — status-code-driven exchange with metadata packing,
|
||||
keep-alive, and abort support.
|
||||
5. **Incremental check** — the client sends `STATUS_CHECK` + path + size +
|
||||
mtime and, with `--checksum`, a whole-file content checksum (`xxh128` by
|
||||
default; selectable via `--checksum-choice`/`--cc`, seeded by
|
||||
`--checksum-seed`); the server compares against the destination. Can be
|
||||
batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with sleep throttling on
|
||||
64 KiB write chunks.
|
||||
7. **Metadata restoration** — mode via `chmod()`/`fchmod()`, times via
|
||||
`utimensat()`/`futimens()`, and ownership only with an identity flag via
|
||||
fd-relative `fchown()`/`fchownat()`.
|
||||
8. **`--delete`** — the sender tracks all sent paths; the receiver walks the
|
||||
destination tree and removes unlisted files and directories.
|
||||
9. **SSH transport** — `socketpair()` + `fork()` + `execvp("ssh", ...)` with
|
||||
`ControlMaster` and port support.
|
||||
10. **TLS transport** — OpenSSL `SSL_CTX` with TLS 1.2 minimum, mutual CA
|
||||
verification, and transparent `SSL_read()`/`SSL_write()` via
|
||||
`io_set_ssl()`.
|
||||
11. **Path traversal protection** — `has_path_traversal()` rejects any file
|
||||
path containing `..` components, preventing directory escape attacks.
|
||||
12. **Connection limiting** — the server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100).
|
||||
13. **Keep-alive** — idle connections receive periodic `STATUS_KEEPALIVE` to
|
||||
detect half-open TCP connections.
|
||||
14. **Abort handling** — `SIGINT` sets an abort flag; the next protocol
|
||||
operation sends `STATUS_ABORT` for clean server cleanup.
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically
|
||||
renamed via `rename()`, preventing partial files.
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir`
|
||||
(or the same directory with a `~` suffix), preserving the original.
|
||||
|
||||
## Security Features
|
||||
|
||||
@@ -264,6 +387,8 @@ Received files are written to a temporary path (suffixed with `.tmp`) and then a
|
||||
- C11 compiler
|
||||
- CMake >= 3.22
|
||||
- zstd library
|
||||
- zlib library
|
||||
- lz4 library
|
||||
- OpenSSL (development headers and libraries)
|
||||
- pthreads
|
||||
- SSH client (for SSH transport mode only)
|
||||
@@ -272,12 +397,12 @@ Received files are written to a temporary path (suffixed with `.tmp`) and then a
|
||||
|
||||
**Ubuntu/Debian:**
|
||||
```bash
|
||||
sudo apt install cmake build-essential libzstd-dev libssl-dev openssh-client
|
||||
sudo apt install cmake build-essential libzstd-dev zlib1g-dev liblz4-dev libssl-dev openssh-client
|
||||
```
|
||||
|
||||
**Nix:**
|
||||
```bash
|
||||
nix-shell # provides zstd, openssl, cmake, gcc
|
||||
nix-shell # provides zstd, zlib, lz4, openssl, cmake, gcc
|
||||
```
|
||||
|
||||
## Building
|
||||
@@ -297,22 +422,34 @@ cmake --build build -j$(nproc)
|
||||
|
||||
### SSH transfer
|
||||
|
||||
The remote host must have `fastsync-server` available in `PATH`, or use
|
||||
The remote host must have `fastsync-server` available in `PATH` (install or
|
||||
copy the built `./build/server` there as `fastsync-server`), or use
|
||||
`--fastsync-server-path`. SSH starts `fastsync-server --stdio` in its remote
|
||||
working directory, so use a destination below that directory unless the
|
||||
remote server is otherwise configured with a matching authorized root.
|
||||
|
||||
The remote `--stdio` server argv is composed by the client, so it must never
|
||||
be trusted to opt a root receiver into super-user activities: `--allow-super`
|
||||
is rejected with `--stdio` and super stays off on that path. Operators
|
||||
exposing `fastsync-server --stdio` over SSH must use a forced command (e.g. an
|
||||
`authorized_keys` `command=` entry) if the default must hold.
|
||||
|
||||
```bash
|
||||
ssh user@host 'mkdir -p destination'
|
||||
./build/client /path/to/source user@host:destination
|
||||
```
|
||||
|
||||
FastSync is **push-only**: the source (first argument) is always a local
|
||||
directory and only the destination may be remote. A remote source such as
|
||||
`client user@host:src ./local` (a "pull") is intentionally not supported; see
|
||||
[RSYNC_COMPAT.md](RSYNC_COMPAT.md#direction).
|
||||
|
||||
### TCP transfer
|
||||
|
||||
Start the FastSync server:
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to -p 8080
|
||||
./build/server --destination-root /path/to -p 8080 --allow-unauthenticated
|
||||
```
|
||||
|
||||
Then run the client:
|
||||
@@ -327,8 +464,13 @@ Plain TCP requires the explicit `--allow-unauthenticated` server option. Use TLS
|
||||
authenticated network connections.
|
||||
|
||||
### TLS transfer
|
||||
|
||||
Server TLS requires `--cert`, `--key`, `--ca`, and `--client-cn`; the client
|
||||
requires `--cert`, `--key`, and `--ca`.
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem -p 8443
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem \
|
||||
--ca ca.pem --client-cn client -p 8443
|
||||
./build/client --tls --cert client.pem --key client-key.pem --ca ca.pem \
|
||||
--server-host example.com --server-port 8443 \
|
||||
--source-dir /path/to/source --dest-dir /path/to/destination \
|
||||
@@ -341,32 +483,32 @@ These examples show the intended rsync-style workflow. Options marked as
|
||||
FastSync-native are optional performance or transport extensions.
|
||||
|
||||
```bash
|
||||
#Basic synchronization
|
||||
# Basic synchronization
|
||||
./build/client /source/ /destination/
|
||||
|
||||
#Archive - style synchronization(current FastSync archive behavior)
|
||||
# Archive-style synchronization (current FastSync archive behavior)
|
||||
./build/client -a /source/ user@host:destination/
|
||||
|
||||
#Preview a transfer without changing the destination
|
||||
# Preview a transfer without changing the destination
|
||||
./build/client -n /source/ /destination/
|
||||
|
||||
#Exclude temporary and object files
|
||||
# Exclude temporary and object files
|
||||
./build/client --exclude '*.tmp' --exclude '*.o' \
|
||||
/source/ user@host:destination/
|
||||
|
||||
#Remove destination entries not present in the source
|
||||
# Remove destination entries not present in the source
|
||||
./build/client --delete /source/ user@host:destination/
|
||||
|
||||
#Skip unchanged files using size and modification time
|
||||
# Skip unchanged files using size and modification time
|
||||
./build/client --incremental /source/ user@host:destination/
|
||||
|
||||
#Verify content when size and time are not sufficient
|
||||
# Verify content when size and time are not sufficient
|
||||
./build/client --incremental --checksum /source/ user@host:destination/
|
||||
|
||||
#Preserve supported mode and timestamp metadata
|
||||
./build/client -M /source/ user@host:destination/
|
||||
# Preserve supported mode and timestamp metadata
|
||||
./build/client --preserve /source/ user@host:destination/
|
||||
|
||||
#Keep backups of overwritten destination files
|
||||
# Keep backups of overwritten destination files
|
||||
./build/client --backup --backup-dir backups \
|
||||
/source/ user@host:destination/
|
||||
```
|
||||
@@ -379,11 +521,11 @@ features without changing the meaning of ordinary compatibility options.
|
||||
| Option | Purpose |
|
||||
|---|---|
|
||||
| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming zstd compression, levels 1-22. |
|
||||
| `--compress-level <n>` | Set the zstd compression level. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` and `none`. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming compression (default `zstd`), levels 1-22. |
|
||||
| `--compress-level <n>` | Set the compression level (1-22). Omitted, each codec uses its rsync default: zstd 3, zlib/zlibx 6, lz4 ignored. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, and `auto`; `zlib`/`zlibx` share the same literal-only zlib path. |
|
||||
| `--zl <n>` | Alias for `--compress-level`. |
|
||||
| `--skip-compress <list>` | Skip compression for comma-separated suffixes; incompatible with `--chunk-serialization`. |
|
||||
| `--skip-compress <list>` | Skip compression for `/`- or `,`-separated suffixes; defaults to rsync 3.4.1's built-in list. Incompatible with `--chunk-serialization`. |
|
||||
| `--compress-threads <n>` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. |
|
||||
| `--chunk-size <bytes>` | Set the transfer chunk size. |
|
||||
| `--chunk-serialization` | Enable FastSync chunk serialization (long form only; `-s` is rsync's `--secluded-args`). |
|
||||
@@ -394,11 +536,11 @@ features without changing the meaning of ordinary compatibility options.
|
||||
| `--server-host <host>` | Select the TCP server host. |
|
||||
| `--server-port <port>` | Select the TCP server port (`--port <port>` and `--port=<port>` are rsync-friendly aliases). |
|
||||
| `--tls` | Enable TLS for TCP transport. |
|
||||
| `--bwlimit <KB/s>` | Apply token-bucket bandwidth limiting. |
|
||||
| `--progress` | Show transfer progress and throughput. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout (positive seconds). Omit to keep the built-in 30 s socket / 60 s protocol defaults. |
|
||||
| `--contimeout <seconds>` | Set connection timeout. |
|
||||
| `--bwlimit <RATE>` | Apply token-bucket bandwidth limiting with rsync's exact `parse_size_arg` grammar (bare = KiB/s, `K`/`M`/`G`/`T`/`P` binary, `KB`/`MB` decimal, `KiB`/`MiB` binary, decimals quantized to whole KiB, `0`/empty = no limit; also paces `--sendfile` transfers). |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; `Number of files`/`Number of created files` carry rsync's per-type breakdown (deleted files are a single total). |
|
||||
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout. Default `0` = disabled (matching rsync); `0` disables it. |
|
||||
| `--contimeout <seconds>` | Connection timeout (default 60, matching rsync); `0` disables it. |
|
||||
|
||||
Short-option conflicts with rsync have been resolved for the CLI namespace
|
||||
(Phase 7): `-c` is now rsync's `--checksum`, `-m` is `--prune-empty-dirs`, `-M`
|
||||
@@ -407,7 +549,8 @@ is `--remote-option`, `-f` is `--filter`, `-s` is `--secluded-args`, `-p` is
|
||||
long-form-only or new shorts: multithreading is `-j`/`--threads`, metadata
|
||||
is `--preserve`, sendfile is `--sendfile`, chunk serialization is
|
||||
`--chunk-serialization`, timeout is `--timeout`, and SSH port is `--ssh-port`.
|
||||
`-a`/`--archive` is now real rsync archive (`-rlptgoD`).
|
||||
`-a`/`--archive` is now rsync archive `-rlptgoD` (owner/group implied, but the
|
||||
receiver still needs privilege to apply them).
|
||||
|
||||
`--secluded-args` (and its short form `-s`) is accepted as a compatibility
|
||||
no-op. It does not change FastSync's transport or protocol behavior, because
|
||||
@@ -419,42 +562,103 @@ remote SSH argv is already built injection-safe.
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials. |
|
||||
| `-n`, `--dry-run` | Scan and report without writing files. |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. |
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated. |
|
||||
| `-n`, `--dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer. |
|
||||
| `--incremental` | Skip files matching destination size and mtime. Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `-c, --checksum` | Verify content by checksum (implies the incremental quick-check). Algorithm selectable with `--checksum-choice`. |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh64`/`xxhash` (default), `xxh3`, `xxh128`, `md5`, or `auto`. |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync. |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime. |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match. |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver. |
|
||||
| `--ignore-existing` | Skip files that already exist on the receiver; like rsync it does not apply to directories or symlinks. |
|
||||
| `-@, --modify-window <sec>` | Modification-time tolerance (seconds) for the incremental/basis quick-check; `0` requires an exact mtime match. |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing (`--no-whole-file` clears it). |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`). |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents (aliases `--old-dirs`/`--old-d`). |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root. |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root). |
|
||||
| `-0, --from0` | Treat entries in `--files-from` files as NUL-delimited instead of newline-delimited. |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer (the fixed `.fastsync-stage` staging name diverges from rsync; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`; a basis of any size is supported, streamed in bounded chunks — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination (the copy is streamed, so a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win; a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--verify-basis` | FastSync-only: require a basis hit to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync). |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk). |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`). |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch). |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-during (matching rsync's `--del`): extras are removed per directory as the transfer proceeds, so destination space is freed progressively. Scoped to the synchronized directories, so `--files-from` subsets are safe. |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). |
|
||||
| `--delete-during`, `--del` | Delete each directory's extras as that directory is processed (implies `--delete`). Since protocol 2.24.0 the sender streams a per-directory `STATUS_DELETE_PLAN` frame as it reaches each source directory; this is also the default timing of a plain `--delete`. |
|
||||
| `--delete-delay` | Record extras per directory during the scan but remove them only after a successful transfer (implies `--delete`). Uses the same per-directory `STATUS_DELETE_PLAN` frames as `--delete-during`, applied late. |
|
||||
| `--delete-commit` | FastSync-only: atomic delete-after timing (only after the whole transfer succeeded). |
|
||||
| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected). |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync. |
|
||||
| `--force` | Allow an incoming file/symlink to replace a destination directory (also during `--delay-updates` publication). |
|
||||
| `--exclude <pattern>` | Exclude matching paths. Repeatable. |
|
||||
| `--include <pattern>` | Include matching paths. Repeatable. |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file. |
|
||||
| `--include-from <file>` | Read include patterns from a file. |
|
||||
| `-f, --filter=RULE` | Add an rsync-style filter rule (`+`/`-`, `include`/`exclude`, `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R`, `clear`/`!`, and modifiers; repeatable). |
|
||||
| `--max-size <bytes>` | Skip files larger than the limit. |
|
||||
| `--min-size <bytes>` | Skip files smaller than the limit. |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth;
|
||||
zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.|
|
||||
| `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` |
|
||||
Back up overwritten files.| | `--backup - dir<dir>` | Store backups under a separate directory.|
|
||||
| `--suffix<suffix>` | Set the backup filename suffix.| | `--partial` |
|
||||
Select partial - transfer handling. On failed/interrupted writes the
|
||||
already-written temp file is retained (best-effort) for resumption.|
|
||||
With `--partial --partial-dir <dir>`, completed files are written under the
|
||||
partial directory and installed atomically. | | `--partial - dir<dir>` |
|
||||
Set a relative partial - transfer directory below the server destination root.
|
||||
Use with `--partial`. |
|
||||
| `--inplace` | Write directly to the destination instead of using a temporary file. |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units; default 1G; `0` = no local limit). |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth; zero means unlimited. |
|
||||
| `-b, --backup` | Back up overwritten files. |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install (confined to the receive root: relative resolves below it, absolute must canonicalize inside it; `EXDEV` falls back to a non-atomic copy). |
|
||||
| `--backup-dir <dir>` | Store backups under a separate directory (requires `--backup`). |
|
||||
| `--suffix <suffix>` | Set the backup filename suffix (default: `~`). |
|
||||
| `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir <dir>`, completed files are written under the partial directory and installed atomically. |
|
||||
| `--partial-dir <dir>` | Set a relative partial-transfer directory below the server destination root. Implies `--partial`. Rejected together with `--inplace` (`--inplace cannot be used with --partial-dir`, matching rsync), because the inplace path bypasses partial/temp staging. |
|
||||
| `--inplace` | Write directly to the destination instead of using a temporary file. Cannot be combined with `--partial-dir`. |
|
||||
| `--fsync` | Fsync every written file before publication. |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree (FastSync-native format, not rsync-interoperable). |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server); FastSync-native format, not rsync-interoperable. |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server); FastSync-native format, not rsync-interoperable. |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes; whatever was already transferred is kept. |
|
||||
| `--stop-at=TIME` | Stop at an absolute time. Accepts rsync's `parse_time` forms (`Y-M-DTh:m`, `Y/M/DTh:m`, `Y-M-D`, `M-D`, `D`, `h:m`, `:m`, `T h:m`; omitted fields resolve to the next matching point in the local timezone), plus `now+N[smhd]` and FastSync's `HH:MM`/`HH:MM:SS` clock-time spelling. An early stop skips the late `--delete` keep-set. |
|
||||
|
||||
### Metadata and links
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--preserve` | Preserve supported file metadata, currently mode and modification time (long form only). |
|
||||
| `-l`, `--links` | Request symlink preservation;
|
||||
link-target transfer remains incomplete. |
|
||||
| `--copy-links` | Copy symlink referents. |
|
||||
| `--safe-links` | Skip symlinks that point outside the transfer tree. |
|
||||
| `--preserve` | Preserve mode and mtime (long form only; equivalent to `-p` + `-t`). Add `-o`/`-g` for owner/group, `-U`/`--atimes` for atime, or an identity flag (`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`) for mapped ownership. |
|
||||
| `-U`, `--atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N`, `--crtimes` | Capture birth time and transmit it; it cannot be applied because no portable filesystem call can set a birth time (documented divergence). |
|
||||
| `-p`, `--perms` | Preserve permission bits. One of the four per-attribute preserve flags (with `-t`/`-o`/`-g`); under `-p` the source mode is copied exactly (group/other-write included; setuid/setgid/sticky included only when super-user activities are permitted, masked under `SUPER_MODE_OFF`/`--no-super`), matching rsync otherwise. |
|
||||
| `-t`, `--times` | Preserve modification times. Independent of the other attributes; `-O`/`--omit-dir-times` suppresses directories only. |
|
||||
| `-O`, `--omit-dir-times` | Do not apply modification times to directories. |
|
||||
| `-J`, `--omit-link-times` | Do not apply times to symlinks. |
|
||||
| `--open-noatime` | Open source files with `O_NOATIME` so reading for a transfer does not update their access time (client-only). |
|
||||
| `-o`, `--owner` | Preserve the source owner (uid). Mapped by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); application is privilege-gated. |
|
||||
| `-g`, `--group` | Preserve the source group (gid). Same name-mapping/numeric-fallback and privilege gating as `-o`. |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group` | Negate each per-attribute flag (also `--no-p`/`--no-t`/`--no-o`/`--no-g`); `--no-preserve` clears all four. |
|
||||
| `-E`, `--executability` | Preserve executable permission bits. |
|
||||
| `-X`, `--xattrs` | Preserve user `user.*` extended attributes. |
|
||||
| `-A`, `--acls` | Preserve POSIX ACLs. |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax, including `D`/`F`/`X` selectors and `s`/`t`); does not imply `-p`. |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files (`USER:GROUP`, `USER`, or `:GROUP`); conflicts with `--usermap`/`--groupmap` on the same side. |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership (`FROM:TO` rules; names, ids, `LOW-HIGH` ranges, `*`, empty-`FROM`). |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership (same syntax as `--usermap`). |
|
||||
| `--numeric-ids` | Mapping modifier: apply the source numeric uid/gid directly instead of mapping by name (combine with `-o`/`-g`, `-a`, or a map). |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP]; requires a privileged receiver. |
|
||||
| `--fake-super` | Record the resolved owner plus full mode/rdev in rsync's reserved `user.rsync.%stat` xattr (rsync 3.4.1 grammar) and replay the permission bits; never performs a real chown. |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes). |
|
||||
| `--no-super` | Forbid those super-user activities even when the receiver is root. |
|
||||
| `-l`, `--links` | Copy symlinks as symlinks; the target is stored verbatim (absolute and `..`-bearing targets included), matching rsync. |
|
||||
| `-L`, `--copy-links` | Copy symlink referents (a broken referent makes the run exit 23, matching rsync). |
|
||||
| `--safe-links` | Skip symlinks whose target points outside the transfer tree (applied on the sender). |
|
||||
| `--copy-unsafe-links` | Copy unsafe symlink referents. |
|
||||
| `--munge-links` | Rewrite stored symlink targets with rsync's `/rsyncd-munged/` marker. |
|
||||
| `-k`, `--copy-dirlinks` | Treat a symlink to a directory as a real directory on the sender. |
|
||||
| `-K`, `--keep-dirlinks` | Follow an existing destination symlink-to-directory (confined to the receive root). |
|
||||
| `-H`, `--hard-links` | Preserve hard-link relationships across the transfer. |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`). |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`). |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets. |
|
||||
| `--copy-devices` | Copy a source device's content as an ordinary regular file on the destination (rsync's non-privileged safe mode) instead of recreating the device node. |
|
||||
| `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). |
|
||||
|
||||
### Output and logging
|
||||
@@ -462,9 +666,18 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--progress` | Show live transfer progress. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `-q`, `--quiet` | Suppress non-error output. |
|
||||
| `--progress` | Show rsync-style per-file progress blocks; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; `Number of files`/`Number of created files` carry rsync's per-type breakdown (deleted files are a single total). |
|
||||
| `-i, --itemize-changes` | Print an rsync-style per-file change line. |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %c %C %i %M %%`). |
|
||||
| `--list-only` | List source files instead of transferring. |
|
||||
| `--outbuf=MODE` | stdout/stderr buffering: `N` (none/unbuffered), `L` (line-buffered), or `B` (block-buffered, default). |
|
||||
| `--log-file <path>` | Write log output to a file. |
|
||||
| `--log-file-format=FORMAT` | Per-file log-line format (requires `--log-file`). |
|
||||
| `--stderr=MODE` | Route logging: `errors` (default), `all`, or `client` (forward the client's diagnostics to the server's stderr over the client-message channel). |
|
||||
| `--msgs2stderr` | Route all messages to stderr (deprecated spelling of `--stderr=all`). |
|
||||
| `--no-msgs2stderr` | Forward the client's diagnostics to the server (deprecated spelling of `--stderr=client`). |
|
||||
| `-V`, `--version` | Print the FastSync protocol version. |
|
||||
| `--help` | Print command usage. |
|
||||
|
||||
@@ -473,30 +686,61 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--ssh-port <port>` | SSH port for the SSH transport (default: 22). Note the short `-p` is now rsync's `--perms`. |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode. |
|
||||
| `-e`, `--rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments). |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode (client-only; never crosses the wire). |
|
||||
| `--rsync-path <path>` | Alias for `--fastsync-server-path`. |
|
||||
| `-M`, `--remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable; rejected for daemon/TCP destinations). |
|
||||
| `--trust-sender` | Receiver-local: trust the remote sender's file list and skip path re-validation (does not affect symlink targets). **On the client this flag alone is inert** — it is never sent on the wire; the server must be started with its own `--trust-sender`, or the client must forward it with `-M--trust-sender` (SSH only). |
|
||||
| `--timeout <sec>` | Socket + per-message I/O timeout; default `0` = disabled. |
|
||||
| `--contimeout <sec>` | Connection timeout; default 60; `0` disables. |
|
||||
| `--source-dir <path>` | Set the source directory explicitly. |
|
||||
| `--dest-dir <path>` | Set the destination directory explicitly. |
|
||||
| `--save-to-disk` | Enable server-side disk persistence. |
|
||||
| `--server-host <host>` | TCP server address. |
|
||||
| `--server-port <port>` | TCP server port. `--port <port>` / `--port=<port>` is an alias. |
|
||||
| `--tls` | Enable TLS. Requires `--cert` and `--key`. |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address. |
|
||||
| `-4`, `--ipv4` | Force IPv4 for destination resolution. |
|
||||
| `-6`, `--ipv6` | Force IPv6 for destination resolution. |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect. |
|
||||
| `--blocking-io` | SSH transport only: leave the socket without read/write timeouts so it blocks naturally (no effect on TCP). |
|
||||
| `--protocol=NUM` | Force the wire protocol version; must equal the current `PROTOCOL_VERSION` (FastSync cannot speak older/virtual wire formats). |
|
||||
| `--old-args` | Accepted for rsync CLI compatibility; no effect (the remote server path is always safely quoted). |
|
||||
| `--iconv=LOCAL[,REMOTE]` | Convert file-name charsets at the wire boundary (`LOCAL` is our names' charset, `REMOTE` the peer's, defaulting to `LOCAL`). |
|
||||
| `--no-iconv` | Disable `--iconv` charset conversion (same as `--iconv=-`). |
|
||||
| `--tls` | Enable TLS. Requires `--cert`, `--key`, and `--ca`. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--ca <path>` | CA file for peer verification (always required with `--tls`). |
|
||||
|
||||
## Server Options
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--stdio` | Serve one SSH connection over standard input/output. |
|
||||
| `-p <port>` | TCP listen port. |
|
||||
| `--daemon` | Run as a persistent daemon listener using a module config file; the daemon default port is 873 (unlike `-p`, which defaults to 8080). |
|
||||
| `--config=FILE` | Daemon config file (default: `~/.config/fastsync/fastsyncd.conf`, else `/etc/fastsyncd.conf`). Requires `--daemon`. |
|
||||
| `--dparam=KEY=VALUE` | Override one global config key on the command line. Requires `--daemon`. |
|
||||
| `--no-detach` | Stay in the foreground (default detaches to the background when running `--daemon`). |
|
||||
| `-p, --port <port>` | TCP listen port (default: 8080, range: 1–65535). |
|
||||
| `--tls` | Enable TLS. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root;
|
||||
defaults to the current directory. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. |
|
||||
| `--cert <path>` | TLS certificate file (PEM). |
|
||||
| `--key <path>` | TLS private key file (PEM). |
|
||||
| `--ca <path>` | CA file for peer verification (PEM). |
|
||||
| `--client-cn <name>` | TLS client certificate CN; mandatory with `--tls` (the server verifies the client CN). |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root; defaults to the current directory. |
|
||||
| `--address <addr>` | Bind the listening socket to this address. |
|
||||
| `-4`, `--ipv4` | Bind an IPv4 socket (default). |
|
||||
| `-6`, `--ipv6` | Bind an IPv6 socket. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. This also gates `--force` (which can recursively replace/remove a destination directory tree). |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. Rejected with `--stdio` (the SSH remote argv is client-composed; use a forced command if the default must hold). No effect when not root. Daemon modules opt in per module with `client owner = yes`. |
|
||||
| `--trust-sender` | Trust the remote sender's file list: skip the receiver's up-front path-traversal re-validation (fewer checks, faster, potentially unsafe; off by default). It does not affect symlink targets, which are stored verbatim either way. A client `--trust-sender` is never sent over the wire — the server must set this flag itself, or the client must forward it via `-M--trust-sender`. |
|
||||
| `--no-super` | Operator veto: never attempt super-user activities (ownership, device nodes) even as root, and refuse any client `--copy-as`/`--super` request. |
|
||||
| `--allow-unauthenticated` | Permit plaintext/anonymous network clients; an auth-required module still accepts only opted-in loopback plaintext. |
|
||||
| `--iconv=LOCAL[,REMOTE]` | Declare this server's LOCAL charset for file-name conversion. |
|
||||
| `--password-file=FILE` | Credential store for modules that declare `auth users`. Requires `--daemon`. FastSync-native SCRAM/PBKDF2 format, not rsync-interoperable. |
|
||||
| `--early-input=FILE` | Second credential store layered over `--password-file`. Requires `--daemon`. FastSync-native format, not rsync-interoperable. |
|
||||
| `--hash-credentials <file>` | Read `<file>`'s `user:password` lines and print PBKDF2 credential-store lines to stdout, then exit. Cannot be combined with `--daemon` or `--stdio`. FastSync-native, not rsync-interoperable. |
|
||||
| `--iterations N` | PBKDF2 iteration count for `--hash-credentials` (default 600000, range 100000–10000000). Requires `--hash-credentials`. FastSync-native, not rsync-interoperable. |
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--help` | Print server usage. |
|
||||
|
||||
@@ -506,18 +750,56 @@ defaults to the current directory. |
|
||||
implicit global section, then `[module]` sections). Besides `port`, `motd file`,
|
||||
and `address`, the global section accepts:
|
||||
|
||||
- `max connections = N` — cap on concurrent connections, default 100. The
|
||||
- `max connections = N` — global cap on concurrent connections, default 100. The
|
||||
listener enforces it; `0`, negative, and non-numeric values are parse errors.
|
||||
- `max connections per host = N` — cap on concurrent connections from a single
|
||||
source IP, default 0 (unlimited). Enforced across all forked connection
|
||||
children through a shared registry.
|
||||
- `auth failure delay = MS` — milliseconds to sleep after a failed
|
||||
authentication, default 500. `0` disables it and the value is capped at 60000,
|
||||
authentication, default 500. `0` disables it and the value is capped at 5000,
|
||||
so online password guessing is rate-limited per connection. Successful auths
|
||||
are never delayed.
|
||||
- `auth lockout threshold = N` — number of failed authentications from one source
|
||||
IP before that source is locked out, default 10; `0` disables the lockout. The
|
||||
failure counter is shared across every connection child, so the lockout holds
|
||||
even when the next attempt is handled by a different forked child.
|
||||
- `auth lockout duration = SECONDS` — how long a locked-out source is refused
|
||||
(default 300). A locked-out client is refused before any SCRAM challenge is
|
||||
sent; a successful authentication clears the counter.
|
||||
- `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access
|
||||
patterns.
|
||||
|
||||
A `[module]` may also set `max connections` (parsed and validated but not
|
||||
enforced per module — the global cap applies to the whole listener) and its own
|
||||
`hosts allow`/`hosts deny`.
|
||||
A `[module]` requires `path`, and may also set `read only`, `write only`,
|
||||
`client owner`, `auth users`, `max connections` (0 = unlimited; enforced per
|
||||
module across all connection children), and its own `hosts allow`/`hosts deny`.
|
||||
|
||||
Like rsync, a module is **read-only by default**: a bare `[module]` with only a
|
||||
`path` refuses a write transfer. Opt a module into writability explicitly with
|
||||
`read only = no` or `write only = yes`; a global `read only` value in the
|
||||
section before the first `[module]` sets the default for later modules, and a
|
||||
module's own `read only`/`write only = yes` always wins over it. An
|
||||
rsync-style `write only = yes` is mapped to writability because FastSync is
|
||||
push-only (a module can never be read from the network).
|
||||
|
||||
The per-host cap and the shared auth lockout identify a source by its numeric
|
||||
peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local
|
||||
client shares that one address, so counting or locking them out would let one
|
||||
local process deny service to all the others. The per-module and global
|
||||
`max connections` caps still apply to loopback. Because the key is the peer IP,
|
||||
`max connections per host` and `auth lockout` also cannot distinguish clients
|
||||
behind the same NAT, proxy, or reverse-proxy address — they share one budget and
|
||||
one lockout counter, so an over-aggressive lockout can affect unrelated users
|
||||
behind that address. Prefer TLS client certificates (`--client-cn`) plus
|
||||
`hosts allow`/`hosts deny` for per-client policy when clients share an address,
|
||||
and size `auth lockout threshold` accordingly.
|
||||
|
||||
The shared per-source table has a bounded lifetime: an entry with no live
|
||||
connection is reclaimed once its lockout has expired, or after it has been idle
|
||||
(300 s). If every entry is still live or locked, a new source is admitted without
|
||||
per-host accounting (fail open) and a rate-limited warning is logged; the
|
||||
per-module cap and host ACLs still apply. The occupancy counters are re-derived
|
||||
from the shared slot table after every child exit, so a child killed mid-transfer
|
||||
(or mid-registration) cannot leak a slot or an occupancy count.
|
||||
|
||||
Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR
|
||||
(`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs
|
||||
@@ -551,7 +833,7 @@ before the module list, before authentication, and the connecting peer address
|
||||
|
||||
## Protocol and Security
|
||||
|
||||
FastSync protocol version `2.20.0` is shared by the client and server. The
|
||||
FastSync protocol version `2.30.0` is shared by the client and server. The
|
||||
current protocol is sender-driven and includes configuration negotiation,
|
||||
including the maximum allocation limit, incremental checks, checksums,
|
||||
manifests, keep-alives, abort handling, per-file remove-source results, and
|
||||
@@ -603,10 +885,9 @@ mandates `--client-cn`, so a TLS connection to an auth-required module always
|
||||
has its client CN verified (`--client-cn` matches the certificate's CN only, not
|
||||
a subjectAltName, which is acceptable for a private CA).
|
||||
|
||||
TLS provides encrypted TCP transport. Supplying `--ca` enables certificate
|
||||
verification; without it, traffic is encrypted but peer identity is not
|
||||
verified. Use certificate verification for deployments where authentication
|
||||
matters. The default TCP transport is not encrypted.
|
||||
TLS provides encrypted TCP transport. Both the client and the server require
|
||||
`--ca` together with `--tls`, so peer certificates are always verified
|
||||
(`SSL_VERIFY_PEER`, depth 4). The default TCP transport is not encrypted.
|
||||
|
||||
The receiver protects its destination root with path validation, `openat()`
|
||||
directory traversal, `O_NOFOLLOW`, temporary files, and atomic renames. Delete
|
||||
@@ -617,18 +898,29 @@ operations require the server's explicit `--allow-delete` policy.
|
||||
The project will reach the drop-in replacement goal in stages:
|
||||
|
||||
1. Correct rsync option meanings, including short options, combined options,
|
||||
and `--option=value` syntax.
|
||||
and `--option=value` syntax — **done** in the rsync-parity wave: `-r`/`-b`/
|
||||
`-L`/`-B`, short-option clustering (`-av`, `-aAX`, `-rlpt`), and attached
|
||||
values (`-B1000`, `-essh`, `-MOPT`) all parse.
|
||||
2. Add differential tests that compare FastSync and rsync contents, metadata,
|
||||
links, deletes, filters, dry runs, and exit codes.
|
||||
3. Make `-a` implement the expected recursive, links, permissions, times,
|
||||
owner/group, and supported special-file behavior.
|
||||
4. Complete symlink, sparse-file, metadata, delete-policy, and resumable-write
|
||||
semantics.
|
||||
links, deletes, filters, dry runs, and exit codes — **done** for the
|
||||
completion wave's scope; the tests live in `tests/integration/` and skip
|
||||
cleanly when rsync is unavailable.
|
||||
3. `-a` implements full rsync `-rlptgoD`; under `-p` the source mode is copied
|
||||
exactly, including group/other-write bits, with setuid/setgid/sticky copied
|
||||
only when super-user activities are permitted (masked under
|
||||
`SUPER_MODE_OFF`/`--no-super`). Ownership application stays privilege-gated,
|
||||
as in rsync.
|
||||
4. Symlink (verbatim storage), sparse-file, metadata, delete-policy (including
|
||||
`--max-delete` partial + exit 25, per-directory `--delete-during`/
|
||||
`--delete-delay`), codecs, and resumable-write semantics are implemented;
|
||||
remaining work is the documented edge cases, which the **Parity Completion
|
||||
Wave** section of `RSYNC_COMPAT.md` enumerates honestly.
|
||||
5. Add rsync remote-shell and daemon protocol interoperability.
|
||||
6. Keep FastSync performance options as negotiated, optional extensions.
|
||||
|
||||
The exhaustive implementation matrix and compatibility notes are in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md); each row is classified as parity, caveat,
|
||||
or divergent.
|
||||
|
||||
## Testing
|
||||
|
||||
@@ -641,7 +933,7 @@ Run the unit test binary:
|
||||
Run the Python integration suite:
|
||||
|
||||
```bash
|
||||
python3 -m pytest tests/
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
For stricter local validation:
|
||||
@@ -665,10 +957,13 @@ rsync protocol or filesystem-semantic compatibility.
|
||||
|
||||
## Performance Guidance
|
||||
|
||||
- Use `-m` for workloads with many files or enough CPU parallelism.
|
||||
- Use `-c` or `-z` when network bandwidth is more constrained than CPU.
|
||||
- Use `-j`/`--threads` for workloads with many files or enough CPU parallelism
|
||||
(`-m` is `--prune-empty-dirs`).
|
||||
- Use `-z` when network bandwidth is more constrained than CPU (`-c` is
|
||||
`--checksum`, not a bandwidth option).
|
||||
- Tune `--chunk-size` for file sizes, memory limits, and network latency.
|
||||
- Use `-f` for large uncompressed TCP transfers where zero-copy I/O helps.
|
||||
- Use `--sendfile` for large uncompressed TCP transfers where zero-copy I/O
|
||||
helps (`-f` is `--filter`).
|
||||
- Use `--incremental` to avoid retransmitting unchanged files.
|
||||
- Use `--delta` for changed files when both endpoints are FastSync peers.
|
||||
- Use `--bwlimit` when sharing a link with other traffic.
|
||||
|
||||
+733
-283
File diff suppressed because it is too large.
Load diff
+399
-62
@@ -3,19 +3,24 @@
|
||||
|
||||
Compares FastSync configs against rsync (no compression) and rsync+zstd.
|
||||
Data is ~75% random/incompressible and ~25% structured/compressible by default,
|
||||
controllable via --random-ratio.
|
||||
controllable via --random-ratio. Transfers are verified by default (source and
|
||||
destination must match) so a fast-but-broken copy is never counted.
|
||||
|
||||
Usage:
|
||||
python3 benchmark/bench.py
|
||||
python3 benchmark/bench.py --runs 5 --profiles lan wan
|
||||
python3 benchmark/bench.py --random-ratio 0.5 --size-mb 50
|
||||
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
|
||||
python3 benchmark/bench.py --warm --runs 3
|
||||
python3 benchmark/bench.py --output json
|
||||
"""
|
||||
import argparse
|
||||
import filecmp
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import random
|
||||
import shlex
|
||||
import shutil
|
||||
import socket
|
||||
import statistics
|
||||
@@ -25,7 +30,10 @@ import tempfile
|
||||
import time
|
||||
|
||||
PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, "build")
|
||||
DEFAULT_BUILD_DIR = "build-bench"
|
||||
# Populated by configure_build_dirs(); default to the dedicated bench dir so
|
||||
# importing this module never depends on the user's existing build/ tree.
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, DEFAULT_BUILD_DIR)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
BENCH_DIR = os.path.join(PROJECT_ROOT, "bench_data")
|
||||
@@ -57,6 +65,7 @@ RSYNC_CONFIGS = [
|
||||
{"name": "rsync -z --zstd", "flags": ["-z", "--zc", "zstd"],"tool": "rsync"},
|
||||
]
|
||||
|
||||
|
||||
class RsyncDaemon:
|
||||
"""Manages an rsync daemon for network-fair benchmarking."""
|
||||
|
||||
@@ -118,6 +127,12 @@ STRUCTURED_FILES = {
|
||||
"nested/another.txt": b"another nested file\n" * 50,
|
||||
}
|
||||
|
||||
# Repeated text used to synthesize genuinely compressible filler of any size.
|
||||
COMPRESSIBLE_TEXT = (
|
||||
b"FastSync benchmark payload: the quick brown fox jumps over the lazy dog. "
|
||||
b"0123456789 ABCDEFGHIJKLMNOPQRSTUVWXYZ abcdefghijklmnopqrstuvwxyz\n"
|
||||
)
|
||||
|
||||
|
||||
class Progress:
|
||||
"""Simple progress bar with ETA."""
|
||||
@@ -152,35 +167,127 @@ class Progress:
|
||||
sys.stderr.flush()
|
||||
|
||||
|
||||
def write_compressible(path, nbytes):
|
||||
"""Write exactly nbytes of highly compressible, repeated text content."""
|
||||
if nbytes <= 0:
|
||||
return
|
||||
block = COMPRESSIBLE_TEXT * (max(1, 8192 // len(COMPRESSIBLE_TEXT)) + 1)
|
||||
remaining = nbytes
|
||||
with open(path, "wb") as f:
|
||||
while remaining > 0:
|
||||
piece = block if remaining >= len(block) else block[:remaining]
|
||||
f.write(piece)
|
||||
remaining -= len(piece)
|
||||
|
||||
|
||||
def generate_bench_data(source_dir, size_mb=25, random_ratio=0.75):
|
||||
"""Generate test data. ~random_ratio is incompressible, rest is structured."""
|
||||
"""Generate test data honouring the requested random/compressible split.
|
||||
|
||||
Exactly ``random_ratio * target`` bytes are incompressible random data and
|
||||
the remainder is genuinely compressible structured/repeated content. The
|
||||
measured byte counts are returned so callers can report the real mix.
|
||||
"""
|
||||
if os.path.exists(source_dir):
|
||||
shutil.rmtree(source_dir)
|
||||
os.makedirs(source_dir)
|
||||
|
||||
target = size_mb * 1024 * 1024
|
||||
structured_budget = int(target * (1 - random_ratio))
|
||||
written = 0
|
||||
random_budget = int(target * random_ratio)
|
||||
compressible_budget = target - random_budget
|
||||
compressible_written = 0
|
||||
random_written = 0
|
||||
files = 0
|
||||
|
||||
# A handful of fixed, human-meaningful files (directories, small files, a
|
||||
# binary blob) as long as they fit inside the compressible budget.
|
||||
for rel_path, content in STRUCTURED_FILES.items():
|
||||
if written >= structured_budget:
|
||||
if compressible_written + len(content) > compressible_budget:
|
||||
break
|
||||
full_path = os.path.join(source_dir, rel_path)
|
||||
os.makedirs(os.path.dirname(full_path), exist_ok=True)
|
||||
with open(full_path, "wb") as f:
|
||||
f.write(content)
|
||||
written += len(content)
|
||||
compressible_written += len(content)
|
||||
files += 1
|
||||
|
||||
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
|
||||
i = 0
|
||||
while written < target:
|
||||
chunk_size = min(5 * 1024 * 1024, target - written)
|
||||
with open(os.path.join(source_dir, f"bulk/file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk_size))
|
||||
written += chunk_size
|
||||
i += 1
|
||||
# Fill the rest of the compressible share with generated repeated content.
|
||||
if compressible_written < compressible_budget:
|
||||
os.makedirs(os.path.join(source_dir, "compressible"), exist_ok=True)
|
||||
i = 0
|
||||
while compressible_written < compressible_budget:
|
||||
chunk = min(1024 * 1024, compressible_budget - compressible_written)
|
||||
write_compressible(os.path.join(source_dir, "compressible", f"text_{i}.dat"), chunk)
|
||||
compressible_written += chunk
|
||||
files += 1
|
||||
i += 1
|
||||
|
||||
return written
|
||||
# Incompressible share.
|
||||
if random_written < random_budget:
|
||||
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
|
||||
i = 0
|
||||
while random_written < random_budget:
|
||||
chunk = min(5 * 1024 * 1024, random_budget - random_written)
|
||||
with open(os.path.join(source_dir, "bulk", f"file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk))
|
||||
random_written += chunk
|
||||
files += 1
|
||||
i += 1
|
||||
|
||||
return {
|
||||
"total_bytes": compressible_written + random_written,
|
||||
"compressible_bytes": compressible_written,
|
||||
"random_bytes": random_written,
|
||||
"files": files,
|
||||
}
|
||||
|
||||
|
||||
def list_relative_files(root):
|
||||
"""Return the set of file paths (relative to root) under a directory."""
|
||||
found = set()
|
||||
for dirpath, _dirnames, filenames in os.walk(root):
|
||||
for name in filenames:
|
||||
full = os.path.join(dirpath, name)
|
||||
found.add(os.path.relpath(full, root))
|
||||
return found
|
||||
|
||||
|
||||
def verify_transfer(source_dir, dest_dir):
|
||||
"""Recursively check dest matches source (paths, sizes, content).
|
||||
|
||||
Returns (ok, detail). Content is compared byte-for-byte, never hashed, so
|
||||
collisions are impossible. This is intentionally not part of the timing.
|
||||
"""
|
||||
if not os.path.isdir(dest_dir):
|
||||
return False, "destination directory missing"
|
||||
src_files = list_relative_files(source_dir)
|
||||
dst_files = list_relative_files(dest_dir)
|
||||
if src_files != dst_files:
|
||||
missing = src_files - dst_files
|
||||
extra = dst_files - src_files
|
||||
return False, f"path set mismatch (missing {len(missing)}, extra {len(extra)})"
|
||||
for rel in sorted(src_files):
|
||||
src = os.path.join(source_dir, rel)
|
||||
dst = os.path.join(dest_dir, rel)
|
||||
if os.path.getsize(src) != os.path.getsize(dst):
|
||||
return False, f"size mismatch: {rel}"
|
||||
if not filecmp.cmp(src, dst, shallow=False):
|
||||
return False, f"content mismatch: {rel}"
|
||||
return True, ""
|
||||
|
||||
|
||||
def percentile(values, pct):
|
||||
"""Linear-interpolation percentile (matches numpy's default method)."""
|
||||
if not values:
|
||||
return None
|
||||
ordered = sorted(values)
|
||||
if len(ordered) == 1:
|
||||
return ordered[0]
|
||||
rank = (len(ordered) - 1) * (pct / 100.0)
|
||||
low = math.floor(rank)
|
||||
high = math.ceil(rank)
|
||||
if low == high:
|
||||
return ordered[int(rank)]
|
||||
return ordered[low] + (ordered[high] - ordered[low]) * (rank - low)
|
||||
|
||||
|
||||
def find_free_port():
|
||||
@@ -208,18 +315,45 @@ def wait_proc(proc, timeout=5):
|
||||
proc.wait()
|
||||
|
||||
|
||||
def _tc_base_cmd():
|
||||
"""Return the command prefix for tc, honouring root vs sudo."""
|
||||
tc = shutil.which("tc")
|
||||
if not tc:
|
||||
raise RuntimeError(
|
||||
"tc (iproute2) not found in PATH; install iproute2 to use network profiles")
|
||||
if os.geteuid() == 0:
|
||||
return [tc]
|
||||
sudo = shutil.which("sudo")
|
||||
if sudo:
|
||||
return [sudo, tc]
|
||||
raise RuntimeError(
|
||||
"applying network limits requires root or sudo; "
|
||||
"re-run as root or install sudo")
|
||||
|
||||
|
||||
def _run_tc(args, check=True):
|
||||
return subprocess.run(_tc_base_cmd() + args, check=check, capture_output=True)
|
||||
|
||||
|
||||
def netem_apply(delay=None, jitter=None, throughput=None, loss=None):
|
||||
"""Apply tc/netem rules to loopback. Pass None to skip a parameter."""
|
||||
netem_reset()
|
||||
cmd = ["sudo", "tc", "qdisc", "add", "dev", "lo", "root", "netem"]
|
||||
params = []
|
||||
if throughput:
|
||||
cmd += ["rate", throughput]
|
||||
params += ["rate", throughput]
|
||||
if delay:
|
||||
cmd += ["delay", delay, jitter or "0ms"]
|
||||
params += ["delay", delay, jitter or "0ms"]
|
||||
if loss:
|
||||
cmd += ["loss", loss]
|
||||
if len(cmd) > 6:
|
||||
subprocess.run(cmd, check=True, capture_output=True)
|
||||
params += ["loss", loss]
|
||||
if not params:
|
||||
return
|
||||
try:
|
||||
_run_tc(["qdisc", "add", "dev", "lo", "root", "netem"] + params)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
detail = exc.stderr.decode(errors="replace").strip() if exc.stderr else str(exc)
|
||||
raise RuntimeError(f"failed to apply network profile via tc/netem: {detail}") from exc
|
||||
except RuntimeError:
|
||||
raise
|
||||
|
||||
|
||||
def netem_apply_profile(profile_name):
|
||||
@@ -236,7 +370,11 @@ def netem_apply_profile(profile_name):
|
||||
|
||||
|
||||
def netem_reset():
|
||||
subprocess.run("sudo tc qdisc del dev lo root".split(), capture_output=True)
|
||||
"""Best-effort removal of any loopback qdisc. Always safe to call."""
|
||||
try:
|
||||
_run_tc(["qdisc", "del", "dev", "lo", "root"], check=False)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def run_fastsync(source_dir, dest_dir, flags, port):
|
||||
@@ -287,7 +425,81 @@ def run_transfer(config, source_dir, dest_dir, port=None, rsync_daemon=None):
|
||||
return run_fastsync(source_dir, dest_dir, config["flags"], port)
|
||||
|
||||
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=None):
|
||||
def apply_incremental_changes(source_dir, target_bytes):
|
||||
"""Add and modify a few files so a warm transfer has real work to do.
|
||||
|
||||
Returns a mutation record (changed byte count plus enough data to revert
|
||||
and re-apply it) so every warm run can start from a pristine source.
|
||||
"""
|
||||
modified_n = 3
|
||||
added_n = 2
|
||||
per_file = max(4096, target_bytes // (modified_n + added_n))
|
||||
modified = {}
|
||||
added = {}
|
||||
changed = 0
|
||||
|
||||
existing = sorted(list_relative_files(source_dir))
|
||||
if existing:
|
||||
step = max(1, len(existing) // modified_n)
|
||||
for rel in existing[::step][:modified_n]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
original_size = os.path.getsize(path)
|
||||
with open(path, "ab") as f:
|
||||
f.write(random.randbytes(per_file))
|
||||
modified[rel] = (original_size, per_file)
|
||||
changed += per_file
|
||||
|
||||
for i in range(added_n):
|
||||
os.makedirs(os.path.join(source_dir, "incremental"), exist_ok=True)
|
||||
rel = os.path.join("incremental", f"new_{i}.dat")
|
||||
write_compressible(os.path.join(source_dir, rel), per_file)
|
||||
added[rel] = per_file
|
||||
changed += per_file
|
||||
|
||||
return {"changed": changed, "modified": modified, "added": added}
|
||||
|
||||
|
||||
def revert_incremental_changes(source_dir, mutation):
|
||||
"""Undo apply_incremental_changes so the source is pristine again."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (original_size, _appended) in mutation["modified"].items():
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
with open(path, "r+b") as f:
|
||||
f.truncate(original_size)
|
||||
for rel in mutation["added"]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
os.remove(path)
|
||||
|
||||
|
||||
def reapply_incremental_changes(source_dir, mutation):
|
||||
"""Re-apply a mutation after an untimed pristine seed transfer."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (_original_size, appended) in mutation["modified"].items():
|
||||
with open(os.path.join(source_dir, rel), "ab") as f:
|
||||
f.write(random.randbytes(appended))
|
||||
for rel, size in mutation["added"].items():
|
||||
write_compressible(os.path.join(source_dir, rel), size)
|
||||
|
||||
|
||||
def expected_received_root(dest_dir, source_dir, tool):
|
||||
"""Where a tool places transferred files inside dest_dir.
|
||||
|
||||
FastSync mirrors the absolute source path under dest_dir (see the
|
||||
integration suite's get_dest_received_dir); rsync copies the source tree
|
||||
contents directly into dest_dir.
|
||||
"""
|
||||
if tool == "rsync":
|
||||
return dest_dir
|
||||
return os.path.join(dest_dir, os.path.abspath(source_dir).lstrip(os.sep))
|
||||
|
||||
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
measure_bytes, verify=True, warm=False, mutation=None,
|
||||
progress=None):
|
||||
"""Run benchmark for all configs, returns list of results."""
|
||||
is_limited = profile_name != "unlimited"
|
||||
has_rsync = any(c["tool"] == "rsync" for c in configs)
|
||||
@@ -303,7 +515,10 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
results = []
|
||||
for config in configs:
|
||||
times = []
|
||||
invalid = 0
|
||||
for run_idx in range(runs):
|
||||
if warm:
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
if os.path.exists(dest_dir):
|
||||
shutil.rmtree(dest_dir)
|
||||
os.makedirs(dest_dir, exist_ok=True)
|
||||
@@ -311,15 +526,33 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
port = find_free_port()
|
||||
server = None
|
||||
try:
|
||||
if config["tool"] == "fastsync":
|
||||
if config["tool"] == "fastsync" or warm:
|
||||
server = subprocess.Popen(
|
||||
SERVER_CMD + ["-p", str(port)],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
)
|
||||
wait_for_port(port)
|
||||
|
||||
if warm:
|
||||
seed = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if seed is None:
|
||||
invalid += 1
|
||||
sys.stderr.write(" warm-mode seeding failed; run not counted\n")
|
||||
continue
|
||||
reapply_incremental_changes(source_dir, mutation)
|
||||
|
||||
t = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if t is not None:
|
||||
if t is None:
|
||||
invalid += 1
|
||||
elif verify:
|
||||
root = expected_received_root(dest_dir, source_dir, config["tool"])
|
||||
ok, detail = verify_transfer(source_dir, root)
|
||||
if ok:
|
||||
times.append(t)
|
||||
else:
|
||||
invalid += 1
|
||||
sys.stderr.write(f" verification FAILED ({detail}); run not counted\n")
|
||||
else:
|
||||
times.append(t)
|
||||
finally:
|
||||
if server:
|
||||
@@ -332,15 +565,22 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
"config": config["name"],
|
||||
"tool": config["tool"],
|
||||
"profile": profile_name,
|
||||
"warm": warm,
|
||||
"runs": len(times),
|
||||
"invalid": invalid,
|
||||
"times": [round(t, 4) for t in times],
|
||||
}
|
||||
if times:
|
||||
entry["p50"] = round(statistics.median(times), 4)
|
||||
entry["p95"] = round(sorted(times)[int(len(times) * 0.95)], 4) if len(times) > 1 else entry["p50"]
|
||||
p50 = percentile(times, 50)
|
||||
p95 = percentile(times, 95)
|
||||
entry["p50"] = round(p50, 4)
|
||||
entry["p95"] = round(p95, 4)
|
||||
entry["min"] = round(min(times), 4)
|
||||
entry["max"] = round(max(times), 4)
|
||||
entry["stdev"] = round(statistics.stdev(times), 4) if len(times) > 1 else 0.0
|
||||
if measure_bytes:
|
||||
entry["throughput_mbps"] = round(
|
||||
(measure_bytes / (1024 * 1024)) / p50, 3)
|
||||
results.append(entry)
|
||||
return results
|
||||
finally:
|
||||
@@ -350,44 +590,59 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
netem_reset()
|
||||
|
||||
|
||||
def print_table(results, total_bytes, random_ratio):
|
||||
def print_table(results, measure_bytes, stats, warm):
|
||||
"""Print results as a human-readable table grouped by profile."""
|
||||
profiles = {}
|
||||
for r in results:
|
||||
profiles.setdefault(r["profile"], []).append(r)
|
||||
|
||||
total = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total * 100 if total else 0
|
||||
rand_pct = stats["random_bytes"] / total * 100 if total else 0
|
||||
|
||||
for profile, entries in profiles.items():
|
||||
params = NETWORK_PROFILES.get(profile, {})
|
||||
print(f"\n{'=' * 85}")
|
||||
print(f"\n{'=' * 95}")
|
||||
print(f" Profile: {profile.upper()}")
|
||||
if params.get("rate"):
|
||||
print(f" Network: {params['rate']}, {params['delay']} +/- {params['jitter']}, loss {params['loss']}")
|
||||
else:
|
||||
print(f" Network: unlimited")
|
||||
print(f" Data: {total_bytes / (1024*1024):.1f} MB ({random_ratio*100:.0f}% random, {(1-random_ratio)*100:.0f}% compressible)")
|
||||
print(f"{'=' * 85}")
|
||||
print(f" Data: {total / (1024*1024):.1f} MB "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)")
|
||||
if warm:
|
||||
print(f" Mode: warm (incremental) — measured {measure_bytes / (1024*1024):.2f} MB "
|
||||
f"changed after an untimed full seed")
|
||||
else:
|
||||
print(" Mode: cold (full copy)")
|
||||
print(f"{'=' * 95}")
|
||||
|
||||
fs_entries = [e for e in entries if e.get("tool") == "fastsync"]
|
||||
rsync_entries = [e for e in entries if e.get("tool") == "rsync"]
|
||||
|
||||
header = (f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} "
|
||||
f"{'stdev':>8} {'MB/s':>9} {'runs':>5} {'bad':>4}")
|
||||
rule = (f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} "
|
||||
f"{'-' * 8} {'-' * 9} {'-' * 5} {'-' * 4}")
|
||||
|
||||
if fs_entries:
|
||||
print(f"\n FastSync:")
|
||||
print(f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
print(header)
|
||||
print(rule)
|
||||
for e in sorted(fs_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if rsync_entries:
|
||||
print(f"\n rsync:")
|
||||
print(f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
print(header)
|
||||
print(rule)
|
||||
for e in sorted(rsync_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if params.get("rate_bps") and fs_entries and rsync_entries:
|
||||
fs_best = min((e["p50"] for e in fs_entries if "p50" in e), default=None)
|
||||
rsync_best = min((e["p50"] for e in rsync_entries if "p50" in e), default=None)
|
||||
theoretical = total_bytes / params["rate_bps"]
|
||||
theoretical = measure_bytes / params["rate_bps"]
|
||||
if fs_best and rsync_best:
|
||||
print(f"\n Theoretical max (line rate): {theoretical:.4f}s")
|
||||
print(f" FastSync best: {fs_best:.4f}s ({theoretical/fs_best:.2f}x vs line rate)")
|
||||
@@ -397,10 +652,43 @@ def print_table(results, total_bytes, random_ratio):
|
||||
|
||||
def _print_entry(e):
|
||||
if "p50" in e:
|
||||
tp = f"{e['throughput_mbps']:.2f}" if "throughput_mbps" in e else "N/A"
|
||||
print(f" {e['config']:<38} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} {e['runs']:>5}")
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} "
|
||||
f"{tp:>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
else:
|
||||
print(f" {e['config']:<38} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {e['runs']:>5}")
|
||||
print(f" {e['config']:<38} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} "
|
||||
f"{'N/A':>8} {'N/A':>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
|
||||
|
||||
def configure_build_dirs(build_dir):
|
||||
"""Install the selected build directory and derived binary paths."""
|
||||
global BUILD_DIR, SERVER_CMD, CLIENT_CMD
|
||||
if not os.path.isabs(build_dir):
|
||||
build_dir = os.path.join(PROJECT_ROOT, build_dir)
|
||||
BUILD_DIR = os.path.abspath(build_dir)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
|
||||
|
||||
def build_project():
|
||||
"""Configure (Release) and build into the dedicated bench build dir."""
|
||||
if shutil.which("cmake") is None:
|
||||
sys.stderr.write("cmake not found in PATH; cannot build\n")
|
||||
sys.exit(1)
|
||||
os.makedirs(BUILD_DIR, exist_ok=True)
|
||||
configure = ["cmake", "-B", BUILD_DIR, "-S", PROJECT_ROOT,
|
||||
"-DCMAKE_BUILD_TYPE=Release"]
|
||||
result = subprocess.run(configure, capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("CMake configure failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
jobs = str(os.cpu_count() or 1)
|
||||
result = subprocess.run(["cmake", "--build", BUILD_DIR, "-j", jobs],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("Build failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def main():
|
||||
@@ -417,12 +705,20 @@ Custom network limits (--delay/--jitter/--throughput) override profiles.
|
||||
|
||||
Data mix:
|
||||
Default is ~75%% random/incompressible + ~25%% structured/compressible,
|
||||
reflecting typical real-world file sets.
|
||||
reflecting typical real-world file sets. The actual mix is measured and
|
||||
reported. Transfers are verified (destination must match source) unless
|
||||
--no-verify is given.
|
||||
|
||||
Warm mode:
|
||||
--warm seeds the destination with an untimed full copy of a pristine base,
|
||||
then measures only the incremental transfer after modifying a few files.
|
||||
|
||||
Examples:
|
||||
%(prog)s --profiles wan --runs 5
|
||||
%(prog)s --throughput 50mbit --delay 30ms --jitter 5ms
|
||||
%(prog)s --random-ratio 0.5 --size-mb 100
|
||||
%(prog)s --warm --runs 3 --no-rsync
|
||||
%(prog)s --dry-run --size-mb 4 --random-ratio 0.25
|
||||
""")
|
||||
parser.add_argument("--runs", type=int, default=3,
|
||||
help="Number of runs per config (default: 3)")
|
||||
@@ -430,7 +726,8 @@ Examples:
|
||||
choices=list(NETWORK_PROFILES.keys()),
|
||||
help="Predefined network profiles (default: unlimited)")
|
||||
parser.add_argument("--configs", nargs="+", default=None,
|
||||
help="Custom FastSync config flags")
|
||||
help="Custom FastSync config flags (shell-quoted, e.g. "
|
||||
"\"-j -z --chunk-serialization\")")
|
||||
parser.add_argument("--size-mb", type=int, default=25,
|
||||
help="Test data size in MB (default: 25)")
|
||||
parser.add_argument("--random-ratio", type=float, default=0.75,
|
||||
@@ -445,6 +742,14 @@ Examples:
|
||||
help="Custom packet loss (e.g. 1%%)")
|
||||
parser.add_argument("--no-rsync", action="store_true",
|
||||
help="Skip rsync comparison")
|
||||
parser.add_argument("--no-verify", action="store_true",
|
||||
help="Skip source/destination verification after each run")
|
||||
parser.add_argument("--warm", action="store_true",
|
||||
help="Incremental mode: seed dest first, measure only changes")
|
||||
parser.add_argument("--build-dir", default=DEFAULT_BUILD_DIR,
|
||||
help=f"Build directory (default: {DEFAULT_BUILD_DIR})")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="Only generate data and report its composition, then exit")
|
||||
parser.add_argument("--progress", action="store_true",
|
||||
help="Show progress bar with ETA")
|
||||
parser.add_argument("--output", choices=["table", "json"], default="table",
|
||||
@@ -453,14 +758,48 @@ Examples:
|
||||
help="Don't clean up test data")
|
||||
args = parser.parse_args()
|
||||
|
||||
if not 0.0 <= args.random_ratio <= 1.0:
|
||||
parser.error("--random-ratio must be between 0.0 and 1.0")
|
||||
if args.size_mb <= 0:
|
||||
parser.error("--size-mb must be positive")
|
||||
|
||||
configure_build_dirs(args.build_dir)
|
||||
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
stats = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
total_bytes = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
rand_pct = stats["random_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB in {stats['files']} files "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)",
|
||||
file=sys.stderr)
|
||||
|
||||
if args.dry_run:
|
||||
print(f"size_mb={args.size_mb} random_ratio={args.random_ratio:.4f} "
|
||||
f"total_bytes={stats['total_bytes']} "
|
||||
f"compressible_bytes={stats['compressible_bytes']} "
|
||||
f"random_bytes={stats['random_bytes']} files={stats['files']}")
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
return
|
||||
|
||||
# Warm mode: keep a pristine base copy, then mutate the live source.
|
||||
base_dir = None
|
||||
measure_bytes = total_bytes
|
||||
mutation = None
|
||||
if args.warm:
|
||||
change_target = max(64 * 1024, min(int(total_bytes * 0.01), 4 * 1024 * 1024))
|
||||
mutation = apply_incremental_changes(source_dir, change_target)
|
||||
measure_bytes = mutation["changed"]
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
print(f"Warm mode: each run seeds a full copy, then measures "
|
||||
f"{measure_bytes / (1024*1024):.3f} MB of add/change deltas", file=sys.stderr)
|
||||
|
||||
# Build (Release: benchmarking a debug build is meaningless)
|
||||
print("Building (Release)...", file=sys.stderr)
|
||||
configure = (f"cmake -B {BUILD_DIR} -S {PROJECT_ROOT} "
|
||||
f"-DCMAKE_BUILD_TYPE=Release > /dev/null 2>&1")
|
||||
if os.system(configure) != 0:
|
||||
print("CMake configure failed", file=sys.stderr); sys.exit(1)
|
||||
if os.system(f"cmake --build {BUILD_DIR} -j$(nproc) > /dev/null 2>&1") != 0:
|
||||
print("Build failed", file=sys.stderr); sys.exit(1)
|
||||
print(f"Building (Release) into {BUILD_DIR}...", file=sys.stderr)
|
||||
build_project()
|
||||
|
||||
# Determine active profile for display
|
||||
has_custom_net = args.delay or args.jitter or args.throughput or args.loss
|
||||
@@ -480,19 +819,10 @@ Examples:
|
||||
else:
|
||||
profiles_to_run = args.profiles or ["unlimited"]
|
||||
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
total_bytes = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
compressible_pct = (1 - args.random_ratio) * 100
|
||||
random_pct = args.random_ratio * 100
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB "
|
||||
f"({random_pct:.0f}% random, {compressible_pct:.0f}% compressible)",
|
||||
file=sys.stderr)
|
||||
|
||||
# Build config list
|
||||
# Build config list (shlex so quoted/space-separated flags survive)
|
||||
if args.configs:
|
||||
fastsync_configs = [{"name": c, "flags": c.split(), "tool": "fastsync"} for c in args.configs]
|
||||
fastsync_configs = [{"name": c, "flags": shlex.split(c), "tool": "fastsync"}
|
||||
for c in args.configs]
|
||||
else:
|
||||
fastsync_configs = list(FASTSYNC_CONFIGS)
|
||||
|
||||
@@ -509,9 +839,16 @@ Examples:
|
||||
all_results = []
|
||||
try:
|
||||
for profile in profiles_to_run:
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile, progress)
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile,
|
||||
measure_bytes, verify=not args.no_verify,
|
||||
warm=args.warm, mutation=mutation,
|
||||
progress=progress)
|
||||
all_results.extend(results)
|
||||
except RuntimeError as exc:
|
||||
sys.stderr.write(f"error: {exc}\n")
|
||||
sys.exit(1)
|
||||
finally:
|
||||
netem_reset()
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
|
||||
@@ -519,7 +856,7 @@ Examples:
|
||||
if args.output == "json":
|
||||
print(json.dumps(all_results, indent=2))
|
||||
else:
|
||||
print_table(all_results, total_bytes, args.random_ratio)
|
||||
print_table(all_results, measure_bytes, stats, args.warm)
|
||||
print()
|
||||
|
||||
|
||||
|
||||
@@ -8,3 +8,6 @@ markers =
|
||||
daemon_detach: real double-fork backgrounding path (--daemon without
|
||||
--no-detach); slower/fragile, so it runs in the full suite but not the
|
||||
fast PR gate
|
||||
parity: differential rsync-parity case (full set; runs on push to
|
||||
dev/main)
|
||||
parity_ci: fast differential rsync-parity subset (runs on the PR gate)
|
||||
@@ -0,0 +1 @@
|
||||
nested
|
||||
@@ -0,0 +1 @@
|
||||
nested
|
||||
@@ -3,26 +3,59 @@
|
||||
}:
|
||||
|
||||
pkgs.mkShell {
|
||||
# Development shell for FastSync. Provides the host-side toolchain needed to
|
||||
# build, lint, unit-test, integration-test and benchmark the project.
|
||||
# It deliberately does NOT build on entry: run the CMake commands in README.md
|
||||
# (or use the CI Docker image for exact CI parity).
|
||||
nativeBuildInputs = with pkgs; [
|
||||
# build
|
||||
gcc
|
||||
cmake
|
||||
gnumake
|
||||
pkg-config
|
||||
# lint / static analysis (matches CI)
|
||||
clang-tools # clang-format
|
||||
cppcheck
|
||||
# tests
|
||||
(python3.withPackages (ps: with ps; [ pytest pytest-xdist psutil ]))
|
||||
openssh # SSH transport integration tests
|
||||
# debugging
|
||||
gdb
|
||||
valgrind
|
||||
# coverage
|
||||
lcov
|
||||
# benchmark tooling
|
||||
rsync
|
||||
iproute2 # tc/netem for network shaping
|
||||
# misc
|
||||
git
|
||||
curl
|
||||
nodejs
|
||||
nixpkgs-fmt
|
||||
docker
|
||||
tea
|
||||
];
|
||||
|
||||
buildInputs = with pkgs; [
|
||||
zstd
|
||||
zlib
|
||||
lz4
|
||||
openssl
|
||||
(python3.withPackages (ps: with ps; [ pytest ]))
|
||||
];
|
||||
|
||||
# The CMake configure step fetches xxHash via FetchContent, which needs
|
||||
# network access; NIX_ENFORCE_PURITY must be off so the sandbox does not block.
|
||||
NIX_ENFORCE_PURITY = 0;
|
||||
|
||||
shellHook = ''
|
||||
export NIX_ENFORCE_PURITY=0
|
||||
cmake -B build
|
||||
export PATH="$PWD/build:$PATH"
|
||||
# Make an existing build tree available on PATH, but never build here.
|
||||
if [ -d "$PWD/build" ]; then
|
||||
export PATH="$PWD/build:$PATH"
|
||||
fi
|
||||
echo "FastSync dev shell ready."
|
||||
echo " Build: cmake -B build -S . && cmake --build build -j\$(nproc)"
|
||||
echo " Unit: ./build/tests"
|
||||
echo " CI parity: docker run --rm --user \"\$(id -u):\$(id -g)\" -v \"\$PWD:/workspace\" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 ..."
|
||||
'';
|
||||
}
|
||||
+627
-157
@@ -1,5 +1,8 @@
|
||||
#include "change_list.h"
|
||||
#include "checksum.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
@@ -7,21 +10,7 @@
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
/* Itemize code emitted for a transferred regular file.
|
||||
*
|
||||
* Layout (rsync-compatible 11-char item): `>f` marks a regular file that was
|
||||
* transferred to the remote host; the trailing nine markers are, in order,
|
||||
* c(hecksum) s(ize) t(ime) p(erms) o(wner) g(roup) u(ser/acl) a(ttrs) x(attrs).
|
||||
* Every marker is `+` (FastSync does not compare each attribute on the
|
||||
* receiving side, so a sent file is reported as fully updated). Files that
|
||||
* are already up to date print no line at all, matching rsync's single -i
|
||||
* which only itemizes changes.
|
||||
*
|
||||
* Because the scanner only yields regular-file transfer candidates, `>d`
|
||||
* (directory) lines are never produced; directories are not transferred as
|
||||
* items by FastSync. */
|
||||
#define ITEMIZE_SENT_FILE ">f+++++++++"
|
||||
#include <unistd.h>
|
||||
|
||||
typedef struct {
|
||||
char* data;
|
||||
@@ -80,103 +69,23 @@ static bool strbuf_append(StrBuf* buf, const char* text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool strbuf_append_ull(StrBuf* buf, unsigned long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
static bool strbuf_append_longlong(StrBuf* buf, long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%lld", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
bool change_list_enabled(const Config* config) {
|
||||
return config != NULL && (config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL));
|
||||
(config->log_file != NULL && config->log_file_format != NULL) ||
|
||||
(config->info_level & LOG_INFO_NAME) != 0);
|
||||
}
|
||||
|
||||
char* change_render_itemize(const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
const char* code = event->is_directory ? ">d+++++++++" : ITEMIZE_SENT_FILE;
|
||||
StrBuf line = {0};
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append(&line, " ") &&
|
||||
strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
/* Emitted once, lazily, ahead of the first --info=name entry: rsync prints the
|
||||
* transfer-root `./` name line when the root directory is (re)created. */
|
||||
static bool name_root_printed = false;
|
||||
|
||||
void change_reset_name_root(void) {
|
||||
name_root_printed = false;
|
||||
}
|
||||
|
||||
static const char* leaf_name(const char* path) {
|
||||
if (path == NULL)
|
||||
return "";
|
||||
const char* slash = strrchr(path, '/');
|
||||
return slash != NULL && slash[1] != '\0' ? slash + 1 : path;
|
||||
}
|
||||
/* ---- Itemize code ---- */
|
||||
|
||||
char* change_render_format(const char* format, const ChangeEvent* event) {
|
||||
if (format == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = strbuf_append(&line, leaf_name(event->path));
|
||||
break;
|
||||
case 'l':
|
||||
ok = strbuf_append_ull(&line, event->size);
|
||||
break;
|
||||
case 'b':
|
||||
ok = strbuf_append_ull(&line, event->bytes_sent);
|
||||
break;
|
||||
case 'M':
|
||||
ok = strbuf_append_longlong(&line, (long long)event->mtime_sec);
|
||||
break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* Format a mode as an `ls -l` permission string, e.g. `-rw-r--r--`. */
|
||||
/* Format the permission bits as an `ls -l` string, e.g. `-rw-r--r--`. */
|
||||
static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[0] = S_ISDIR(mode) ? 'd'
|
||||
: S_ISLNK(mode) ? 'l'
|
||||
@@ -198,29 +107,138 @@ static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[10] = '\0';
|
||||
}
|
||||
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime,
|
||||
const char* path) {
|
||||
char permission[11];
|
||||
mode_to_ls_string(mode, permission);
|
||||
char date[32];
|
||||
struct tm broken_down;
|
||||
if (localtime_r(&mtime, &broken_down) != NULL) {
|
||||
if (strftime(date, sizeof(date), "%Y/%m/%d %H:%M:%S", &broken_down) == 0)
|
||||
snprintf(date, sizeof(date), "?");
|
||||
} else {
|
||||
snprintf(date, sizeof(date), "?");
|
||||
static char itemize_type_char(const ChangeEvent* event) {
|
||||
if (event->is_directory)
|
||||
return 'd';
|
||||
if (event->is_symlink)
|
||||
return 'L';
|
||||
if (event->is_special) {
|
||||
if (S_ISCHR(event->mode) || S_ISBLK(event->mode))
|
||||
return 'D';
|
||||
return 'S';
|
||||
}
|
||||
return 'f';
|
||||
}
|
||||
|
||||
static bool times_match(const Config* config, const ChangeEvent* event) {
|
||||
if (!event->dest.known || !event->dest.existed)
|
||||
return false;
|
||||
if (event->mtime_sec == event->dest.mtime_sec) {
|
||||
/* A regular file's sub-second mtime IS preserved by the receiver, so an nsec
|
||||
difference is a real change. A directory or symlink has no preserved
|
||||
sub-second mtime (rsync's quick-check compares whole seconds there), so a
|
||||
nanosecond-only difference must not render a spurious `.d..t` / `.L..t`. */
|
||||
if (event->is_directory || event->is_symlink || event->is_special)
|
||||
return true;
|
||||
return event->mtime_nsec == event->dest.mtime_nsec;
|
||||
}
|
||||
long long delta = (long long)event->mtime_sec - (long long)event->dest.mtime_sec;
|
||||
if (delta < 0)
|
||||
delta = -delta;
|
||||
return delta <= (long long)config->modify_window;
|
||||
}
|
||||
|
||||
/* Fill the 11-character itemize code (10 chars + NUL). `created` means the
|
||||
* destination entry did not exist, so every attribute marker is `+`. */
|
||||
static void itemize_code(const Config* config, const ChangeEvent* event, char code[12]) {
|
||||
bool known = event->dest.known;
|
||||
bool created = !known || !event->dest.existed;
|
||||
char update;
|
||||
if (event->is_hardlink)
|
||||
update = 'h';
|
||||
else if (created)
|
||||
update = (event->is_directory || event->is_symlink || event->is_special) ? 'c' : '>';
|
||||
else if (event->is_directory)
|
||||
/* rsync: an existing directory that only has attribute changes carries no
|
||||
transfer, so the update column is `.` rather than `>`. */
|
||||
update = '.';
|
||||
else if (event->is_symlink)
|
||||
/* rsync: an existing symlink whose target is unchanged is a `.` update
|
||||
(attributes only); a changed target is `c` (the link value changed). */
|
||||
update = event->dest.target_matches ? '.' : 'c';
|
||||
else
|
||||
update = '>';
|
||||
code[0] = update;
|
||||
code[1] = itemize_type_char(event);
|
||||
if (created) {
|
||||
for (int i = 0; i < 9; i++)
|
||||
code[2 + i] = '+';
|
||||
code[11] = '\0';
|
||||
return;
|
||||
}
|
||||
/* rsync's value/checksum column: `c` for a symlink whose target changed (the
|
||||
link value is the compared content); no destination digest is available for
|
||||
a regular file. */
|
||||
bool value_diff = event->is_symlink && !event->dest.target_matches;
|
||||
/* rsync itemizes size only for regular files: a directory's st_size and a
|
||||
symlink's target length are not compared. */
|
||||
bool size_diff = !event->is_directory && !event->is_symlink && !event->is_special &&
|
||||
event->size != event->dest.size;
|
||||
/* rsync itemizes the time column only when -t/--times is in effect. */
|
||||
bool time_diff = config->preserve_times && !times_match(config, event);
|
||||
bool perms_diff = (event->mode & 07777) != (event->dest.mode & 07777);
|
||||
bool owner_diff = event->uid != (uid_t)event->dest.uid;
|
||||
bool group_diff = event->gid != (gid_t)event->dest.gid;
|
||||
code[2] = value_diff ? 'c' : '.';
|
||||
code[3] = size_diff ? 's' : '.';
|
||||
code[4] = time_diff ? 't' : '.';
|
||||
code[5] = (config->preserve_perms && perms_diff) ? 'p' : '.';
|
||||
code[6] = (config->preserve_owner && owner_diff) ? 'o' : '.';
|
||||
code[7] = (config->preserve_group && group_diff) ? 'g' : '.';
|
||||
code[8] = '.'; /* reserved */
|
||||
code[9] = '.'; /* acl: not compared */
|
||||
code[10] = '.';
|
||||
code[11] = '\0';
|
||||
}
|
||||
|
||||
/* True when the itemized destination entry is unchanged, i.e. rsync would print
|
||||
* no line at all. Reuses itemize_code so suppression is exactly consistent
|
||||
* with what would have been rendered: the update column must be `.` and every
|
||||
* attribute column must be `.`. */
|
||||
static bool itemize_is_unchanged(const Config* config, const ChangeEvent* event) {
|
||||
if (!event->dest.known || !event->dest.existed)
|
||||
return false;
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
if (code[0] != '.')
|
||||
return false;
|
||||
for (int i = 2; i < 11; i++) {
|
||||
if (code[i] != '.')
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* rsync %n: the transfer-relative name, with a trailing slash for directories.
|
||||
* The transfer root is `.` (so `%n` renders `./`), matching rsync's root entry. */
|
||||
static bool append_name(StrBuf* buf, const ChangeEvent* event) {
|
||||
const char* name = event->name != NULL ? event->name : "";
|
||||
if (event->is_directory && name[0] == '\0')
|
||||
return strbuf_append(buf, "./");
|
||||
if (!strbuf_append(buf, name))
|
||||
return false;
|
||||
if (event->is_directory && name[strlen(name) - 1] != '/')
|
||||
return strbuf_append_char(buf, '/');
|
||||
return true;
|
||||
}
|
||||
|
||||
/* rsync %L: " -> target" for a symlink, " => target" for a hard link, else "". */
|
||||
static bool append_link_suffix(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (event->is_symlink && event->symlink_target != NULL)
|
||||
return strbuf_append(buf, " -> ") && strbuf_append(buf, event->symlink_target);
|
||||
if (event->is_hardlink && event->hardlink_target != NULL)
|
||||
return strbuf_append(buf, " => ") && strbuf_append(buf, event->hardlink_target);
|
||||
return true;
|
||||
}
|
||||
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
StrBuf line = {0};
|
||||
char size_field[32];
|
||||
int written = snprintf(size_field, sizeof(size_field), "%llu", size);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, size_field) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, date) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, path != NULL ? path : "");
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append_char(&line, ' ') &&
|
||||
append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
@@ -228,6 +246,276 @@ char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* rsync's `--info=name` line for an updated entry: the transfer-relative name
|
||||
* (trailing slash for directories) plus the ` -> target` / ` => target` link
|
||||
* suffix. `--info=name` does not alter an itemize/out-format run. */
|
||||
static char* change_render_name(const ChangeEvent* event) {
|
||||
StrBuf line = {0};
|
||||
bool ok = append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* rsync's `--info=name2` line for an unchanged entry: `NAME is uptodate`. */
|
||||
static char* change_render_name_uptodate(const ChangeEvent* event) {
|
||||
char* name = change_render_name(event);
|
||||
if (name == NULL)
|
||||
return NULL;
|
||||
size_t length = strlen(name);
|
||||
char* line = malloc(length + sizeof(" is uptodate"));
|
||||
if (line == NULL) {
|
||||
free(name);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(line, name, length);
|
||||
memcpy(line + length, " is uptodate", sizeof(" is uptodate"));
|
||||
free(name);
|
||||
return line;
|
||||
}
|
||||
|
||||
/* ---- --out-format / --log-file-format ---- */
|
||||
|
||||
/* rsync 3.4.1's `%C` uses the negotiated TRANSFER checksum (the first name of a
|
||||
* two-name "transfer,pre-transfer" --checksum-choice), not the pre-transfer
|
||||
* whole-file digest FastSync compares against on the wire. The default "auto"
|
||||
* resolves to xxh128, so an explicit selection and the default both render the
|
||||
* selected algorithm's digest. */
|
||||
static ChecksumAlgo out_format_checksum_algo(const Config* config) {
|
||||
return (ChecksumAlgo)config->cli.checksum_transfer_algo;
|
||||
}
|
||||
|
||||
/* Render a digest as rsync's sum_as_hex: xxh128 prints the HIGH 64-bit half
|
||||
* before the low half, and xxh64/xxh3 print their 64-bit value big-endian; every
|
||||
* other algorithm prints its bytes in order. */
|
||||
static void digest_to_hex(ChecksumAlgo algo, const uint8_t* digest, size_t len, char* out) {
|
||||
if (algo == CHECKSUM_ALGO_XXH128 && len == 16) {
|
||||
uint64_t low = 0;
|
||||
uint64_t high = 0;
|
||||
memcpy(&low, digest, sizeof(low));
|
||||
memcpy(&high, digest + 8, sizeof(high));
|
||||
snprintf(out, len * 2 + 1, "%016llx%016llx", (unsigned long long)high, (unsigned long long)low);
|
||||
return;
|
||||
}
|
||||
if ((algo == CHECKSUM_ALGO_XXH64 || algo == CHECKSUM_ALGO_XXH3) && len == 8) {
|
||||
uint64_t value = 0;
|
||||
memcpy(&value, digest, sizeof(value));
|
||||
snprintf(out, len * 2 + 1, "%016llx", (unsigned long long)value);
|
||||
return;
|
||||
}
|
||||
static const char hex[] = "0123456789abcdef";
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
out[i * 2] = hex[(digest[i] >> 4) & 0xf];
|
||||
out[i * 2 + 1] = hex[digest[i] & 0xf];
|
||||
}
|
||||
out[len * 2] = '\0';
|
||||
}
|
||||
|
||||
static bool format_uses_checksum(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0')
|
||||
break;
|
||||
if (token == 'C')
|
||||
return true;
|
||||
p += 2;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Fill event->checksum/checksum_known for a transferred regular file. A
|
||||
* non-regular entry (or a hard-link sibling) leaves checksum_known false, which
|
||||
* renders as spaces like rsync. */
|
||||
static void fill_event_checksum(const Config* config, const File* file, ChangeEvent* event) {
|
||||
if (file == NULL || file->is_dir || file->is_symlink || file->is_special ||
|
||||
(file->link_group != 0 && !file->link_first))
|
||||
return;
|
||||
if (!format_uses_checksum(config->out_format) && !format_uses_checksum(config->log_file_format))
|
||||
return;
|
||||
if (file->path == NULL)
|
||||
return;
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
/* rsync renders `--checksum-choice=none` as a blank 2-character column. */
|
||||
if (algo == CHECKSUM_ALGO_NONE)
|
||||
return;
|
||||
uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
|
||||
size_t len = 0;
|
||||
/* rsync's %C is the transfer checksum, which is always seeded with 0 (it is
|
||||
* independent of --checksum-seed, as rsync 3.4.1 demonstrates). */
|
||||
if (!checksum_digest_file(algo, 0, file->path, digest, sizeof(digest), &len))
|
||||
return;
|
||||
digest_to_hex(algo, digest, len, event->checksum);
|
||||
event->checksum_known = true;
|
||||
}
|
||||
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event) {
|
||||
if (format == NULL || event == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'i': {
|
||||
if (event->deleted) {
|
||||
/* rsync's ITEM_DELETED itemize code: `*deleting ` (11 chars). */
|
||||
ok = strbuf_append(&line, "*deleting ");
|
||||
break;
|
||||
}
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
ok = strbuf_append(&line, code);
|
||||
break;
|
||||
}
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = append_name(&line, event);
|
||||
break;
|
||||
case 'L':
|
||||
ok = append_link_suffix(&line, event);
|
||||
break;
|
||||
case 'l': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->size);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'b': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_sent);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'c': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_read);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'C': {
|
||||
if (event->checksum_known) {
|
||||
ok = strbuf_append(&line, event->checksum);
|
||||
} else {
|
||||
/* rsync pads a non-regular / untransferred / `none` entry with spaces;
|
||||
`none` renders as a blank 2-character column. */
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
int width = algo == CHECKSUM_ALGO_NONE ? 2 : checksum_digest_len(algo) * 2;
|
||||
for (int i = 0; i < width && ok; i++)
|
||||
ok = strbuf_append_char(&line, ' ');
|
||||
}
|
||||
} break;
|
||||
case 'M': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(event->mtime_sec, true, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 't': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(time(NULL), false, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 'o':
|
||||
ok = strbuf_append(&line, "send");
|
||||
break;
|
||||
case 'p': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%ld", (long)getpid());
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'B': {
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
ok = strbuf_append(&line, permission + 1);
|
||||
} break;
|
||||
case 'U': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->uid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'G': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->gid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- --list-only ---- */
|
||||
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event) {
|
||||
(void)config;
|
||||
if (event == NULL)
|
||||
return NULL;
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
char date[32];
|
||||
if (!format_rsync_datetime(event->mtime_sec, false, date, sizeof(date)))
|
||||
snprintf(date, sizeof(date), "?");
|
||||
StrBuf line = {0};
|
||||
char size_field[40];
|
||||
char grouped[32];
|
||||
if (!format_big_num(event->size, false, grouped, sizeof(grouped))) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
int written = snprintf(size_field, sizeof(size_field), "%15s", grouped);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
const char* name = event->name != NULL && event->name[0] != '\0' ? event->name : ".";
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append(&line, size_field) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, date) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, name);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- Event emission ---- */
|
||||
|
||||
static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_output) {
|
||||
char* escaped = output_escape(line, eight_bit_output);
|
||||
if (escaped != NULL) {
|
||||
@@ -242,20 +530,49 @@ static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_ou
|
||||
void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
if (event->decision == CHANGE_UP_TO_DATE)
|
||||
return;
|
||||
bool to_stdout = config->itemize_changes || config->out_format != NULL;
|
||||
bool to_log = config->log_file != NULL && config->log_file_format != NULL;
|
||||
bool progress_active = config->show_progress || (config->info_level & LOG_INFO_PROGRESS);
|
||||
if (event->decision == CHANGE_UP_TO_DATE) {
|
||||
/* --info=name2 prints `NAME is uptodate` for entries the receiver already
|
||||
had. An itemize/out-format run reports them through its own format (or
|
||||
not at all), the progress stream has no frame for them, and neither the
|
||||
itemize nor the log-file stream previously reported an up-to-date entry,
|
||||
so nothing else here changes. */
|
||||
if (!to_stdout && (config->info_level & LOG_INFO_NAME_UPTODATE) != 0 && !progress_active) {
|
||||
char* line = change_render_name_uptodate(event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (to_stdout) {
|
||||
char* line = config->out_format != NULL ? change_render_format(config->out_format, event)
|
||||
: change_render_itemize(event);
|
||||
char* line = config->out_format != NULL
|
||||
? change_render_format(config->out_format, config, event)
|
||||
: change_render_itemize(config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
} else if ((config->info_level & LOG_INFO_NAME) != 0 && !progress_active) {
|
||||
/* --info=name without -i/--out-format: print the updated entry's name. The
|
||||
--progress path owns the name line when progress output is active (it
|
||||
emits the same names before the progress frames), so do not duplicate.
|
||||
The transfer-root `./` line precedes the first such name. */
|
||||
if (!name_root_printed) {
|
||||
name_root_printed = true;
|
||||
fputs("./\n", stdout);
|
||||
}
|
||||
char* line = change_render_name(event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
if (to_log) {
|
||||
char* line = change_render_format(config->log_file_format, event);
|
||||
char* line = change_render_format(config->log_file_format, config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(config->log_file, line, config->eight_bit_output);
|
||||
free(line);
|
||||
@@ -266,9 +583,6 @@ void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
static bool format_uses_mtime(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
/* Mirror change_render_format's tokenizer: "%%" is a literal percent (so
|
||||
* "%%M" does NOT expand %M) and unknown "%X" escapes consume both chars.
|
||||
* This keeps the optional stat() fallback below in step with the renderer. */
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
@@ -284,46 +598,202 @@ static bool format_uses_mtime(const char* format) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
/* Relative path of an entry below the transfer root (no leading slash). Uses
|
||||
* the sender-side send_path override when present (bare-relative -R layout). */
|
||||
static char* relative_name(const Config* config, const File* file) {
|
||||
const char* full = file_wire_path(file);
|
||||
if (file->send_path != NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL || full == NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, full, root_len) == 0) {
|
||||
if (full[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (full[root_len] == '/')
|
||||
return str_dup(full + root_len + 1);
|
||||
}
|
||||
return str_dup(full);
|
||||
}
|
||||
|
||||
/* rsync %f long form: the source argument as typed (leading '/' removed,
|
||||
* trailing '/' removed, leading "./" removed) joined to the relative name. */
|
||||
static char* display_name(const Config* config, const char* name) {
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL)
|
||||
return str_dup(name != NULL ? name : "");
|
||||
const char* p = root;
|
||||
while (*p == '/')
|
||||
p++;
|
||||
if (p[0] == '.' && p[1] == '/')
|
||||
p += 2;
|
||||
size_t root_len = strlen(p);
|
||||
while (root_len > 0 && p[root_len - 1] == '/')
|
||||
root_len--;
|
||||
size_t name_len = name != NULL ? strlen(name) : 0;
|
||||
if (root_len == 0 && name_len == 0)
|
||||
return str_dup("");
|
||||
char* out = malloc(root_len + (root_len > 0 && name_len > 0 ? 1 : 0) + name_len + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t offset = 0;
|
||||
if (root_len > 0) {
|
||||
memcpy(out, p, root_len);
|
||||
offset = root_len;
|
||||
}
|
||||
if (root_len > 0 && name_len > 0)
|
||||
out[offset++] = '/';
|
||||
if (name_len > 0)
|
||||
memcpy(out + offset, name, name_len);
|
||||
out[offset + name_len] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
static void fill_event_from_file(const Config* config, const File* file, ChangeEvent* event,
|
||||
char** name_out, char** path_out) {
|
||||
char* name = relative_name(config, file);
|
||||
char* path = display_name(config, name);
|
||||
event->name = name;
|
||||
event->path = path;
|
||||
*name_out = name;
|
||||
*path_out = path;
|
||||
if (file->metadata != NULL) {
|
||||
event->mtime_sec = file->metadata->mtime_sec;
|
||||
event->mtime_nsec = file->metadata->mtime_nsec;
|
||||
event->mode = file->metadata->mode;
|
||||
event->uid = file->metadata->uid;
|
||||
event->gid = file->metadata->gid;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0) {
|
||||
event->mtime_sec = st.st_mtime;
|
||||
event->mtime_nsec = st.st_mtim.tv_nsec;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
/* The displayed path is the one transmitted (with -R + --files-from this is
|
||||
the bare relative destination path); the metadata fallback below still
|
||||
stats the local absolute path. */
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = false;
|
||||
event.is_special = false;
|
||||
event.is_hardlink = false;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
/* FastSync has no wire-byte counter yet, so %b reports the source length
|
||||
* that had to be delivered (always equal to %l); the actual bytes written
|
||||
* to the socket (compressed/delta) are not measured. */
|
||||
event.bytes_sent = event.size;
|
||||
if (file->metadata != NULL) {
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
/* Best-effort fallback for %M when no metadata was captured (no -M): the
|
||||
* path is stat()ed just to fill the field, and any failure leaves 0. */
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0)
|
||||
event.mtime_sec = st.st_mtime;
|
||||
event.dest = file->dest_state;
|
||||
if (file->is_symlink) {
|
||||
event.is_symlink = true;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.size = file->symlink_target != NULL ? strlen(file->symlink_target) : 0;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->is_special) {
|
||||
event.is_special = true;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->link_group != 0 && !file->link_first) {
|
||||
event.is_hardlink = true;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.bytes_sent = 0;
|
||||
} else {
|
||||
event.bytes_sent = bytes_sent;
|
||||
/* rsync's %c is the block-checksum bytes received for the file. Even a
|
||||
* whole-file transfer (no basis; --append/--inplace included) receives
|
||||
* rsync's 16-byte sum header, so rsync reports 16; a dry run transfers
|
||||
* nothing and reports 0. FastSync's whole-file path has no sum header, so
|
||||
* report rsync's value for parity. With delta enabled the real received
|
||||
* bytes are kept, but FastSync's signature framing differs from rsync's so
|
||||
* those stay numerically divergent. */
|
||||
bool delta_active = config->use_delta && !config->whole_file;
|
||||
event.bytes_read = (!config->dry_run && !delta_active) ? 16 : bytes_read;
|
||||
}
|
||||
change_emit(config, &event);
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (file->is_symlink) {
|
||||
/* Output parity (protocol 2.30.0): an unchanged symlink is silent, like
|
||||
rsync's quick check. The itemize/log stream suppresses it only when every
|
||||
attribute matches; the name stream suppresses it whenever the link target
|
||||
is unchanged (rsync names a symlink only when it relinks or creates it). */
|
||||
bool itemize_output = config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL);
|
||||
bool suppress = itemize_output
|
||||
? itemize_is_unchanged(config, &event)
|
||||
: (event.dest.known && event.dest.existed && event.dest.target_matches);
|
||||
if (suppress)
|
||||
event.decision = CHANGE_UP_TO_DATE;
|
||||
}
|
||||
if (name != NULL && path != NULL) {
|
||||
fill_event_checksum(config, file, &event);
|
||||
change_emit(config, &event);
|
||||
}
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
if (file == NULL)
|
||||
return;
|
||||
unsigned long long payload = file->data != NULL ? file->data->size : 0;
|
||||
change_emit_file_sent_bytes(config, file, payload, 0);
|
||||
}
|
||||
|
||||
void change_emit_file_uptodate(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.decision = CHANGE_UP_TO_DATE;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = file->is_symlink;
|
||||
event.is_special = file->is_special;
|
||||
event.is_hardlink = file->link_group != 0 && !file->link_first;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
void change_emit_dir_sent(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = true;
|
||||
event.size = 0;
|
||||
event.bytes_sent = 0;
|
||||
if (file->metadata != NULL)
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
change_emit(config, &event);
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
/* Output parity (protocol 2.30.0): suppress a directory rsync would leave
|
||||
silent. The itemize/log stream suppresses it only when every attribute
|
||||
matches (`.d.........`); the name stream suppresses any pre-existing
|
||||
directory (rsync names a directory only when it is created). */
|
||||
bool itemize_output = config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL);
|
||||
bool suppress = itemize_output ? itemize_is_unchanged(config, &event)
|
||||
: (event.dest.known && event.dest.existed);
|
||||
/* The transfer root's line is an unconditional FastSync residual (rsync keys
|
||||
it off the root's own attribute change); keep emitting it. */
|
||||
bool is_root = event.name != NULL && event.name[0] == '\0';
|
||||
if (suppress && !is_root)
|
||||
event.decision = CHANGE_UP_TO_DATE;
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
+61
-25
@@ -2,7 +2,9 @@
|
||||
#define CHANGE_LIST_H
|
||||
|
||||
#include "config.h"
|
||||
#include "checksum.h"
|
||||
#include "file_types.h"
|
||||
#include "format.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
@@ -26,42 +28,59 @@ typedef enum {
|
||||
} ChangeDecision;
|
||||
|
||||
typedef struct {
|
||||
const char* path; /* full source path */
|
||||
const char* path; /* long-form display path (rsync %f) */
|
||||
const char* name; /* transfer-relative path (rsync %n), no trailing slash */
|
||||
ChangeDecision decision;
|
||||
bool is_directory;
|
||||
unsigned long long size; /* source file length in bytes */
|
||||
/* The number of bytes reported for a sent file. FastSync has no wire-byte
|
||||
* counter, so this is always the source length (== size / %l); actual
|
||||
* post-compression/delta bytes on the wire are not counted. */
|
||||
unsigned long long bytes_sent;
|
||||
time_t mtime_sec; /* 0 when unknown */
|
||||
bool is_symlink;
|
||||
bool is_special;
|
||||
bool is_hardlink; /* a hard-link sibling (linked, no data sent) */
|
||||
bool deleted; /* a would-delete report (-n --delete); no source file */
|
||||
const char* symlink_target;
|
||||
const char* hardlink_target;
|
||||
unsigned long long size; /* source file length in bytes */
|
||||
unsigned long long bytes_sent; /* wire bytes actually transferred (rsync %b) */
|
||||
unsigned long long bytes_read; /* wire bytes read back for this file (rsync %c) */
|
||||
/* rsync %C: whole-file checksum hex for a transferred regular file. Only
|
||||
* filled when the active format uses %C (checksum_known == false otherwise,
|
||||
* which renders as spaces like rsync for non-regular entries). */
|
||||
bool checksum_known;
|
||||
char checksum[CHECKSUM_MAX_DIGEST_LEN * 2 + 1];
|
||||
time_t mtime_sec;
|
||||
long mtime_nsec;
|
||||
mode_t mode;
|
||||
uid_t uid;
|
||||
gid_t gid;
|
||||
/* Receiver-reported pre-transfer destination state (OutputDestState.known is
|
||||
* false when no report was requested/received). */
|
||||
OutputDestState dest;
|
||||
} ChangeEvent;
|
||||
|
||||
/* True when any output mode is active and per-file events matter. */
|
||||
bool change_list_enabled(const Config* config);
|
||||
|
||||
/* Render the rsync-style itemize line for a transferred file:
|
||||
* `>f+++++++++ <path>`
|
||||
* The 11-char code is `>f` (regular file transferred to the remote host)
|
||||
* followed by c/s/t/p/o/g/u/a/x markers that are all `+` (value will be set
|
||||
* / differs) because FastSync does not separately compare checksums, size,
|
||||
* mtime, perms, owner, group, uid, acl, or xattr on the receiving side, so a
|
||||
* sent file is reported as fully updated. Up-to-date files print no line
|
||||
* (rsync single `-i` only shows changes). Caller frees the result. */
|
||||
char* change_render_itemize(const ChangeEvent* event);
|
||||
/* Render the rsync-style itemize line for a transferred item
|
||||
* (`%i %n%L`): `>f+++++++++ sub/b.txt`. Caller frees the result. */
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Expand an --out-format/--log-file-format template. Tokens:
|
||||
* %f full source path %b "bytes sent" == the source length (%l);
|
||||
* %n leaf (base) name actual post-compression/delta wire bytes
|
||||
* %l file length in bytes are not counted
|
||||
* %M mtime in whole seconds %% a literal percent sign
|
||||
/* Expand an --out-format/--log-file-format template. Supported tokens:
|
||||
* %i itemize code %n transfer-relative name (dir: trailing /)
|
||||
* %f long display path %l file length in bytes
|
||||
* %b wire bytes transferred %c block-checksum bytes received (rsync: 16
|
||||
* for a whole-file transfer, 0 for a dry run)
|
||||
* %C whole-file checksum hex (xxh128 by default; spaces for non-regular)
|
||||
* %M mtime (YYYY/MM/DD-HH:MM:SS)
|
||||
* %t current time %o operation ("send"/"del.")
|
||||
* %p pid %B permission bits without the type char
|
||||
* %U uid %G gid
|
||||
* %L " -> target" / " => target" %% a literal percent sign
|
||||
* Unknown %X sequences are preserved verbatim. Caller frees the result. */
|
||||
char* change_render_format(const char* format, const ChangeEvent* event);
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Render one --list-only long-listing entry:
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 <path>`
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 sub/b.txt`
|
||||
* (ls -l style columns; mtime in the local time zone). Caller frees it. */
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime, const char* path);
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Emit an event to every active destination:
|
||||
* stdout: --itemize-changes line, or the --out-format expansion when set;
|
||||
@@ -69,10 +88,27 @@ char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime
|
||||
* CHANGE_UP_TO_DATE events produce no output. */
|
||||
void change_emit(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent. */
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent. `bytes_sent`
|
||||
* is the process-wide wire-byte delta for this file (rsync's %b) and `bytes_read`
|
||||
* the received bytes used for the delta handshake; pass 0 when unknown. For a
|
||||
* whole-file transfer %c is pinned to rsync's 16-byte sum header regardless. */
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent, deriving
|
||||
* the wire byte counts from the source payload length. */
|
||||
void change_emit_file_sent(const Config* config, const File* file);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
void change_emit_dir_sent(const Config* config, const File* file);
|
||||
|
||||
/* Build and emit a CHANGE_UP_TO_DATE event for a file the receiver already had.
|
||||
* With --info=name2 it renders rsync's "NAME is uptodate" line (no output
|
||||
* otherwise). */
|
||||
void change_emit_file_uptodate(const Config* config, const File* file);
|
||||
|
||||
/* Reset the lazy transfer-root `./` line emitted ahead of the first
|
||||
* --info=name entry. Call once at the start of a transfer. */
|
||||
void change_reset_name_root(void);
|
||||
|
||||
#endif
|
||||
+1314
-217
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,679 @@
|
||||
#include "client_send_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "change_list.h"
|
||||
#include "charset.h"
|
||||
#include "config.h"
|
||||
#include "data.h"
|
||||
#include "delta.h"
|
||||
#include "file.h"
|
||||
#include "format.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "scanner.h"
|
||||
#include "transport_tls.h"
|
||||
#include "utils.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
/* True when --dry-run should contact a receiver rather than running the
|
||||
* client-side local manifest. Any target a real run would reach over the wire
|
||||
* selects the server-contacting path: a remote (SSH host:path), a daemon
|
||||
* (host::module/path), an explicit --server-host, --server-port/--port, TLS, or
|
||||
* a source-bind --address. A plain local destination (none of these) keeps the
|
||||
* original client-side behavior, which never dials the default 127.0.0.1:8080. */
|
||||
bool dry_run_targets_server(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
if (config->transport == TRANSPORT_SSH)
|
||||
return true;
|
||||
if (config->module && config->module[0] != '\0')
|
||||
return true;
|
||||
if (config->cli.server_host_set || config->cli.server_port_set)
|
||||
return true;
|
||||
if (config->use_tls)
|
||||
return true;
|
||||
if (config->address != NULL)
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk) {
|
||||
if (!manifest)
|
||||
return true;
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
const char* path = file_wire_path(chunk->items[i]);
|
||||
if (*path == '/')
|
||||
path++;
|
||||
char* entry = str_dup(path);
|
||||
if (!entry) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate manifest entry");
|
||||
return false;
|
||||
}
|
||||
if (!array_list_add(manifest, entry)) {
|
||||
free(entry);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */
|
||||
int send_dry_run_manifest(const Config* config) {
|
||||
int skipped = 0;
|
||||
ArrayList* missing_dest = NULL;
|
||||
if (!client_prepare_files_from(config, &missing_dest, &skipped))
|
||||
return -1;
|
||||
PreparedScanner prepared;
|
||||
if (!prepare_scanner(config, 0, &prepared)) {
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
return -1;
|
||||
}
|
||||
DirectoryScanner* scanner =
|
||||
directory_scanner_create_with_options(config->send_directory, &prepared.options);
|
||||
if (!scanner) {
|
||||
prepared_scanner_destroy(&prepared);
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
return -1;
|
||||
}
|
||||
Chunk* chunk;
|
||||
int file_count = 0;
|
||||
unsigned long long total_bytes = 0;
|
||||
char size_buffer[32];
|
||||
if (!config->quiet)
|
||||
printf("Dry run: files to be transferred\n");
|
||||
while ((chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
if (!config->quiet) {
|
||||
char* escaped_path =
|
||||
output_escape(file_wire_path(chunk->items[i]), config->eight_bit_output);
|
||||
if (!escaped_path) {
|
||||
chunk_destroy(chunk);
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
return -1;
|
||||
}
|
||||
if (config->human_readable)
|
||||
printf(
|
||||
" %s (%s)\n", escaped_path,
|
||||
display_bytes(chunk->items[i]->data->size, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf(" %s (%zu bytes)\n", escaped_path, chunk->items[i]->data->size);
|
||||
free(escaped_path);
|
||||
}
|
||||
total_bytes += chunk->items[i]->data->size;
|
||||
file_count++;
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
/* --delete-missing-args: the missing entries' destination mirrors render as
|
||||
would-be deletions (rsync's dry-run also lists its *deleting lines). */
|
||||
if (missing_dest && !config->quiet) {
|
||||
for (int i = 0; i < missing_dest->size; i++) {
|
||||
char* escaped = output_escape((char*)missing_dest->items[i], config->eight_bit_output);
|
||||
printf(" %s (missing; would be deleted)\n", escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
}
|
||||
}
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
if (!config->quiet) {
|
||||
if (config->human_readable)
|
||||
printf("Total: %d files, %s\n", file_count,
|
||||
display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
char* name; /* transfer-relative name ("" == the source root) */
|
||||
mode_t mode;
|
||||
unsigned long long size;
|
||||
time_t mtime;
|
||||
long mtime_nsec;
|
||||
bool is_dir;
|
||||
bool is_symlink;
|
||||
char* link_target;
|
||||
} ListEntry;
|
||||
|
||||
static void list_entries_destroy(ListEntry* entries, size_t count) {
|
||||
if (entries == NULL)
|
||||
return;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
free(entries[i].name);
|
||||
free(entries[i].link_target);
|
||||
}
|
||||
free(entries);
|
||||
}
|
||||
|
||||
static int compare_list_entries(const void* left, const void* right) {
|
||||
const ListEntry* a = (const ListEntry*)left;
|
||||
const ListEntry* b = (const ListEntry*)right;
|
||||
return strcmp(a->name, b->name);
|
||||
}
|
||||
|
||||
/* Relative path of an entry below `root` ("" for the root itself). Mirrors
|
||||
* change_list's relative_name for list-only rendering. */
|
||||
static char* list_relative_name(const char* root, const char* full) {
|
||||
if (root == NULL || full == NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, full, root_len) == 0) {
|
||||
if (full[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (full[root_len] == '/')
|
||||
return str_dup(full + root_len + 1);
|
||||
}
|
||||
return str_dup(full);
|
||||
}
|
||||
|
||||
/* --list-only: print an ls-style listing of the entries that WOULD be
|
||||
* transferred and exit without contacting the server or writing anything.
|
||||
* Names are transfer-relative (rsync prints `a.txt`, `sub/b.txt`, `.`) and
|
||||
* directory entries are included. Returns 0 on success, 1 on error. */
|
||||
int send_list_only(const Config* config) {
|
||||
int skipped = 0;
|
||||
if (!files_from_list_check(config, NULL, &skipped))
|
||||
return 1;
|
||||
PreparedScanner prepared;
|
||||
if (!prepare_scanner(config, 0, &prepared))
|
||||
return 1;
|
||||
prepared.options.use_metadata = true; /* capture mode + mtime for the listing */
|
||||
prepared.options.list_dirs = true;
|
||||
DirectoryScanner* scanner =
|
||||
directory_scanner_create_with_options(config->send_directory, &prepared.options);
|
||||
if (!scanner) {
|
||||
prepared_scanner_destroy(&prepared);
|
||||
return 1;
|
||||
}
|
||||
ListEntry* entries = NULL;
|
||||
size_t count = 0;
|
||||
size_t capacity = 0;
|
||||
bool oom = false;
|
||||
|
||||
/* rsync lists the source root itself (as "."). Only when the source is a
|
||||
* directory and no --files-from subset is in effect. */
|
||||
if (config->files_from_set == NULL && config->send_directory != NULL) {
|
||||
struct stat st;
|
||||
if (stat(config->send_directory, &st) == 0 && S_ISDIR(st.st_mode)) {
|
||||
capacity = 64;
|
||||
entries = calloc(capacity, sizeof(ListEntry));
|
||||
if (entries == NULL) {
|
||||
oom = true;
|
||||
} else if ((entries[0].name = str_dup("")) == NULL) {
|
||||
/* A NULL name would be dereferenced by qsort/render: fail the listing. */
|
||||
oom = true;
|
||||
} else {
|
||||
entries[0].mode = st.st_mode;
|
||||
entries[0].mtime = st.st_mtime;
|
||||
entries[0].mtime_nsec = st.st_mtim.tv_nsec;
|
||||
entries[0].size = (unsigned long long)st.st_size;
|
||||
entries[0].is_dir = true;
|
||||
count = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Chunk* chunk;
|
||||
while (!oom && (chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* f = chunk->items[i];
|
||||
if (f == NULL)
|
||||
continue;
|
||||
if (count == capacity) {
|
||||
size_t new_capacity = capacity > 0 ? capacity * 2 : 64;
|
||||
if (new_capacity <= capacity) {
|
||||
oom = true;
|
||||
break;
|
||||
}
|
||||
ListEntry* grown = realloc(entries, new_capacity * sizeof(ListEntry));
|
||||
if (!grown) {
|
||||
oom = true;
|
||||
break;
|
||||
}
|
||||
entries = grown;
|
||||
memset(entries + capacity, 0, (new_capacity - capacity) * sizeof(ListEntry));
|
||||
capacity = new_capacity;
|
||||
}
|
||||
char* name = list_relative_name(config->send_directory, file_wire_path(f));
|
||||
if (!name) {
|
||||
oom = true;
|
||||
break;
|
||||
}
|
||||
mode_t mode = 0;
|
||||
time_t mtime = 0;
|
||||
long mtime_nsec = 0;
|
||||
if (f->metadata != NULL) {
|
||||
mode = f->metadata->mode;
|
||||
mtime = f->metadata->mtime_sec;
|
||||
mtime_nsec = f->metadata->mtime_nsec;
|
||||
} else {
|
||||
struct stat st;
|
||||
if (lstat(f->path, &st) == 0) {
|
||||
mode = st.st_mode;
|
||||
mtime = st.st_mtime;
|
||||
mtime_nsec = st.st_mtim.tv_nsec;
|
||||
}
|
||||
}
|
||||
entries[count].name = name;
|
||||
entries[count].mode = mode;
|
||||
entries[count].mtime = mtime;
|
||||
entries[count].mtime_nsec = mtime_nsec;
|
||||
if (f->is_symlink)
|
||||
entries[count].size = f->symlink_target != NULL ? strlen(f->symlink_target) : 0;
|
||||
else if (f->is_dir) {
|
||||
struct stat dir_st;
|
||||
entries[count].size = stat(f->path, &dir_st) == 0 ? (unsigned long long)dir_st.st_size : 0;
|
||||
} else
|
||||
entries[count].size = f->data != NULL ? f->data->size : 0;
|
||||
entries[count].is_dir = f->is_dir;
|
||||
entries[count].is_symlink = f->is_symlink;
|
||||
entries[count].link_target =
|
||||
f->is_symlink && f->symlink_target ? str_dup(f->symlink_target) : NULL;
|
||||
count++;
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
bool failed = oom || directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner);
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
if (failed) {
|
||||
list_entries_destroy(entries, count);
|
||||
if (oom)
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed while listing");
|
||||
return 1;
|
||||
}
|
||||
if (count > 1)
|
||||
qsort(entries, count, sizeof(ListEntry), compare_list_entries);
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.name = entries[i].name;
|
||||
event.path = entries[i].name;
|
||||
event.mode = entries[i].mode;
|
||||
event.size = entries[i].size;
|
||||
event.mtime_sec = entries[i].mtime;
|
||||
event.mtime_nsec = entries[i].mtime_nsec;
|
||||
event.is_directory = entries[i].is_dir;
|
||||
event.is_symlink = entries[i].is_symlink;
|
||||
event.symlink_target = entries[i].link_target;
|
||||
char* line = change_render_list_line(config, &event);
|
||||
if (line != NULL) {
|
||||
char* escaped = output_escape(line, config->eight_bit_output);
|
||||
printf("%s\n", escaped != NULL ? escaped : line);
|
||||
free(escaped);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
list_entries_destroy(entries, count);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Send the delete manifest to the server. Returns 0 on success, -1 on
|
||||
failure. It carries FOUR path sections (keep-set paths, protected excluded
|
||||
prefixes, --delete-missing-args exact-delete paths, and the destination-
|
||||
relative directories the sender synchronized this run) followed by the
|
||||
protocol-2.30.0 per-directory filter-rule block (`per_dir_rules`, the rules
|
||||
the scan compiled from each directory's merge files).
|
||||
When --delete-excluded is given `protected` is empty: excluded destination
|
||||
mirrors are then ordinary extras and are removed. When
|
||||
--delete-missing-args is active `missing_args` holds the destination mirrors
|
||||
of missing --files-from entries: each is an explicit receiver-side deletion
|
||||
request, independent of the extras walk. `synced_dirs` confines the extras
|
||||
walk to entries directly inside a synchronized directory. A NULL
|
||||
keep-set / protected / missing / dirs list transmits an empty section. All
|
||||
four sections are unbounded on the sender; the receiver enforces
|
||||
MAX_MANIFEST_ENTRIES per section and a single MAX_MANIFEST_BYTES budget
|
||||
shared across the sections, rejecting (with STATUS_ERROR) an over-budget
|
||||
frame. A heavily filtered source whose exclusion list is large therefore
|
||||
fails the run cleanly on the receiver rather than being truncated. */
|
||||
int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs,
|
||||
const FilterRuleList* per_dir_rules) {
|
||||
if (!send_status(fd, STATUS_MANIFEST))
|
||||
return -1;
|
||||
int keep_count = manifest ? manifest->size : 0;
|
||||
if (!send_int(fd, keep_count))
|
||||
return -1;
|
||||
for (int i = 0; i < keep_count; i++) {
|
||||
if (!send_wire_str(fd, (char*)manifest->items[i]))
|
||||
return -1;
|
||||
}
|
||||
/* The receiver has ONE protected-prefix section; filter-excluded prefixes
|
||||
(dropped under --delete-excluded) and size-pruned prefixes (always
|
||||
protected) are concatenated into it. */
|
||||
int protected_count =
|
||||
(protected_prefixes ? protected_prefixes->size : 0) + (size_skipped ? size_skipped->size : 0);
|
||||
if (!send_int(fd, protected_count))
|
||||
return -1;
|
||||
if (protected_prefixes) {
|
||||
for (int i = 0; i < protected_prefixes->size; i++) {
|
||||
if (!send_wire_str(fd, (char*)protected_prefixes->items[i]))
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
if (size_skipped) {
|
||||
for (int i = 0; i < size_skipped->size; i++) {
|
||||
if (!send_wire_str(fd, (char*)size_skipped->items[i]))
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
int missing_count = missing_args ? missing_args->size : 0;
|
||||
if (!send_int(fd, missing_count))
|
||||
return -1;
|
||||
for (int i = 0; i < missing_count; i++) {
|
||||
if (!send_wire_str(fd, (char*)missing_args->items[i]))
|
||||
return -1;
|
||||
}
|
||||
int dirs_count = synced_dirs ? synced_dirs->size : 0;
|
||||
if (!send_int(fd, dirs_count))
|
||||
return -1;
|
||||
for (int i = 0; i < dirs_count; i++) {
|
||||
if (!send_wire_str(fd, (char*)synced_dirs->items[i]))
|
||||
return -1;
|
||||
}
|
||||
/* Protocol 2.30.0: the receiver-side per-directory filter rules discovered by
|
||||
the sender's scan, so the whole-tree commit walker can shield a
|
||||
destination-only entry that matches only a per-directory merge rule. */
|
||||
if (!delete_filter_dir_rules_send(fd, per_dir_rules))
|
||||
return -1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Transmit the keep-set manifest and wait for the receiver's verdict. Used by
|
||||
--delete-before/--delete-during, where the extras are removed on the receiver
|
||||
BEFORE the first byte of file data is sent: the receiver acknowledges with
|
||||
STATUS_OK once the bounded delete committed, or STATUS_ERROR if it could not
|
||||
(in which case the sender aborts without streaming any data). The ACK may
|
||||
take much longer than an ordinary per-message round trip because the receiver
|
||||
performs the whole bounded deletion walk (up to MAX_SERVER_DELETE_COUNT
|
||||
unlinks) before replying, so the wait uses a generous explicit deadline
|
||||
instead of the default 60 s receive window. */
|
||||
#define DELETE_ACK_TIMEOUT_SEC 3600
|
||||
/* While waiting for the (potentially slow) receiver-side deletion, send a
|
||||
* STATUS_KEEPALIVE at most this often so the connection is demonstrably alive
|
||||
* and neither side's per-message timeout trips. */
|
||||
#define DELETE_ACK_KEEPALIVE_SEC 10
|
||||
|
||||
bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args,
|
||||
ArrayList* synced_dirs, const FilterRuleList* per_dir_rules) {
|
||||
if (!client || !manifest)
|
||||
return false;
|
||||
if (send_delete_manifest(client->file_descriptor, manifest, protected_prefixes, size_skipped,
|
||||
missing_args, synced_dirs, per_dir_rules) != 0)
|
||||
return false;
|
||||
Status ack;
|
||||
/* The wait is long (up to an hour) and runs inline on this thread: a helper
|
||||
* thread would race the non-thread-safe protocol send path, so keepalives are
|
||||
* emitted from this wait loop itself. A Ctrl-C/SIGTERM abort flag also ends
|
||||
* the wait; the caller then best-effort sends STATUS_ABORT. */
|
||||
if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC,
|
||||
DELETE_ACK_KEEPALIVE_SEC, client_abort_pending)) {
|
||||
/* A Ctrl-C/SIGTERM abort ends the wait above; tell the receiver before the
|
||||
caller tears the connection down (best-effort). */
|
||||
if (client_abort_pending()) {
|
||||
log_info_message(LOG_INFO_MISC,
|
||||
"Abort requested while awaiting delete ack; sending STATUS_ABORT");
|
||||
send_status(client->file_descriptor, STATUS_ABORT);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (ack != STATUS_OK) {
|
||||
log_server_rejection("Server failed to delete files before the transfer");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Server-contacting --dry-run. Connects to the configured remote/daemon and
|
||||
* runs the normal per-file incremental decision WITHOUT transmitting any file
|
||||
* data: the receiver (which also sees dry_run=true on the wire) answers
|
||||
* STATUS_OK for an up-to-date file and STATUS_DRY_RUN_TRANSFER for a file it
|
||||
* would otherwise write, mutating nothing on either side. The would-transfer
|
||||
* set and the same trailer as the local dry-run are printed. A
|
||||
* --compare-dest exact basis hit with no destination copy is reported as a
|
||||
* skip by the receiver.
|
||||
*
|
||||
* Only regular files take the receiver-consulted check; directory / symlink /
|
||||
* special / hard-link-sibling entries have no per-file content check, so they
|
||||
* are reported conservatively as would-transfer and their frames are never
|
||||
* sent (which is what keeps the receiver mutation-free). --delete* is
|
||||
* deliberately NOT transmitted in dry-run, so no deletion can occur; the
|
||||
* would-delete manifest report is a documented follow-up.
|
||||
*
|
||||
* Returns 0 on success, 1 on error. */
|
||||
int send_dry_run_remote(Config* config) {
|
||||
int from_skipped = 0;
|
||||
ArrayList* missing_args = NULL;
|
||||
if (!client_prepare_files_from(config, &missing_args, &from_skipped))
|
||||
return 1;
|
||||
if (missing_args)
|
||||
array_list_delete(missing_args);
|
||||
/* A live session may follow, so arm graceful abort handling. */
|
||||
client_set_abort_armed(true);
|
||||
ProtocolSession session;
|
||||
Client* client = client_connect_and_bind_session(config, &session);
|
||||
if (!client) {
|
||||
client_set_abort_armed(false);
|
||||
return 1;
|
||||
}
|
||||
|
||||
int ret = 1;
|
||||
bool partial = false;
|
||||
time_t dry_start = time(NULL);
|
||||
ReceiverStats dry_stats;
|
||||
memset(&dry_stats, 0, sizeof(dry_stats));
|
||||
PreparedScanner prepared;
|
||||
memset(&prepared, 0, sizeof(prepared));
|
||||
DirectoryScanner* scanner = NULL;
|
||||
ArrayList* dry_manifest = NULL;
|
||||
ArrayList* dry_dirs = NULL;
|
||||
ArrayList* dry_excluded = NULL;
|
||||
ArrayList* dry_size_skipped = NULL;
|
||||
FilterRuleList* dry_per_dir = NULL;
|
||||
if (!config_send(client->file_descriptor, config))
|
||||
goto dry_fail;
|
||||
receive_daemon_motd(client, config);
|
||||
if (!prepare_scanner(config, 0, &prepared))
|
||||
goto dry_fail;
|
||||
/* -n --delete: build the same keep-set manifest, protected prefixes, and
|
||||
synchronized-directory scope a real run would send, so the receiver's
|
||||
read-only extras walk enumerates exactly the deletions a real run makes. */
|
||||
if (config->use_delete) {
|
||||
dry_manifest = array_list_create(free);
|
||||
dry_dirs = array_list_create(free);
|
||||
dry_size_skipped = array_list_create(free);
|
||||
dry_per_dir = filter_rule_list_create();
|
||||
if (!dry_manifest || !dry_dirs || !dry_size_skipped || !dry_per_dir)
|
||||
goto dry_fail;
|
||||
prepared.options.per_dir_rules = dry_per_dir;
|
||||
if (!config->delete_excluded) {
|
||||
dry_excluded = array_list_create(free);
|
||||
if (!dry_excluded)
|
||||
goto dry_fail;
|
||||
prepared.options.excluded_paths = dry_excluded;
|
||||
}
|
||||
prepared.options.size_skipped_paths = dry_size_skipped;
|
||||
/* A --files-from subset confines the extras walk to the directories the
|
||||
scan synchronized; a full recursive transfer marks the root itself. */
|
||||
if (config->files_from_set == NULL) {
|
||||
char* root_marker = delete_scope_root_marker(config);
|
||||
if (!root_marker || !array_list_add(dry_dirs, root_marker)) {
|
||||
free(root_marker);
|
||||
goto dry_fail;
|
||||
}
|
||||
} else {
|
||||
prepared.options.synced_dirs = dry_dirs;
|
||||
}
|
||||
}
|
||||
scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options);
|
||||
if (!scanner)
|
||||
goto dry_fail;
|
||||
|
||||
int file_count = 0;
|
||||
unsigned long long total_bytes = 0;
|
||||
char size_buffer[32];
|
||||
if (!config->quiet)
|
||||
printf("Dry run: files to be transferred\n");
|
||||
Chunk* chunk;
|
||||
while ((chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
if (dry_manifest && !add_chunk_to_manifest(dry_manifest, chunk)) {
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* f = chunk->items[i];
|
||||
if (!f)
|
||||
continue;
|
||||
unsigned long long fsize = f->data ? f->data->size : 0;
|
||||
bool would;
|
||||
if (f->is_dir || f->is_symlink || f->is_special ||
|
||||
(f->link_group != 0 && !f->link_first && f->hardlink_target != NULL)) {
|
||||
/* No receiver-side content check exists for these frame types; a real
|
||||
run would (re)create them, so report would-transfer and send no
|
||||
frame (the receiver must stay mutation-free). */
|
||||
would = true;
|
||||
} else if (fsize > MAX_RECEIVE_WHOLE_FILE_SIZE && !config->use_incremental &&
|
||||
!config_has_basis(config)) {
|
||||
/* A non-incremental run streams a >whole-file-limit source without the
|
||||
STATUS_CHECK handshake, so no read-only receiver decision is possible
|
||||
(and none is needed: a real run would transfer it). */
|
||||
would = true;
|
||||
} else {
|
||||
DeltaSignature* sig = NULL;
|
||||
unsigned long long resume_offset = 0;
|
||||
int rc = incremental_check(client, f, config, &sig, &resume_offset);
|
||||
delta_signature_destroy(sig);
|
||||
if (rc < 0) {
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
if (rc == 1)
|
||||
continue; /* up to date; nothing to report */
|
||||
if (rc != 4) {
|
||||
log_message(LOG_LEVEL_ERROR, "Unexpected receiver reply during dry-run");
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
would = true;
|
||||
}
|
||||
if (would) {
|
||||
if (!config->quiet) {
|
||||
char* escaped_path = output_escape(file_wire_path(f), config->eight_bit_output);
|
||||
if (!escaped_path) {
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
if (config->human_readable)
|
||||
printf(" %s (%s)\n", escaped_path,
|
||||
display_bytes(fsize, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf(" %s (%llu bytes)\n", escaped_path, fsize);
|
||||
free(escaped_path);
|
||||
}
|
||||
total_bytes += fsize;
|
||||
file_count++;
|
||||
}
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
bool io_error = directory_scanner_had_io_error(scanner);
|
||||
if (directory_scanner_failed(scanner))
|
||||
goto dry_fail;
|
||||
if (io_error)
|
||||
log_message(LOG_LEVEL_WARNING, "source scan hit an unreadable directory");
|
||||
/* Send the keep-set manifest (no data frames) so the receiver can enumerate
|
||||
the destination extras; an early-timing delete ACKs before it will accept
|
||||
the terminal FINISHED. */
|
||||
bool early_delete = config->use_delete && config_delete_timing_early(config);
|
||||
if (dry_manifest) {
|
||||
if (send_delete_manifest(client->file_descriptor, dry_manifest, dry_excluded, dry_size_skipped,
|
||||
NULL, dry_dirs, dry_per_dir) != 0)
|
||||
goto dry_fail;
|
||||
if (early_delete) {
|
||||
Status ack;
|
||||
if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC,
|
||||
DELETE_ACK_KEEPALIVE_SEC, client_abort_pending) ||
|
||||
ack != STATUS_OK)
|
||||
goto dry_fail;
|
||||
}
|
||||
}
|
||||
/* Terminate the stream so the receiver emits its success frame; no data frame
|
||||
is ever sent in dry-run. */
|
||||
if (!send_status(client->file_descriptor, STATUS_FINISHED))
|
||||
goto dry_fail;
|
||||
Status status;
|
||||
if (!receive_status(client->file_descriptor, &status))
|
||||
goto dry_fail;
|
||||
if (status == STATUS_STATS) {
|
||||
ArrayList* would_delete = array_list_create(free);
|
||||
if (!would_delete)
|
||||
goto dry_fail;
|
||||
if (!receive_stats_record(client->file_descriptor, &dry_stats, would_delete)) {
|
||||
array_list_delete(would_delete);
|
||||
goto dry_fail;
|
||||
}
|
||||
print_delete_reports(config, would_delete);
|
||||
array_list_delete(would_delete);
|
||||
if (!receive_status(client->file_descriptor, &status))
|
||||
goto dry_fail;
|
||||
}
|
||||
/* A per-entry receiver failure is rsync's PARTIAL transfer (exit 23), not a
|
||||
hard failure: a dry run transfers nothing, but keep the verdict consistent
|
||||
with the normal path instead of treating it as a protocol error. */
|
||||
if (status == STATUS_PARTIAL) {
|
||||
partial = true;
|
||||
} else if (status != STATUS_OK) {
|
||||
goto dry_fail;
|
||||
}
|
||||
if (!config->quiet) {
|
||||
if (config->human_readable)
|
||||
printf("Total: %d files, %s\n", file_count,
|
||||
display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB);
|
||||
}
|
||||
{
|
||||
TransferStats dry_transfer;
|
||||
memset(&dry_transfer, 0, sizeof(dry_transfer));
|
||||
dry_transfer.flist_reg = (unsigned long long)file_count;
|
||||
dry_transfer.total_file_size = total_bytes;
|
||||
dry_transfer.transferred_regular = (unsigned long long)file_count;
|
||||
dry_transfer.transferred_file_size = total_bytes;
|
||||
dry_transfer.literal_data = total_bytes;
|
||||
report_transfer_stats(config, &dry_transfer, dry_start, &dry_stats);
|
||||
}
|
||||
ret = io_error ? 1 : (partial ? 23 : 0);
|
||||
|
||||
dry_fail:
|
||||
if (dry_manifest)
|
||||
array_list_delete(dry_manifest);
|
||||
if (dry_dirs)
|
||||
array_list_delete(dry_dirs);
|
||||
if (dry_excluded)
|
||||
array_list_delete(dry_excluded);
|
||||
if (dry_size_skipped)
|
||||
array_list_delete(dry_size_skipped);
|
||||
if (dry_per_dir)
|
||||
filter_rule_list_free(dry_per_dir);
|
||||
if (scanner)
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
disconnect_transfer_client(client);
|
||||
protocol_session_unbind();
|
||||
client_set_abort_armed(false);
|
||||
return ret;
|
||||
}
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,470 @@
|
||||
#include "client_send_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "charset.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "file_list.h"
|
||||
#include "filter.h"
|
||||
#include "hardlink.h"
|
||||
#include "log.h"
|
||||
#include "scanner.h"
|
||||
#include "utils.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Build the scanner options for one scan. Returns false and logs on failure. */
|
||||
bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out) {
|
||||
if (!out)
|
||||
return false;
|
||||
out->base_filters = NULL;
|
||||
out->hardlinks = NULL;
|
||||
out->relative_prefix = NULL;
|
||||
memset(&out->options, 0, sizeof(out->options));
|
||||
|
||||
int rule_count = config->filters ? config->filters->size : 0;
|
||||
const char** texts = NULL;
|
||||
if (rule_count > 0) {
|
||||
texts = malloc((size_t)rule_count * sizeof(char*));
|
||||
if (!texts) {
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed for filter rules");
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < rule_count; i++)
|
||||
texts[i] = (const char*)config->filters->items[i];
|
||||
}
|
||||
if (rule_count > 0 || config->cvs_exclude) {
|
||||
char err[160];
|
||||
out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude,
|
||||
config->delete_excluded, err, sizeof(err));
|
||||
free(texts);
|
||||
if (!out->base_filters) {
|
||||
log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err);
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
free(texts);
|
||||
}
|
||||
|
||||
ScannerOptions* options = &out->options;
|
||||
options->use_metadata = config->use_metadata;
|
||||
options->preserve_atimes = config->preserve_atimes;
|
||||
options->preserve_crtimes = config->preserve_crtimes;
|
||||
options->preserve_xattrs = config->preserve_xattrs;
|
||||
options->preserve_acls = config->preserve_acls;
|
||||
options->chunk_size = config->chunk_size;
|
||||
/* --exclude/--include are compiled, in command-line order, into the SAME
|
||||
* ordered filter rule list as --filter/-f (see config_add_selection_rule), so
|
||||
* the legacy per-kind arrays are deliberately NOT passed to the scanner:
|
||||
* doing so would re-apply them with the old "excludes first, then includes as
|
||||
* a mandatory whitelist" precedence and defeat rsync's first-match-wins
|
||||
* ordering. The arrays remain populated purely for the Config API surface. */
|
||||
options->exclude_patterns = NULL;
|
||||
options->exclude_count = 0;
|
||||
options->include_patterns = NULL;
|
||||
options->include_count = 0;
|
||||
options->max_size = config->max_size;
|
||||
options->min_size = config->min_size;
|
||||
options->max_depth = config->max_depth;
|
||||
options->num_threads = num_threads;
|
||||
options->follow_symlinks = config->follow_symlinks;
|
||||
options->copy_links = config->copy_links;
|
||||
options->safe_links = config->safe_links;
|
||||
options->copy_unsafe_links = config->copy_unsafe_links;
|
||||
options->copy_dirlinks = config->copy_dirlinks;
|
||||
options->munge_links = config->munge_links;
|
||||
options->checksum = config->checksum;
|
||||
options->one_file_system = config->one_file_system;
|
||||
options->preserve_devices = config->preserve_devices;
|
||||
options->preserve_specials = config->preserve_specials;
|
||||
options->copy_devices = config->copy_devices;
|
||||
options->file_list = (const FileListSet*)config->files_from_set;
|
||||
options->base_filters = out->base_filters;
|
||||
options->per_dir_filters = config->per_dir_filter;
|
||||
options->delete_excluded = config->delete_excluded;
|
||||
options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2;
|
||||
options->dirs = config->dirs;
|
||||
options->relative = config->relative;
|
||||
/* A real recursive transfer recreates empty source directories (rsync
|
||||
parity); low-level scanner users leave this off. */
|
||||
options->emit_empty_dirs = true;
|
||||
/* --no-implied-dirs only has meaning with -R (rsync): without it the option
|
||||
is a documented no-op, so the scanner must not suppress directory
|
||||
metadata. */
|
||||
options->no_implied_dirs = config->no_implied_dirs && config->relative;
|
||||
/* -R/--relative outside --files-from reconstructs every destination path from
|
||||
* the source spec (rsync's '/./' cut point). With --files-from the listed
|
||||
* entry already supplies the bare relative path, so no prefix is built. */
|
||||
if (config->relative && config->files_from_set == NULL && config->send_directory) {
|
||||
out->relative_prefix = scanner_relative_prefix(config->send_directory);
|
||||
if (!out->relative_prefix) {
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed building --relative path prefix");
|
||||
filter_rule_list_free(out->base_filters);
|
||||
out->base_filters = NULL;
|
||||
return false;
|
||||
}
|
||||
options->relative_prefix = out->relative_prefix;
|
||||
}
|
||||
options->prune_empty_dirs = config->prune_empty_dirs;
|
||||
options->ignore_io_errors = config->ignore_errors;
|
||||
options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args;
|
||||
options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet;
|
||||
options->note_mount = (config->info_level & LOG_INFO_MOUNT) != 0 && !config->quiet;
|
||||
options->send_directory = config->send_directory;
|
||||
options->eight_bit_output = config->eight_bit_output;
|
||||
options->excluded_paths = NULL;
|
||||
options->excluded_mutex = NULL;
|
||||
options->size_skipped_paths = NULL;
|
||||
options->synced_dirs = NULL;
|
||||
options->hardlinks = NULL;
|
||||
/* Set by the real send paths; NULL for the metadata-only scans (progress
|
||||
pre-count, batch) that must not perturb the sender's --stats counter. */
|
||||
options->dir_count = NULL;
|
||||
/* P7 Wave D: capture source directory metadata when a directory attribute is
|
||||
requested (-p for modes, -t for times unless -O omits them). Whether they
|
||||
are APPLIED is decided receiver-side. */
|
||||
options->capture_dir_times = dir_metadata_should_capture(config);
|
||||
options->dir_entries = NULL;
|
||||
options->dir_entries_mutex = NULL;
|
||||
if (config->preserve_hard_links) {
|
||||
out->hardlinks = hardlink_table_create();
|
||||
if (!out->hardlinks) {
|
||||
filter_rule_list_free(out->base_filters);
|
||||
out->base_filters = NULL;
|
||||
return false;
|
||||
}
|
||||
options->hardlinks = out->hardlinks;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void prepared_scanner_destroy(PreparedScanner* prepared) {
|
||||
if (!prepared)
|
||||
return;
|
||||
filter_rule_list_free(prepared->base_filters);
|
||||
prepared->base_filters = NULL;
|
||||
hardlink_table_destroy(prepared->hardlinks);
|
||||
prepared->hardlinks = NULL;
|
||||
free(prepared->relative_prefix);
|
||||
prepared->relative_prefix = NULL;
|
||||
}
|
||||
|
||||
/* -R/--relative implied directories: rsync transmits the metadata of the
|
||||
* parent directories implied by the source path (every prefix component above
|
||||
* the source root) so the receiver applies their attributes to the created
|
||||
* parents. FastSync's scan only covers the source root and below, so append
|
||||
* one metadata-only directory entry per implied ancestor. --no-implied-dirs
|
||||
* suppresses this exactly like rsync. A missing ancestor is never fatal. */
|
||||
bool append_implied_dir_times(const Config* config, ArrayList* dir_entries) {
|
||||
if (!dir_entries || !config->relative || config->files_from_set != NULL ||
|
||||
config->no_implied_dirs || !config->send_directory)
|
||||
return true;
|
||||
char* prefix = scanner_relative_prefix(config->send_directory);
|
||||
if (!prefix)
|
||||
return true;
|
||||
int ncomp = 0;
|
||||
for (const char* s = prefix; *s;) {
|
||||
while (*s == '/')
|
||||
s++;
|
||||
if (!*s)
|
||||
break;
|
||||
while (*s && *s != '/')
|
||||
s++;
|
||||
ncomp++;
|
||||
}
|
||||
if (ncomp <= 1) {
|
||||
free(prefix);
|
||||
return true;
|
||||
}
|
||||
char* fs = str_dup(config->send_directory);
|
||||
if (!fs) {
|
||||
free(prefix);
|
||||
return true;
|
||||
}
|
||||
size_t flen = strlen(fs);
|
||||
while (flen > 1 && fs[flen - 1] == '/')
|
||||
fs[--flen] = '\0';
|
||||
bool ok = true;
|
||||
/* Walk the source path upwards one component at a time (fs is truncated in
|
||||
place, so each step targets the next implied ancestor). */
|
||||
for (int depth = ncomp - 2; depth >= 0 && ok; depth--) {
|
||||
char* slash = strrchr(fs, '/');
|
||||
if (!slash || slash == fs)
|
||||
break;
|
||||
*slash = '\0';
|
||||
char* p = prefix;
|
||||
int c = 0;
|
||||
while (c <= depth) {
|
||||
while (*p == '/')
|
||||
p++;
|
||||
while (*p && *p != '/')
|
||||
p++;
|
||||
c++;
|
||||
}
|
||||
char saved = *p;
|
||||
*p = '\0';
|
||||
struct stat st;
|
||||
if (stat(fs, &st) == 0 && S_ISDIR(st.st_mode)) {
|
||||
File* file = file_create(fs);
|
||||
if (!file) {
|
||||
ok = false;
|
||||
} else {
|
||||
file->is_dir = true;
|
||||
file->metadata =
|
||||
file_metadata_create(fs, &st, config->preserve_atimes, config->preserve_crtimes);
|
||||
file->send_path = str_dup(prefix);
|
||||
if (!file->metadata || !file->send_path || !array_list_add(dir_entries, file)) {
|
||||
file_destroy(file);
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
*p = saved;
|
||||
}
|
||||
free(fs);
|
||||
free(prefix);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* The delete-walk root scope for a full (non---files-from) transfer: rsync
|
||||
* confines --delete to the directories it actually transferred. A plain
|
||||
* recursive run mirrors the source under the receive root, so "." (the whole
|
||||
* tree) is correct; an -R run transfers only the reconstructed prefix subtree,
|
||||
* so the walk is scoped to that prefix instead. Returns a malloc'd wire path
|
||||
* (or "."), or NULL on allocation failure. */
|
||||
char* delete_scope_root_marker(const Config* config) {
|
||||
if (config->relative && config->files_from_set == NULL && config->send_directory) {
|
||||
char* prefix = scanner_relative_prefix(config->send_directory);
|
||||
if (!prefix)
|
||||
return NULL;
|
||||
if (prefix[0] != '\0')
|
||||
return prefix;
|
||||
free(prefix);
|
||||
}
|
||||
return str_dup(".");
|
||||
}
|
||||
|
||||
/* The -R destination prefix that confines a per-directory delete walk, or NULL
|
||||
* when the whole receive root is in scope. The marker was installed into
|
||||
* `synced_dirs` by delete_scope_root_marker(); for a plain recursive transfer
|
||||
* it is "." (whole root) and for --files-from the list is not a single prefix. */
|
||||
const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs) {
|
||||
if (!config || config->files_from_set != NULL || !config->relative || !config->send_directory)
|
||||
return NULL;
|
||||
if (!synced_dirs || synced_dirs->size != 1)
|
||||
return NULL;
|
||||
const char* marker = (const char*)synced_dirs->items[0];
|
||||
if (marker[0] == '\0' || strcmp(marker, ".") == 0)
|
||||
return NULL;
|
||||
return marker;
|
||||
}
|
||||
|
||||
/* The destination-relative mirror path for a missing --files-from entry: where
|
||||
a PRESENT entry with the same name would have been written. With -R that is
|
||||
the entry's bare relative path (the bare wire path the receiver uses);
|
||||
otherwise it is the full source mirror below the destination root
|
||||
(`send_directory` joined to the entry, leading '/' stripped), exactly the
|
||||
path the manifest records for a present sibling. Returns an owned string, or
|
||||
NULL on allocation failure. */
|
||||
static char* files_from_missing_dest_path(const Config* config, const char* entry) {
|
||||
if (config->relative)
|
||||
return str_dup(entry);
|
||||
char* joined = path_cat(config->send_directory, entry);
|
||||
if (!joined)
|
||||
return NULL;
|
||||
const char* rel = *joined == '/' ? joined + 1 : joined;
|
||||
char* dup = str_dup(rel);
|
||||
free(joined);
|
||||
return dup;
|
||||
}
|
||||
|
||||
/* --files-from semantics: every listed entry must resolve under the source
|
||||
* root, otherwise rsync reports a hard error instead of silently transferring
|
||||
* nothing. An entry of "." (the whole tree) and listed-but-empty directories
|
||||
* are valid. An empty list is valid too: rsync transfers nothing and exits 0.
|
||||
* With --ignore-missing-args
|
||||
* (implied by --delete-missing-args) a listed-but-missing entry is instead
|
||||
* skipped: nothing is transferred for it, it never enters the keep-set and the
|
||||
* run succeeds for the rest (an all-missing non-empty list succeeds
|
||||
* transferring nothing, matching rsync). With --delete-missing-args
|
||||
* `missing_dest` (when non-NULL) collects the entry's destination-relative
|
||||
* mirror for the receiver's exact-deletion request. Runs before any
|
||||
* transfer so the failure/skip is surfaced uniformly in the single-threaded,
|
||||
* -m, dry-run and --list-only paths. */
|
||||
bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out) {
|
||||
*skipped_out = 0;
|
||||
const FileListSet* set = (const FileListSet*)config->files_from_set;
|
||||
if (!set)
|
||||
return true;
|
||||
if (!config->send_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "--files-from requires a source directory");
|
||||
return false;
|
||||
}
|
||||
if (set->count == 0) {
|
||||
/* rsync treats an empty --files-from list as "nothing to transfer" and
|
||||
exits 0 (the source directory is still a valid source arg), so this is
|
||||
not an error. Nothing passes the (empty) allow-set, so no file is sent
|
||||
and no keep-set entry is produced. */
|
||||
return true;
|
||||
}
|
||||
bool ignore = config->ignore_missing_args || config->delete_missing_args;
|
||||
for (int i = 0; i < set->count; i++) {
|
||||
const char* entry = set->entries[i];
|
||||
if (entry[0] == '\0')
|
||||
continue; /* "." == list the whole tree */
|
||||
char* full = path_cat(config->send_directory, entry);
|
||||
if (!full) {
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from");
|
||||
return false;
|
||||
}
|
||||
struct stat st;
|
||||
if (lstat(full, &st) != 0) {
|
||||
free(full);
|
||||
if (ignore) {
|
||||
(*skipped_out)++;
|
||||
char* escaped_entry = output_escape(entry, log_get_8_bit_output());
|
||||
log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'",
|
||||
escaped_entry ? escaped_entry : "<allocation failed>");
|
||||
free(escaped_entry);
|
||||
if (config->delete_missing_args && missing_dest) {
|
||||
char* mirror = files_from_missing_dest_path(config, entry);
|
||||
if (!mirror || !array_list_add(missing_dest, mirror)) {
|
||||
free(mirror);
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
char* escaped_entry = output_escape(entry, log_get_8_bit_output());
|
||||
char* escaped_src = output_escape(config->send_directory, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_ERROR, "--files-from entry '%s' not found in source '%s'",
|
||||
escaped_entry ? escaped_entry : "<allocation failed>",
|
||||
escaped_src ? escaped_src : "<allocation failed>");
|
||||
free(escaped_entry);
|
||||
free(escaped_src);
|
||||
return false;
|
||||
}
|
||||
free(full);
|
||||
}
|
||||
if (*skipped_out > 0) {
|
||||
if (config->delete_missing_args) {
|
||||
/* --list-only never deletes and a --dry-run only shows intent, so the
|
||||
summary must not claim a real deletion happened in those modes. */
|
||||
if (config->list_only)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--delete-missing-args: %d missing --files-from entr%s skipped (--list-only "
|
||||
"never deletes)",
|
||||
*skipped_out, *skipped_out == 1 ? "y" : "ies");
|
||||
else if (config->dry_run)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--delete-missing-args: %d missing --files-from entr%s would be deleted from "
|
||||
"the destination (dry run)",
|
||||
*skipped_out, *skipped_out == 1 ? "y" : "ies");
|
||||
else
|
||||
log_message(
|
||||
LOG_LEVEL_WARNING,
|
||||
"--delete-missing-args: %d missing --files-from entr%s will be deleted from the "
|
||||
"destination",
|
||||
*skipped_out, *skipped_out == 1 ? "y" : "ies");
|
||||
} else if (config->ignore_missing_args)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out,
|
||||
*skipped_out == 1 ? "y" : "ies");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Walk the whole source tree once collecting only destination-relative wire
|
||||
paths, loading and sending nothing. --delete-before/--delete-during need the
|
||||
complete keep-set manifest before the first data byte, so it is built by a
|
||||
dedicated pre-scan pass and transmitted early; the data pass then re-scans
|
||||
with a fresh scanner. --delete-before additionally replays this very scan as
|
||||
its data pass (rsync's single file list), so `chunks_out` (optional) retains
|
||||
the scanned Chunk objects for the caller to send instead of destroying them;
|
||||
the caller owns the list and must give it a chunk_destroy destructor. A
|
||||
source I/O error is fatal unless the options carry --ignore-errors, in which
|
||||
case the scan continues past the unreadable directory and *io_error_out
|
||||
reports it (the caller still performs the deletion but reports the run as
|
||||
errored). */
|
||||
bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest,
|
||||
DeletePlanSender* plans, bool* io_error_out,
|
||||
unsigned long long* non_dir_count_out, ArrayList* chunks_out,
|
||||
bool emit_nonreg) {
|
||||
if (io_error_out)
|
||||
*io_error_out = false;
|
||||
if (non_dir_count_out)
|
||||
*non_dir_count_out = 0;
|
||||
ScannerOptions local = *options;
|
||||
/* The pre-scan is normally a paths-only pass with no client output: it must
|
||||
not emit --info=nonreg lines because the data pass re-scans and emits them
|
||||
once. When the caller replays this scan as the data pass (--delete-before)
|
||||
there is no later scan, so it opts in and the lines are emitted here. */
|
||||
local.note_nonreg = emit_nonreg && options->note_nonreg;
|
||||
DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local);
|
||||
if (!scanner)
|
||||
return false;
|
||||
bool ok = true;
|
||||
Chunk* chunk;
|
||||
while ((chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
if (non_dir_count_out) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
const File* f = chunk->items[i];
|
||||
if (f && !f->is_dir)
|
||||
(*non_dir_count_out)++;
|
||||
}
|
||||
}
|
||||
if (manifest && !add_chunk_to_manifest(manifest, chunk)) {
|
||||
ok = false;
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
if (plans) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* f = chunk->items[i];
|
||||
if (!f)
|
||||
continue;
|
||||
const char* path = file_wire_path(f);
|
||||
if (!delete_plan_sender_add(plans, path, f->is_dir)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!ok) {
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (chunks_out) {
|
||||
/* Retain the chunk for the caller's data pass; ownership moves with it. */
|
||||
if (!array_list_add(chunks_out, chunk)) {
|
||||
ok = false;
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
}
|
||||
if (ok) {
|
||||
/* Keep every traversed source directory, including empty ones, so a plan
|
||||
no longer removes the destination directory itself. Their own plans are
|
||||
emitted after the data stream (no file frame triggers them). */
|
||||
if (plans && options->plan_dirs) {
|
||||
for (int i = 0; i < options->plan_dirs->size; i++) {
|
||||
if (!delete_plan_sender_add(plans, (const char*)options->plan_dirs->items[i], true)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (ok && directory_scanner_failed(scanner))
|
||||
ok = false;
|
||||
if (io_error_out)
|
||||
*io_error_out = directory_scanner_had_io_error(scanner);
|
||||
directory_scanner_destroy(scanner);
|
||||
return ok;
|
||||
}
|
||||
+1019
-1061
File diff suppressed because it is too large.
Load diff
@@ -22,7 +22,13 @@ void client_set_abort_armed(bool armed);
|
||||
* never free it, and the caller retains ownership (freeing it with
|
||||
* config_delete() once the call returns). */
|
||||
int send_files(Config* config);
|
||||
int send_files_multithreaded(Config** config);
|
||||
int send_files_multithreaded(Config* config);
|
||||
/* rsync's --ignore-errors deletion gate: with no I/O error during the scan the
|
||||
* deletion phase always proceeds; with one it is suppressed unless
|
||||
* `--ignore-errors` was given. Exposed so the decision can be unit-tested
|
||||
* without a privileged (mode-000) source directory. See client_send.c. */
|
||||
bool ignore_errors_allows_delete(const Config* config, bool had_io_error);
|
||||
|
||||
/* Phase 6 residual-batch (client-only). See client_send.c. */
|
||||
int write_batch_from_source(const Config* config, const char* batch_path);
|
||||
int apply_batch_to_dest(const Config* config, const char* batch_path, const char* dest_root);
|
||||
|
||||
@@ -0,0 +1,126 @@
|
||||
#ifndef CLIENT_SEND_INTERNAL_H
|
||||
#define CLIENT_SEND_INTERNAL_H
|
||||
|
||||
/* Declarations shared between the client_send.c transfer orchestration and the
|
||||
* reporting (client_report.c), scanner-preparation (client_scan.c) and
|
||||
* manifest/list/dry-run (client_manifest.c) translation units that were split
|
||||
* out of it. Nothing here is part of the public client_send.h facade. */
|
||||
|
||||
#include "array_list.h"
|
||||
#include "client_send.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "delta.h"
|
||||
#include "format.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "scanner.h"
|
||||
#include <stdatomic.h>
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <time.h>
|
||||
|
||||
/* One mebibyte in bytes; the unit used by the --stats/--progress lines.
|
||||
Always cast to double when dividing so the output stays fractional. */
|
||||
#define BYTES_PER_MIB (1024ULL * 1024ULL)
|
||||
|
||||
/* Compiled scanner inputs that are shared read-only across scanner instances
|
||||
* and, in -m mode, across worker threads. `base_filters` owns the compiled
|
||||
* command-line + -C rules; the FileListSet allow-set lives in the Config.
|
||||
* `hardlinks` owns the --hard-links/-H link-group detection table (NULL when
|
||||
* off) and is shared (mutex-guarded) across every scanner/worker of one scan. */
|
||||
typedef struct {
|
||||
ScannerOptions options;
|
||||
FilterRuleList* base_filters; /* owned; may be NULL */
|
||||
HardLinkTable* hardlinks; /* owned; may be NULL */
|
||||
char* relative_prefix; /* owned -R prefix; may be NULL */
|
||||
} PreparedScanner;
|
||||
|
||||
/* client_scan.c */
|
||||
bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out);
|
||||
void prepared_scanner_destroy(PreparedScanner* prepared);
|
||||
bool append_implied_dir_times(const Config* config, ArrayList* dir_entries);
|
||||
char* delete_scope_root_marker(const Config* config);
|
||||
const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs);
|
||||
bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out);
|
||||
bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest,
|
||||
DeletePlanSender* plans, bool* io_error_out,
|
||||
unsigned long long* non_dir_count_out, ArrayList* chunks_out,
|
||||
bool emit_nonreg);
|
||||
|
||||
/* client_report.c */
|
||||
void log_server_rejection(const char* context);
|
||||
const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer,
|
||||
size_t buffer_size);
|
||||
unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries,
|
||||
atomic_ullong* counter);
|
||||
void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start,
|
||||
const ReceiverStats* recv);
|
||||
void transfer_stats_note_entry(TransferStats* stats, const File* file);
|
||||
void transfer_stats_note_transferred(TransferStats* stats, const File* file);
|
||||
bool info_flag_enabled(const Config* config, LogInfoFlag flag);
|
||||
void print_delete_reports(const Config* config, const ArrayList* paths);
|
||||
const char* delete_display_path(const Config* config, const char* path);
|
||||
bool progress_requested(const Config* config);
|
||||
void client_progress_cleanup(void);
|
||||
void client_progress_begin(const Config* config);
|
||||
void client_progress_file(const Config* config, const File* file);
|
||||
void client_progress_name(const Config* config, const File* file);
|
||||
/* Emit a transferred entry's ancestor directories (as -i/--out-format change
|
||||
* lines or --progress name lines) before the entry's own line. */
|
||||
void client_change_emit_ancestors(const Config* config, const File* file);
|
||||
/* Output parity (protocol 2.30.0): probe each not-yet-known ancestor directory's
|
||||
* pre-transfer destination state before the entry that first triggers it is
|
||||
* sent. Returns false on a protocol/transport error. */
|
||||
bool client_change_probe_ancestors(const Config* config, const File* file, int fd);
|
||||
/* Mark a transferred directory entry as already reported, and flush the
|
||||
* itemize lines for changed directories that had no transferred child. */
|
||||
void client_change_mark_dir(const Config* config, const File* file);
|
||||
void client_change_emit_pending_dirs(const Config* config, int fd);
|
||||
void client_progress_uptodate(const Config* config, const File* file);
|
||||
void client_progress_prepare(const Config* config, const ArrayList* plan_dirs,
|
||||
unsigned long long plan_non_dir_count);
|
||||
bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete);
|
||||
/* --stderr=client diagnostic channel (client_report.c): install the queueing
|
||||
* log sink for a transfer, mark the session live, flush queued diagnostics over
|
||||
* the wire at a frame boundary, and tear the sink down. */
|
||||
void client_messages_install(void);
|
||||
void client_messages_activate(bool active);
|
||||
void client_flush_client_messages(int fd);
|
||||
void client_messages_end(void);
|
||||
|
||||
/* client_send.c */
|
||||
void receive_daemon_motd(Client* client, const Config* config);
|
||||
Client* connect_transfer_client(const Config* config);
|
||||
/* Connect the configured transport and install `session` on it: init with the
|
||||
* socket fd pair, apply the I/O timeout and (when negotiated) the TLS object,
|
||||
* then bind the session to this thread. Returns the connected client, or NULL
|
||||
* after logging the connect failure. The caller owns the client and must keep
|
||||
* `session` alive until it calls protocol_session_unbind(). */
|
||||
Client* client_connect_and_bind_session(const Config* config, ProtocolSession* session);
|
||||
/* Shared --files-from/--delete-missing-args preamble for the send entry points:
|
||||
* when --delete-missing-args is set, allocate the list that
|
||||
* files_from_list_check fills with the destination mirrors of missing entries;
|
||||
* then validate the --files-from list. On success returns true and stores the
|
||||
* (possibly NULL) owned list in *missing_args_out plus the skipped count; on
|
||||
* failure returns false after freeing the list. */
|
||||
bool client_prepare_files_from(const Config* config, ArrayList** missing_args_out,
|
||||
int* skipped_out);
|
||||
void disconnect_transfer_client(Client* client);
|
||||
int incremental_check(Client* client, File* file, const Config* config, DeltaSignature** out_sig,
|
||||
unsigned long long* resume_offset);
|
||||
|
||||
/* client_manifest.c */
|
||||
bool dry_run_targets_server(const Config* config);
|
||||
bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk);
|
||||
int send_dry_run_manifest(const Config* config);
|
||||
int send_list_only(const Config* config);
|
||||
int send_dry_run_remote(Config* config);
|
||||
int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs,
|
||||
const FilterRuleList* per_dir_rules);
|
||||
bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args,
|
||||
ArrayList* synced_dirs, const FilterRuleList* per_dir_rules);
|
||||
|
||||
#endif
|
||||
@@ -22,6 +22,19 @@ bool validate_config(const Config* config) {
|
||||
"--write-batch, --only-write-batch, and --read-batch are mutually exclusive");
|
||||
return false;
|
||||
}
|
||||
/* A dry-run of a local batch apply is not meaningful: --read-batch bypasses
|
||||
the client-side scan/server decision entirely, so dry-run would have no
|
||||
wire state to report (and must not be used as a mutation escape hatch).
|
||||
--only-write-batch likewise never contacts a receiver. --write-batch DOES
|
||||
run a live transfer but additionally mutates the filesystem by emitting the
|
||||
batch file, so a dry-run must not write it either. Reject all three up
|
||||
front instead of silently ignoring --dry-run. */
|
||||
if (config->dry_run && (read_batch || only_write_batch || write_batch)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--dry-run cannot be combined with --read-batch, --only-write-batch, or "
|
||||
"--write-batch; a dry-run must not mutate anything, including batch files");
|
||||
return false;
|
||||
}
|
||||
if (read_batch) {
|
||||
if (!config->receive_root_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "--read-batch requires a destination directory");
|
||||
@@ -40,18 +53,35 @@ bool validate_config(const Config* config) {
|
||||
return false;
|
||||
}
|
||||
if (config->compression_threads > 0 && !config->use_compression) {
|
||||
log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-c or -z)");
|
||||
log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-z/--compress)");
|
||||
return false;
|
||||
}
|
||||
if (config->transport == TRANSPORT_SSH && config->use_sendfile) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport");
|
||||
return false;
|
||||
}
|
||||
/* -M/--remote-option appends an option to the REMOTE server's argv, which
|
||||
* only exists on the SSH (user@host:path) transport. A daemon
|
||||
* (host::module/path) or local TCP destination has no remote command line,
|
||||
* so the option would be silently ignored; reject it by name instead. */
|
||||
if (config->remote_option_count > 0 && config->transport != TRANSPORT_SSH) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"-M/--remote-option is only valid with the SSH transport (user@host:path); it "
|
||||
"cannot be used with a daemon (host::module/path) or local TCP destination");
|
||||
return false;
|
||||
}
|
||||
/* -4 and -6 are mutually exclusive: a socket address family cannot be both. */
|
||||
if (config->ipv4 && config->ipv6) {
|
||||
log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive");
|
||||
return false;
|
||||
}
|
||||
/* rsync 3.4.1 rejects --inplace together with --partial-dir (exit 1): the
|
||||
inplace write path bypasses partial staging, so a partial-dir name would be
|
||||
silently ignored. Match rsync's message and refuse before any I/O. */
|
||||
if (config->inplace && config->partial_dir) {
|
||||
log_message(LOG_LEVEL_ERROR, "--inplace cannot be used with --partial-dir");
|
||||
return false;
|
||||
}
|
||||
if (config->log_file_format && !config->log_file) {
|
||||
log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file");
|
||||
return false;
|
||||
@@ -79,6 +109,16 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s", invariants_error);
|
||||
return false;
|
||||
}
|
||||
/* The receiver rejects a protect-rule block with more than MAX_FILTER_RULES
|
||||
entries as an opaque protocol error; reject an over-limit --filter set here,
|
||||
before any network I/O, with an actionable message. send_protect_entries()
|
||||
re-checks the final built count because cvs-exclude / merge rules can
|
||||
expand it beyond config->filters->size. */
|
||||
if (config->filters && config->filters->size > MAX_FILTER_RULES) {
|
||||
log_message(LOG_LEVEL_ERROR, "too many filter rules: %d (maximum %d)", config->filters->size,
|
||||
MAX_FILTER_RULES);
|
||||
return false;
|
||||
}
|
||||
/* --protocol: FastSync has exactly one wire format, so the forced version
|
||||
must equal the current PROTOCOL_VERSION exactly. Rejected here, before any
|
||||
network I/O, rather than letting the server hit its own mismatch check. */
|
||||
|
||||
+640
-1125
File diff suppressed because it is too large.
Load diff
+125
-16
@@ -10,6 +10,7 @@
|
||||
#include "stop_condition.h"
|
||||
#include <dirent.h>
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdatomic.h>
|
||||
#include <sys/types.h>
|
||||
#include <threads.h>
|
||||
@@ -18,6 +19,11 @@
|
||||
* keeps one transfer from spawning an unbounded pool on a very large machine. */
|
||||
#define MAX_SCANNER_THREADS 256
|
||||
|
||||
/* Depth of the scanner's work queues: the sequential scanner's pending-directory
|
||||
* stack and the parallel scanner's result queue. Bounds memory for a very wide
|
||||
* or very deep tree while leaving ample headroom for normal scans. */
|
||||
#define SCANNER_RESULT_QUEUE_CAP 100
|
||||
|
||||
typedef struct {
|
||||
bool use_metadata;
|
||||
/* Phase 4 metadata capture: -U/--atimes and -N/--crtimes tell the scanner to
|
||||
@@ -50,7 +56,7 @@ typedef struct {
|
||||
bool copy_dirlinks;
|
||||
bool munge_links;
|
||||
bool checksum;
|
||||
bool one_file_system;
|
||||
int one_file_system;
|
||||
/* Phase 4 special/devices: whether device nodes (--devices) and special files
|
||||
* (--specials) are preserved via recreation, and whether --copy-devices
|
||||
* copies a device's content as an ordinary regular file. */
|
||||
@@ -63,8 +69,23 @@ typedef struct {
|
||||
const FileListSet* file_list; /* --files-from allow-set, or NULL */
|
||||
const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */
|
||||
bool per_dir_filters; /* -F: read .rsync-filter per directory */
|
||||
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
|
||||
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
|
||||
/* --delete-excluded: per-directory plain rules become sender-only, so they no
|
||||
longer protect the receiver from deletion. */
|
||||
bool delete_excluded;
|
||||
/* -FF: also exclude the per-directory filter files themselves from the
|
||||
transfer (single -F transfers them). */
|
||||
bool exclude_per_dir_filter_files;
|
||||
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
|
||||
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
|
||||
/* -R/--relative outside --files-from: the destination-relative path prefix
|
||||
* reconstructed from the source spec (rsync's '/./' cut point), or NULL when
|
||||
* -R is off or --files-from is in use (the bare-relative path then comes from
|
||||
* the listed entry). Borrowed read-only; owned by client_send. */
|
||||
const char* relative_prefix;
|
||||
/* --list-only: emit an is_dir File for every traversed directory (the listing
|
||||
* includes directory entries, matching rsync). Client-only; never set on a
|
||||
* real transfer, which relies on implicit parent creation. */
|
||||
bool list_dirs;
|
||||
/* --prune-empty-dirs (long only): in --dirs mode an empty source directory's
|
||||
explicit entry is omitted from the transfer file list (so nothing is
|
||||
created at the destination and it can be pruned by --delete); explicitly
|
||||
@@ -73,20 +94,69 @@ typedef struct {
|
||||
bool prune_empty_dirs;
|
||||
/* Delete-excluded protection sink (optional): when non-NULL the scanner
|
||||
* appends the destination-relative path of every entry it prunes because a
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules, the legacy
|
||||
* --exclude/--include layer, and --max-size/--min-size). The sender turns
|
||||
* this list into the manifest's protected prefixes so `--delete` leaves the
|
||||
* destination mirror of excluded source paths alone (rsync's default), and
|
||||
* empties it when --delete-excluded opts back into deleting them. NOT
|
||||
* recorded for --files-from subset pruning (whose delete semantics stay
|
||||
* keep-set-only) or for -R/--files-from relative wire paths. When
|
||||
* `excluded_mutex` is non-NULL it is taken around every append (the parallel
|
||||
* scanner shares one list across its worker threads). */
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules and the legacy
|
||||
* --exclude/--include layer). The sender turns this list into the manifest's
|
||||
* protected prefixes so `--delete` leaves the destination mirror of excluded
|
||||
* source paths alone (rsync's default), and drops it when --delete-excluded
|
||||
* opts back into deleting them. NOT recorded for --files-from subset pruning
|
||||
* (whose delete semantics derive from the synchronized-directory set) or for
|
||||
* -R/--files-from relative wire paths. When `excluded_mutex` is non-NULL it
|
||||
* is taken around every append (the parallel scanner shares one list across
|
||||
* its worker threads). */
|
||||
ArrayList* excluded_paths;
|
||||
mtx_t* excluded_mutex;
|
||||
/* --ignore-errors: an unreadable directory during the scan is recorded as an
|
||||
* I/O error and skipped instead of aborting the scan. Client-only. */
|
||||
/* Size-prune protection sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every entry it skipped because of
|
||||
* --max-size/--min-size. rsync never deletes a size-skipped source mirror,
|
||||
* even under --delete-excluded, so the sender always transmits this list as
|
||||
* protected prefixes (unlike excluded_paths, which --delete-excluded drops).
|
||||
* Guarded by `excluded_mutex` like excluded_paths. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Synchronized-directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it is about to traverse
|
||||
* that lies inside a --files-from listed directory (or of every traversed
|
||||
* directory when there is no list). The sender sends this set with the delete
|
||||
* manifest so the receiver confines its extras walk to synchronized
|
||||
* directories, exactly like rsync; the receive root is the "." sentinel.
|
||||
* Guarded by `excluded_mutex`. */
|
||||
ArrayList* synced_dirs;
|
||||
/* Per-directory filter-rule sink (optional): when non-NULL the scanner appends
|
||||
* a deep copy of every rule it reads from a per-directory merge file, each
|
||||
* carrying its owner directory and no-inherit flag (see filter.h). The delete
|
||||
* carriers transmit them so the receiver re-derives the per-directory
|
||||
* protect/risk set for destination-only entries. Guarded by `excluded_mutex`
|
||||
* like the other sinks. */
|
||||
FilterRuleList* per_dir_rules;
|
||||
/* Delete-plan directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it traverses (except the
|
||||
* receive root). The per-directory --delete-during/--delete-delay plan
|
||||
* builder uses this to keep an empty in-scope source directory (rsync keeps
|
||||
* it) and to emit its plan after the data stream, when no file frame would
|
||||
* otherwise trigger it. Guarded by `excluded_mutex`. */
|
||||
ArrayList* plan_dirs;
|
||||
/* --ignore-errors: an unreadable subdirectory no longer aborts the scan (it
|
||||
* is always skipped so the rest of the tree transfers); this flag is kept so
|
||||
* the client can distinguish the option state when deciding deletion policy.
|
||||
* Client-only. */
|
||||
bool ignore_io_errors;
|
||||
/* --info=nonreg: print rsync's `skipping non-regular file "NAME"` line for a
|
||||
* non-regular entry that is not being preserved. Client-only. */
|
||||
bool note_nonreg;
|
||||
/* --info=mount: print rsync's `[sender] skipping mount-point dir NAME` when
|
||||
* -xx drops a mount-point directory. Client-only. */
|
||||
bool note_mount;
|
||||
/* --stats directory accounting for a `-r` run (no -t/-p): a shared counter of
|
||||
* traversed directories that are NOT otherwise represented by an inline
|
||||
* directory entry (rsync still counts every directory in `Number of files`).
|
||||
* Incremented when a directory is opened and decremented when an empty
|
||||
* directory is emitted inline (so it is counted exactly once). Atomic
|
||||
* because the parallel scanner's workers share it; NULL disables the
|
||||
* accounting. Client-only. */
|
||||
atomic_ullong* dir_count;
|
||||
/* Source root and 8-bit-output policy used to render a `--info=nonreg` name
|
||||
* relative to the transfer root. Borrowed read-only. */
|
||||
const char* send_directory;
|
||||
bool eight_bit_output;
|
||||
/* --ignore-missing-args (implied by --delete-missing-args): an explicitly
|
||||
* --files-from-listed entry that does not exist under the source is skipped
|
||||
* instead of failing (the --dirs generator is the only scanner path that
|
||||
@@ -114,6 +184,17 @@ typedef struct {
|
||||
bool capture_dir_times;
|
||||
ArrayList* dir_entries;
|
||||
mtx_t* dir_entries_mutex;
|
||||
/* Recreate empty source directories on a recursive transfer: emit a
|
||||
* payload-less directory entry for every traversed directory that produced
|
||||
* no transferred/descended child. Off by default so low-level scanner users
|
||||
* (unit helpers, --list-only) see only the historical file list; the real
|
||||
* sender sets it in prepare_scanner. */
|
||||
bool emit_empty_dirs;
|
||||
/* --no-implied-dirs with -R + --files-from: a directory that is only an
|
||||
* implied parent of a listed entry (not itself listed, nor below a listed
|
||||
* directory) must not carry source metadata; it is created with default
|
||||
* attributes at the destination, matching rsync. */
|
||||
bool no_implied_dirs;
|
||||
} ScannerOptions;
|
||||
|
||||
/* Internal per-scanner filter state. FilterNode chains represent the ordered
|
||||
@@ -131,6 +212,22 @@ typedef struct {
|
||||
int current_depth;
|
||||
dev_t root_dev;
|
||||
bool failed;
|
||||
/* rsync-order traversal: each opened directory's entries are inspected once
|
||||
and buffered (an internal SortedEntry[] owned here) sorted as rsync's flist
|
||||
orders them -- non-directories ascending, then directories ascending. The
|
||||
entries are walked in order and child directories are collected in
|
||||
`pending_dirs` (an ArrayList of DirEntry*, owned here) and pushed onto the
|
||||
LIFO `directories` stack in reverse at directory exhaustion, so the emitted
|
||||
stream is depth-first like rsync. `sorted_*` are reset per directory. */
|
||||
void* sorted_entries;
|
||||
size_t sorted_count;
|
||||
size_t sorted_index;
|
||||
void* pending_dirs;
|
||||
/* Recursive scan: whether the open directory yielded any transferred or
|
||||
descended entry. When it did not, closing it emits a directory entry so
|
||||
the empty source directory is recreated at the destination (rsync
|
||||
parity). */
|
||||
bool current_dir_produced;
|
||||
/* Phase 2 (files-from / filter layer). */
|
||||
char* root_path; /* transfer root (fs path) for rel computation */
|
||||
char* current_rel; /* rel path of the open directory ("" == root) */
|
||||
@@ -149,6 +246,10 @@ typedef struct {
|
||||
--ignore-errors the scan continues past it and the caller decides what to
|
||||
do; `failed` is reserved for fatal errors that always abort the scan. */
|
||||
bool io_error;
|
||||
/* The transfer ROOT could not be opened. It is always fatal, even under
|
||||
--ignore-errors, but the client still maps it to rsync's partial-transfer
|
||||
exit (23) rather than a generic failure. */
|
||||
bool root_io_error;
|
||||
} DirectoryScanner;
|
||||
|
||||
typedef struct {
|
||||
@@ -168,7 +269,8 @@ typedef struct {
|
||||
int completed;
|
||||
Chunk* initial_chunk;
|
||||
ProtocolSession* allocation_session;
|
||||
FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
|
||||
FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
|
||||
const ScannerOptions* options; /* borrowed scan options (--info=nonreg output) */
|
||||
} ParallelScanner;
|
||||
|
||||
DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata,
|
||||
@@ -187,13 +289,20 @@ void directory_scanner_destroy(DirectoryScanner* scanner);
|
||||
/* --one-file-system (-x) decision: a directory entry may be descended into
|
||||
* only when the option is disabled or the entry lives on the same device as
|
||||
* the transfer root. Exposed so tests can exercise the rule directly. */
|
||||
bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entry_device);
|
||||
bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device);
|
||||
|
||||
/* Relative path of an on-disk path below `root` ("" == the root itself, NULL
|
||||
* when `fs_path` is not under `root`). Handles trailing slashes and a root of
|
||||
* "/". Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_path_relative(const char* root, const char* fs_path);
|
||||
|
||||
/* -R/--relative destination-relative prefix reconstructed from a source spec:
|
||||
* the path after rsync's first '.' path component (the '/./' cut point), with
|
||||
* leading/trailing slashes removed, or the whole spec (normalized) when there
|
||||
* is no cut. Returns "" for the receive root, or NULL when `spec` is NULL or
|
||||
* allocation fails. Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_relative_prefix(const char* spec);
|
||||
|
||||
ParallelScanner* parallel_scanner_create_with_options(const char* root_directory,
|
||||
const ScannerOptions* options,
|
||||
ProtocolSession* allocation_session);
|
||||
|
||||
@@ -0,0 +1,793 @@
|
||||
#include "log.h"
|
||||
#include "scanner.h"
|
||||
#include "scanner_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "file.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/sysmacros.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <limits.h>
|
||||
|
||||
#include "xattr.h"
|
||||
|
||||
/* A chain node: `own` holds the .rsync-filter rules of one directory, `parent`
|
||||
* the context that directory inherited (nearest ancestor with a filter file).
|
||||
* The chain for a directory's contents runs from that directory's own node up
|
||||
* to the root; the command-line base rules are evaluated after the whole
|
||||
* chain. */
|
||||
struct FilterNode {
|
||||
FilterNode* parent;
|
||||
FilterRuleList* own;
|
||||
};
|
||||
|
||||
void filter_node_destroy(void* item) {
|
||||
if (item) {
|
||||
FilterNode* node = (FilterNode*)item;
|
||||
if (node->own)
|
||||
filter_rule_list_free(node->own);
|
||||
free(node);
|
||||
}
|
||||
}
|
||||
|
||||
FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) {
|
||||
FilterNode* node = malloc(sizeof(FilterNode));
|
||||
if (!node)
|
||||
return NULL;
|
||||
node->parent = parent;
|
||||
node->own = own;
|
||||
return node;
|
||||
}
|
||||
|
||||
/* Evaluate a rule chain for one entry. rsync precedence, highest first: the
|
||||
* innermost (current) directory's .rsync-filter rules, then each ancestor's,
|
||||
* then the root's, and finally the command-line base rules (--filter/-C). The
|
||||
* sender-side verdict decides whether the entry is hidden from the transfer;
|
||||
* the receiver-side verdict decides whether its destination mirror is protected
|
||||
* from --delete. Each side takes the FIRST matching rule independently. */
|
||||
typedef struct {
|
||||
bool hide; /* sender-side exclude matched */
|
||||
bool protect; /* receiver-side exclude matched */
|
||||
} FilterOutcome;
|
||||
|
||||
static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel,
|
||||
const char* leaf, bool is_dir, FilterOutcome* out) {
|
||||
memset(out, 0, sizeof(*out));
|
||||
bool sender_decided = false;
|
||||
bool receiver_decided = false;
|
||||
const FilterNode* n = node;
|
||||
while (!sender_decided || !receiver_decided) {
|
||||
const FilterRuleList* list = n ? n->own : base;
|
||||
if (list) {
|
||||
if (!sender_decided) {
|
||||
FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER);
|
||||
if (action != FILTER_ACTION_NONE) {
|
||||
out->hide = action == FILTER_ACTION_EXCLUDE;
|
||||
sender_decided = true;
|
||||
}
|
||||
}
|
||||
if (!receiver_decided) {
|
||||
FilterAction action =
|
||||
filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER);
|
||||
if (action != FILTER_ACTION_NONE) {
|
||||
out->protect = action == FILTER_ACTION_PROTECT;
|
||||
receiver_decided = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!n)
|
||||
break;
|
||||
n = n->parent;
|
||||
}
|
||||
}
|
||||
|
||||
static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel,
|
||||
const char* leaf, bool is_dir, bool exclude_filter_files,
|
||||
bool* protect_out) {
|
||||
/* -FF: per-directory .rsync-filter files are never transferred (single -F
|
||||
transfers them, matching rsync). */
|
||||
if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) {
|
||||
if (protect_out)
|
||||
*protect_out = false;
|
||||
return false;
|
||||
}
|
||||
FilterOutcome outcome;
|
||||
chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome);
|
||||
if (protect_out)
|
||||
*protect_out = outcome.protect;
|
||||
return !outcome.hide;
|
||||
}
|
||||
|
||||
void dir_entry_destroy(void* item) {
|
||||
if (item) {
|
||||
DirEntry* de = (DirEntry*)item;
|
||||
free(de->path);
|
||||
free(de);
|
||||
}
|
||||
}
|
||||
|
||||
DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context) {
|
||||
DirEntry* de = malloc(sizeof(DirEntry));
|
||||
if (!de)
|
||||
return NULL;
|
||||
de->path = str_dup(path);
|
||||
if (!de->path) {
|
||||
free(de);
|
||||
return NULL;
|
||||
}
|
||||
de->depth = depth;
|
||||
de->context = context;
|
||||
return de;
|
||||
}
|
||||
|
||||
/* Apply rsync's symlink-resolution precedence to one S_ISLNK entry:
|
||||
* --copy-links dereferences every symlink;
|
||||
* --copy-unsafe-links dereferences only targets unsafe_symlink() flags;
|
||||
* -k/--copy-dirlinks dereferences only a symlink whose referent is a dir;
|
||||
* --safe-links (receiver-side in rsync; modelled here) ignores an unsafe
|
||||
* target that would otherwise be carried; with --munge-links
|
||||
* every stored target becomes absolute, so --safe-links then
|
||||
* ignores every symlink, exactly as rsync documents;
|
||||
* -l/--links carries the link.
|
||||
* `link_rel` is the symlink's transfer-relative path (incl. name) and is used
|
||||
* only for the lexical unsafe test. `target` receives the raw link value. */
|
||||
LinkAction scanner_link_action(const ScannerOptions* options, const char* path,
|
||||
const char* link_rel, char* target, size_t target_size) {
|
||||
if (!options->follow_symlinks && !options->copy_links && !options->safe_links &&
|
||||
!options->copy_unsafe_links && !options->copy_dirlinks)
|
||||
return LINK_ACTION_SKIP;
|
||||
ssize_t length = readlink(path, target, target_size - 1);
|
||||
if (length < 0)
|
||||
return LINK_ACTION_SKIP;
|
||||
target[length] = '\0';
|
||||
|
||||
bool unsafe = file_symlink_unsafe(target, link_rel);
|
||||
if (options->copy_links || (options->copy_unsafe_links && unsafe))
|
||||
return LINK_ACTION_DEREF;
|
||||
if (options->copy_dirlinks) {
|
||||
struct stat ref;
|
||||
if (stat(path, &ref) == 0 && S_ISDIR(ref.st_mode))
|
||||
return LINK_ACTION_DEREF;
|
||||
}
|
||||
if (options->safe_links && (unsafe || options->munge_links))
|
||||
return LINK_ACTION_SKIP_PROTECTED;
|
||||
if (!options->follow_symlinks || target[0] == '\0')
|
||||
return LINK_ACTION_SKIP;
|
||||
return LINK_ACTION_CARRY;
|
||||
}
|
||||
|
||||
/* --one-file-system (-x) decision. Only directories can carry a different
|
||||
* device than their parent (mount points), so this is checked when a child
|
||||
* directory is about to be descended into. */
|
||||
bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device) {
|
||||
return one_file_system <= 0 || entry_device == root_device;
|
||||
}
|
||||
|
||||
/* Build a payload-less directory File carrying the captured metadata (when
|
||||
* requested). Used by -x mount-point emission and --list-only directory
|
||||
* entries. Returns NULL on allocation failure. */
|
||||
File* scanner_build_dir_file(const char* path, const struct stat* stats,
|
||||
const ScannerOptions* options) {
|
||||
File* dir = file_create(path);
|
||||
if (dir == NULL)
|
||||
return NULL;
|
||||
dir->is_dir = true;
|
||||
if (options->use_metadata) {
|
||||
dir->metadata =
|
||||
file_metadata_create(dir->path, stats, options->preserve_atimes, options->preserve_crtimes);
|
||||
if (!dir->metadata) {
|
||||
file_destroy(dir);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
return dir;
|
||||
}
|
||||
|
||||
/* Relative path of an on-disk path below `root`. The transfer root may be
|
||||
* given with a trailing slash; the returned rel path never has one and is ""
|
||||
* for the root itself. A root of "/" is handled (its children start at "/").
|
||||
* Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_path_relative(const char* root, const char* fs_path) {
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, fs_path, root_len) != 0)
|
||||
return NULL;
|
||||
if (root_len == 1 && root[0] == '/') {
|
||||
if (fs_path[1] == '\0')
|
||||
return str_dup("");
|
||||
return str_dup(fs_path + 1);
|
||||
}
|
||||
if (fs_path[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (fs_path[root_len] != '/')
|
||||
return NULL;
|
||||
return str_dup(fs_path + root_len + 1);
|
||||
}
|
||||
|
||||
/* -R/--relative destination-relative prefix reconstructed from a source spec:
|
||||
* everything after the first '.' path component (rsync's '/./' cut point),
|
||||
* with leading/trailing slashes removed; or the whole spec (normalized) when
|
||||
* there is no cut. Returns "" for the receive root. Exposed for tests. */
|
||||
char* scanner_relative_prefix(const char* spec) {
|
||||
if (!spec || spec[0] == '\0')
|
||||
return NULL;
|
||||
const char* after = spec;
|
||||
if (spec[0] == '.' && spec[1] == '/') {
|
||||
after = spec + 2;
|
||||
} else {
|
||||
const char* cut = strstr(spec, "/./");
|
||||
if (cut)
|
||||
after = cut + 3;
|
||||
}
|
||||
size_t cap = strlen(spec) + 1;
|
||||
char* out = malloc(cap);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t len = 0;
|
||||
for (const char* s = after; *s;) {
|
||||
while (*s == '/')
|
||||
s++;
|
||||
const char* comp = s;
|
||||
while (*s && *s != '/')
|
||||
s++;
|
||||
size_t clen = (size_t)(s - comp);
|
||||
if (clen == 0 || (clen == 1 && comp[0] == '.'))
|
||||
continue;
|
||||
if (len)
|
||||
out[len++] = '/';
|
||||
memcpy(out + len, comp, clen);
|
||||
len += clen;
|
||||
}
|
||||
out[len] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Relative path of a child entry below the current directory. */
|
||||
char* child_rel_path(const char* parent_rel, const char* name) {
|
||||
if (!parent_rel || parent_rel[0] == '\0')
|
||||
return str_dup(name);
|
||||
return path_cat(parent_rel, name);
|
||||
}
|
||||
|
||||
/* Destination-relative wire path for an entry under an -R prefix. */
|
||||
char* scanner_prefix_send_path(const char* prefix, const char* rel) {
|
||||
if (prefix[0] == '\0')
|
||||
return str_dup(rel);
|
||||
if (rel[0] == '\0')
|
||||
return str_dup(prefix);
|
||||
return path_cat(prefix, rel);
|
||||
}
|
||||
|
||||
/* Apply the --files-from allow-set and the filter layer to one entry. On
|
||||
* return `*protect_out` is true when a receiver-side rule protects the entry's
|
||||
* destination mirror from deletion. */
|
||||
bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base,
|
||||
const FilterNode* node, const char* rel, const char* leaf, bool is_dir,
|
||||
bool per_dir_filters, bool exclude_filter_files, bool* protect_out) {
|
||||
if (protect_out)
|
||||
*protect_out = false;
|
||||
if (file_list && !file_list_affects(file_list, rel))
|
||||
return false;
|
||||
if (base || per_dir_filters)
|
||||
return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out);
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to
|
||||
* read xattrs is non-fatal: the file is transferred without them. A symlink
|
||||
* entry reads the LINK's own xattrs (never the referent's) with the no-follow
|
||||
* variant; on Linux the VFS refuses xattrs on symlinks, so that yields NULL. */
|
||||
void scanner_capture_xattrs_opts(const ScannerOptions* options, File* file) {
|
||||
if (!options || !file || !(options->preserve_xattrs || options->preserve_acls))
|
||||
return;
|
||||
file->xattrs = file->is_symlink ? xattr_capture_path_nofollow(file->path, options->preserve_acls)
|
||||
: xattr_capture_path(file->path, options->preserve_acls);
|
||||
}
|
||||
|
||||
void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) {
|
||||
if (!scanner)
|
||||
return;
|
||||
scanner_capture_xattrs_opts(&scanner->options, file);
|
||||
}
|
||||
|
||||
/* Apply --hard-links (-H) detection to one regular File. On a sibling (a
|
||||
* later member of an already-seen source inode) the File keeps the group id
|
||||
* and the first member's wire path but carries NO data payload (size 0); the
|
||||
* first member is left untouched (data present, link_first). Returns false on
|
||||
* allocation failure (the caller marks the scan failed); the File stays usable
|
||||
* either way. */
|
||||
bool scanner_assign_hardlink(HardLinkTable* table, File* file, const struct stat* stats) {
|
||||
if (!table || !file || !stats)
|
||||
return true;
|
||||
int gid;
|
||||
bool is_first;
|
||||
char* first_path = NULL;
|
||||
if (!hardlink_table_assign(table, file_wire_path(file), stats->st_dev, stats->st_ino, &gid,
|
||||
&is_first, &first_path))
|
||||
return false;
|
||||
file->link_group = gid;
|
||||
file->link_first = is_first;
|
||||
if (!is_first) {
|
||||
file->hardlink_target = first_path;
|
||||
file->data->size = 0;
|
||||
} else {
|
||||
free(first_path);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Phase 4 special/devices decision for one non-regular entry, matching rsync:
|
||||
- a char/block device is RECREATED as a node under -D/--devices, unless
|
||||
--copy-devices asks for its content to be copied into a regular file;
|
||||
- a FIFO/socket is RECREATED under --specials;
|
||||
- when the matching flag is absent the entry is SKIPPED ("skipping
|
||||
non-regular file"), exactly like rsync's default, instead of being
|
||||
silently copied as a zero-length regular file;
|
||||
- anything else (regular/directory) is left to the normal data path. */
|
||||
ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials,
|
||||
bool copy_devices, File* file, const struct stat* stats) {
|
||||
if (!file || !stats)
|
||||
return SCANNER_SPECIAL_REGULAR;
|
||||
bool is_device = S_ISCHR(stats->st_mode) || S_ISBLK(stats->st_mode);
|
||||
bool is_fifo = S_ISFIFO(stats->st_mode);
|
||||
bool is_socket = S_ISSOCK(stats->st_mode);
|
||||
if (!is_device && !is_fifo && !is_socket)
|
||||
return SCANNER_SPECIAL_REGULAR;
|
||||
if (is_device && copy_devices)
|
||||
return SCANNER_SPECIAL_REGULAR; /* copy device content as a regular file */
|
||||
bool preserve = is_device ? preserve_devices : preserve_specials;
|
||||
if (!preserve)
|
||||
return SCANNER_SPECIAL_SKIP;
|
||||
file->is_special = true;
|
||||
file->data->size = 0;
|
||||
file->data->data = NULL;
|
||||
if (is_device) {
|
||||
file->rdev_major = (int32_t)major(stats->st_rdev);
|
||||
file->rdev_minor = (int32_t)minor(stats->st_rdev);
|
||||
}
|
||||
return SCANNER_SPECIAL_RECREATE;
|
||||
}
|
||||
|
||||
/* Append `rel` to the caller's exclusion sink, taking `mtx` when shared across
|
||||
parallel worker threads. Returns false on allocation failure (list left
|
||||
unchanged). */
|
||||
bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) {
|
||||
if (!list)
|
||||
return true;
|
||||
char* dup = str_dup(rel);
|
||||
if (!dup)
|
||||
return false;
|
||||
if (mtx)
|
||||
mtx_lock(mtx);
|
||||
bool ok = array_list_add(list, dup);
|
||||
if (mtx)
|
||||
mtx_unlock(mtx);
|
||||
if (!ok)
|
||||
free(dup);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Record one pruned filesystem path in a delete-protection sink. The stored
|
||||
form is the entry's wire/destination-relative path (a single leading '/'
|
||||
removed, exactly how manifest keep entries are stored), so the receiver's
|
||||
walker prefixes match the destination layout. An allocation failure is a
|
||||
fatal scan error. */
|
||||
static void scanner_record_protected(DirectoryScanner* scanner, const char* fs_path,
|
||||
ArrayList* sink) {
|
||||
if (!sink || !fs_path)
|
||||
return;
|
||||
const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path;
|
||||
if (!excluded_sink_append(sink, scanner->options.excluded_mutex, rel))
|
||||
scanner->failed = true;
|
||||
}
|
||||
|
||||
/* rsync's `--info=nonreg` line for a non-regular entry that is not being
|
||||
* preserved: `skipping non-regular file "NAME"`. The name is the path relative
|
||||
* to the transfer root, so it matches rsync's displayed name. */
|
||||
void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path) {
|
||||
if (!options || !options->note_nonreg || !fs_path)
|
||||
return;
|
||||
const char* rel = utils_strip_transfer_root(fs_path, options->send_directory);
|
||||
char* escaped = output_escape(rel, options->eight_bit_output);
|
||||
printf("skipping non-regular file \"%s\"\n", escaped ? escaped : rel);
|
||||
free(escaped);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
/* Construct one non-directory File from an inspected entry. Shared by the
|
||||
* sequential and parallel scanners so entry construction has a single
|
||||
* implementation: data size (or carried symlink), -R wire path, special/devices
|
||||
* classification, hardlink group, metadata and xattr capture all happen here in
|
||||
* the same order for both. See the declaration for the ownership contract. */
|
||||
ScannerBuildStatus scanner_build_file_entry(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, File** out_file, bool* failed) {
|
||||
*out_file = NULL;
|
||||
if (failed)
|
||||
*failed = false;
|
||||
File* file = file_create(inspected->path);
|
||||
if (!file) {
|
||||
/* The File never existed, so drop the not-yet-transferred symlink target
|
||||
here; the caller's entry teardown would otherwise double-free it. */
|
||||
free(inspected->link_target);
|
||||
inspected->link_target = NULL;
|
||||
return SCANNER_BUILD_FAIL_CONTINUE;
|
||||
}
|
||||
if (inspected->is_symlink) {
|
||||
file->is_symlink = true;
|
||||
file->symlink_target = inspected->link_target;
|
||||
inspected->link_target = NULL;
|
||||
} else {
|
||||
file->data->size = inspected->stats.st_size;
|
||||
}
|
||||
/* -R + --files-from uses the bare transfer-relative path; -R without
|
||||
--files-from prefixes it. Plain scans keep the source path. */
|
||||
bool relative_mode = options->relative && options->file_list != NULL;
|
||||
if (relative_mode) {
|
||||
file->send_path = str_dup(rel);
|
||||
} else if (options->relative_prefix) {
|
||||
file->send_path = scanner_prefix_send_path(options->relative_prefix, rel);
|
||||
}
|
||||
if ((relative_mode || options->relative_prefix) && !file->send_path) {
|
||||
file_destroy(file);
|
||||
return SCANNER_BUILD_FAIL_BREAK;
|
||||
}
|
||||
/* --devices/--specials: a device/FIFO/socket entry marked for preservation
|
||||
becomes a node to recreate (is_special, no data, rdev captured); an
|
||||
unrequested non-regular entry is skipped (rsync default). */
|
||||
ScannerSpecial special =
|
||||
scanner_prepare_special(options->preserve_devices, options->preserve_specials,
|
||||
options->copy_devices, file, &inspected->stats);
|
||||
if (special == SCANNER_SPECIAL_SKIP) {
|
||||
scanner_note_nonreg(options, file->path);
|
||||
file_destroy(file);
|
||||
return SCANNER_BUILD_SKIP;
|
||||
}
|
||||
if (options->hardlinks && S_ISREG(inspected->stats.st_mode) &&
|
||||
!scanner_assign_hardlink(options->hardlinks, file, &inspected->stats)) {
|
||||
/* Allocation failure is non-fatal to this entry (it is still emitted) but
|
||||
marks the scan failed, matching the historical inlined behaviour. */
|
||||
if (failed)
|
||||
*failed = true;
|
||||
}
|
||||
if (options->use_metadata) {
|
||||
file->metadata = file_metadata_create(file->path, &inspected->stats, options->preserve_atimes,
|
||||
options->preserve_crtimes);
|
||||
if (!file->metadata) {
|
||||
file_destroy(file);
|
||||
return SCANNER_BUILD_FAIL_BREAK;
|
||||
}
|
||||
}
|
||||
/* A hardlink sibling carries no data, so it carries no xattrs. */
|
||||
if (!(file->link_group != 0 && !file->link_first))
|
||||
scanner_capture_xattrs_opts(options, file);
|
||||
*out_file = file;
|
||||
return SCANNER_BUILD_OK;
|
||||
}
|
||||
|
||||
/* rsync 3.4.1's `--info=mount` line, emitted when `-xx` drops a mount-point
|
||||
* directory: `[sender] skipping mount-point dir NAME` (the client is the
|
||||
* sender). Plain `-x` keeps the empty directory and prints nothing, matching
|
||||
* rsync. */
|
||||
void scanner_note_mount(const ScannerOptions* options, const char* fs_path) {
|
||||
if (!options || !options->note_mount || !fs_path)
|
||||
return;
|
||||
const char* rel = utils_strip_transfer_root(fs_path, options->send_directory);
|
||||
char* escaped = output_escape(rel, options->eight_bit_output);
|
||||
printf("[sender] skipping mount-point dir %s\n", escaped ? escaped : rel);
|
||||
free(escaped);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
/* --debug=filter: a selection/filter decision dropped an entry. */
|
||||
void scanner_note_filter(const ScannerOptions* options, const char* name) {
|
||||
if (!options || !log_debug_enabled(LOG_DEBUG_FILTER) || !name)
|
||||
return;
|
||||
log_debug_message(LOG_DEBUG_FILTER, "filter: excluded %s", name);
|
||||
}
|
||||
|
||||
/* Account for a directory that will not be represented by an inline directory
|
||||
* entry. Paired with scanner_dir_count_uncount for empty directories that are
|
||||
* emitted inline, so every traversed directory is counted exactly once. */
|
||||
void scanner_dir_count_count(const ScannerOptions* options) {
|
||||
if (options && options->dir_count)
|
||||
atomic_fetch_add(options->dir_count, 1);
|
||||
}
|
||||
|
||||
void scanner_dir_count_uncount(const ScannerOptions* options) {
|
||||
if (options && options->dir_count)
|
||||
atomic_fetch_sub(options->dir_count, 1);
|
||||
}
|
||||
|
||||
/* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */
|
||||
void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) {
|
||||
scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths);
|
||||
}
|
||||
|
||||
/* A --max-size/--min-size prune (always protected, even under --delete-excluded). */
|
||||
void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path) {
|
||||
scanner_record_protected(scanner, fs_path, scanner->options.size_skipped_paths);
|
||||
}
|
||||
|
||||
/* The destination-relative coordinate the receiver's delete walkers match
|
||||
against for an entry at `fs_path` (with `rel` its path relative to the
|
||||
transfer root, "" for the root): `relative_prefix + rel` under -R+--relative,
|
||||
the bare relative path under -R+--files-from, else the source path with a
|
||||
leading '/' removed, with "." for the receive root. Shared by the
|
||||
synchronized-directory sink and the mirrored per-directory rule owners so
|
||||
both live in the same coordinate system. Returns an owned string, or NULL on
|
||||
allocation failure. */
|
||||
char* scanner_dest_rel_path(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode) {
|
||||
char* prefixed = NULL;
|
||||
const char* dest;
|
||||
if (relative_mode) {
|
||||
dest = rel;
|
||||
} else if (options->relative_prefix) {
|
||||
prefixed = scanner_prefix_send_path(options->relative_prefix, rel);
|
||||
if (!prefixed)
|
||||
return NULL;
|
||||
dest = prefixed;
|
||||
} else {
|
||||
dest = fs_path;
|
||||
}
|
||||
if (dest[0] == '/')
|
||||
dest++;
|
||||
if (dest[0] == '\0')
|
||||
dest = ".";
|
||||
char* out = str_dup(dest);
|
||||
free(prefixed);
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Record a directory the scan synchronized. `fs_path` is its absolute path and
|
||||
`rel` its path relative to the transfer root ("" for the root); the stored
|
||||
form matches the wire layout (see scanner_dest_rel_path). Returns false on
|
||||
allocation failure. */
|
||||
bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode) {
|
||||
if (!options->synced_dirs && !options->plan_dirs)
|
||||
return true;
|
||||
if (!file_list_dir_in_scope(options->file_list, rel))
|
||||
return true;
|
||||
char* dest = scanner_dest_rel_path(options, fs_path, rel, relative_mode);
|
||||
if (!dest)
|
||||
return false;
|
||||
bool ok = true;
|
||||
if (options->synced_dirs)
|
||||
ok = excluded_sink_append(options->synced_dirs, options->excluded_mutex, dest);
|
||||
/* The delete-plan keep set needs an entry for every traversed source
|
||||
directory, including empty ones, so its destination mirror is kept rather
|
||||
than deleted as an extra; the receive root (".") is implicit. */
|
||||
if (ok && options->plan_dirs && strcmp(dest, ".") != 0)
|
||||
ok = excluded_sink_append(options->plan_dirs, options->excluded_mutex, dest);
|
||||
free(dest);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Read every per-directory filter file that applies to `dir_path` (its
|
||||
* .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a
|
||||
* fresh list. Returns NULL on allocation/parse failure (message in `err`);
|
||||
* returns an empty list (and *any_exists=false) when no file exists. */
|
||||
FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path,
|
||||
const char* rel, bool relative_mode, bool* any_exists, char* err,
|
||||
size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
const FilterRuleList* base = options->base_filters;
|
||||
bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0);
|
||||
if (any_exists)
|
||||
*any_exists = false;
|
||||
if (!have_names)
|
||||
return NULL;
|
||||
FilterRuleList* own = filter_rule_list_create();
|
||||
if (!own) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false};
|
||||
bool exists = false;
|
||||
if (options->per_dir_filters) {
|
||||
if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size))
|
||||
goto fail;
|
||||
if (exists && any_exists)
|
||||
*any_exists = true;
|
||||
}
|
||||
if (base) {
|
||||
for (int i = 0; i < base->dir_merge_count; i++) {
|
||||
if (!filter_dir_merge_append(own, dir_path, &base->dir_merges[i], rel, &opts, &exists, err,
|
||||
err_size))
|
||||
goto fail;
|
||||
if (exists && any_exists)
|
||||
*any_exists = true;
|
||||
}
|
||||
}
|
||||
/* Mirror the directory's rules into the delete-carrier sink so the receiver
|
||||
* can reconstruct its per-directory protect/risk set. The mirrored rules
|
||||
* carry the destination-relative owner coordinate (not the transfer-root-
|
||||
* relative one the sender's own evaluation uses) so the receiver's delete
|
||||
* walkers, which match against receive-root-relative paths, find them. */
|
||||
if (options->per_dir_rules && own->count > 0) {
|
||||
char* owner = scanner_dest_rel_path(options, dir_path, rel, relative_mode);
|
||||
if (!owner)
|
||||
goto fail;
|
||||
mtx_t* mtx = options->excluded_mutex;
|
||||
if (mtx)
|
||||
mtx_lock(mtx);
|
||||
for (int i = 0; i < own->count; i++) {
|
||||
FilterRule* copy = filter_rule_clone(own->items[i]);
|
||||
if (!copy || !filter_rule_set_owner(copy, owner) ||
|
||||
!filter_rule_list_add(options->per_dir_rules, copy)) {
|
||||
filter_rule_free(copy);
|
||||
if (mtx)
|
||||
mtx_unlock(mtx);
|
||||
free(owner);
|
||||
goto fail;
|
||||
}
|
||||
}
|
||||
if (mtx)
|
||||
mtx_unlock(mtx);
|
||||
free(owner);
|
||||
}
|
||||
return own;
|
||||
fail:
|
||||
filter_rule_list_free(own);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Merge the open directory's own per-directory filter files (the default
|
||||
* .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the
|
||||
* base rule list) into the inherited context, returning the context used for
|
||||
* this directory's entries. On a parse error the scanner is marked failed.
|
||||
* Returns 0 on success, -1 on failure. */
|
||||
int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) {
|
||||
char err[256];
|
||||
bool any_exists = false;
|
||||
FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path,
|
||||
scanner->current_rel ? scanner->current_rel : "",
|
||||
scanner->relative_mode, &any_exists, err, sizeof(err));
|
||||
if (!own) {
|
||||
/* read_dir_filters() leaves `err` set on a parse/allocation failure even
|
||||
when an earlier merge file in the same directory existed (any_exists true);
|
||||
key off the error text rather than any_exists so an invalid per-directory
|
||||
filter file can never be silently ignored. */
|
||||
if (err[0] == '\0') {
|
||||
scanner->current_node = (FilterNode*)inherited;
|
||||
return 0;
|
||||
}
|
||||
char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", err);
|
||||
free(escaped_path);
|
||||
scanner->failed = true;
|
||||
return -1;
|
||||
}
|
||||
if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) {
|
||||
FilterNode* node = filter_node_alloc((FilterNode*)inherited, own);
|
||||
if (!node || !array_list_add(scanner->filter_nodes, node)) {
|
||||
filter_node_destroy(node);
|
||||
scanner->failed = true;
|
||||
return -1;
|
||||
}
|
||||
scanner->current_node = node;
|
||||
} else {
|
||||
filter_rule_list_free(own);
|
||||
scanner->current_node = (FilterNode*)inherited;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Inspect symlinks, resolve the entry type, and apply file filters once for both scanners.
|
||||
* `link_rel` is the entry's path relative to the transfer root (including its
|
||||
* name), used for the lexical rsync unsafe-symlink test. */
|
||||
int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir,
|
||||
const char* link_rel, const char* name, ScannerEntry* entry) {
|
||||
entry->excluded = false;
|
||||
entry->size_excluded = false;
|
||||
entry->referent_error = false;
|
||||
entry->is_symlink = false;
|
||||
entry->link_target = NULL;
|
||||
entry->path = path_cat(containing_dir, name);
|
||||
if (!entry->path)
|
||||
return -1;
|
||||
|
||||
struct stat link_stats;
|
||||
if (lstat(entry->path, &link_stats) != 0) {
|
||||
free(entry->path);
|
||||
return 0;
|
||||
}
|
||||
if (!S_ISLNK(link_stats.st_mode))
|
||||
goto regular;
|
||||
|
||||
char link_target[4096];
|
||||
switch (scanner_link_action(options, entry->path, link_rel, link_target, sizeof(link_target))) {
|
||||
case LINK_ACTION_SKIP:
|
||||
goto skip;
|
||||
case LINK_ACTION_SKIP_PROTECTED:
|
||||
/* --safe-links ignored the link, but rsync still counts it as present in
|
||||
the transfer, so its destination mirror survives --delete. Record it as
|
||||
an excluded path (the same delete-protection channel as a filter prune). */
|
||||
entry->excluded = true;
|
||||
goto skip;
|
||||
case LINK_ACTION_DEREF:
|
||||
if (stat(entry->path, &entry->stats) != 0) {
|
||||
/* rsync reports "symlink has no referent" and continues with a partial
|
||||
transfer (exit 23); record the error so the run exits 23 too. */
|
||||
char* escaped = output_escape(entry->path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "symlink has no referent: %s",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
entry->referent_error = true;
|
||||
goto skip;
|
||||
}
|
||||
entry->is_directory = S_ISDIR(entry->stats.st_mode);
|
||||
if (entry->is_directory)
|
||||
return 1;
|
||||
goto apply_filters;
|
||||
case LINK_ACTION_CARRY:
|
||||
break;
|
||||
}
|
||||
|
||||
/* Carry the link as a symlink. --munge-links is applied by the RECEIVER (it
|
||||
prefixes every stored target with /rsyncd-munged/); when the SOURCE already
|
||||
holds a munged value the sender strips it so the receiver re-munges a clean
|
||||
target, round-tripping a munged tree exactly like rsync. */
|
||||
entry->is_symlink = true;
|
||||
entry->stats = link_stats;
|
||||
entry->is_directory = false;
|
||||
entry->link_target = str_dup(link_target);
|
||||
if (!entry->link_target)
|
||||
goto skip;
|
||||
if (options->munge_links)
|
||||
file_symlink_unmunge(entry->link_target);
|
||||
goto apply_filters;
|
||||
|
||||
regular:
|
||||
/* Not a symlink: the lstat() above already described this entry, and lstat
|
||||
and stat are identical for every non-symlink, so reuse that result instead
|
||||
of issuing a redundant stat() on the scanner hot path. stat() is still
|
||||
used on the dereference paths above/below for actual symlinks (copy-links,
|
||||
safe/copy-unsafe links, and -k symlinks-to-directories). */
|
||||
entry->stats = link_stats;
|
||||
entry->is_directory = S_ISDIR(link_stats.st_mode);
|
||||
if (entry->is_directory)
|
||||
return 1;
|
||||
|
||||
apply_filters:
|
||||
for (int i = 0; i < options->exclude_count; i++)
|
||||
if (glob_match(options->exclude_patterns[i], name)) {
|
||||
entry->excluded = true;
|
||||
goto skip;
|
||||
}
|
||||
if (options->include_count > 0) {
|
||||
bool included = false;
|
||||
for (int i = 0; i < options->include_count; i++)
|
||||
if (glob_match(options->include_patterns[i], name))
|
||||
included = true;
|
||||
if (!included) {
|
||||
entry->excluded = true;
|
||||
goto skip;
|
||||
}
|
||||
}
|
||||
if ((options->max_size > 0 && (unsigned long long)entry->stats.st_size > options->max_size) ||
|
||||
(options->min_size > 0 && (unsigned long long)entry->stats.st_size < options->min_size)) {
|
||||
entry->excluded = true;
|
||||
entry->size_excluded = true;
|
||||
goto skip;
|
||||
}
|
||||
return 1;
|
||||
|
||||
skip:
|
||||
free(entry->path);
|
||||
entry->path = NULL;
|
||||
free(entry->link_target);
|
||||
entry->link_target = NULL;
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,132 @@
|
||||
#ifndef SCANNER_INTERNAL_H
|
||||
#define SCANNER_INTERNAL_H
|
||||
|
||||
/* Internal declarations shared between the scanner translation units
|
||||
* (scanner_filter.c, scanner.c, scanner_parallel.c). Nothing here is part of
|
||||
* the public scanner façade (scanner.h); every symbol stays internal to the
|
||||
* client module. */
|
||||
|
||||
#include "array_list.h"
|
||||
#include "file.h"
|
||||
#include "scanner.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
typedef struct {
|
||||
char* path;
|
||||
int depth;
|
||||
FilterNode* context; /* inherited per-directory filter context */
|
||||
} DirEntry;
|
||||
|
||||
/* How rsync's readlink_stat()/generator resolves one source symlink. */
|
||||
typedef enum {
|
||||
LINK_ACTION_SKIP, /* not transferred (no link option) */
|
||||
LINK_ACTION_SKIP_PROTECTED, /* ignored as unsafe by --safe-links; rsync keeps
|
||||
it in the transfer, so its destination mirror
|
||||
must be protected from --delete */
|
||||
LINK_ACTION_DEREF, /* follow the referent (--copy-links, an unsafe
|
||||
target under --copy-unsafe-links, or -k dir) */
|
||||
LINK_ACTION_CARRY, /* transmit the link itself (-l) */
|
||||
} LinkAction;
|
||||
|
||||
typedef struct {
|
||||
char* path;
|
||||
struct stat stats;
|
||||
bool is_directory;
|
||||
/* True when the entry should be carried through as a SYMLINK (is_symlink)
|
||||
rather than a dereferenced file/directory. When true, `link_target` holds
|
||||
the owned target string to transmit (sender-munged under --munge-links);
|
||||
ownership transfers to the File built from this entry. */
|
||||
bool is_symlink;
|
||||
char* link_target;
|
||||
/* True when the entry was pruned by a user selection rule (--filter/-C/per-dir
|
||||
rules or the --exclude/--include layer) rather than skipped for another
|
||||
reason (unreadable, symlink policy, not applicable). */
|
||||
bool excluded;
|
||||
/* True when the entry was skipped specifically by --max-size/--min-size.
|
||||
Size pruning protects the destination mirror even under --delete-excluded,
|
||||
so it is recorded into a separate sink from `excluded`. */
|
||||
bool size_excluded;
|
||||
/* True when a symlink selected for dereferencing (-L/--copy-links or an
|
||||
unsafe target under --copy-unsafe-links) had no usable referent (a broken
|
||||
link or a stat() failure). rsync still reports this as a partial transfer
|
||||
(exit 23) even though the entry is skipped, so the scanner records it as a
|
||||
non-fatal I/O error. */
|
||||
bool referent_error;
|
||||
} ScannerEntry;
|
||||
|
||||
typedef enum {
|
||||
SCANNER_SPECIAL_REGULAR, /* ordinary file: transfer content */
|
||||
SCANNER_SPECIAL_RECREATE, /* is_special node to recreate on the receiver */
|
||||
SCANNER_SPECIAL_SKIP, /* non-regular entry not requested: skip */
|
||||
} ScannerSpecial;
|
||||
|
||||
/* Result of scanner_build_file_entry(). The two failure variants preserve the
|
||||
* sequential scanner's historical distinction between a failure before the
|
||||
* File existed (which kept walking the directory) and one afterwards (which cut
|
||||
* the chunk short); both mark the scan failed. */
|
||||
typedef enum {
|
||||
SCANNER_BUILD_OK, /* File built; caller owns it */
|
||||
SCANNER_BUILD_SKIP, /* non-regular entry not preserved; no File */
|
||||
SCANNER_BUILD_FAIL_CONTINUE, /* failed before the File existed */
|
||||
SCANNER_BUILD_FAIL_BREAK, /* failed after the File existed */
|
||||
} ScannerBuildStatus;
|
||||
|
||||
/* scanner_filter.c */
|
||||
void filter_node_destroy(void* item);
|
||||
FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own);
|
||||
void dir_entry_destroy(void* item);
|
||||
DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context);
|
||||
LinkAction scanner_link_action(const ScannerOptions* options, const char* path,
|
||||
const char* link_rel, char* target, size_t target_size);
|
||||
File* scanner_build_dir_file(const char* path, const struct stat* stats,
|
||||
const ScannerOptions* options);
|
||||
char* child_rel_path(const char* parent_rel, const char* name);
|
||||
char* scanner_prefix_send_path(const char* prefix, const char* rel);
|
||||
char* scanner_dest_rel_path(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode);
|
||||
bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base,
|
||||
const FilterNode* node, const char* rel, const char* leaf, bool is_dir,
|
||||
bool per_dir_filters, bool exclude_filter_files, bool* protect_out);
|
||||
void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file);
|
||||
void scanner_capture_xattrs_opts(const ScannerOptions* options, File* file);
|
||||
bool scanner_assign_hardlink(HardLinkTable* table, File* file, const struct stat* stats);
|
||||
ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials,
|
||||
bool copy_devices, File* file, const struct stat* stats);
|
||||
/* Build one non-directory transfer File from an inspected entry. `rel` is the
|
||||
* entry's transfer-root-relative path (used for the -R wire path); `inspected`
|
||||
* supplies the on-disk path, stats and (for a carried symlink) the target whose
|
||||
* ownership transfers to the File. Populates data size, send_path, special-node
|
||||
* state, hardlink group, metadata and xattrs. On SCANNER_BUILD_OK the caller
|
||||
* owns *out_file; on SCANNER_BUILD_SKIP it is NULL and the entry is dropped; on
|
||||
* either failure it is NULL and the caller must mark the scan failed. `*failed`
|
||||
* additionally reports a non-fatal hardlink-table allocation failure, in which
|
||||
* case a usable File is still returned. */
|
||||
ScannerBuildStatus scanner_build_file_entry(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, File** out_file, bool* failed);
|
||||
bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel);
|
||||
void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path);
|
||||
void scanner_note_mount(const ScannerOptions* options, const char* fs_path);
|
||||
void scanner_note_filter(const ScannerOptions* options, const char* name);
|
||||
void scanner_dir_count_count(const ScannerOptions* options);
|
||||
void scanner_dir_count_uncount(const ScannerOptions* options);
|
||||
void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path);
|
||||
void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path);
|
||||
bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode);
|
||||
FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path,
|
||||
const char* rel, bool relative_mode, bool* any_exists, char* err,
|
||||
size_t err_size);
|
||||
int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited);
|
||||
int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir,
|
||||
const char* link_rel, const char* name, ScannerEntry* entry);
|
||||
|
||||
/* scanner.c */
|
||||
bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path,
|
||||
const char* fs_path, bool relative_mode, const char* relative_prefix,
|
||||
bool preserve_atimes, bool preserve_crtimes, bool preserve_xattrs,
|
||||
bool preserve_acls, bool no_implied_dirs,
|
||||
const FileListSet* file_list);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,681 @@
|
||||
#include "log.h"
|
||||
#include "scanner.h"
|
||||
#include "scanner_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "file.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/sysmacros.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <limits.h>
|
||||
|
||||
#include "xattr.h"
|
||||
|
||||
typedef struct {
|
||||
ParallelScanner* ps;
|
||||
char** dirs;
|
||||
int dir_count;
|
||||
char* root_dir; /* the transfer root, for relative-path computation */
|
||||
ScannerOptions options;
|
||||
ProtocolSession* allocation_session;
|
||||
} ParallelWorkerArg;
|
||||
|
||||
static int parallel_worker_thread(void* arg) {
|
||||
ParallelWorkerArg* wa = (ParallelWorkerArg*)arg;
|
||||
ProtocolSession* allocation_session = wa->allocation_session;
|
||||
if (allocation_session)
|
||||
protocol_session_bind(allocation_session);
|
||||
for (int i = 0; i < wa->dir_count; i++) {
|
||||
DirectoryScanner* ds = directory_scanner_create_with_options(wa->dirs[i], &wa->options);
|
||||
if (!ds) {
|
||||
mtx_lock(&wa->ps->result_mutex);
|
||||
wa->ps->failed = true;
|
||||
atomic_store(&wa->ps->cancelled, true);
|
||||
cnd_broadcast(&wa->ps->result_not_empty);
|
||||
cnd_broadcast(&wa->ps->result_not_full);
|
||||
mtx_unlock(&wa->ps->result_mutex);
|
||||
for (int j = i; j < wa->dir_count; j++)
|
||||
free(wa->dirs[j]);
|
||||
break;
|
||||
}
|
||||
/* Root .rsync-filter rules (parsed by the parallel scanner) apply to the
|
||||
* contents of every assigned subdirectory. Relative paths (used by the
|
||||
* allow-set and per-directory rules) are computed against the transfer
|
||||
* root, not the subdirectory the worker is seeded with. Exclusion
|
||||
* recording shares one caller-owned list across the workers. */
|
||||
free(ds->root_path);
|
||||
ds->root_path = str_dup(wa->root_dir);
|
||||
ds->seed_node = wa->ps->root_filter_node;
|
||||
ds->options.excluded_mutex = &wa->ps->result_mutex;
|
||||
Chunk* chunk;
|
||||
while ((chunk = directory_scanner_next(ds)) != NULL) {
|
||||
if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex,
|
||||
&wa->ps->result_not_empty, &wa->ps->result_not_full,
|
||||
&wa->ps->cancelled)) {
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (directory_scanner_failed(ds)) {
|
||||
mtx_lock(&wa->ps->result_mutex);
|
||||
wa->ps->failed = true;
|
||||
atomic_store(&wa->ps->cancelled, true);
|
||||
cnd_broadcast(&wa->ps->result_not_empty);
|
||||
cnd_broadcast(&wa->ps->result_not_full);
|
||||
mtx_unlock(&wa->ps->result_mutex);
|
||||
} else if (directory_scanner_had_io_error(ds)) {
|
||||
/* --ignore-errors path: an unreadable directory was skipped, not fatal. */
|
||||
mtx_lock(&wa->ps->result_mutex);
|
||||
wa->ps->io_error = true;
|
||||
mtx_unlock(&wa->ps->result_mutex);
|
||||
}
|
||||
directory_scanner_destroy(ds);
|
||||
free(wa->dirs[i]);
|
||||
}
|
||||
ParallelScanner* ps = wa->ps;
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
mtx_lock(&ps->result_mutex);
|
||||
ps->completed++;
|
||||
if (ps->completed >= ps->expected_threads) {
|
||||
ps->done = true;
|
||||
cnd_signal(&ps->result_not_empty);
|
||||
}
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
if (allocation_session)
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
|
||||
static void parallel_scanner_creation_failed(ParallelScanner* ps) {
|
||||
mtx_lock(&ps->result_mutex);
|
||||
ps->failed = true;
|
||||
atomic_store(&ps->cancelled, true);
|
||||
ps->expected_threads = ps->created_threads;
|
||||
if (ps->completed >= ps->expected_threads)
|
||||
ps->done = true;
|
||||
cnd_broadcast(&ps->result_not_empty);
|
||||
cnd_broadcast(&ps->result_not_full);
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
}
|
||||
|
||||
/* Initialize result queue and synchronization primitives. Returns true on success. */
|
||||
static bool parallel_scanner_init(ParallelScanner* ps) {
|
||||
ps->result_queue = queue_create(SCANNER_RESULT_QUEUE_CAP, chunk_destroy);
|
||||
if (!ps->result_queue)
|
||||
return false;
|
||||
atomic_init(&ps->cancelled, false);
|
||||
int init = 0;
|
||||
bool ok = true;
|
||||
if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success)
|
||||
ok = false;
|
||||
if (ok) {
|
||||
init++;
|
||||
if (cnd_init(&ps->result_not_empty) != thrd_success)
|
||||
ok = false;
|
||||
}
|
||||
if (ok) {
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
if (cnd_init(&ps->result_not_full) != thrd_success)
|
||||
ok = false;
|
||||
}
|
||||
if (!ok) {
|
||||
if (init >= 3)
|
||||
cnd_destroy(&ps->result_not_full);
|
||||
if (init >= 2)
|
||||
cnd_destroy(&ps->result_not_empty);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&ps->result_mutex);
|
||||
queue_destroy(ps->result_queue);
|
||||
ps->result_queue = NULL;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Split files into chunks of roughly chunk_size bytes. Returns the first chunk (also stored
|
||||
* chunks beyond the first are enqueued on `queue`). Nulls out consumed entries in `files`.
|
||||
* Sets *failed on allocation/enqueue errors. */
|
||||
static Chunk* batch_files(ArrayList* files, unsigned long long chunk_size, Queue* queue,
|
||||
bool* failed) {
|
||||
Chunk* first = NULL;
|
||||
if (files->size <= 0)
|
||||
return NULL;
|
||||
ArrayList* batch = array_list_create(NULL);
|
||||
if (!batch) {
|
||||
*failed = true;
|
||||
return NULL;
|
||||
}
|
||||
unsigned long long batch_size = 0;
|
||||
for (int i = 0; i < files->size; i++) {
|
||||
File* f = (File*)files->items[i];
|
||||
if (!array_list_add(batch, f)) {
|
||||
*failed = true;
|
||||
break;
|
||||
}
|
||||
batch_size += f->data->size;
|
||||
if (batch_size >= chunk_size || i == files->size - 1) {
|
||||
void** items = array_list_to_array(batch);
|
||||
if (!items) {
|
||||
*failed = true;
|
||||
array_list_delete(batch);
|
||||
batch = NULL;
|
||||
break;
|
||||
}
|
||||
Chunk* c = chunk_create((File**)items, batch->size);
|
||||
free(items);
|
||||
if (!c) {
|
||||
*failed = true;
|
||||
array_list_delete(batch);
|
||||
batch = NULL;
|
||||
break;
|
||||
}
|
||||
int batch_start = i - batch->size + 1;
|
||||
for (int j = batch_start; j <= i; j++)
|
||||
files->items[j] = NULL;
|
||||
batch->item_destroyer = NULL;
|
||||
array_list_delete(batch);
|
||||
batch = NULL;
|
||||
if (!first) {
|
||||
first = c;
|
||||
} else {
|
||||
if (!queue_enqueue(queue, c)) {
|
||||
chunk_destroy(c);
|
||||
*failed = true;
|
||||
}
|
||||
}
|
||||
if (i < files->size - 1) {
|
||||
batch = array_list_create(NULL);
|
||||
if (!batch) {
|
||||
*failed = true;
|
||||
break;
|
||||
}
|
||||
batch_size = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (batch) {
|
||||
batch->item_destroyer = NULL;
|
||||
array_list_delete(batch);
|
||||
}
|
||||
return first;
|
||||
}
|
||||
|
||||
/* Record the delete-protection mirror of a root entry that
|
||||
* scanner_inspect_entry() skipped (inspection == 0): a dereferenced symlink
|
||||
* with no referent is a partial-transfer I/O error and a user-selection or size
|
||||
* prune protects the entry's destination mirror. */
|
||||
static void scan_root_record_skipped(const ScannerOptions* options, const char* root_directory,
|
||||
const char* name, const ScannerEntry* inspected,
|
||||
ParallelScanner* ps) {
|
||||
if (inspected->referent_error)
|
||||
ps->io_error = true;
|
||||
ArrayList* sink = NULL;
|
||||
if (inspected->excluded)
|
||||
sink = inspected->size_excluded ? options->size_skipped_paths : options->excluded_paths;
|
||||
if (!sink)
|
||||
return;
|
||||
/* A root-level prune protects the destination mirror of the entry's wire
|
||||
path: under -R + --files-from that is the bare relative name, otherwise it
|
||||
is the full source path with a leading '/' removed (matching the
|
||||
send_path/file_wire_path the scanner hands the sender). */
|
||||
if (options->relative && options->file_list != NULL) {
|
||||
if (!excluded_sink_append(sink, options->excluded_mutex, name))
|
||||
ps->failed = true;
|
||||
} else if (options->relative_prefix) {
|
||||
char* wrel = scanner_prefix_send_path(options->relative_prefix, name);
|
||||
if (!wrel) {
|
||||
ps->failed = true;
|
||||
} else {
|
||||
if (!excluded_sink_append(sink, options->excluded_mutex, wrel))
|
||||
ps->failed = true;
|
||||
free(wrel);
|
||||
}
|
||||
} else {
|
||||
char* abs_path = path_cat(root_directory, name);
|
||||
if (!abs_path) {
|
||||
ps->failed = true;
|
||||
} else {
|
||||
const char* rel = *abs_path == '/' ? abs_path + 1 : abs_path;
|
||||
if (!excluded_sink_append(sink, options->excluded_mutex, rel))
|
||||
ps->failed = true;
|
||||
free(abs_path);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Record the delete-protection mirror of a root entry dropped by the
|
||||
* --files-from allow-set or a filter rule. Returns false only when the -R
|
||||
* prefix could not be built (the caller must abandon the entry immediately);
|
||||
* other allocation failures mark the scan failed but let the caller continue to
|
||||
* the filter-notice step, matching the historical inlined flow. */
|
||||
static bool scan_root_record_protection(const ScannerOptions* options, const char* rel,
|
||||
const char* name, const char* cur_path, bool protect,
|
||||
bool passes, bool use_rel, ParallelScanner* ps) {
|
||||
if (passes && !protect)
|
||||
return true;
|
||||
/* --files-from subset pruning is not a filter exclusion; -R bare-wire-path
|
||||
exclusions are never recorded (see ScannerOptions.excluded_paths). */
|
||||
bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel);
|
||||
if ((!files_from_prune && !use_rel) || protect) {
|
||||
const char* rel_path;
|
||||
char* prefixed = NULL;
|
||||
if (use_rel) {
|
||||
/* -R + --files-from: the destination/wire path is the bare relative
|
||||
name, not the source path. */
|
||||
rel_path = rel;
|
||||
} else if (options->relative_prefix) {
|
||||
prefixed = scanner_prefix_send_path(options->relative_prefix, name);
|
||||
if (!prefixed)
|
||||
return false;
|
||||
rel_path = prefixed;
|
||||
} else {
|
||||
rel_path = *cur_path == '/' ? cur_path + 1 : cur_path;
|
||||
}
|
||||
if (options->excluded_paths &&
|
||||
!excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path))
|
||||
ps->failed = true;
|
||||
free(prefixed);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Root-level directory node: apply -x/--one-file-system and either emit the
|
||||
* mount-point directory (plain -x) or queue the directory for a worker. */
|
||||
static void scan_root_dir(const ScannerOptions* options, const char* cur_path, const char* rel,
|
||||
const struct stat* st, ArrayList* root_files, ArrayList* subdirs,
|
||||
dev_t root_dev, ParallelScanner* ps) {
|
||||
if (!scanner_same_filesystem(options->one_file_system, root_dev, st->st_dev)) {
|
||||
if (options->one_file_system > 1) {
|
||||
/* -xx: drop the mount-point directory entirely (rsync) and print the
|
||||
--info=mount line when enabled. */
|
||||
scanner_note_mount(options, cur_path);
|
||||
return;
|
||||
}
|
||||
/* -x/--one-file-system: emit the mount-point directory entry (empty) but do
|
||||
not descend into it (see the sequential scanner for the same rule). */
|
||||
File* mount = scanner_build_dir_file(cur_path, st, options);
|
||||
if (!mount) {
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
if (options->relative_prefix) {
|
||||
mount->send_path = scanner_prefix_send_path(options->relative_prefix, rel);
|
||||
if (!mount->send_path) {
|
||||
file_destroy(mount);
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (!array_list_add(root_files, mount)) {
|
||||
file_destroy(mount);
|
||||
ps->failed = true;
|
||||
}
|
||||
return;
|
||||
}
|
||||
char* dir = str_dup(cur_path);
|
||||
if (!dir || !array_list_add(subdirs, dir)) {
|
||||
free(dir);
|
||||
ps->failed = true;
|
||||
}
|
||||
}
|
||||
|
||||
/* Build a non-directory root entry through the shared construction path and add
|
||||
* it to `root_files`. A non-regular entry the options do not preserve is
|
||||
* dropped by the builder (which prints rsync's nonreg line); an allocation
|
||||
* failure marks the scan failed. */
|
||||
static void scan_root_add_non_dir(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, ArrayList* root_files, ParallelScanner* ps) {
|
||||
File* file = NULL;
|
||||
bool failed = false;
|
||||
ScannerBuildStatus status = scanner_build_file_entry(options, inspected, rel, &file, &failed);
|
||||
if (failed || status == SCANNER_BUILD_FAIL_CONTINUE || status == SCANNER_BUILD_FAIL_BREAK)
|
||||
ps->failed = true;
|
||||
if (status != SCANNER_BUILD_OK)
|
||||
return;
|
||||
if (!array_list_add(root_files, file)) {
|
||||
file_destroy(file);
|
||||
ps->failed = true;
|
||||
}
|
||||
}
|
||||
|
||||
/* Regular file or carried symlink at the transfer root. */
|
||||
static void scan_root_file(const ScannerOptions* options, ScannerEntry* inspected, const char* rel,
|
||||
ArrayList* root_files, ParallelScanner* ps) {
|
||||
scan_root_add_non_dir(options, inspected, rel, root_files, ps);
|
||||
}
|
||||
|
||||
/* Device/FIFO/socket at the transfer root: recreated under --devices/--specials,
|
||||
* otherwise dropped by the shared builder. */
|
||||
static void scan_root_special(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, ArrayList* root_files, ParallelScanner* ps) {
|
||||
scan_root_add_non_dir(options, inspected, rel, root_files, ps);
|
||||
}
|
||||
|
||||
/* Scan one root-directory entry into either the subdirs or files list. */
|
||||
static void scan_root_entry(const ScannerOptions* options, const FilterNode* root_node,
|
||||
const char* root_directory, const struct dirent* entry,
|
||||
ArrayList* root_files, ArrayList* subdirs, dev_t root_dev,
|
||||
ParallelScanner* ps) {
|
||||
ScannerEntry inspected;
|
||||
int inspection =
|
||||
scanner_inspect_entry(options, root_directory, entry->d_name, entry->d_name, &inspected);
|
||||
if (inspection < 0) {
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
if (inspection == 0) {
|
||||
scan_root_record_skipped(options, root_directory, entry->d_name, &inspected, ps);
|
||||
return;
|
||||
}
|
||||
char* cur_path = inspected.path;
|
||||
char* rel = str_dup(entry->d_name);
|
||||
if (!rel) {
|
||||
ps->failed = true;
|
||||
goto done;
|
||||
}
|
||||
bool is_dir = inspected.is_directory;
|
||||
bool protect = false;
|
||||
bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel,
|
||||
entry->d_name, is_dir, options->per_dir_filters,
|
||||
options->exclude_per_dir_filter_files, &protect);
|
||||
/* -R + --files-from: root-level files keep their bare relative send path. */
|
||||
bool use_rel = options->relative && options->file_list != NULL;
|
||||
if (!passes || protect) {
|
||||
if (!scan_root_record_protection(options, rel, entry->d_name, cur_path, protect, passes,
|
||||
use_rel, ps)) {
|
||||
ps->failed = true;
|
||||
goto done;
|
||||
}
|
||||
if (!passes) {
|
||||
scanner_note_filter(options, entry->d_name);
|
||||
goto done;
|
||||
}
|
||||
}
|
||||
if (is_dir) {
|
||||
scan_root_dir(options, cur_path, rel, &inspected.stats, root_files, subdirs, root_dev, ps);
|
||||
} else if (S_ISCHR(inspected.stats.st_mode) || S_ISBLK(inspected.stats.st_mode) ||
|
||||
S_ISFIFO(inspected.stats.st_mode) || S_ISSOCK(inspected.stats.st_mode)) {
|
||||
scan_root_special(options, &inspected, rel, root_files, ps);
|
||||
} else {
|
||||
scan_root_file(options, &inspected, rel, root_files, ps);
|
||||
}
|
||||
done:
|
||||
free(rel);
|
||||
free(cur_path);
|
||||
free(inspected.link_target);
|
||||
}
|
||||
|
||||
/* Scan the root directory itself, collecting root files and subdirectories.
|
||||
* Returns false if the root directory could not be opened. */
|
||||
static bool scan_root_directory(ParallelScanner* ps, const char* root_directory,
|
||||
const ScannerOptions* options, const FilterNode* root_node,
|
||||
dev_t root_dev, ArrayList* root_files, ArrayList* subdirs) {
|
||||
DIR* dir = opendir(root_directory);
|
||||
if (!dir) {
|
||||
log_perror("Could not open root directory for parallel scan");
|
||||
return false;
|
||||
}
|
||||
/* The parallel scanner opens the transfer root directly (not through
|
||||
open_next_directory), so record it as synchronized here. */
|
||||
if (!scanner_record_synced_dir(options, root_directory, "",
|
||||
options->relative && options->file_list != NULL)) {
|
||||
closedir(dir);
|
||||
ps->failed = true;
|
||||
return false;
|
||||
}
|
||||
log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", root_directory);
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
scan_root_entry(options, root_node, root_directory, entry, root_files, subdirs, root_dev, ps);
|
||||
}
|
||||
closedir(dir);
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Spawn worker threads, one per group of subdirectories. */
|
||||
static void spawn_parallel_workers(ParallelScanner* ps, ArrayList* subdirs,
|
||||
const ScannerOptions* options, const char* root_directory,
|
||||
unsigned long long cs) {
|
||||
if (subdirs->size <= 0)
|
||||
return;
|
||||
int n = options->num_threads > 0 ? options->num_threads : 4;
|
||||
if (n > subdirs->size)
|
||||
n = subdirs->size;
|
||||
|
||||
ps->num_threads = n;
|
||||
ps->expected_threads = n;
|
||||
ps->threads = calloc(n, sizeof(thrd_t));
|
||||
if (!ps->threads) {
|
||||
ps->num_threads = 0;
|
||||
ps->expected_threads = 0;
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
int dirs_per_thread = subdirs->size / n;
|
||||
int remainder = subdirs->size % n;
|
||||
int start = 0;
|
||||
ps->num_threads = 0;
|
||||
for (int t = 0; t < n; t++) {
|
||||
int count = dirs_per_thread + (t < remainder ? 1 : 0);
|
||||
if (count == 0)
|
||||
break;
|
||||
ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg));
|
||||
if (!wa) {
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
wa->ps = ps;
|
||||
wa->dirs = calloc(count, sizeof(char*));
|
||||
wa->root_dir = str_dup(root_directory);
|
||||
if (!wa->dirs || !wa->root_dir) {
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
bool dup_ok = true;
|
||||
for (int j = 0; j < count; j++) {
|
||||
wa->dirs[j] = str_dup((char*)subdirs->items[start + j]);
|
||||
if (!wa->dirs[j])
|
||||
dup_ok = false;
|
||||
}
|
||||
if (!dup_ok) {
|
||||
for (int j = 0; j < count; j++)
|
||||
free(wa->dirs[j]);
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
wa->dir_count = count;
|
||||
wa->options = *options;
|
||||
wa->options.chunk_size = cs;
|
||||
wa->allocation_session = ps->allocation_session;
|
||||
start += count;
|
||||
if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) {
|
||||
for (int j = 0; j < count; j++)
|
||||
free(wa->dirs[j]);
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
ps->num_threads++;
|
||||
ps->created_threads++;
|
||||
}
|
||||
}
|
||||
|
||||
ParallelScanner* parallel_scanner_create_with_options(const char* root_directory,
|
||||
const ScannerOptions* options,
|
||||
ProtocolSession* allocation_session) {
|
||||
if (!root_directory || !options)
|
||||
return NULL;
|
||||
ParallelScanner* ps = calloc(1, sizeof(ParallelScanner));
|
||||
if (!ps)
|
||||
return NULL;
|
||||
if (!parallel_scanner_init(ps)) {
|
||||
free(ps);
|
||||
return NULL;
|
||||
}
|
||||
ps->allocation_session = allocation_session;
|
||||
ps->options = options;
|
||||
|
||||
ArrayList* root_files = array_list_create(file_destroy);
|
||||
ArrayList* subdirs = array_list_create(free);
|
||||
if (!root_files || !subdirs) {
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
dev_t root_dev = 0;
|
||||
if (options->one_file_system) {
|
||||
struct stat root_stats;
|
||||
if (stat(root_directory, &root_stats) != 0) {
|
||||
log_perror("Could not stat source directory");
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
root_dev = root_stats.st_dev;
|
||||
}
|
||||
|
||||
/* Build the root directory's per-directory filter context once; workers seed
|
||||
* their scanners with it so per-dir rules behave identically to the sequential
|
||||
* scanner. */
|
||||
FilterNode* root_node = NULL;
|
||||
{
|
||||
char err[256];
|
||||
bool any_exists = false;
|
||||
FilterRuleList* own = read_dir_filters(options, root_directory, "",
|
||||
options->relative && options->file_list != NULL,
|
||||
&any_exists, err, sizeof(err));
|
||||
if (!own) {
|
||||
/* A parse/allocation failure must fail the scan even when an earlier
|
||||
merge file in the same directory existed (see the sequential scanner). */
|
||||
if (err[0] != '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err);
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
/* no files exist: leave root_node NULL */
|
||||
} else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) {
|
||||
root_node = filter_node_alloc(NULL, own);
|
||||
if (!root_node) {
|
||||
filter_rule_list_free(own);
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
} else {
|
||||
filter_rule_list_free(own);
|
||||
}
|
||||
}
|
||||
ps->root_filter_node = root_node;
|
||||
|
||||
if (!scan_root_directory(ps, root_directory, options, root_node, root_dev, root_files, subdirs)) {
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
/* The root itself is a traversed directory (rsync counts it in
|
||||
`Number of files`); the worker DirectoryScanners account for every
|
||||
subdirectory below it. */
|
||||
scanner_dir_count_count(options);
|
||||
/* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the
|
||||
transfer root itself (it hands the root's immediate subdirectories to
|
||||
workers), so capture the root's directory time here. */
|
||||
if (options->capture_dir_times &&
|
||||
!scanner_capture_dir_time(
|
||||
options->dir_entries, options->dir_entries_mutex, root_directory, root_directory,
|
||||
options->relative && options->file_list != NULL, options->relative_prefix,
|
||||
options->preserve_atimes, options->preserve_crtimes, options->preserve_xattrs,
|
||||
options->preserve_acls, options->no_implied_dirs, options->file_list)) {
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
unsigned long long cs = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE;
|
||||
ps->initial_chunk = batch_files(root_files, cs, ps->result_queue, &ps->failed);
|
||||
array_list_delete(root_files);
|
||||
|
||||
spawn_parallel_workers(ps, subdirs, options, root_directory, cs);
|
||||
array_list_delete(subdirs);
|
||||
return ps;
|
||||
}
|
||||
|
||||
Chunk* parallel_scanner_next(ParallelScanner* ps) {
|
||||
if (ps->initial_chunk) {
|
||||
Chunk* c = ps->initial_chunk;
|
||||
ps->initial_chunk = NULL;
|
||||
return c;
|
||||
}
|
||||
if (ps->num_threads == 0) {
|
||||
mtx_lock(&ps->result_mutex);
|
||||
if (!queue_is_empty(ps->result_queue)) {
|
||||
Chunk* chunk = queue_dequeue(ps->result_queue);
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
return chunk;
|
||||
}
|
||||
ps->done = true;
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
return NULL;
|
||||
}
|
||||
Chunk* chunk = queue_dequeue_multithreaded(
|
||||
ps->result_queue, &ps->result_mutex, &ps->result_not_empty, &ps->result_not_full, &ps->done);
|
||||
return chunk;
|
||||
}
|
||||
|
||||
bool parallel_scanner_failed(const ParallelScanner* ps) {
|
||||
return ps == NULL || ps->failed;
|
||||
}
|
||||
|
||||
bool parallel_scanner_had_io_error(const ParallelScanner* ps) {
|
||||
return ps != NULL && ps->io_error;
|
||||
}
|
||||
|
||||
void parallel_scanner_destroy(ParallelScanner* ps) {
|
||||
if (!ps)
|
||||
return;
|
||||
mtx_lock(&ps->result_mutex);
|
||||
ps->done = true;
|
||||
atomic_store(&ps->cancelled, true);
|
||||
cnd_broadcast(&ps->result_not_empty);
|
||||
cnd_broadcast(&ps->result_not_full);
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
for (int i = 0; i < ps->num_threads; i++)
|
||||
thrd_join(ps->threads[i], NULL);
|
||||
free(ps->threads);
|
||||
if (ps->root_filter_node)
|
||||
filter_node_destroy(ps->root_filter_node);
|
||||
if (ps->initial_chunk)
|
||||
chunk_destroy(ps->initial_chunk);
|
||||
queue_destroy(ps->result_queue);
|
||||
mtx_destroy(&ps->result_mutex);
|
||||
cnd_destroy(&ps->result_not_empty);
|
||||
cnd_destroy(&ps->result_not_full);
|
||||
free(ps);
|
||||
}
|
||||
+175
-92
@@ -19,20 +19,30 @@ void print_usage(void) {
|
||||
printf("\n");
|
||||
printf("Options:\n");
|
||||
printf(" -c, --checksum Verify content by checksum instead of size+mtime\n");
|
||||
printf(" -z, --compress [level] Enable compression (level 1-22, default 5)\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, metadata,\n");
|
||||
printf(" devices and specials (not compression/multithreading)\n");
|
||||
printf(" -z, --compress [level] Enable compression. The default level is\n");
|
||||
printf(" per-codec: zstd 3 (range 1-22), zlib/zlibx 6, lz4\n");
|
||||
printf(" ignores the level\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, perms, times,\n");
|
||||
printf(" owner, group, devices and specials; not\n");
|
||||
printf(" compression/multithreading\n");
|
||||
printf(" -r, --recursive Recurse into directories (FastSync is always recursive)\n");
|
||||
printf(" --inc-recursive Accepted for rsync CLI compatibility; no effect (FastSync\n");
|
||||
printf(" always performs a full scan, so the destination is identical)\n");
|
||||
printf(" --no-inc-recursive Accepted for rsync CLI compatibility; no effect\n");
|
||||
printf(" -n, --dry-run Show what would be transferred\n");
|
||||
printf(" --remove-source-files Remove regular source files after successful transfer\n");
|
||||
printf(" -p, --perms Preserve permission bits (part of the metadata bundle)\n");
|
||||
printf(" -p, --perms Preserve permission bits\n");
|
||||
printf(" -t, --times Preserve modification times\n");
|
||||
printf(" -o, --owner Preserve owner (uid)\n");
|
||||
printf(" -g, --group Preserve group (gid)\n");
|
||||
printf(" --ssh-port <port> SSH port (default: 22)\n");
|
||||
printf(" -e, --rsh <command> Remote shell to launch on the client for the SSH\n");
|
||||
printf(" transport (default: ssh). The command may include\n");
|
||||
printf(" arguments, e.g. -e \"ssh -p 2222\"\n");
|
||||
printf(" --rsync-path <path> Alias for --fastsync-server-path (path to the\n");
|
||||
printf(" fastsync server binary on the remote side)\n");
|
||||
printf(" --blocking-io Leave the SSH transport socket without read/write\n");
|
||||
printf(" timeouts so it blocks naturally\n");
|
||||
printf(" --blocking-io SSH transport only: leave the socket without read/write\n");
|
||||
printf(" timeouts so it blocks naturally (no effect on TCP)\n");
|
||||
printf(" --outbuf=MODE stdout/stderr buffering: N (none/unbuffered),\n");
|
||||
printf(" L (line-buffered), or B (block-buffered, default)\n");
|
||||
printf(" --progress Show transfer progress\n");
|
||||
@@ -44,6 +54,7 @@ void print_usage(void) {
|
||||
printf(" converted before transmission and back on receipt; a\n");
|
||||
printf(" name that cannot be represented in the target charset\n");
|
||||
printf(" fails that transfer cleanly (rsync-compatible)\n");
|
||||
printf(" --no-iconv Disable --iconv charset conversion (same as --iconv=-)\n");
|
||||
printf(" --protocol=NUM Force the wire protocol version (must equal the current\n");
|
||||
printf(" PROTOCOL_VERSION; FastSync cannot speak older/virtual\n");
|
||||
printf(" wire formats)\n");
|
||||
@@ -54,24 +65,27 @@ void print_usage(void) {
|
||||
printf(" Emit the batch file only (no destination, no server)\n");
|
||||
printf(" --read-batch=FILE Apply the batch file to the destination (no source, no\n");
|
||||
printf(" server); takes only the destination as an argument\n");
|
||||
printf(" NOTE: the FastSync batch format is NOT interoperable with rsync's batch\n");
|
||||
printf(" files (different container format); do not mix the two tools.\n");
|
||||
printf(" --delete Delete files on receiver not in source\n");
|
||||
printf(" (default timing: delete only after the whole\n");
|
||||
printf(" transfer has succeeded)\n");
|
||||
printf(" (default timing: delete-during, like rsync --del)\n");
|
||||
printf(" --delete-before Delete extras before the transfer starts\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-during Delete extras once the keep-set manifest is known,\n");
|
||||
printf(" before the data is applied (implies --delete)\n");
|
||||
printf(" --delete-during Delete a directory's extras as that directory is\n");
|
||||
printf(" processed (implies --delete)\n");
|
||||
printf(" --del Alias for --delete-during\n");
|
||||
printf(" --delete-delay Delete extras only after a successful transfer\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-delay Record the extras during the scan but remove them\n");
|
||||
printf(" only after a successful transfer (implies --delete)\n");
|
||||
printf(" --delete-after Delete only after the whole transfer succeeded\n");
|
||||
printf(" (the default --delete timing; implies --delete)\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-commit FastSync-only: restore the late whole-tree commit\n");
|
||||
printf(" (identical to --delete-after; implies --delete)\n");
|
||||
printf(" --delete-excluded Also delete destination files that were excluded on\n");
|
||||
printf(" the source (default protects them, matching rsync)\n");
|
||||
printf(" --max-delete=NUM Never delete more than NUM destination entries per run;\n");
|
||||
printf(" if the extras would exceed NUM, nothing is deleted and\n");
|
||||
printf(" the run fails with a clear error (implies --delete only\n");
|
||||
printf(" when used with it)\n");
|
||||
printf(" --max-delete=NUM Delete at most NUM destination entries per run; if the\n");
|
||||
printf(" extras exceed NUM, the rest are skipped and the run is\n");
|
||||
printf(" reported as partial (exit 25, matching rsync). Only\n");
|
||||
printf(" applies together with --delete\n");
|
||||
printf(" --ignore-errors Continue (and still delete) when a source directory is\n");
|
||||
printf(" unreadable during the scan, instead of aborting with no\n");
|
||||
printf(" deletion\n");
|
||||
@@ -83,10 +97,11 @@ void print_usage(void) {
|
||||
printf(" entry's destination mirror receiver-side. Independent of\n");
|
||||
printf(" --delete (it does not imply --delete; a non-empty directory\n");
|
||||
printf(" mirror is removed only with --force or --delete)\n");
|
||||
printf(" -m, --prune-empty-dirs Do not transfer empty directory entries (--dirs mode);\n");
|
||||
printf(" recursive transfers never send empty dirs\n");
|
||||
printf(" -m, --prune-empty-dirs Do not create empty directories (a recursive transfer\n");
|
||||
printf(" otherwise recreates them, like rsync)\n");
|
||||
printf(" Note: each timing flag implies --delete. Combining a timing flag with\n");
|
||||
printf(" --no-delete (in either order) is rejected as a config error.\n");
|
||||
printf(" --no-delete (in either order) is rejected as a config error, as is more\n");
|
||||
printf(" than one timing flag.\n");
|
||||
printf(" --ignore-existing Skip files that already exist on receiver\n");
|
||||
printf(" --delay-updates Put updated files into place only at the end of transfer\n");
|
||||
printf(" --dirs, -d, --old-dirs, --old-d Transfer the named directory entries without\n");
|
||||
@@ -96,24 +111,28 @@ void print_usage(void) {
|
||||
printf(" -R, --relative With --files-from, preserve each listed entry's relative path\n");
|
||||
printf(" below the destination root instead of mirroring the full\n");
|
||||
printf(" source path (no effect without --files-from)\n");
|
||||
printf(" --no-implied-dirs With -R --files-from, refuse to place a listed file whose\n");
|
||||
printf(" parent directory is not itself listed\n");
|
||||
printf(" --no-implied-dirs With -R, do not apply the source metadata of a listed file's\n");
|
||||
printf(" implied parent directories (they are still created with\n");
|
||||
printf(" default attributes)\n");
|
||||
printf(" --mkpath Create the destination root directory on the server when it\n");
|
||||
printf(" does not exist yet\n");
|
||||
printf(" --exclude <pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file> Read include patterns from file\n");
|
||||
printf(" --exclude <pattern>, --exclude=<pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern>, --include=<pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file>, --exclude-from=<file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file>, --include-from=<file> Read include patterns from file\n");
|
||||
printf(" --files-from <file> Read the source file list from FILE (paths relative to the "
|
||||
"source root)\n");
|
||||
printf(" -0, --from0 Entries in --files-from are NUL-delimited\n");
|
||||
printf(" -f, --filter=RULE rsync-style filter rule (+/- include/exclude; repeatable;\n");
|
||||
printf(" both --filter=RULE and the -f RULE / -f=RULE short forms work)\n");
|
||||
printf(" -f, --filter=RULE rsync-style filter rule: exclude/- include/+ hide/H show/S\n");
|
||||
printf(" protect/P risk/R merge/. dir-merge/: clear/! with modifiers\n");
|
||||
printf(" (repeatable; --filter=RULE and -f RULE / -f=RULE both work)\n");
|
||||
printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n");
|
||||
printf(" -F Apply per-directory .rsync-filter files during the scan\n");
|
||||
printf(" -F Apply per-directory .rsync-filter files; repeated -FF also\n");
|
||||
printf(" excludes the .rsync-filter files themselves\n");
|
||||
printf(" --max-size <n> Skip files larger than n bytes\n");
|
||||
printf(" --min-size <n> Skip files smaller than n bytes\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G)\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G; 0 = no limit,\n");
|
||||
printf(" matching rsync)\n");
|
||||
printf(" --incremental Skip files unchanged since last transfer\n");
|
||||
printf(" --size-only Skip incremental files matching in size, ignoring mtime\n");
|
||||
printf(" -I, --ignore-times Transfer files even when size and mtime match\n");
|
||||
@@ -127,22 +146,29 @@ void print_usage(void) {
|
||||
printf(" into the destination instead of transferring its data\n");
|
||||
printf(" --link-dest <dir> Like --copy-dest, but hard-links the unchanged file from DIR\n");
|
||||
printf(" into the destination (repeatable; earlier DIRs win)\n");
|
||||
printf(" --verify-basis FastSync-only: require a basis hit's content to match the\n");
|
||||
printf(" source by whole-file digest instead of trusting rsync's\n");
|
||||
printf(" size+mtime (or --size-only) quick-check\n");
|
||||
printf(" --checksum-choice, --cc <alg> Whole-file checksum algorithm for --incremental/\n");
|
||||
printf(" --checksum compares (xxh64/xxhash or md5; default xxh64 with\n");
|
||||
printf(" seed 0). The seed comes from --checksum-seed\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash64 digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. The\n");
|
||||
printf(" digest algorithm and seed must match on sender and receiver\n");
|
||||
printf(" --checksum compares. Accepted: xxh128 (default), xxh3, xxh64\n");
|
||||
printf(" (aka xxhash), md5, md4, sha1, or none. A two-name\n");
|
||||
printf(" 'transfer,pre-transfer' form is accepted like rsync; 'none' as\n");
|
||||
printf(" the pre-transfer algorithm is rejected with --checksum\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. A seed\n");
|
||||
printf(" of 0 (the default) is randomized per transfer, exactly like\n");
|
||||
printf(" rsync, and the chosen seed is sent to the receiver\n");
|
||||
printf(" --delta Delta transfer for changed files (requires --incremental)\n");
|
||||
printf(" -W, --whole-file Transfer changed files without delta processing\n");
|
||||
printf(" --no-whole-file rsync spelling that clears -W/--whole-file\n");
|
||||
printf(" -y, --fuzzy Use a similar-named file already in the destination\n");
|
||||
printf(" directory as the delta basis when the destination has no\n");
|
||||
printf(" usable file at the exact path (saves bandwidth; implies\n");
|
||||
printf(" --incremental and --delta; inert with --whole-file,\n");
|
||||
printf(" --no-delta, or --no-incremental)\n");
|
||||
printf(" --no-fuzzy Disable --fuzzy\n");
|
||||
printf(" --delta-block <n>, --block-size <n>\n");
|
||||
printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT);
|
||||
printf(" -B <n>, --block-size <n>, --delta-block <n>\n");
|
||||
printf(" Delta block size in bytes (default: %u)\n", DELTA_BLOCK_SIZE_DEFAULT);
|
||||
printf(" --delta-max <n> Max file size for delta transfer (default: %llu)\n",
|
||||
DELTA_MAX_FILE_SIZE);
|
||||
printf(" -j, --threads[=N] Enable the multithreaded scanner/loader/sender\n");
|
||||
@@ -152,52 +178,71 @@ void print_usage(void) {
|
||||
printf(" --chunk-serialization Enable chunk serialization (long form only)\n");
|
||||
printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n");
|
||||
printf(" SSH argv is already built injection-safe)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm (default: zstd)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only;\n");
|
||||
printf(" -f is bound to --filter, not --sendfile)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm: zstd (default), lz4, zlib,\n");
|
||||
printf(" zlibx, none, or auto\n");
|
||||
printf(" --zc <alg> Alias for --compress-choice\n");
|
||||
printf(" -v, --verbose Enable debug logging\n");
|
||||
printf(" -q, --quiet Suppress non-error output\n");
|
||||
printf(" --debug=FLAGS Fine-grained debug logging (use --debug=help for flags)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,misc,skip,stats,all,none\n");
|
||||
printf(" none suppresses info even with --verbose\n");
|
||||
printf(" --preserve Preserve file metadata (long form only)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,name,misc,skip,stats,all,none\n");
|
||||
printf(" (use --info=help for flags; none suppresses --verbose)\n");
|
||||
printf(" --preserve Preserve permissions and times (= -pt; long form only)\n");
|
||||
printf(" --no-perms Negate -p/--perms\n");
|
||||
printf(" --no-times Negate -t/--times\n");
|
||||
printf(" --no-owner Negate -o/--owner\n");
|
||||
printf(" --no-group Negate -g/--group\n");
|
||||
printf(" --no-preserve Disable metadata preservation (negates --preserve)\n");
|
||||
printf(" -E, --executability Preserve executable permission bits\n");
|
||||
printf(" -U, --atimes Preserve access times\n");
|
||||
printf(" -N, --crtimes Capture birth time; cannot be applied (documented\n");
|
||||
printf(" divergence)\n");
|
||||
printf(" -O, --omit-dir-times Do not apply modification times to directories\n");
|
||||
printf(" -J, --omit-link-times Do not apply times to symlinks\n");
|
||||
printf(" --open-noatime Open source files with O_NOATIME so reading for a\n");
|
||||
printf(" transfer does not update their access time\n");
|
||||
printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n");
|
||||
printf(" privileged security.*/trusted.* namespaces are\n");
|
||||
printf(" never captured or applied)\n");
|
||||
printf(" -A, --acls Preserve POSIX ACLs (the system.posix_acl_* xattrs;\n");
|
||||
printf(" setting an ACL the receiver is not permitted to\n");
|
||||
printf(" set is warned and skipped, never fatal)\n");
|
||||
printf(" --fake-super Store the source uid/gid/mode/mtime in a reserved\n");
|
||||
printf(" user.fastsync.stat xattr on each written file and\n");
|
||||
printf(" re-apply it (fd-relative) on a privileged run; the\n");
|
||||
printf(" recording format diverges from rsync's user.rsync.%%stat%%\n");
|
||||
printf(" --fake-super Store the source mode/rdev/uid/gid in rsync's\n");
|
||||
printf(" reserved user.rsync.%%stat xattr on each written\n");
|
||||
printf(" file (interoperable with rsync); it never performs a\n");
|
||||
printf(" real chown, so an unprivileged receiver records the\n");
|
||||
printf(" privileged stat for a later restore\n");
|
||||
printf(" --super Permit the receiver to attempt super-user activities\n");
|
||||
printf(" (char/block device-node creation, --write-devices)\n");
|
||||
printf(" within the confined receive root. Never elevates\n");
|
||||
printf(" privileges and never bypasses confinement; ownership\n");
|
||||
printf(" is still applied only with an explicit identity flag\n");
|
||||
printf(" (--numeric-ids/--chown/--usermap/--groupmap/--copy-as)\n");
|
||||
printf(" is still applied only with -o/--owner, -g/--group, or an\n");
|
||||
printf(" explicit identity flag (--chown/--usermap/--groupmap/\n");
|
||||
printf(" --copy-as); --numeric-ids only changes how ids map\n");
|
||||
printf(" --no-super Forbid those super-user activities even when the\n");
|
||||
printf(" receiver is running as root\n");
|
||||
printf(" --chmod <changes> Modify transferred permissions (rsync syntax)\n");
|
||||
printf(" --numeric-ids Do not map uid/gid by name: use the source numeric\n");
|
||||
printf(" ids directly when applying ownership\n");
|
||||
printf(
|
||||
" --chmod <changes> Modify new/transferred permissions (rsync syntax; implies no -p)\n");
|
||||
printf(" --numeric-ids Map uid/gid by id instead of by name (a modifier, not\n");
|
||||
printf(" an ownership request: combine with -o/-g or a map)\n");
|
||||
printf(" --usermap=MAP Map usernames when applying ownership: comma-separated\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM/TO are names\n");
|
||||
printf(" (resolved on the source machine), * (match any /\n");
|
||||
printf(" current user), or @N numeric ids. e.g. *:nobody\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM is a name (from\n");
|
||||
printf(" the source), an id, an inclusive LOW-HIGH range, *\n");
|
||||
printf(" (any id), or empty (ids with no name). TO is an id, *\n");
|
||||
printf(" (current user), or a name resolved on the receiver.\n");
|
||||
printf(" e.g. 0-99:nobody,*:normal (cannot mix with --chown)\n");
|
||||
printf(" --groupmap=MAP Map group names when applying ownership (same syntax)\n");
|
||||
printf(" --chown=USER:GROUP Override the ownership of transferred files. Forms:\n");
|
||||
printf(" USER:GROUP, USER (owner only), :GROUP (group only); a\n");
|
||||
printf(" value of * means the current/root user as appropriate.\n");
|
||||
printf(" Names resolve on the source machine; @N for numerics.\n");
|
||||
printf(" (Metadata is enabled with --preserve; -M now means\n");
|
||||
printf(" rsync's --remote-option.)\n");
|
||||
printf(" (Implies owner/group metadata; -M now means rsync's\n");
|
||||
printf(" --remote-option.)\n");
|
||||
printf(" --copy-as=USER[:GROUP] Force every written entry (files, dirs, symlinks\n");
|
||||
printf(" and special nodes) to USER[:GROUP], resolved on the\n");
|
||||
printf(" source machine like --chown. Requires a privileged\n");
|
||||
printf(" (root) receiver and implies --preserve; an\n");
|
||||
printf(" (root) receiver and implies owner/group metadata; an\n");
|
||||
printf(" unprivileged receiver refuses the transfer. Never\n");
|
||||
printf(" switches process credentials (safe-subset; see\n");
|
||||
printf(" RSYNC_COMPAT.md). A daemon refuses it.\n");
|
||||
@@ -209,65 +254,88 @@ void print_usage(void) {
|
||||
printf(" --server-port <n> Server port (default: 8080)\n");
|
||||
printf(" --port <n> Alias for --server-port\n");
|
||||
printf(" --password-file <f> Authenticate a host::module/path daemon destination.\n");
|
||||
printf(" The file's first user:password line supplies the\n");
|
||||
printf(" username and password (only a SHA-256 digest of the\n");
|
||||
printf(" password is sent; keep the file mode 0600)\n");
|
||||
printf(" FastSync-native SCRAM/PBKDF2 credential scheme (NOT\n");
|
||||
printf(" rsync's --password-file): the file's first user:password\n");
|
||||
printf(" line supplies the username and password; no password or\n");
|
||||
printf(" reusable digest is sent (keep the file mode 0600)\n");
|
||||
printf(" --no-motd Suppress display of the daemon's MOTD (the server\n");
|
||||
printf(" still sends it; the client just does not show it)\n");
|
||||
printf(" --bwlimit <KB/s> Bandwidth limit in kilobytes per second\n");
|
||||
printf(" --bwlimit=RATE Limit socket I/O bandwidth (default unit KiB/s,\n");
|
||||
printf(" rsync-style: 0 = no limit; K/M/G/T/P suffixes are\n");
|
||||
printf(" binary, KB/MB decimal, KiB/MiB binary; decimals allowed)\n");
|
||||
printf(" --tls Enable TLS encryption\n");
|
||||
printf(" --cert <path> TLS certificate file (PEM)\n");
|
||||
printf(" --key <path> TLS private key file (PEM)\n");
|
||||
printf(" --ca <path> TLS CA certificate file (PEM)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 30; long form only)\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 10)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 0 = disabled, matching\n");
|
||||
printf(" rsync). 0 disables it; --no-timeout is the same\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 60, matching\n");
|
||||
printf(" rsync); 0 disables it (--no-contimeout)\n");
|
||||
printf(" --stop-after=MINS Stop the transfer after MINS minutes (a positive\n");
|
||||
printf(" integer); whatever was already transferred is kept\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time: HH:MM, HH:MM:SS, or\n");
|
||||
printf(" now+N[smhd] (a time already in the past stops the\n");
|
||||
printf(" transfer immediately; client-only). An early stop\n");
|
||||
printf(" skips the late --delete keep-set so it cannot delete\n");
|
||||
printf(" source mirrors that were not yet scanned\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time. Accepts rsync's date form\n");
|
||||
printf(" (Y-M-DTh:m, Y/M/DTh:m, abbreviable fields such as 12-31,\n");
|
||||
printf(" 14:00, :59, 1) plus FastSync's HH:MM[:SS] and now+N[smhd]\n");
|
||||
printf(" (a time already in the past stops the transfer\n");
|
||||
printf(" immediately; client-only). An early stop skips the late\n");
|
||||
printf(" --delete keep-set so it cannot delete source mirrors that\n");
|
||||
printf(" were not yet scanned\n");
|
||||
printf(" --address <ip> Bind the outgoing client socket to this source address\n");
|
||||
printf(" -4, --ipv4 Force IPv4 for destination resolution\n");
|
||||
printf(" -6, --ipv6 Force IPv6 for destination resolution\n");
|
||||
printf(" --sockopts=OPTS Comma-separated OPT=VAL socket options applied before connect:\n");
|
||||
printf(" TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR\n");
|
||||
printf(" --backup Backup existing files before overwriting\n");
|
||||
printf(" -b, --backup Backup existing files before overwriting\n");
|
||||
printf(" --backup-dir <dir> Directory for backups (requires --backup)\n");
|
||||
printf(" --suffix <str> Backup suffix (default: ~)\n");
|
||||
printf(" --stats Print transfer statistics at end\n");
|
||||
printf(" -i, --itemize-changes Print an rsync-style per-file change line\n");
|
||||
printf(" --out-format=FORMAT Output format for changed files (%%f %%n %%l %%b %%M %%%%)\n");
|
||||
printf(" --out-format=FORMAT Output format (%%f %%n %%l %%b %%c %%C %%i %%M %%%%)\n");
|
||||
printf(" --list-only List source files instead of transferring\n");
|
||||
printf(" --log-file-format=FORMAT Per-file log line format (needs --log-file)\n");
|
||||
printf(" -h, --human-readable Print byte sizes in human-readable form\n");
|
||||
printf(" --max-depth <n> Maximum directory depth (0=unlimited)\n");
|
||||
printf(" -x, --one-file-system Do not cross filesystem boundaries\n");
|
||||
printf(" --log-file <path> Write log messages to file\n");
|
||||
printf(" --stderr=MODE Route logging to stderr: errors or all\n");
|
||||
printf(" --log-file <path>, --log-file=<path> Write log messages to file\n");
|
||||
printf(" --stderr=MODE Route logging: errors (default), all, or client\n");
|
||||
printf(" (forward the client's diagnostics to the server's\n");
|
||||
printf(" stderr)\n");
|
||||
printf(" --msgs2stderr Route all messages to stderr (deprecated spelling of\n");
|
||||
printf(" --stderr=all)\n");
|
||||
printf(" --no-msgs2stderr Forward the client's diagnostics to the server\n");
|
||||
printf(" (deprecated spelling of --stderr=client)\n");
|
||||
printf(" --partial Keep partial files on interrupted transfer\n");
|
||||
printf(" --partial-dir <dir> Directory for partial files\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install\n");
|
||||
printf(" --partial-dir <dir> Directory for partial files (implies --partial)\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install.\n");
|
||||
printf(" Confined to the receive root: a relative dir resolves below\n");
|
||||
printf(" it and an absolute/traversal dir is rejected. The dir must\n");
|
||||
printf(" already exist; a different filesystem falls back to a\n");
|
||||
printf(" non-atomic copy instead of aborting\n");
|
||||
printf(" --fastsync-server-path <path>\n");
|
||||
printf(" Path to fastsync-server on remote (default: fastsync-server)\n");
|
||||
printf(" --old-args Accepted for rsync CLI compatibility; no effect (the\n");
|
||||
printf(" remote server path is always safely quoted now)\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation over SSH\n");
|
||||
printf(" (repeatable; each value is single-quote-escaped on the remote\n");
|
||||
printf(" command line; empty values and values with control characters\n");
|
||||
printf(" are rejected; -M OPT, -M=OPT and --remote-option=OPT work)\n");
|
||||
printf(" --trust-sender Trust the remote sender's file list: the receiver skips its\n");
|
||||
printf(" own up-front path-traversal/containment re-validation of the\n");
|
||||
printf(" incoming file list (fewer checks, faster, potentially unsafe).\n");
|
||||
printf(" Local receiver policy: never sent to the peer, off by default\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation. SSH\n");
|
||||
printf(" transport ONLY (user@host:path): a daemon (host::module) or\n");
|
||||
printf(" local TCP destination rejects it (no remote command line to\n");
|
||||
printf(" append to). Repeatable; each value is single-quote-escaped on\n");
|
||||
printf(" the remote command line; empty values and values with control\n");
|
||||
printf(" characters are rejected; -M OPT, -M=OPT and\n");
|
||||
printf(" --remote-option=OPT work\n");
|
||||
printf(" --trust-sender RECEIVER-LOCAL policy: trust the remote sender's file list\n");
|
||||
printf(" and skip the receiver's own up-front path-traversal/\n");
|
||||
printf(" containment re-validation of the incoming list (fewer checks,\n");
|
||||
printf(" faster, potentially unsafe). It is never sent to the peer, so\n");
|
||||
printf(" for a push it must be enabled on the receiving SERVER\n");
|
||||
printf(" (fastsync-server --trust-sender) or forwarded with\n");
|
||||
printf(" -M--trust-sender; the client flag alone has no effect\n");
|
||||
printf(" -l, --links Copy symlinks as symlinks\n");
|
||||
printf(" --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks that point outside transfer tree\n");
|
||||
printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n");
|
||||
printf(" -L, --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks whose target points outside the tree\n");
|
||||
printf(" --copy-unsafe-links Copy unsafe symlinks (outside tree) as referent files\n");
|
||||
printf(" -k, --copy-dirlinks Transform symlinks to directories into real dirs\n");
|
||||
printf(" -K, --keep-dirlinks Keep an existing symlink-to-dir as that dir\n");
|
||||
printf(" --munge-links Munge symlink targets on the wire (sender)\n");
|
||||
printf(" --munge-links Munge stored symlink targets (/rsyncd-munged/) on the receiver\n");
|
||||
printf(" -H, --hard-links Preserve hard-link relationships across the transfer\n");
|
||||
printf(" -S, --sparse Handle sparse files efficiently\n");
|
||||
printf(
|
||||
@@ -275,8 +343,7 @@ void print_usage(void) {
|
||||
printf(
|
||||
" --devices Recreate device nodes on the destination (privileged; skipped when\n");
|
||||
printf(" the receiver lacks CAP_MKNOD)\n");
|
||||
printf(" --specials Recreate special files (FIFOs) on the destination (sockets "
|
||||
"skipped)\n");
|
||||
printf(" --specials Recreate special files (FIFOs, sockets) on the destination\n");
|
||||
printf(" --copy-devices Copy a source device's content as a regular file instead\n");
|
||||
printf(" --write-devices Write received data into an existing destination device node\n");
|
||||
printf(" --inplace Update files in-place (no temp+rename)\n");
|
||||
@@ -287,9 +354,12 @@ void print_usage(void) {
|
||||
printf(" --append-verify Like --append, but verifies the retained prefix checksum\n");
|
||||
printf(" before appending (falls back to a full transfer on mismatch)\n");
|
||||
printf(" --fsync Fsync every written file before publication\n");
|
||||
printf(" --compress-level <n> Compression level (default: 5)\n");
|
||||
printf(" --compress-level <n> Compression level (per-codec default: zstd 3,\n");
|
||||
printf(" zlib/zlibx 6, lz4 ignores it)\n");
|
||||
printf(" --zl <n> Alias for --compress-level\n");
|
||||
printf(" --skip-compress=LIST Skip compression for comma-separated suffixes\n");
|
||||
printf(" --skip-compress=LIST Skip compression for suffixes in LIST (separated by\n");
|
||||
printf(" '/' as in rsync, or ','); a leading dot is optional. The\n");
|
||||
printf(" default is rsync 3.4.1's built-in skip-compress list\n");
|
||||
printf(" --compress-threads <n> Compression worker threads (requires zstd threaded support)\n");
|
||||
printf(" --no-OPTION Disable a supported boolean option\n");
|
||||
printf(" --help Show this help\n");
|
||||
@@ -297,7 +367,20 @@ void print_usage(void) {
|
||||
}
|
||||
|
||||
void print_debug_usage(void) {
|
||||
printf("Supported debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n");
|
||||
printf("Emitting debug flags: IO,PROTO,PACK,UTIL,FLIST,DEL,HASH,DELTASUM,\n");
|
||||
printf("RECV,FILTER,SEND,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): ACL,BACKUP,BIND,CHDIR,\n");
|
||||
printf("CONNECT,CMD,DUP,EXIT,FUZZY,GENR,HLINK,ICONV,NSTR,OWN,TIME.\n");
|
||||
printf("Flags may be comma-separated, for example: --debug=io,proto\n");
|
||||
printf("Other rsync debug flags are unsupported and rejected.\n");
|
||||
printf("An optional level suffix is accepted (e.g. --debug=io2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
|
||||
void print_info_usage(void) {
|
||||
printf("Emitting info flags: COPY,MISC,SKIP,STATS,DEL,REMOVE,NAME,FLIST,\n");
|
||||
printf("NONREG,PROGRESS,MOUNT,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): BACKUP,SYMS,SYMSAFE.\n");
|
||||
printf("Flags may be comma-separated, for example: --info=name,stats\n");
|
||||
printf("An optional level suffix is accepted (e.g. --info=stats2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
@@ -3,5 +3,6 @@
|
||||
|
||||
void print_usage(void);
|
||||
void print_debug_usage(void);
|
||||
void print_info_usage(void);
|
||||
|
||||
#endif
|
||||
+702
-160
@@ -3,6 +3,7 @@
|
||||
#include "charset.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "delay_updates.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
@@ -10,9 +11,13 @@
|
||||
#include "metadata.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code) {
|
||||
if (!outcomes)
|
||||
@@ -43,17 +48,110 @@ void receiver_outcomes_destroy(ReceiverOutcomes* outcomes) {
|
||||
/* End-of-transfer success frame. When --remove-source-files was negotiated
|
||||
each processed data file is acknowledged first (STATUS_NEXT = written,
|
||||
STATUS_OK = skipped) so the sender never removes a source the receiver did
|
||||
not actually store. The frame always ends with a plain STATUS_OK. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes) {
|
||||
not actually store. The frame ends with `final_status` (STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped). */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status) {
|
||||
if (!config->remove_source_files)
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
size_t count = outcomes ? outcomes->count : 0;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
Status per_file = outcomes->entries[i] == FILE_SAVE_WRITTEN ? STATUS_NEXT : STATUS_OK;
|
||||
Status per_file;
|
||||
if (outcomes->entries[i] == FILE_SAVE_WRITTEN)
|
||||
per_file = STATUS_NEXT;
|
||||
else if (outcomes->entries[i] == FILE_SAVE_FAILED)
|
||||
per_file = STATUS_ERROR;
|
||||
else
|
||||
per_file = STATUS_OK;
|
||||
if (!send_status(fd, per_file))
|
||||
return false;
|
||||
}
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
}
|
||||
|
||||
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
|
||||
const struct ArrayList* would_delete,
|
||||
const struct ArrayList* deleted_paths) {
|
||||
if (!config->report_stats)
|
||||
return true;
|
||||
ReceiverStats local;
|
||||
memset(&local, 0, sizeof(local));
|
||||
const ReceiverStats* out = stats ? stats : &local;
|
||||
/* The path list carries the dry-run would-delete set for a -n run and the
|
||||
actually-removed set for a real --info=del run. */
|
||||
const struct ArrayList* paths =
|
||||
config->dry_run ? would_delete : (config->report_deletes ? deleted_paths : NULL);
|
||||
size_t count = paths ? (size_t)paths->size : 0;
|
||||
if (count > (size_t)MAX_MANIFEST_ENTRIES)
|
||||
count = MAX_MANIFEST_ENTRIES;
|
||||
ReceiverStats record = *out;
|
||||
record.would_delete_count = count;
|
||||
if (!send_status(fd, STATUS_STATS) || !format_stats_send(fd, &record) ||
|
||||
!send_int(fd, (int)count))
|
||||
return false;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
const char* path = (const char*)paths->items[i];
|
||||
if (!send_wire_str(fd, path ? path : ""))
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Add a delete commit's tally to the sink's end-of-transfer wire counters (when
|
||||
the sink reports them). Runs on the receiving thread, so no locking. */
|
||||
static void receiver_tally_deleted(const ReceiverSink* sink, size_t deleted) {
|
||||
if (sink && sink->stats && deleted > 0)
|
||||
sink->stats->deleted_files += deleted;
|
||||
}
|
||||
|
||||
/* Observer for --info=del/--stats: record each truly-removed destination-
|
||||
relative path (when the context carries a path list) and tally it by type
|
||||
(when it carries a stats record), so the terminal STATUS_STATS frame can list
|
||||
the paths and render rsync's per-type `Number of deleted files` breakdown. A
|
||||
failed append is best-effort (the deletion already happened; output is
|
||||
cosmetic). Shared by the single-threaded receiver and the -m pipeline's
|
||||
deferred commit. */
|
||||
void receiver_record_deleted_path(void* context, const char* rel_path, DeleteEntryType type) {
|
||||
ReceiverDeleteContext* del = context;
|
||||
if (!del || !rel_path)
|
||||
return;
|
||||
if (del->stats) {
|
||||
switch (type) {
|
||||
case DELETE_ENTRY_DIR:
|
||||
del->stats->deleted_dir++;
|
||||
break;
|
||||
case DELETE_ENTRY_LINK:
|
||||
del->stats->deleted_link++;
|
||||
break;
|
||||
case DELETE_ENTRY_SPECIAL:
|
||||
del->stats->deleted_special++;
|
||||
break;
|
||||
default:
|
||||
del->stats->deleted_reg++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
ArrayList* paths = del->deleted_paths;
|
||||
if (!paths)
|
||||
return;
|
||||
/* Bound the retained list like the keep-set manifest: only MAX_MANIFEST_ENTRIES
|
||||
paths are ever transmitted in the terminal STATUS_STATS frame, so recording
|
||||
more only grows memory. A hostile/huge deletion set is therefore capped. */
|
||||
if ((size_t)paths->size >= (size_t)MAX_MANIFEST_ENTRIES)
|
||||
return;
|
||||
char* copy = str_dup(rel_path);
|
||||
if (copy && !array_list_add(paths, copy))
|
||||
free(copy);
|
||||
}
|
||||
|
||||
/* Install the delete observer (and its context) for one commit when the sink
|
||||
carries a stats record or a path list. Returns NULL when neither is needed,
|
||||
so the delete engines skip the observer entirely. */
|
||||
static DeletePathObserver receiver_delete_observer(const ReceiverSink* sink,
|
||||
ReceiverDeleteContext* del) {
|
||||
del->stats = sink ? sink->stats : NULL;
|
||||
del->deleted_paths = sink ? sink->deleted_paths : NULL;
|
||||
return (del->stats || del->deleted_paths) ? receiver_record_deleted_path : NULL;
|
||||
}
|
||||
|
||||
static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) {
|
||||
@@ -213,6 +311,7 @@ static bool status_counts_as_progress(Status status) {
|
||||
case STATUS_ABORT:
|
||||
case STATUS_CHECK_BATCH:
|
||||
case STATUS_DIR_TIMES:
|
||||
case STATUS_CLIENT_MSG:
|
||||
return false;
|
||||
default:
|
||||
return true;
|
||||
@@ -242,21 +341,410 @@ static bool receiver_note_status(const struct timespec* session_start,
|
||||
}
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) {
|
||||
return receiver_process_pending(config, file_descriptor, sink, NULL);
|
||||
return receiver_process_pending_ctx(config, file_descriptor, sink, NULL, NULL, NULL);
|
||||
}
|
||||
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans) {
|
||||
return receiver_process_pending_ctx(config, file_descriptor, sink, pending_manifest,
|
||||
pending_plans, NULL);
|
||||
}
|
||||
|
||||
/* Per-connection state threaded through the status handlers below. The parked
|
||||
keep-set / per-directory session live here so one teardown helper can release
|
||||
them on every exit path. */
|
||||
typedef struct {
|
||||
Config* config;
|
||||
int fd;
|
||||
const ReceiverSink* sink;
|
||||
DeleteManifest** pending_manifest;
|
||||
DeletePlanSession** pending_plans;
|
||||
/* Parked keep-set for the late/commit timing. Every exit path frees it
|
||||
exactly once; the only exception is the successful FINISHED handoff, which
|
||||
transfers ownership to *pending_manifest (used by the -m receiver). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* Per-directory delete session for --delete-during/--delete-delay. During the
|
||||
loop it applies plans inline (during) or snapshots their extras (delay); on
|
||||
a successful FINISHED it is either committed here or handed to
|
||||
*pending_plans so the -m caller commits after its disk writer drained. */
|
||||
DeletePlanSession* plan_session;
|
||||
/* Observer context for the per-directory delete session, which outlives the
|
||||
frame handler; must stay alive until the session commits. For a session
|
||||
handed to the caller (pending_plans) this points at a caller-owned
|
||||
long-lived context; otherwise it points at an internal stack context. */
|
||||
ReceiverDeleteContext* delete_ctx;
|
||||
bool early_delete;
|
||||
bool per_dir_delete;
|
||||
bool delete_limit_noted;
|
||||
} ReceiverPendingState;
|
||||
|
||||
/* Outcome of one frame handler. NEXT reads the following status frame; FAIL
|
||||
tears the connection down without a peer STATUS_ERROR; ERROR tears it down
|
||||
and (when the sink owns error reporting) emits STATUS_ERROR. */
|
||||
typedef enum {
|
||||
RECEIVER_STEP_NEXT,
|
||||
RECEIVER_STEP_FAIL,
|
||||
RECEIVER_STEP_ERROR,
|
||||
} ReceiverStep;
|
||||
|
||||
static ReceiverStep receiver_handle_keepalive(ReceiverPendingState* state) {
|
||||
if (!send_status(state->fd, STATUS_KEEPALIVE))
|
||||
return RECEIVER_STEP_FAIL;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
/* rsync --stderr=client: a client diagnostic forwarded over the wire. Read the
|
||||
* bounded string and log it through the normal destination/level gate. The
|
||||
* body is peer-controlled text: log_client_message() escapes every
|
||||
* non-printable byte (newlines, CR, ANSI ESC, ...) before writing, so a hostile
|
||||
* client cannot forge log lines or inject terminal control sequences. A
|
||||
* malformed string (over-long or embedded NUL) is a framing error and tears the
|
||||
* connection down. */
|
||||
static ReceiverStep receiver_handle_client_msg(ReceiverPendingState* state) {
|
||||
char* message = receive_str(state->fd);
|
||||
if (!message)
|
||||
return RECEIVER_STEP_FAIL;
|
||||
if (message[0] != '\0')
|
||||
log_client_message(message);
|
||||
free(message);
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_abort(ReceiverPendingState* state) {
|
||||
(void)state;
|
||||
log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up");
|
||||
return RECEIVER_STEP_FAIL;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_check(ReceiverPendingState* state) {
|
||||
bool skipped = false;
|
||||
bool would_transfer = false;
|
||||
File* file = receive_incremental_check_ex(state->fd, state->config, &skipped, &would_transfer);
|
||||
if (state->config->dry_run) {
|
||||
/* Server-contacting --dry-run: the reply has already been sent
|
||||
(STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and
|
||||
nothing may be stored. Both flags false means a genuine protocol
|
||||
error (STATUS_ERROR already sent or sent by receive_error below). */
|
||||
if (!skipped && !would_transfer)
|
||||
return RECEIVER_STEP_ERROR;
|
||||
} else if (!skipped && (!file || !state->sink->store_file(file, state->sink->context))) {
|
||||
return RECEIVER_STEP_ERROR;
|
||||
}
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_chunk(ReceiverPendingState* state) {
|
||||
Chunk* chunk = receive_chunk_data(state->fd, state->config);
|
||||
if (!chunk || !receiver_process_chunk(chunk, state->sink))
|
||||
return RECEIVER_STEP_ERROR;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_check_batch(ReceiverPendingState* state) {
|
||||
if (!receiver_process_batch(state->config, state->fd))
|
||||
return RECEIVER_STEP_FAIL;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
/* Probe a destination entry's pre-transfer state for the output-parity
|
||||
dest-info report (protocol 2.30.0). `wire_path` is the destination-relative
|
||||
path; `incoming_target` is non-NULL only for a symlink probe, in which case
|
||||
the on-disk link target is compared with the target the receiver is about to
|
||||
store (after --munge-links). The final component is never followed and the
|
||||
parent walk is confined below the receive root. Returns false only on an
|
||||
allocation/secure-walk failure; a missing entry is reported as existed=false. */
|
||||
static bool receiver_probe_dest_state(const Config* config, const char* wire_path,
|
||||
const char* incoming_target, OutputDestState* out) {
|
||||
memset(out, 0, sizeof(*out));
|
||||
out->known = true;
|
||||
if (!config || !wire_path || wire_path[0] == '\0')
|
||||
return false;
|
||||
char* full = path_cat(config->receive_root_directory, wire_path);
|
||||
if (!full)
|
||||
return false;
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(full, &leaf, false);
|
||||
free(full);
|
||||
if (parent_fd < 0) {
|
||||
/* A missing/unreachable parent means the entry cannot exist yet. */
|
||||
free(leaf);
|
||||
return true;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) == 0) {
|
||||
out->existed = true;
|
||||
out->size = (unsigned long long)st.st_size;
|
||||
out->mtime_sec = (long long)st.st_mtime;
|
||||
#ifdef __linux__
|
||||
out->mtime_nsec = st.st_mtim.tv_nsec;
|
||||
#endif
|
||||
out->mode = (uint32_t)st.st_mode;
|
||||
out->uid = (int32_t)st.st_uid;
|
||||
out->gid = (int32_t)st.st_gid;
|
||||
if (incoming_target && S_ISLNK(st.st_mode)) {
|
||||
char target_buf[PATH_MAX];
|
||||
ssize_t n = readlinkat(parent_fd, leaf, target_buf, sizeof(target_buf) - 1);
|
||||
if (n >= 0) {
|
||||
target_buf[n] = '\0';
|
||||
char* expected =
|
||||
config->munge_links ? file_symlink_munge(incoming_target) : str_dup(incoming_target);
|
||||
if (expected) {
|
||||
out->target_matches = strcmp(target_buf, expected) == 0;
|
||||
free(expected);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
return true;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_mkdir(ReceiverPendingState* state) {
|
||||
const Config* config = state->config;
|
||||
int fd = state->fd;
|
||||
if (config->report_dest_info) {
|
||||
int probe = 0;
|
||||
if (!receive_int(fd, &probe) || (probe != 0 && probe != 1))
|
||||
return RECEIVER_STEP_FAIL;
|
||||
if (probe) {
|
||||
/* Probe-only frame: report the destination state and create nothing. */
|
||||
char* path = receive_wire_str(fd);
|
||||
if (!path || path[0] == '\0' || (!file_get_trust_sender() && has_path_traversal(path))) {
|
||||
free(path);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return RECEIVER_STEP_FAIL;
|
||||
}
|
||||
OutputDestState info;
|
||||
bool ok = receiver_probe_dest_state(config, path, NULL, &info);
|
||||
free(path);
|
||||
if (!ok || !send_status(fd, STATUS_DEST_INFO) || !format_dest_state_send(fd, &info))
|
||||
return RECEIVER_STEP_FAIL;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
}
|
||||
File* dir = file_receive_directory(fd, config);
|
||||
if (!dir)
|
||||
return RECEIVER_STEP_ERROR;
|
||||
if (config->report_dest_info) {
|
||||
OutputDestState info;
|
||||
bool ok = receiver_probe_dest_state(config, file_wire_path(dir), NULL, &info);
|
||||
if (!ok || !send_status(fd, STATUS_DEST_INFO) || !format_dest_state_send(fd, &info)) {
|
||||
file_destroy(dir);
|
||||
return RECEIVER_STEP_FAIL;
|
||||
}
|
||||
}
|
||||
if (!state->sink->store_file(dir, state->sink->context))
|
||||
return RECEIVER_STEP_ERROR;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_dir_times(const ReceiverPendingState* state) {
|
||||
if (!receiver_process_dir_times(state->fd, state->config, state->sink))
|
||||
return RECEIVER_STEP_ERROR;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_hardlink(ReceiverPendingState* state) {
|
||||
File* file = file_receive_hardlink(state->fd);
|
||||
if (!file || !state->sink->store_file(file, state->sink->context))
|
||||
return RECEIVER_STEP_ERROR;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_symlink(ReceiverPendingState* state) {
|
||||
const Config* config = state->config;
|
||||
File* sym = file_receive_symlink(state->fd, config);
|
||||
if (!sym)
|
||||
return RECEIVER_STEP_ERROR;
|
||||
if (config->report_dest_info) {
|
||||
OutputDestState info;
|
||||
bool ok = receiver_probe_dest_state(config, file_wire_path(sym), sym->symlink_target, &info);
|
||||
if (!ok || !send_status(state->fd, STATUS_DEST_INFO) ||
|
||||
!format_dest_state_send(state->fd, &info)) {
|
||||
file_destroy(sym);
|
||||
return RECEIVER_STEP_FAIL;
|
||||
}
|
||||
}
|
||||
if (!state->sink->store_file(sym, state->sink->context))
|
||||
return RECEIVER_STEP_ERROR;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_special(ReceiverPendingState* state) {
|
||||
File* file = file_receive_special(state->fd);
|
||||
if (!file || !state->sink->store_file(file, state->sink->context))
|
||||
return RECEIVER_STEP_ERROR;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_manifest(ReceiverPendingState* state) {
|
||||
Config* config = state->config;
|
||||
int fd = state->fd;
|
||||
const ReceiverSink* sink = state->sink;
|
||||
DeleteManifest* manifest = receive_manifest_entries(fd);
|
||||
if (!manifest)
|
||||
return RECEIVER_STEP_FAIL; /* receive_manifest_entries already sent STATUS_ERROR */
|
||||
if (config->dry_run) {
|
||||
/* Server-contacting --dry-run mutates nothing, so a keep-set manifest
|
||||
is consumed and discarded. The early-delete mode still needs its ACK
|
||||
so a sender blocked on the delete handshake is not left hanging.
|
||||
When would-delete reporting is armed, enumerate (read-only) the
|
||||
destination extras so the terminal STATUS_STATS frame can list them. */
|
||||
if (config->use_delete && sink->would_delete) {
|
||||
size_t count = 0;
|
||||
if (!manifest_would_delete_list(config, manifest, sink->would_delete, &count))
|
||||
log_message(LOG_LEVEL_WARNING, "dry-run: could not enumerate would-delete paths");
|
||||
}
|
||||
delete_manifest_free(manifest);
|
||||
if (state->early_delete && !send_status(fd, STATUS_OK))
|
||||
return RECEIVER_STEP_FAIL;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
if (state->early_delete) {
|
||||
/* --delete-before: the whole-tree manifest is authoritative the moment
|
||||
it arrives, before any file data. Delete now and acknowledge so the
|
||||
sender only starts streaming once the deletion committed (or failed).
|
||||
A later transfer failure does not restore these deletions. A
|
||||
--max-delete-capped commit still succeeds and the transfer proceeds;
|
||||
the terminal success frame reports the cap. */
|
||||
size_t deleted = 0;
|
||||
ReceiverDeleteContext delctx;
|
||||
DeletePathObserver observer = receiver_delete_observer(sink, &delctx);
|
||||
DeleteCommitResult deletion =
|
||||
(config->use_delete || config->delete_missing_args)
|
||||
? manifest_delete_all_observed(config, manifest, &deleted, observer, &delctx)
|
||||
: DELETE_COMMIT_OK;
|
||||
receiver_tally_deleted(sink, deleted);
|
||||
delete_manifest_free(manifest);
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return RECEIVER_STEP_FAIL;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
if (!send_status(fd, STATUS_OK))
|
||||
return RECEIVER_STEP_FAIL;
|
||||
} else if (config->use_delete || config->delete_missing_args) {
|
||||
/* Plain --delete / --delete-after and the --delete-missing-args
|
||||
exact-path deletions: hold the manifest and commit it only after
|
||||
STATUS_FINISHED. The per-directory modes never send this frame. */
|
||||
if (state->deferred_manifest) {
|
||||
log_message(LOG_LEVEL_ERROR, "Received a second delete manifest");
|
||||
delete_manifest_free(state->deferred_manifest);
|
||||
state->deferred_manifest = NULL;
|
||||
delete_manifest_free(manifest);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return RECEIVER_STEP_FAIL;
|
||||
}
|
||||
state->deferred_manifest = manifest;
|
||||
} else {
|
||||
delete_manifest_free(manifest);
|
||||
}
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_delete_plan(ReceiverPendingState* state) {
|
||||
Config* config = state->config;
|
||||
int fd = state->fd;
|
||||
const ReceiverSink* sink = state->sink;
|
||||
if (!state->per_dir_delete) {
|
||||
log_message(LOG_LEVEL_ERROR, "Received a per-directory delete plan without a per-dir "
|
||||
"delete timing");
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return RECEIVER_STEP_FAIL;
|
||||
}
|
||||
if (!state->plan_session) {
|
||||
state->plan_session = delete_plan_session_create(config);
|
||||
if (state->plan_session && (sink->stats || sink->deleted_paths))
|
||||
delete_plan_session_set_delete_observer(state->plan_session, receiver_record_deleted_path,
|
||||
state->delete_ctx);
|
||||
}
|
||||
if (!state->plan_session || delete_plan_session_receive(state->plan_session, config, fd) != 0)
|
||||
return RECEIVER_STEP_FAIL;
|
||||
if (delete_plan_session_limit_reached(state->plan_session) && !state->delete_limit_noted &&
|
||||
sink->note_delete_limit) {
|
||||
sink->note_delete_limit(sink->context);
|
||||
state->delete_limit_noted = true;
|
||||
}
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
static ReceiverStep receiver_handle_file(ReceiverPendingState* state) {
|
||||
File* file = file_receive(state->config, state->fd);
|
||||
if (!file) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to receive file");
|
||||
return RECEIVER_STEP_ERROR;
|
||||
}
|
||||
if (!state->sink->store_file(file, state->sink->context))
|
||||
return RECEIVER_STEP_ERROR;
|
||||
return RECEIVER_STEP_NEXT;
|
||||
}
|
||||
|
||||
/* One dispatch per admitted frame type; STATUS_NEXT (and any other
|
||||
data-bearing status) falls through to the regular file receiver. */
|
||||
static ReceiverStep receiver_dispatch_status(ReceiverPendingState* state, Status status) {
|
||||
switch (status) {
|
||||
case STATUS_KEEPALIVE:
|
||||
return receiver_handle_keepalive(state);
|
||||
case STATUS_CLIENT_MSG:
|
||||
return receiver_handle_client_msg(state);
|
||||
case STATUS_ABORT:
|
||||
return receiver_handle_abort(state);
|
||||
case STATUS_CHECK:
|
||||
return receiver_handle_check(state);
|
||||
case STATUS_CHUNK:
|
||||
return receiver_handle_chunk(state);
|
||||
case STATUS_CHECK_BATCH:
|
||||
return receiver_handle_check_batch(state);
|
||||
case STATUS_MKDIR:
|
||||
return receiver_handle_mkdir(state);
|
||||
case STATUS_DIR_TIMES:
|
||||
return receiver_handle_dir_times(state);
|
||||
case STATUS_HARDLINK:
|
||||
return receiver_handle_hardlink(state);
|
||||
case STATUS_SYMLINK:
|
||||
return receiver_handle_symlink(state);
|
||||
case STATUS_SPECIAL:
|
||||
return receiver_handle_special(state);
|
||||
case STATUS_MANIFEST:
|
||||
return receiver_handle_manifest(state);
|
||||
case STATUS_DELETE_PLAN:
|
||||
return receiver_handle_delete_plan(state);
|
||||
default:
|
||||
return receiver_handle_file(state);
|
||||
}
|
||||
}
|
||||
|
||||
/* Release the parked keep-set / per-directory session exactly once on every
|
||||
failure exit. Never commit a deletion for a failed stream. */
|
||||
static void receiver_drop_pending(ReceiverPendingState* state) {
|
||||
if (state->deferred_manifest) {
|
||||
delete_manifest_free(state->deferred_manifest);
|
||||
state->deferred_manifest = NULL;
|
||||
}
|
||||
if (state->plan_session) {
|
||||
delete_plan_session_destroy(state->plan_session);
|
||||
state->plan_session = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
/* Runs the whole receive loop. The delete manifest may legitimately arrive
|
||||
either FIRST (--delete-before / --delete-during: the sender transmits the
|
||||
validated keep-set before any file data) or LAST (plain --delete /
|
||||
--delete-after / --delete-delay: the manifest closes the data stream). In
|
||||
validated keep-set before any file data) or LAST (--delete-after /
|
||||
--delete-commit / --delete-delay: the manifest closes the data stream). In
|
||||
the early modes the receiver deletes as soon as the manifest has been read
|
||||
and acknowledges with STATUS_OK so the sender only starts streaming once the
|
||||
deletion has committed (or failed); in the late modes the manifest is held
|
||||
and the deletion is committed only after the terminal STATUS_FINISHED proves
|
||||
the whole transfer succeeded. See receiver_process_pending() for how the -m
|
||||
receiver defers that commit until its disk writer has drained. */
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest) {
|
||||
the whole transfer succeeded. A plain --delete defaults to the per-directory
|
||||
delete-during plan mode (no manifest at all). See the per-frame handlers
|
||||
above for how the -m receiver defers that commit until its disk writer has
|
||||
drained. */
|
||||
int receiver_process_pending_ctx(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest,
|
||||
DeletePlanSession** pending_plans,
|
||||
ReceiverDeleteContext* observer_ctx) {
|
||||
Status status;
|
||||
if (!receive_status(file_descriptor, &status))
|
||||
return -1;
|
||||
@@ -269,103 +757,37 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
last_progress = session_start;
|
||||
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
|
||||
return -1;
|
||||
bool early_delete = config_delete_timing_early(config);
|
||||
/* Parked keep-set for the late/commit timing. Every exit path below frees it
|
||||
exactly once; the only exception is the successful FINISHED handoff, which
|
||||
transfers ownership to *pending_manifest (used by the -m receiver). */
|
||||
DeleteManifest* deferred_manifest = NULL;
|
||||
/* A per-directory delete session handed to the caller outlives this stack
|
||||
frame, so its observer context must be caller-owned (observer_ctx); only
|
||||
the default inline-commit case may use the stack context. */
|
||||
ReceiverDeleteContext local_ctx;
|
||||
ReceiverDeleteContext* delete_ctx = observer_ctx ? observer_ctx : &local_ctx;
|
||||
delete_ctx->stats = sink ? sink->stats : NULL;
|
||||
delete_ctx->deleted_paths = sink ? sink->deleted_paths : NULL;
|
||||
ReceiverPendingState state = {
|
||||
.config = config,
|
||||
.fd = file_descriptor,
|
||||
.sink = sink,
|
||||
.pending_manifest = pending_manifest,
|
||||
.pending_plans = pending_plans,
|
||||
.deferred_manifest = NULL,
|
||||
.plan_session = NULL,
|
||||
.delete_ctx = delete_ctx,
|
||||
.early_delete = config_delete_timing_early(config),
|
||||
.per_dir_delete = config_delete_timing_per_dir(config),
|
||||
.delete_limit_noted = false,
|
||||
};
|
||||
bool notify_peer = false;
|
||||
while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK ||
|
||||
status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH ||
|
||||
status == STATUS_MKDIR || status == STATUS_MANIFEST || status == STATUS_HARDLINK ||
|
||||
status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES) {
|
||||
if (status == STATUS_KEEPALIVE) {
|
||||
if (!send_status(file_descriptor, STATUS_KEEPALIVE))
|
||||
goto fail;
|
||||
goto next_status;
|
||||
}
|
||||
if (status == STATUS_ABORT) {
|
||||
log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up");
|
||||
status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES ||
|
||||
status == STATUS_DELETE_PLAN || status == STATUS_CLIENT_MSG) {
|
||||
ReceiverStep step = receiver_dispatch_status(&state, status);
|
||||
if (step == RECEIVER_STEP_FAIL)
|
||||
goto fail;
|
||||
}
|
||||
if (status == STATUS_CHECK) {
|
||||
bool skipped;
|
||||
File* file = receive_incremental_check(file_descriptor, config, &skipped);
|
||||
if (!skipped && (!file || !sink->store_file(file, sink->context)))
|
||||
goto receive_error;
|
||||
} else if (status == STATUS_CHUNK) {
|
||||
Chunk* chunk = receive_chunk_data(file_descriptor, config);
|
||||
if (!chunk || !receiver_process_chunk(chunk, sink))
|
||||
goto receive_error;
|
||||
} else if (status == STATUS_CHECK_BATCH) {
|
||||
if (!receiver_process_batch(config, file_descriptor))
|
||||
goto fail;
|
||||
goto next_status;
|
||||
} else if (status == STATUS_MKDIR) {
|
||||
File* dir = file_receive_directory(file_descriptor, config);
|
||||
if (!dir || !sink->store_file(dir, sink->context))
|
||||
goto receive_error;
|
||||
} else if (status == STATUS_DIR_TIMES) {
|
||||
if (!receiver_process_dir_times(file_descriptor, config, sink))
|
||||
goto receive_error;
|
||||
} else if (status == STATUS_HARDLINK) {
|
||||
File* file = file_receive_hardlink(file_descriptor);
|
||||
if (!file || !sink->store_file(file, sink->context))
|
||||
goto receive_error;
|
||||
} else if (status == STATUS_SYMLINK) {
|
||||
File* sym = file_receive_symlink(file_descriptor, config);
|
||||
if (!sym || !sink->store_file(sym, sink->context))
|
||||
goto receive_error;
|
||||
} else if (status == STATUS_SPECIAL) {
|
||||
File* file = file_receive_special(file_descriptor);
|
||||
if (!file || !sink->store_file(file, sink->context))
|
||||
goto receive_error;
|
||||
} else if (status == STATUS_MANIFEST) {
|
||||
DeleteManifest* manifest = receive_manifest_entries(file_descriptor);
|
||||
if (!manifest)
|
||||
goto fail; /* receive_manifest_entries already sent STATUS_ERROR */
|
||||
if (early_delete) {
|
||||
/* --delete-before / --delete-during: the manifest is authoritative the
|
||||
moment it arrives, before any file data. Delete now and acknowledge
|
||||
so the sender only starts streaming once the deletion committed (or
|
||||
failed). This is the rsync delete-before/delete-during window: a
|
||||
later transfer failure does not restore these deletions. */
|
||||
bool deletion_ok = (config->use_delete || config->delete_missing_args)
|
||||
? manifest_delete_all(config, manifest)
|
||||
: true;
|
||||
delete_manifest_free(manifest);
|
||||
if (!deletion_ok) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (!send_status(file_descriptor, STATUS_OK))
|
||||
goto fail;
|
||||
} else if (config->use_delete || config->delete_missing_args) {
|
||||
/* Plain --delete / --delete-after / --delete-delay and the
|
||||
--delete-missing-args exact-path deletions: hold the manifest and
|
||||
commit it only after STATUS_FINISHED. */
|
||||
if (deferred_manifest) {
|
||||
log_message(LOG_LEVEL_ERROR, "Received a second delete manifest");
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
delete_manifest_free(manifest);
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
deferred_manifest = manifest;
|
||||
} else {
|
||||
delete_manifest_free(manifest);
|
||||
}
|
||||
goto next_status;
|
||||
} else {
|
||||
File* file = file_receive(config, file_descriptor);
|
||||
if (!file) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to receive file");
|
||||
goto receive_error;
|
||||
}
|
||||
if (!sink->store_file(file, sink->context))
|
||||
goto receive_error;
|
||||
}
|
||||
next_status:
|
||||
if (step == RECEIVER_STEP_ERROR)
|
||||
goto receive_error;
|
||||
if (!receive_status(file_descriptor, &status))
|
||||
goto receive_error;
|
||||
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
|
||||
@@ -375,27 +797,75 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status");
|
||||
goto receive_error;
|
||||
}
|
||||
/* --delay-updates: publish every staged file BEFORE the deferred delete
|
||||
commit, matching rsync's --delete-after ordering (all updates land first,
|
||||
then extras are removed). The single-threaded receiver stores files
|
||||
synchronously, so every staged file is complete here. The -m receiver
|
||||
hands both publication and deletion to its caller via
|
||||
pending_manifest/pending_plans; that caller publishes first, after its disk
|
||||
writer has drained. */
|
||||
bool handoff = state.pending_manifest != NULL || state.pending_plans != NULL;
|
||||
if (!handoff && !config->dry_run && config->delay_updates && config->delay_context) {
|
||||
if (!delay_updates_publish(config->delay_context, config)) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
}
|
||||
/* Commit-style (late) deletion: every data frame has been received and the
|
||||
sender proved the whole tree with STATUS_FINISHED. The single-threaded
|
||||
receiver stores files synchronously, so everything is on disk here and the
|
||||
deletion can be committed before the --delay-updates publication in
|
||||
send_success (the walker skips the staging dir, so staged files are never
|
||||
treated as extras). The -m receiver passes `pending_manifest` because its
|
||||
disk writer may still be draining; the caller commits after the writer has
|
||||
joined so no extra file is removed unless the transfer is known to have
|
||||
succeeded. */
|
||||
if (deferred_manifest) {
|
||||
if (pending_manifest) {
|
||||
*pending_manifest = deferred_manifest;
|
||||
deferred_manifest = NULL;
|
||||
receiver stores files synchronously, so everything is on disk here (and a
|
||||
--delay-updates run has already published above). The -m receiver passes
|
||||
`pending_manifest` because its disk writer may still be draining; the
|
||||
caller commits after the writer has joined so no extra file is removed
|
||||
unless the transfer is known to have succeeded. */
|
||||
if (state.deferred_manifest) {
|
||||
if (state.pending_manifest) {
|
||||
*state.pending_manifest = state.deferred_manifest;
|
||||
state.deferred_manifest = NULL;
|
||||
} else {
|
||||
bool deletion_ok = manifest_delete_all(config, deferred_manifest);
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
if (!deletion_ok) {
|
||||
size_t deleted = 0;
|
||||
DeletePathObserver observer = receiver_delete_observer(sink, state.delete_ctx);
|
||||
DeleteCommitResult deletion = manifest_delete_all_observed(
|
||||
config, state.deferred_manifest, &deleted, observer, state.delete_ctx);
|
||||
receiver_tally_deleted(sink, deleted);
|
||||
delete_manifest_free(state.deferred_manifest);
|
||||
state.deferred_manifest = NULL;
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
}
|
||||
}
|
||||
/* Per-directory deletion: --delete-during already applied each plan inline, so
|
||||
this only finishes the missing-args deletions; --delete-delay committed
|
||||
nothing yet and applies its decompressed snapshot here. The -m receiver
|
||||
hands the session to its caller instead, which commits after the disk
|
||||
writer drained. */
|
||||
if (state.plan_session) {
|
||||
if (sink->stats || sink->deleted_paths)
|
||||
delete_plan_session_set_delete_observer(state.plan_session, receiver_record_deleted_path,
|
||||
state.delete_ctx);
|
||||
if (state.pending_plans) {
|
||||
*state.pending_plans = state.plan_session;
|
||||
state.plan_session = NULL;
|
||||
} else if (config->dry_run) {
|
||||
/* Central dry-run no-op: never commit a deletion for a -n run. */
|
||||
delete_plan_session_destroy(state.plan_session);
|
||||
state.plan_session = NULL;
|
||||
} else {
|
||||
DeleteCommitResult deletion = delete_plan_session_commit(state.plan_session, config);
|
||||
bool limit = delete_plan_session_limit_reached(state.plan_session);
|
||||
receiver_tally_deleted(sink, delete_plan_session_deleted(state.plan_session));
|
||||
delete_plan_session_destroy(state.plan_session);
|
||||
state.plan_session = NULL;
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (limit && !state.delete_limit_noted && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
}
|
||||
}
|
||||
if (sink->send_success) {
|
||||
@@ -408,21 +878,14 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
}
|
||||
return 0;
|
||||
|
||||
receive_error:
|
||||
notify_peer = true;
|
||||
fail:
|
||||
/* Failure exits that must not (or already did) report a STATUS_ERROR. The
|
||||
parked keep-set is dropped: never commit a deletion for a failed stream. */
|
||||
if (deferred_manifest) {
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
}
|
||||
return -1;
|
||||
|
||||
receive_error:
|
||||
if (deferred_manifest) {
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
}
|
||||
if (sink->send_error)
|
||||
parked keep-set/session is dropped: never commit a deletion for a failed
|
||||
stream. */
|
||||
receiver_drop_pending(&state);
|
||||
if (notify_peer && sink->send_error)
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
return -1;
|
||||
}
|
||||
@@ -436,29 +899,68 @@ typedef struct {
|
||||
after the whole transfer (and its delete/publication phases) has run so a
|
||||
child write never clobbers a directory mtime. */
|
||||
DirTimeList dir_times;
|
||||
/* Set when a --max-delete commit was capped; the terminal frame then carries
|
||||
STATUS_DELETE_LIMIT so the sender exits 25 like rsync. */
|
||||
bool delete_limit_reached;
|
||||
/* End-of-transfer wire counters (protocol 2.25.0) and the -n/--dry-run
|
||||
--delete would-delete path list collected while processing the manifest. */
|
||||
ReceiverStats stats;
|
||||
ArrayList* would_delete;
|
||||
/* --info=del: actually-removed paths collected during the delete commit. */
|
||||
ArrayList* deleted_paths;
|
||||
/* Per-run count of entries that failed to materialize without aborting the
|
||||
stream (currently ONLY a --devices mknod EPERM/EACCES). A nonzero count
|
||||
makes the terminal frame carry a non-OK status so the client exits
|
||||
non-zero, matching rsync's continue-and-exit-partial behavior. */
|
||||
size_t failed_entries;
|
||||
} ReceiverSaveContext;
|
||||
|
||||
static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
FileSaveResult result = FILE_SAVE_ERROR;
|
||||
if (!context->config->save_to_disk) {
|
||||
bool created = false;
|
||||
unsigned created_dirs = 0;
|
||||
if (context->config->dry_run) {
|
||||
/* Defense in depth: a dry-run receiver mutates nothing even if a data
|
||||
frame reaches the sink (the sender is not supposed to send one). */
|
||||
result = FILE_SAVE_SKIPPED;
|
||||
} else if (!context->config->save_to_disk) {
|
||||
/* Nothing is stored; report the file as not-written so a
|
||||
--remove-source-files sender keeps its source. */
|
||||
result = FILE_SAVE_SKIPPED;
|
||||
} else {
|
||||
result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
|
||||
result = file_save_to_disk_full_ex(context->config->receive_root_directory, file,
|
||||
context->config, &created, &created_dirs);
|
||||
}
|
||||
/* A directory's times are deferred, never applied inline: collect the
|
||||
metadata now and apply it at the end. -O/--omit-dir-times is honored by
|
||||
dir_time_list_apply's caller (see receiver_send_success_frame). */
|
||||
/* Wire-stats tally: bytes reconstructed from the basis file (delta matches)
|
||||
count as matched data in the end-of-transfer report. */
|
||||
if (result != FILE_SAVE_ERROR && file->matched_bytes > 0)
|
||||
context->stats.matched_data += file->matched_bytes;
|
||||
/* --devices parity: a device node the receiver could not mknod (EPERM/EACCES)
|
||||
is counted per-run but does not abort the transfer. The terminal frame
|
||||
turns a nonzero count into a non-OK status so the client exits non-zero. */
|
||||
if (result == FILE_SAVE_FAILED)
|
||||
context->failed_entries++;
|
||||
/* Protocol 2.28.0: receiver-observed literal bytes and the created-entry
|
||||
breakdown (regular/dir/link/special) for the `--stats` report. */
|
||||
if (result == FILE_SAVE_WRITTEN)
|
||||
receiver_stats_note_saved(&context->stats, file, created, created_dirs);
|
||||
/* A directory's metadata is deferred, never applied inline: collect it now
|
||||
and apply it at the end. -O/--omit-dir-times and --preserve_perms/-times
|
||||
are honored by dir_metadata_list_apply's caller (see
|
||||
receiver_send_success_frame). */
|
||||
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_times_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir &&
|
||||
!file->is_special && !file->skip &&
|
||||
/* A dry-run receiver mutates nothing AND records no per-file outcomes: a
|
||||
hostile dry-run client that streamed data frames anyway must not be able to
|
||||
grow `outcomes` without bound (receiver_outcomes_append reallocs uncharged)
|
||||
or force a per-frame ack. */
|
||||
if (!context->config->dry_run && result != FILE_SAVE_ERROR &&
|
||||
context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip &&
|
||||
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
return false;
|
||||
@@ -467,35 +969,75 @@ static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
return result != FILE_SAVE_ERROR;
|
||||
}
|
||||
|
||||
static void receiver_note_delete_limit(void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
/* Terminal status for a run. A capped --delete limit wins (rsync exit 25);
|
||||
otherwise any per-entry failure (for example an unprivileged --devices
|
||||
mknod) makes the terminal frame STATUS_PARTIAL so the client exits 23
|
||||
(rsync's "partial transfer due to error") while still removing the sources
|
||||
it successfully transferred under --remove-source-files. A fatal stream
|
||||
error keeps STATUS_ERROR (a non-23 exit). A clean run keeps STATUS_OK. */
|
||||
static Status receiver_final_status(bool delete_limit_reached, size_t failed_entries) {
|
||||
if (delete_limit_reached)
|
||||
return STATUS_DELETE_LIMIT;
|
||||
return failed_entries > 0 ? STATUS_PARTIAL : STATUS_OK;
|
||||
}
|
||||
|
||||
static bool receiver_send_success_frame(int fd, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
/* --delay-updates: the whole protocol stream (including manifest/delete
|
||||
handling, which ran inside receiver_process) has succeeded and every
|
||||
staged file was fully written. Publish them atomically now, before the
|
||||
success/outcome frame tells a --remove-source-files sender it may delete
|
||||
its sources. */
|
||||
if (context->config->delay_updates && context->config->delay_context) {
|
||||
if (!delay_updates_publish(context->config->delay_context, context->config)) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
/* P7 Wave D: every child is now written and the delete / --delay-updates
|
||||
phases have committed, so it is finally safe to stamp directory times.
|
||||
This runs after the deferred deletion because receiver_process commits it
|
||||
before calling this success frame. */
|
||||
dir_time_list_apply(&context->dir_times, context->config->receive_root_directory);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes);
|
||||
if (context->failed_entries > 0)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"%zu entr%s failed to materialize; continuing (partial transfer)",
|
||||
context->failed_entries, context->failed_entries == 1 ? "y" : "ies");
|
||||
Status final_status =
|
||||
receiver_final_status(context->delete_limit_reached, context->failed_entries);
|
||||
if (!receiver_send_stats_frame(fd, context->config, &context->stats, context->would_delete,
|
||||
context->deleted_paths))
|
||||
return false;
|
||||
/* Server-contacting --dry-run: nothing was staged or written, so there is
|
||||
nothing to publish and no directory times to stamp. */
|
||||
if (context->config->dry_run)
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
/* P7 Wave D: every child is now written and the --delay-updates publication
|
||||
(done in receiver_process before the delete commit) plus the deferred
|
||||
deletion have both committed, so it is finally safe to stamp directory
|
||||
times. */
|
||||
dir_metadata_list_apply(&context->dir_times, context->config->receive_root_directory,
|
||||
context->config);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
}
|
||||
|
||||
int receiver_receive_files(Config* config, int file_descriptor) {
|
||||
ReceiverSaveContext context = {.config = config, .outcomes = {0}};
|
||||
dir_time_list_init(&context.dir_times);
|
||||
ReceiverSink sink = {receiver_save_file, &context, true, true, receiver_send_success_frame};
|
||||
context.would_delete = array_list_create(free);
|
||||
/* report_deletes (--info=del / -i / --out-format under --delete) is the only
|
||||
reason to retain the actually-removed paths; a plain --delete must not
|
||||
str_dup every removal. NULL is handled by every consumer. */
|
||||
context.deleted_paths = config->report_deletes ? array_list_create(free) : NULL;
|
||||
if (!context.would_delete || (config->report_deletes && !context.deleted_paths)) {
|
||||
array_list_delete(context.would_delete);
|
||||
array_list_delete(context.deleted_paths);
|
||||
return -1;
|
||||
}
|
||||
ReceiverSink sink = {receiver_save_file,
|
||||
&context,
|
||||
true,
|
||||
true,
|
||||
receiver_send_success_frame,
|
||||
receiver_note_delete_limit,
|
||||
&context.stats,
|
||||
context.would_delete,
|
||||
context.deleted_paths};
|
||||
int ret = receiver_process(config, file_descriptor, &sink);
|
||||
if (ret != 0 && config->delay_updates && config->delay_context)
|
||||
delay_updates_cleanup(config->delay_context);
|
||||
receiver_outcomes_destroy(&context.outcomes);
|
||||
dir_time_list_free(&context.dir_times);
|
||||
array_list_delete(context.would_delete);
|
||||
array_list_delete(context.deleted_paths);
|
||||
return ret;
|
||||
}
|
||||
+64
-5
@@ -2,8 +2,10 @@
|
||||
#define RECEIVER_H
|
||||
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include <stdbool.h>
|
||||
#include <time.h>
|
||||
|
||||
@@ -21,6 +23,12 @@ typedef struct {
|
||||
|
||||
typedef bool (*ReceiverSuccessFrame)(int fd, void* context);
|
||||
|
||||
/* Records that a --max-delete commit stopped with extras left over, so the
|
||||
caller's terminal success frame can carry STATUS_DELETE_LIMIT instead of
|
||||
STATUS_OK. The commit runs on the receiver thread, so the flag is stored in
|
||||
the sink's own context rather than in a shared global. */
|
||||
typedef void (*ReceiverNoteDeleteLimit)(void* context);
|
||||
|
||||
typedef struct {
|
||||
ReceiverFileSink store_file;
|
||||
void* context;
|
||||
@@ -28,23 +36,74 @@ typedef struct {
|
||||
bool send_success;
|
||||
/* Emits the end-of-transfer success frame. When the sender requested
|
||||
--remove-source-files this includes one per-file status per processed
|
||||
data file followed by the final STATUS_OK; otherwise just STATUS_OK. */
|
||||
data file followed by the final status; otherwise just the final status. */
|
||||
ReceiverSuccessFrame send_success_frame;
|
||||
/* Optional; may be NULL when the sink has no --max-delete handling. */
|
||||
ReceiverNoteDeleteLimit note_delete_limit;
|
||||
/* Optional end-of-transfer wire counters (protocol 2.25.0). When non-NULL
|
||||
and the wire config carries report_stats, the success frame is preceded by
|
||||
a STATUS_STATS record; `would_delete` (optional, receiver-owned strings)
|
||||
carries the -n/--dry-run --delete path list. */
|
||||
ReceiverStats* stats;
|
||||
struct ArrayList* would_delete;
|
||||
/* When --info=del requested it, receiver-owned strings for every path the
|
||||
deletion commit ACTUALLY removed, sent in the terminal STATUS_STATS frame's
|
||||
path list so the sender can print rsync's `deleting PATH` lines. */
|
||||
struct ArrayList* deleted_paths;
|
||||
} ReceiverSink;
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code);
|
||||
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes);
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes);
|
||||
|
||||
/* Delete observer context: `deleted_paths` (optional) receives owned copies of
|
||||
every truly-removed destination-relative path for --info=del; `stats`
|
||||
(optional) receives the per-type `Number of deleted files` tallies for
|
||||
--stats. Both may be NULL, in which case the observer is a no-op. */
|
||||
typedef struct {
|
||||
ReceiverStats* stats;
|
||||
struct ArrayList* deleted_paths;
|
||||
} ReceiverDeleteContext;
|
||||
|
||||
/* DeletePathObserver implementation: records each truly-removed path (when the
|
||||
context carries a path list) and tallies it by type (when it carries a stats
|
||||
record). Shared by the single-threaded receiver and the -m pipeline's
|
||||
deferred commit. */
|
||||
void receiver_record_deleted_path(void* context, const char* rel_path, DeleteEntryType type);
|
||||
|
||||
/* Send the terminal success frame. `final_status` is usually STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status);
|
||||
|
||||
/* Emit STATUS_STATS (a fixed ReceiverStats record plus, when `would_delete` is
|
||||
non-NULL, a count and that many wire strings) when the wire config requested
|
||||
report_stats. A no-op otherwise. */
|
||||
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
|
||||
const struct ArrayList* would_delete,
|
||||
const struct ArrayList* deleted_paths);
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink);
|
||||
/* receiver_process with an escape hatch for the commit-style (late) deletion:
|
||||
when `pending_manifest` is non-NULL the receiver does NOT delete at
|
||||
STATUS_FINISHED itself; instead it stores the owned keep-set manifest there
|
||||
(leaving *pending_manifest untouched on early modes/errors) so the caller can
|
||||
commit the deletion only after its disk writer has fully drained. Pass NULL
|
||||
to keep the default behaviour (delete before the success frame). */
|
||||
commit the deletion only after its disk writer has fully drained. Likewise,
|
||||
when `pending_plans` is non-NULL the --delete-delay per-directory session is
|
||||
handed to the caller instead of being committed at STATUS_FINISHED. Pass NULL
|
||||
for either to keep the default behaviour (delete before the success frame). */
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest);
|
||||
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans);
|
||||
/* receiver_process_pending() with an explicit observer context for a
|
||||
per-directory delete session that is handed to the caller via
|
||||
`pending_plans`. The session outlives this call (the -m pipeline commits it
|
||||
after joining its disk writer), so its observer context must too: pass a
|
||||
long-lived object such as PipelineContextReceiver.delete_ctx. When
|
||||
`delete_ctx` is NULL an internal stack context is used, which is only safe
|
||||
when the session is committed before returning (the default behaviour). */
|
||||
int receiver_process_pending_ctx(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest,
|
||||
DeletePlanSession** pending_plans,
|
||||
ReceiverDeleteContext* delete_ctx);
|
||||
int receiver_receive_files(Config* config, int file_descriptor);
|
||||
|
||||
/* ---- Connection time bounds (anti-slowloris) ----
|
||||
|
||||
@@ -27,6 +27,14 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue*
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->deferred_manifest = NULL;
|
||||
context->deferred_plans = NULL;
|
||||
context->delete_ctx.stats = NULL;
|
||||
context->delete_ctx.deleted_paths = NULL;
|
||||
context->delete_limit_reached = false;
|
||||
context->failed_entries = 0;
|
||||
memset(&context->stats, 0, sizeof(context->stats));
|
||||
context->would_delete = NULL;
|
||||
context->deleted_paths = NULL;
|
||||
atomic_init(&context->cancelled, false);
|
||||
int init = 0;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
|
||||
@@ -39,6 +47,18 @@ PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue*
|
||||
goto fail;
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
context->would_delete = array_list_create(free);
|
||||
if (!context->would_delete)
|
||||
goto fail;
|
||||
/* The actually-removed path list is only needed to render rsync's
|
||||
`deleting PATH` lines, which the client requests via report_deletes
|
||||
(--info=del / -i / --out-format under --delete). A plain --delete run must
|
||||
not allocate it or observe every removal. */
|
||||
if (config->report_deletes) {
|
||||
context->deleted_paths = array_list_create(free);
|
||||
if (!context->deleted_paths)
|
||||
goto fail;
|
||||
}
|
||||
return context;
|
||||
|
||||
fail:
|
||||
@@ -49,6 +69,12 @@ fail:
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&context->mutex);
|
||||
/* Free every list that was already created before the failing allocation:
|
||||
`context` itself is freed below, so they would otherwise leak. */
|
||||
if (context->would_delete)
|
||||
array_list_delete(context->would_delete);
|
||||
if (context->deleted_paths)
|
||||
array_list_delete(context->deleted_paths);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
@@ -57,9 +83,15 @@ void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
|
||||
config_delete(context->config);
|
||||
if (context->deferred_manifest)
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
if (context->deferred_plans)
|
||||
delete_plan_session_destroy(context->deferred_plans);
|
||||
queue_destroy(context->queue);
|
||||
receiver_outcomes_destroy(&context->outcomes);
|
||||
dir_time_list_free(&context->dir_times);
|
||||
if (context->would_delete)
|
||||
array_list_delete(context->would_delete);
|
||||
if (context->deleted_paths)
|
||||
array_list_delete(context->deleted_paths);
|
||||
mtx_destroy(&context->mutex);
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
@@ -132,9 +164,23 @@ bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, Fi
|
||||
|
||||
static bool receiver_enqueue_file(File* file, void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
if (file && file->matched_bytes > 0) {
|
||||
mtx_lock(&context->mutex);
|
||||
context->stats.matched_data += file->matched_bytes;
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
return pipeline_context_receiver_enqueue_file(context, file);
|
||||
}
|
||||
|
||||
/* Early delete modes (--delete-before/--delete-during) commit the manifest
|
||||
inside receiver_process_pending on this thread; record a capped commit so
|
||||
server.c's terminal frame can report STATUS_DELETE_LIMIT. The plain bool is
|
||||
safe: receive_thread writes it before the main thread joins the thread. */
|
||||
static void receiver_pipeline_note_delete_limit(void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static void receiver_thread_fail(PipelineContextReceiver* context) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
@@ -152,9 +198,18 @@ int receive_thread(void* pipeline_context) {
|
||||
const Config* config = context->config;
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL};
|
||||
if (receiver_process_pending((Config*)config, file_descriptor, &sink,
|
||||
&context->deferred_manifest) != 0) {
|
||||
ReceiverSink sink = {receiver_enqueue_file,
|
||||
context,
|
||||
false,
|
||||
false,
|
||||
NULL,
|
||||
receiver_pipeline_note_delete_limit,
|
||||
&context->stats,
|
||||
context->would_delete,
|
||||
context->deleted_paths};
|
||||
if (receiver_process_pending_ctx((Config*)config, file_descriptor, &sink,
|
||||
&context->deferred_manifest, &context->deferred_plans,
|
||||
&context->delete_ctx) != 0) {
|
||||
receiver_thread_fail(context);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
@@ -196,8 +251,30 @@ int write_thread(void* pipeline_context) {
|
||||
}
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
FileSaveResult result = FILE_SAVE_SKIPPED;
|
||||
if (save_to_disk) {
|
||||
result = file_save_to_disk_full(root_directory, file, context->config);
|
||||
bool created = false;
|
||||
unsigned created_dirs = 0;
|
||||
/* Server-contacting --dry-run: never write. The receiver thread does not
|
||||
enqueue anything on the dry-run path, but this keeps the writer thread
|
||||
provably mutation-free if a data frame ever reached it. */
|
||||
bool dry_run = context->config->dry_run;
|
||||
if (save_to_disk && !dry_run) {
|
||||
result =
|
||||
file_save_to_disk_full_ex(root_directory, file, context->config, &created, &created_dirs);
|
||||
if (result == FILE_SAVE_WRITTEN) {
|
||||
/* Protocol 2.28.0: fold the receiver-observed literal bytes and the
|
||||
created-entry type into the shared stats block under its mutex (the
|
||||
receive thread also writes stats.matched_data). */
|
||||
mtx_lock(&context->mutex);
|
||||
receiver_stats_note_saved(&context->stats, file, created, created_dirs);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
/* --devices parity: a device node that could not be mknod'ed is counted
|
||||
per-run but does NOT abort the transfer. */
|
||||
if (result == FILE_SAVE_FAILED) {
|
||||
mtx_lock(&context->mutex);
|
||||
context->failed_entries++;
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
@@ -215,9 +292,9 @@ int write_thread(void* pipeline_context) {
|
||||
/* P7 Wave D: a directory's times are never applied inline (a later child
|
||||
write would clobber them); accumulate the metadata here and let the
|
||||
caller apply it once every writer has drained. */
|
||||
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_times_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
if (!dry_run && result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
@@ -234,8 +311,8 @@ int write_thread(void* pipeline_context) {
|
||||
which sources were actually written versus skipped on the receiver.
|
||||
Explicit directory entries and recreated device/special nodes have no
|
||||
source and are never acknowledged (mirrors receiver.c). */
|
||||
if (context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip &&
|
||||
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
if (!dry_run && context->config->remove_source_files && !file->is_dir && !file->is_special &&
|
||||
!file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
|
||||
@@ -41,10 +41,39 @@ typedef struct PipelineContextReceiver {
|
||||
transfer truly succeeded. NULL in the early delete modes (which delete at
|
||||
the manifest). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* Per-directory delete session for --delete-delay: receive_thread snapshots
|
||||
each plan's extras as it arrives and hands the session here instead of
|
||||
committing while the disk writer may still be draining; server.c commits it
|
||||
after both threads joined. NULL for every other timing. */
|
||||
DeletePlanSession* deferred_plans;
|
||||
/* Observer context for `deferred_plans`. It must outlive the receive thread
|
||||
(the session is committed by server.c after both threads join), so it lives
|
||||
here rather than on receiver_process_pending()'s stack; receive_thread
|
||||
installs it on the session. */
|
||||
ReceiverDeleteContext delete_ctx;
|
||||
/* Set by server.c when the deferred delete commit hit the --max-delete
|
||||
budget; the terminal success frame then carries STATUS_DELETE_LIMIT
|
||||
(rsync exit 25) while the transfer itself still succeeds. */
|
||||
bool delete_limit_reached;
|
||||
/* P7 Wave D: directory metadata collected by write_thread from received
|
||||
directory entries. Only write_thread mutates it (before it joins); the
|
||||
caller (server.c) applies it after the delete/delay-updates phase. */
|
||||
DirTimeList dir_times;
|
||||
/* End-of-transfer wire counters (protocol 2.25.0). receive_thread accumulates
|
||||
matched_data under `mutex`; server.c adds the delete-commit tallies after
|
||||
both threads join and emits the STATUS_STATS frame. */
|
||||
ReceiverStats stats;
|
||||
/* -n/--dry-run --delete would-delete path list, collected by receive_thread
|
||||
and reported in the STATUS_STATS frame. */
|
||||
struct ArrayList* would_delete;
|
||||
/* --info=del actually-removed path list, collected by the deferred delete
|
||||
commit in server.c and reported in the STATUS_STATS frame. */
|
||||
struct ArrayList* deleted_paths;
|
||||
/* Per-run count of entries that failed to materialize without aborting the
|
||||
stream (currently ONLY a --devices mknod EPERM/EACCES). write_thread
|
||||
increments it under `mutex`; server.c turns a nonzero count into a non-OK
|
||||
terminal status so the client exits non-zero. */
|
||||
size_t failed_entries;
|
||||
} PipelineContextReceiver;
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
|
||||
|
||||
+835
-437
File diff suppressed because it is too large.
Load diff
+38
-16
@@ -3,20 +3,12 @@
|
||||
#include "credentials.h"
|
||||
#include "utils.h"
|
||||
#include <limits.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/socket.h>
|
||||
|
||||
static void set_error(char* err, size_t err_size, const char* fmt, ...) {
|
||||
if (!err || err_size == 0)
|
||||
return;
|
||||
va_list args;
|
||||
va_start(args, fmt);
|
||||
vsnprintf(err, err_size, fmt, args);
|
||||
va_end(args);
|
||||
}
|
||||
#define set_error utils_set_error
|
||||
|
||||
void server_cli_options_default(ServerCliOptions* opts) {
|
||||
if (!opts)
|
||||
@@ -179,6 +171,8 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
opts->trust_sender = true;
|
||||
} else if (arg_is(argv[i], "--no-super")) {
|
||||
opts->no_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-super")) {
|
||||
opts->allow_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-unauthenticated")) {
|
||||
opts->allow_unauthenticated = true;
|
||||
} else if (arg_has_value(argv[i], "--iconv", &inline_value)) {
|
||||
@@ -190,14 +184,20 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->iconv_spec = inline_value;
|
||||
} else if (arg_is(argv[i], "-p")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for -p");
|
||||
return -1;
|
||||
} else if (arg_is(argv[i], "-p") || arg_has_value(argv[i], "--port", &inline_value)) {
|
||||
if (inline_value) {
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(inline_value, &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
} else {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for %s", argv[i]);
|
||||
return -1;
|
||||
}
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
}
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
} else {
|
||||
if (arg_has_value(argv[i], "--config", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
@@ -259,6 +259,28 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
set_error(err, err_size, "--hash-credentials cannot be combined with --daemon or --stdio");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->no_super) {
|
||||
set_error(err, err_size, "--allow-super and --no-super are mutually exclusive");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->daemon_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is for a locally-launched standalone TCP server; daemon modules opt "
|
||||
"in per module with 'client owner = yes'");
|
||||
return -1;
|
||||
}
|
||||
/* --stdio is the SSH transport: the remote server argv is composed by the
|
||||
* CLIENT (directly and via --remote-option), so a client could otherwise pass
|
||||
* --allow-super to a root --stdio receiver and defeat the C3 secure default.
|
||||
* Never honor it there; the super mode stays forced OFF. An operator who
|
||||
* must keep the historical permissive behavior over SSH has to launch the
|
||||
* receiver through a forced command, not via client-composed argv. */
|
||||
if (opts->allow_super && opts->stdio_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is not accepted with --stdio (the remote argv is client-composed; "
|
||||
"use a forced command if the default must hold)");
|
||||
return -1;
|
||||
}
|
||||
if (opts->hash_iterations_set && opts->hash_credentials_file == NULL) {
|
||||
set_error(err, err_size, "--iterations requires --hash-credentials");
|
||||
return -1;
|
||||
|
||||
@@ -45,6 +45,17 @@ typedef struct ServerCliOptions {
|
||||
* device-node creation) even when running as root. Applies to --stdio and
|
||||
* --daemon alike; also makes the server refuse any client --copy-as. */
|
||||
bool no_super; /* --no-super */
|
||||
/* --allow-super: locally-launched standalone TCP listener opt-in that keeps
|
||||
* the historical permissive behavior for a PRIVILEGED (root) receiver.
|
||||
* Without it a root standalone server forces SUPER_MODE_OFF, so a client
|
||||
* --devices / --write-devices / --super / ownership request cannot make it
|
||||
* create device nodes, write raw devices, or apply client-chosen ownership.
|
||||
* It is rejected for --stdio: that path's remote argv is composed by the
|
||||
* client (directly and via --remote-option), so it must never opt a root
|
||||
* receiver back into super mode. Non-root receivers are unaffected (the
|
||||
* kernel refuses the confined attempts). The daemon path instead uses the
|
||||
* per-module `client owner = yes` opt-in. */
|
||||
bool allow_super; /* --allow-super */
|
||||
/* --iconv=CONVERT_SPEC: the server's own LOCAL charset declaration. The
|
||||
* client's full spec rides the wire config frame anyway; when the server is
|
||||
* started with its own --iconv, its LOCAL half overrides the local charset
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
ArrayList* array_list_create(void (*item_destroyer)(void* item)) {
|
||||
ArrayList* list = (ArrayList*)protocol_alloc(sizeof(ArrayList));
|
||||
if (list == NULL) {
|
||||
log_perror("ERROR: Could not allocate memory for array list struct");
|
||||
log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not allocate memory for array list struct");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
@@ -47,7 +47,7 @@ static bool array_list_extend(ArrayList* array_list) {
|
||||
new_capacity = INITIAL_ARRAY_SIZE;
|
||||
void* new_items = protocol_realloc(array_list->items, new_capacity * sizeof(void*));
|
||||
if (new_items == NULL) {
|
||||
log_perror("ERROR: Could not reallocate memory for array list items");
|
||||
log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not reallocate memory for array list items");
|
||||
return false;
|
||||
}
|
||||
array_list->items = new_items;
|
||||
@@ -73,7 +73,7 @@ void** array_list_to_array(const ArrayList* array_list) {
|
||||
}
|
||||
void** array = protocol_alloc(array_list->size * sizeof(void*));
|
||||
if (array == NULL) {
|
||||
log_perror("Could not malloc space for array from array list!");
|
||||
log_message(LOG_LEVEL_ERROR, "%s", "Could not malloc space for array from array list!");
|
||||
return NULL;
|
||||
}
|
||||
memcpy(array, array_list->items, array_list->size * sizeof(void*));
|
||||
|
||||
+54
-19
@@ -2,6 +2,7 @@
|
||||
#include "data.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include <errno.h>
|
||||
#include <stdlib.h>
|
||||
@@ -10,11 +11,15 @@
|
||||
|
||||
/* Serialization metadata mode for the batch stream, captured from the config at
|
||||
* batch_write_header time. The header persists it into the file so a batch is
|
||||
* self-describing: batch_read_apply re-reads it from the file (not from the
|
||||
* reading config), so a batch written with -M is applied identically by an
|
||||
* invoking process regardless of its own -M setting. The batch driver is a
|
||||
* single sequential scan pass within one thread, so this module-level flag is
|
||||
* safe. */
|
||||
* self-describing about whether per-entry metadata was CAPTURED in the stream:
|
||||
* batch_read_apply re-reads it from the file (not from the reading config) to
|
||||
* decode the chunk records correctly. Which attributes are actually APPLIED,
|
||||
* however, comes from the INVOKING process's per-attribute config (the
|
||||
* FileAttrPolicy and the dir-metadata gate), so a batch written with -M is NOT
|
||||
* automatically applied identically by an invoking process with a different
|
||||
* -p/-t/-o/-g: --read-batch must be invoked with the same -p/-t/-o/-g as the
|
||||
* write side (rsync requires the same options). The batch driver is a single
|
||||
* sequential scan pass within one thread, so this module-level flag is safe. */
|
||||
static bool batch_metadata_mode = false;
|
||||
|
||||
static bool write_all_bytes(int fd, const void* data, size_t size) {
|
||||
@@ -91,22 +96,37 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
if (fd < 0 || dest_root == NULL || dest_root[0] == '\0')
|
||||
return -1;
|
||||
|
||||
/* Directory metadata is deferred to the end of the apply (a child write would
|
||||
* otherwise clobber its parent's mtime/mode). The batch header's single
|
||||
* metadata bit only says whether metadata is present in the stream; which
|
||||
* attributes are APPLIED comes from the invoking process's config, so
|
||||
* --read-batch must be invoked with the same -p/-t/-o/-g as the write side
|
||||
* (rsync requires the same options). The identity snapshot is activated so
|
||||
* -o/-g and the explicit ownership flags can apply. */
|
||||
DirTimeList dir_times;
|
||||
dir_time_list_init(&dir_times);
|
||||
int result = -1;
|
||||
if (!identity_set_active(config)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not activate the identity policy");
|
||||
goto done;
|
||||
}
|
||||
|
||||
char magic[BATCH_MAGIC_LEN];
|
||||
bool eof = false;
|
||||
if (!read_exact(fd, magic, BATCH_MAGIC_LEN, &eof) || eof ||
|
||||
memcmp(magic, BATCH_MAGIC, BATCH_MAGIC_LEN) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad magic)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char version;
|
||||
if (!read_exact(fd, &version, 1, &eof) || eof || version != BATCH_FORMAT_VERSION) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad or missing format version)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char mode;
|
||||
if (!read_exact(fd, &mode, 1, &eof) || eof || (mode != 0 && mode != 1)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad metadata flag)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
bool use_metadata = mode == 1;
|
||||
|
||||
@@ -114,47 +134,62 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
unsigned long long length;
|
||||
if (!read_exact(fd, &length, sizeof(length), &eof)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated length prefix");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (eof)
|
||||
break; /* clean end of stream */
|
||||
if (length == 0 || length > BATCH_MAX_RECORD) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected record length %llu (valid range 1..%llu)",
|
||||
length, (unsigned long long)BATCH_MAX_RECORD);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
char* record = (char*)malloc((size_t)length);
|
||||
if (record == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not allocate a %llu-byte record", length);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (!read_exact(fd, record, (size_t)length, &eof) || eof) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated chunk record");
|
||||
free(record);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
Data* data = data_create(record, (size_t)length);
|
||||
if (data == NULL)
|
||||
return -1; /* data_create frees `record` on failure */
|
||||
goto done; /* data_create frees `record` on failure */
|
||||
Chunk* chunk = chunk_deserialize(data, use_metadata);
|
||||
data_destroy(data);
|
||||
if (chunk == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected malformed chunk record");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* file = chunk->items[i];
|
||||
chunk->items[i] = NULL;
|
||||
if (file == NULL)
|
||||
continue;
|
||||
FileSaveResult result = file_save_to_disk_full(dest_root, file, config);
|
||||
file_destroy(file);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
FileSaveResult save = file_save_to_disk_full(dest_root, file, config);
|
||||
/* Accumulate directory metadata (when it applies) before the File is
|
||||
* destroyed; applied once the whole stream has been consumed. */
|
||||
if (save != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(config) &&
|
||||
!dir_time_list_add(&dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
chunk_destroy(chunk);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
file_destroy(file);
|
||||
if (save == FILE_SAVE_ERROR) {
|
||||
chunk_destroy(chunk);
|
||||
goto done;
|
||||
}
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
return 0;
|
||||
dir_metadata_list_apply(&dir_times, dest_root, config);
|
||||
result = 0;
|
||||
|
||||
done:
|
||||
identity_clear_active();
|
||||
dir_time_list_free(&dir_times);
|
||||
return result;
|
||||
}
|
||||
+15
-10
@@ -168,9 +168,14 @@ bool charset_spec_valid_direction(const char* from_charset, const char* to_chars
|
||||
return direction_probe_valid(from_charset, to_charset);
|
||||
}
|
||||
|
||||
/* The receiver's real conversion is wire(client REMOTE) -> server-local (the
|
||||
* server's own --iconv LOCAL half, or the client's LOCAL half when the server
|
||||
* has no --iconv). A dedicated pre-ack check so an impossible direction is
|
||||
/* The receiver's conversion is wire charset -> destination charset. rsync's
|
||||
* CONVERT_SPEC is LOCAL,REMOTE and "stays the same whether you're pushing or
|
||||
* pulling", so for a PUSH (FastSync's only direction) the destination end's
|
||||
* charset is the spec's REMOTE half: the client converts LOCAL -> REMOTE on the
|
||||
* sender and the receiver writes the wire bytes verbatim. Only a server that
|
||||
* declares its OWN --iconv (the daemon "charset" analog) has a different local
|
||||
* charset, and then it is that spec's LOCAL half and the receiver converts
|
||||
* wire -> server-local. A dedicated pre-ack check so an impossible direction is
|
||||
* rejected before the connection instead of refusing mid-transfer. */
|
||||
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec) {
|
||||
if (!spec)
|
||||
@@ -180,7 +185,7 @@ bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec)
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
const char* wire = remote;
|
||||
const char* target_local = local;
|
||||
const char* target_local = remote;
|
||||
char* server_local = NULL;
|
||||
char* server_remote = NULL;
|
||||
if (server_spec) {
|
||||
@@ -302,13 +307,13 @@ bool charset_wire_init_receiver(const char* spec, const char* server_spec) {
|
||||
char* remote;
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
/* The wire charset is the client spec's REMOTE half; the local charset is
|
||||
* the client spec's LOCAL half unless the server was itself started with
|
||||
* --iconv naming a different local charset (the server halves above never
|
||||
* travel, so the server's own flag is the only way its local charset can
|
||||
* differ from what the client assumed). */
|
||||
/* The wire charset is the client spec's REMOTE half (rsync's LOCAL,REMOTE
|
||||
* spec stays the same push or pull, so on a push the destination end's
|
||||
* charset is REMOTE and the receiver writes the wire bytes verbatim). Only a
|
||||
* server started with its own --iconv declares a different local charset (the
|
||||
* server halves above never travel), and then it is that spec's LOCAL half. */
|
||||
const char* wire = remote;
|
||||
const char* target_local = local;
|
||||
const char* target_local = remote;
|
||||
char* server_local = NULL;
|
||||
char* server_remote = NULL;
|
||||
if (server_spec) {
|
||||
|
||||
@@ -57,8 +57,9 @@ void charset_conversion_close(void* conversion);
|
||||
/* Process-wide wire conversion. charset_wire_init_sender (client side) opens
|
||||
* LOCAL->REMOTE; charset_wire_init_receiver (server side) opens
|
||||
* wire(REMOTE)->server-local. server_spec is the server's own --iconv, whose
|
||||
* LOCAL half may override the local charset the client assumed; NULL reuses
|
||||
* the client spec's LOCAL half. Both return false on an unsupported spec.
|
||||
* LOCAL half overrides the destination charset; NULL means the destination
|
||||
* charset is the client spec's REMOTE half (rsync's push semantics: the wire
|
||||
* bytes are written verbatim). Both return false on an unsupported spec.
|
||||
* The state is freed with charset_wire_free. */
|
||||
bool charset_wire_init_sender(const char* spec);
|
||||
bool charset_wire_init_receiver(const char* spec, const char* server_spec);
|
||||
@@ -66,9 +67,9 @@ void charset_wire_free(void);
|
||||
bool charset_wire_active(void);
|
||||
|
||||
/* Pre-ack receiver-direction sanity (see charset_wire_init_receiver): true
|
||||
* when the exact wire->server-local conversion the receiver will use (client
|
||||
* spec's REMOTE half into the server's own LOCAL half, or the client's LOCAL
|
||||
* half when the server has no --iconv) opens and produces NUL-free output. */
|
||||
* when the exact wire->destination conversion the receiver will use (client
|
||||
* spec's REMOTE half into the server's own LOCAL half, or REMOTE->REMOTE when
|
||||
* the server has no --iconv) opens and produces NUL-free output. */
|
||||
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec);
|
||||
|
||||
/* Convert a path across the wire in the process direction. Returns a malloc'd
|
||||
|
||||
+369
-17
@@ -1,12 +1,171 @@
|
||||
#include "checksum.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <openssl/evp.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* delta.c owns the single XXH_IMPLEMENTATION that provides the xxHash symbols
|
||||
* for the whole binary; this TU only needs the declarations. */
|
||||
* for the whole binary; this TU only needs the declarations. The streaming
|
||||
* state structs and XXH3_update are exposed only with XXH_STATIC_LINKING_ONLY. */
|
||||
#define XXH_STATIC_LINKING_ONLY
|
||||
#include <xxhash.h>
|
||||
|
||||
/* ---------------------------------------------------------------------------
|
||||
* Self-contained MD4 (RFC 1320). OpenSSL's MD4 lives in the legacy provider
|
||||
* and is not guaranteed present, so FastSync carries its own implementation to
|
||||
* keep --checksum-choice=md4 working on every build.
|
||||
* ------------------------------------------------------------------------- */
|
||||
|
||||
typedef struct {
|
||||
uint32_t state[4];
|
||||
uint64_t bit_count;
|
||||
uint8_t buffer[64];
|
||||
size_t buffer_len;
|
||||
} Md4Ctx;
|
||||
|
||||
static uint32_t md4_rotl(uint32_t x, int n) {
|
||||
return (x << n) | (x >> (32 - n));
|
||||
}
|
||||
|
||||
static void md4_transform(uint32_t state[4], const uint8_t block[64]) {
|
||||
uint32_t x[16];
|
||||
for (int i = 0; i < 16; i++)
|
||||
x[i] = (uint32_t)block[i * 4] | ((uint32_t)block[i * 4 + 1] << 8) |
|
||||
((uint32_t)block[i * 4 + 2] << 16) | ((uint32_t)block[i * 4 + 3] << 24);
|
||||
|
||||
uint32_t a = state[0], b = state[1], c = state[2], d = state[3];
|
||||
|
||||
#define F(x, y, z) (((x) & (y)) | (~(x) & (z)))
|
||||
#define G(x, y, z) (((x) & (y)) | ((x) & (z)) | ((y) & (z)))
|
||||
#define H(x, y, z) ((x) ^ (y) ^ (z))
|
||||
#define ROUND1(a, b, c, d, k, s) a = md4_rotl(a + F(b, c, d) + x[k], s)
|
||||
#define ROUND2(a, b, c, d, k, s) a = md4_rotl(a + G(b, c, d) + x[k] + 0x5a827999u, s)
|
||||
#define ROUND3(a, b, c, d, k, s) a = md4_rotl(a + H(b, c, d) + x[k] + 0x6ed9eba1u, s)
|
||||
|
||||
ROUND1(a, b, c, d, 0, 3);
|
||||
ROUND1(d, a, b, c, 1, 7);
|
||||
ROUND1(c, d, a, b, 2, 11);
|
||||
ROUND1(b, c, d, a, 3, 19);
|
||||
ROUND1(a, b, c, d, 4, 3);
|
||||
ROUND1(d, a, b, c, 5, 7);
|
||||
ROUND1(c, d, a, b, 6, 11);
|
||||
ROUND1(b, c, d, a, 7, 19);
|
||||
ROUND1(a, b, c, d, 8, 3);
|
||||
ROUND1(d, a, b, c, 9, 7);
|
||||
ROUND1(c, d, a, b, 10, 11);
|
||||
ROUND1(b, c, d, a, 11, 19);
|
||||
ROUND1(a, b, c, d, 12, 3);
|
||||
ROUND1(d, a, b, c, 13, 7);
|
||||
ROUND1(c, d, a, b, 14, 11);
|
||||
ROUND1(b, c, d, a, 15, 19);
|
||||
|
||||
ROUND2(a, b, c, d, 0, 3);
|
||||
ROUND2(d, a, b, c, 4, 5);
|
||||
ROUND2(c, d, a, b, 8, 9);
|
||||
ROUND2(b, c, d, a, 12, 13);
|
||||
ROUND2(a, b, c, d, 1, 3);
|
||||
ROUND2(d, a, b, c, 5, 5);
|
||||
ROUND2(c, d, a, b, 9, 9);
|
||||
ROUND2(b, c, d, a, 13, 13);
|
||||
ROUND2(a, b, c, d, 2, 3);
|
||||
ROUND2(d, a, b, c, 6, 5);
|
||||
ROUND2(c, d, a, b, 10, 9);
|
||||
ROUND2(b, c, d, a, 14, 13);
|
||||
ROUND2(a, b, c, d, 3, 3);
|
||||
ROUND2(d, a, b, c, 7, 5);
|
||||
ROUND2(c, d, a, b, 11, 9);
|
||||
ROUND2(b, c, d, a, 15, 13);
|
||||
|
||||
ROUND3(a, b, c, d, 0, 3);
|
||||
ROUND3(d, a, b, c, 8, 9);
|
||||
ROUND3(c, d, a, b, 4, 11);
|
||||
ROUND3(b, c, d, a, 12, 15);
|
||||
ROUND3(a, b, c, d, 2, 3);
|
||||
ROUND3(d, a, b, c, 10, 9);
|
||||
ROUND3(c, d, a, b, 6, 11);
|
||||
ROUND3(b, c, d, a, 14, 15);
|
||||
ROUND3(a, b, c, d, 1, 3);
|
||||
ROUND3(d, a, b, c, 9, 9);
|
||||
ROUND3(c, d, a, b, 5, 11);
|
||||
ROUND3(b, c, d, a, 13, 15);
|
||||
ROUND3(a, b, c, d, 3, 3);
|
||||
ROUND3(d, a, b, c, 11, 9);
|
||||
ROUND3(c, d, a, b, 7, 11);
|
||||
ROUND3(b, c, d, a, 15, 15);
|
||||
|
||||
#undef F
|
||||
#undef G
|
||||
#undef H
|
||||
#undef ROUND1
|
||||
#undef ROUND2
|
||||
#undef ROUND3
|
||||
|
||||
state[0] += a;
|
||||
state[1] += b;
|
||||
state[2] += c;
|
||||
state[3] += d;
|
||||
}
|
||||
|
||||
static void md4_init(Md4Ctx* ctx) {
|
||||
ctx->state[0] = 0x67452301u;
|
||||
ctx->state[1] = 0xefcdab89u;
|
||||
ctx->state[2] = 0x98badcfeu;
|
||||
ctx->state[3] = 0x10325476u;
|
||||
ctx->bit_count = 0;
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
|
||||
static void md4_update(Md4Ctx* ctx, const uint8_t* data, size_t len) {
|
||||
ctx->bit_count += (uint64_t)len * 8;
|
||||
while (len > 0) {
|
||||
size_t space = sizeof(ctx->buffer) - ctx->buffer_len;
|
||||
size_t take = len < space ? len : space;
|
||||
memcpy(ctx->buffer + ctx->buffer_len, data, take);
|
||||
ctx->buffer_len += take;
|
||||
data += take;
|
||||
len -= take;
|
||||
if (ctx->buffer_len == sizeof(ctx->buffer)) {
|
||||
md4_transform(ctx->state, ctx->buffer);
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void md4_final(Md4Ctx* ctx, uint8_t out[16]) {
|
||||
uint64_t bit_count = ctx->bit_count;
|
||||
uint8_t pad = 0x80;
|
||||
md4_update(ctx, &pad, 1);
|
||||
uint8_t zero = 0;
|
||||
while (ctx->buffer_len != 56)
|
||||
md4_update(ctx, &zero, 1);
|
||||
uint8_t length_le[8];
|
||||
for (int i = 0; i < 8; i++)
|
||||
length_le[i] = (uint8_t)((bit_count >> (8 * i)) & 0xff);
|
||||
md4_update(ctx, length_le, sizeof(length_le));
|
||||
for (int i = 0; i < 4; i++) {
|
||||
out[i * 4] = (uint8_t)(ctx->state[i] & 0xff);
|
||||
out[i * 4 + 1] = (uint8_t)((ctx->state[i] >> 8) & 0xff);
|
||||
out[i * 4 + 2] = (uint8_t)((ctx->state[i] >> 16) & 0xff);
|
||||
out[i * 4 + 3] = (uint8_t)((ctx->state[i] >> 24) & 0xff);
|
||||
}
|
||||
}
|
||||
|
||||
/* One-shot EVP digest (md5/sha1). Returns false when OpenSSL refuses. */
|
||||
static bool evp_digest(const EVP_MD* md, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
static const uint8_t empty = 0;
|
||||
const void* input = data ? data : ∅
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_Digest(input, size, out, &digest_len, md, NULL) != 1)
|
||||
return false;
|
||||
if (digest_len > out_capacity)
|
||||
return false;
|
||||
*out_len = digest_len;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
@@ -14,30 +173,166 @@ bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t
|
||||
if (data == NULL && size != 0)
|
||||
return false;
|
||||
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64: {
|
||||
uint64_t digest = XXH64(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5) {
|
||||
case CHECKSUM_ALGO_XXH3: {
|
||||
uint64_t digest = XXH3_64bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_XXH128: {
|
||||
XXH128_hash_t digest = XXH3_128bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
/* md5 takes no seed; the caller's seed is deliberately ignored (documented
|
||||
* in RSYNC_COMPAT.md). OpenSSL's one-shot EVP_Digest needs a non-NULL
|
||||
* buffer even for an empty input, so map a NULL data + size==0 to an empty
|
||||
* buffer. */
|
||||
static const uint8_t empty = 0;
|
||||
const void* input = data ? data : ∅
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_Digest(input, size, out, &digest_len, EVP_md5(), NULL) != 1)
|
||||
return false;
|
||||
if (digest_len > out_capacity)
|
||||
return false;
|
||||
*out_len = digest_len;
|
||||
* in RSYNC_COMPAT.md). */
|
||||
return evp_digest(EVP_md5(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_MD4: {
|
||||
Md4Ctx ctx;
|
||||
md4_init(&ctx);
|
||||
md4_update(&ctx, (const uint8_t*)data, size);
|
||||
md4_final(&ctx, out);
|
||||
*out_len = 16;
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
/* sha1 takes no seed; the caller's seed is deliberately ignored. */
|
||||
return evp_digest(EVP_sha1(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
/* No checksum requested: an empty digest is the successful result. */
|
||||
*out_len = 0;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!path || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
|
||||
int fd = open(path, O_RDONLY | O_CLOEXEC);
|
||||
if (fd < 0)
|
||||
return false;
|
||||
|
||||
bool ok = checksum_digest_fd(algo, seed, fd, out, out_capacity, out_len);
|
||||
close(fd);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len) {
|
||||
if (fd < 0 || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
|
||||
if (algo == CHECKSUM_ALGO_NONE) {
|
||||
/* No checksum requested: nothing to read; an empty digest succeeds. */
|
||||
*out_len = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
uint8_t buffer[64 * 1024];
|
||||
bool ok = false;
|
||||
lseek(fd, 0, SEEK_SET);
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5 || algo == CHECKSUM_ALGO_SHA1) {
|
||||
const EVP_MD* md = algo == CHECKSUM_ALGO_MD5 ? EVP_md5() : EVP_sha1();
|
||||
EVP_MD_CTX* ctx = EVP_MD_CTX_new();
|
||||
if (!ctx)
|
||||
return false;
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_DigestInit_ex(ctx, md, NULL) == 1) {
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (EVP_DigestUpdate(ctx, buffer, (size_t)got) != 1) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
if (ok && EVP_DigestFinal_ex(ctx, out, &digest_len) == 1 && digest_len <= out_capacity)
|
||||
*out_len = digest_len;
|
||||
else
|
||||
ok = false;
|
||||
}
|
||||
EVP_MD_CTX_free(ctx);
|
||||
return ok;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD4) {
|
||||
Md4Ctx ctx;
|
||||
md4_init(&ctx);
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0)
|
||||
md4_update(&ctx, buffer, (size_t)got);
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
if (ok) {
|
||||
md4_final(&ctx, out);
|
||||
*out_len = 16;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
XXH64_state_t xxh64;
|
||||
XXH3_state_t* xxh3 = NULL;
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
XXH64_reset(&xxh64, seed);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3 || algo == CHECKSUM_ALGO_XXH128) {
|
||||
xxh3 = XXH3_createState();
|
||||
if (!xxh3)
|
||||
return false;
|
||||
if (algo == CHECKSUM_ALGO_XXH3)
|
||||
XXH3_64bits_reset_withSeed(xxh3, seed);
|
||||
else
|
||||
XXH3_128bits_reset_withSeed(xxh3, seed);
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64)
|
||||
XXH64_update(&xxh64, buffer, (size_t)got);
|
||||
else if (XXH3_64bits_update(xxh3, buffer, (size_t)got) == XXH_ERROR) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
|
||||
if (ok) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
uint64_t digest = XXH64_digest(&xxh64);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3) {
|
||||
uint64_t digest = XXH3_64bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else {
|
||||
XXH128_hash_t digest = XXH3_128bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
}
|
||||
}
|
||||
if (xxh3)
|
||||
XXH3_freeState(xxh3);
|
||||
return ok;
|
||||
}
|
||||
|
||||
int checksum_algo_from_name(const char* name) {
|
||||
@@ -45,8 +340,18 @@ int checksum_algo_from_name(const char* name) {
|
||||
return -1;
|
||||
if (strcasecmp(name, "xxh64") == 0 || strcasecmp(name, "xxhash") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH64;
|
||||
if (strcasecmp(name, "xxh3") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH3;
|
||||
if (strcasecmp(name, "xxh128") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH128;
|
||||
if (strcasecmp(name, "md5") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD5;
|
||||
if (strcasecmp(name, "md4") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD4;
|
||||
if (strcasecmp(name, "sha1") == 0)
|
||||
return (int)CHECKSUM_ALGO_SHA1;
|
||||
if (strcasecmp(name, "none") == 0)
|
||||
return (int)CHECKSUM_ALGO_NONE;
|
||||
return -1;
|
||||
}
|
||||
|
||||
@@ -54,22 +359,69 @@ const char* checksum_algo_name(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
return "xxh64";
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return "xxh3";
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
return "xxh128";
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
return "md5";
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return "md4";
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return "sha1";
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return "none";
|
||||
}
|
||||
return "<unknown>";
|
||||
}
|
||||
|
||||
bool checksum_algo_valid(int algo) {
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5;
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5 ||
|
||||
algo == (int)CHECKSUM_ALGO_XXH3 || algo == (int)CHECKSUM_ALGO_XXH128 ||
|
||||
algo == (int)CHECKSUM_ALGO_MD4 || algo == (int)CHECKSUM_ALGO_SHA1 ||
|
||||
algo == (int)CHECKSUM_ALGO_NONE;
|
||||
}
|
||||
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return 8;
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return 16;
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return 20;
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static ChecksumAlgo compiled_checksum_preference_first(void) {
|
||||
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
|
||||
* resolves to xxh128. */
|
||||
static const ChecksumAlgo preference[] = {
|
||||
CHECKSUM_ALGO_XXH128, CHECKSUM_ALGO_XXH3, CHECKSUM_ALGO_XXH64, CHECKSUM_ALGO_MD5,
|
||||
CHECKSUM_ALGO_MD4, CHECKSUM_ALGO_SHA1, CHECKSUM_ALGO_NONE,
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(preference) / sizeof(preference[0]); i++) {
|
||||
if (checksum_algo_valid((int)preference[i]))
|
||||
return preference[i];
|
||||
}
|
||||
return CHECKSUM_ALGO_XXH64;
|
||||
}
|
||||
|
||||
int checksum_choice_resolve(void) {
|
||||
bool specified = false;
|
||||
int env = env_choice_first("RSYNC_CHECKSUM_LIST", checksum_algo_from_name, &specified);
|
||||
if (specified)
|
||||
return env; /* -1 = the list named no supported checksum */
|
||||
return (int)compiled_checksum_preference_first();
|
||||
}
|
||||
|
||||
ChecksumAlgo checksum_negotiate_default(void) {
|
||||
int resolved = checksum_choice_resolve();
|
||||
return resolved >= 0 ? (ChecksumAlgo)resolved : compiled_checksum_preference_first();
|
||||
}
|
||||
+54
-10
@@ -8,18 +8,35 @@
|
||||
/* Whole-file content-digest algorithms selectable with --checksum-choice and
|
||||
* seeded with --checksum-seed. The ids are the values actually placed on the
|
||||
* wire (config frame), so they must be kept stable and validated on receive.
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the default and is byte-for-byte what FastSync
|
||||
* computed before these options existed (xxHash64 with seed 0). */
|
||||
typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the historical FastSync default and its numeric
|
||||
* value is preserved. The full set mirrors the algorithms rsync 3.4.1 can be
|
||||
* built with; every one of them is implemented here. */
|
||||
typedef enum {
|
||||
CHECKSUM_ALGO_XXH64 = 0,
|
||||
CHECKSUM_ALGO_MD5 = 1,
|
||||
CHECKSUM_ALGO_XXH3 = 2,
|
||||
CHECKSUM_ALGO_XXH128 = 3,
|
||||
CHECKSUM_ALGO_MD4 = 4,
|
||||
CHECKSUM_ALGO_SHA1 = 5,
|
||||
CHECKSUM_ALGO_NONE = 6
|
||||
} ChecksumAlgo;
|
||||
|
||||
/* md5 digest is 16 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 16
|
||||
/* FastSync's negotiated default (rsync 3.4.1 auto-negotiates xxh128 first).
|
||||
* The wire default for Config->checksum_algo is this value. */
|
||||
#define CHECKSUM_ALGO_DEFAULT CHECKSUM_ALGO_XXH128
|
||||
|
||||
/* sha1 digest is 20 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 20
|
||||
|
||||
/* Compute the whole-file digest of the first `size` bytes of `data`.
|
||||
*
|
||||
* - CHECKSUM_ALGO_XXH64: xxHash64(data, size, seed) (full 64-bit seed).
|
||||
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP.
|
||||
* md5 has no seed, so `seed` is ignored (documented).
|
||||
* - CHECKSUM_ALGO_XXH3: XXH3_64bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_XXH128: XXH3_128bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_MD4: md4(data, size), self-contained RFC 1320 (seed ignored).
|
||||
* - CHECKSUM_ALGO_SHA1: sha1(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_NONE: no digest; *out_len is 0 and nothing is written.
|
||||
* - `size == 0` hashes the empty input (plus its seed), not a NULL input.
|
||||
*
|
||||
* Writes up to `out_capacity` bytes into `out`, storing the digest length in
|
||||
@@ -28,9 +45,23 @@ typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Streaming whole-file digest: hash the contents of `path` without holding the
|
||||
* whole file in memory. Same digest/capacity contract as checksum_digest.
|
||||
* Returns false on open/read failure or an undersized buffer. */
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Descriptor form of the streaming digest: rewinds `fd` to the start and hashes
|
||||
* to EOF without closing it. Used by the --verify-basis path to hash an
|
||||
* already-open, root-confined basis descriptor. Same contract as
|
||||
* checksum_digest_file. */
|
||||
bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len);
|
||||
|
||||
/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "xxh64" and "xxhash" (both map to CHECKSUM_ALGO_XXH64, rsync's
|
||||
* xxhash spelling) and "md5". Returns -1 for any unsupported name. */
|
||||
* Accepts "xxh64"/"xxhash", "xxh3", "xxh128", "md5", "md4", "sha1", "none".
|
||||
* "auto" is not an algorithm here; the caller resolves it to the negotiated
|
||||
* default. Returns -1 for any unrecognized name. */
|
||||
int checksum_algo_from_name(const char* name);
|
||||
|
||||
/* Canonical name of an algorithm (used in CLI error messages). */
|
||||
@@ -39,7 +70,20 @@ const char* checksum_algo_name(ChecksumAlgo algo);
|
||||
/* True when `algo` is a supported id (used by config receive validation). */
|
||||
bool checksum_algo_valid(int algo);
|
||||
|
||||
/* Digest length in bytes for an algorithm (xxx64 = 8, md5 = 16). */
|
||||
/* Digest length in bytes for an algorithm (xxh64/xxh3 = 8,
|
||||
* md5/md4/xxh128 = 16, sha1 = 20, none = 0). */
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list that is
|
||||
* supported on this build (rsync 3.4.1's `--version` order:
|
||||
* xxh128 xxh3 xxh64 md5 md4 sha1 none). Used to resolve "auto". */
|
||||
ChecksumAlgo checksum_negotiate_default(void);
|
||||
|
||||
/* Resolve "auto" the way rsync does: the first supported name in
|
||||
* RSYNC_CHECKSUM_LIST (whitespace-separated, client half ends at '&'), then the
|
||||
* compiled-in preference order when the variable is unset/blank. Returns -1
|
||||
* when the variable is set but names no supported checksum (rsync's failed
|
||||
* negotiation), otherwise a valid ChecksumAlgo id. */
|
||||
int checksum_choice_resolve(void);
|
||||
|
||||
#endif /* CHECKSUM_H */
|
||||
+170
-77
@@ -1,90 +1,183 @@
|
||||
#include "chmod.h"
|
||||
#include "file.h"
|
||||
#include <stddef.h>
|
||||
#include <string.h>
|
||||
|
||||
static bool parse_clause(mode_t* mode, const char* begin, const char* end) {
|
||||
const char* p = begin;
|
||||
unsigned who = 0;
|
||||
while (p < end && strchr("ugoa", *p)) {
|
||||
if (*p == 'a')
|
||||
who = 7;
|
||||
else
|
||||
who |= *p == 'u' ? 1U : (*p == 'g' ? 2U : 4U);
|
||||
p++;
|
||||
}
|
||||
if (who == 0)
|
||||
who = 7;
|
||||
if (p == end || (*p != '+' && *p != '-' && *p != '='))
|
||||
return false;
|
||||
char operation = *p++;
|
||||
mode_t bits = 0;
|
||||
while (p < end) {
|
||||
mode_t bit;
|
||||
switch (*p++) {
|
||||
case 'r':
|
||||
bit = 4;
|
||||
break;
|
||||
case 'w':
|
||||
bit = 2;
|
||||
break;
|
||||
case 'x':
|
||||
bit = 1;
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
bits |= bit;
|
||||
}
|
||||
for (unsigned class_index = 0; class_index < 3; class_index++) {
|
||||
unsigned class_bit = 1U << class_index;
|
||||
if (!(who & class_bit))
|
||||
continue;
|
||||
mode_t shift = (mode_t)((2U - class_index) * 3U);
|
||||
mode_t mask = (mode_t)(7U << shift);
|
||||
mode_t class_bits = (mode_t)(bits << shift);
|
||||
if (operation == '+')
|
||||
*mode |= class_bits;
|
||||
else if (operation == '-')
|
||||
*mode &= ~class_bits;
|
||||
else
|
||||
*mode = (*mode & ~mask) | class_bits;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
/* rsync's --chmod parser (parse_chmod + tweak_mode). A single clause is
|
||||
* applied as it is completed, so repeated clauses and repeated --chmod options
|
||||
* (joined with commas by the CLI) accumulate exactly like rsync. The D/F
|
||||
* selectors restrict a clause to directories/files; X adds execute only to
|
||||
* directories or files that were already executable. */
|
||||
|
||||
#define CHMOD_BITS 07777
|
||||
#define CHMOD_FLAG_X_KEEP (1U << 0)
|
||||
#define CHMOD_FLAG_DIRS_ONLY (1U << 1)
|
||||
#define CHMOD_FLAG_FILES_ONLY (1U << 2)
|
||||
|
||||
enum chmod_op { CHMOD_OP_ADD = 1, CHMOD_OP_SUB, CHMOD_OP_EQ, CHMOD_OP_SET };
|
||||
enum chmod_state {
|
||||
CHMOD_STATE_ERROR,
|
||||
CHMOD_STATE_1ST_HALF,
|
||||
CHMOD_STATE_2ND_HALF,
|
||||
CHMOD_STATE_OCTAL
|
||||
};
|
||||
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result) {
|
||||
if (!spec || !*spec || !result)
|
||||
return false;
|
||||
bool numeric = true;
|
||||
size_t length = strlen(spec);
|
||||
if (length > 4)
|
||||
numeric = false;
|
||||
for (size_t i = 0; i < length && numeric; i++)
|
||||
numeric = spec[i] >= '0' && spec[i] <= '7';
|
||||
if (numeric) {
|
||||
if (length == 0 || length > 4)
|
||||
return false;
|
||||
mode_t parsed = 0;
|
||||
for (size_t i = 0; i < length; i++)
|
||||
parsed = (mode_t)((parsed << 3) | (spec[i] - '0'));
|
||||
*result = parsed;
|
||||
return true;
|
||||
}
|
||||
|
||||
const mode_t nonperm = mode & ~(mode_t)CHMOD_BITS;
|
||||
const bool initially_executable = (mode & 0111) != 0;
|
||||
mode_t changed = mode;
|
||||
const char* begin = spec;
|
||||
while (*begin) {
|
||||
const char* end = strchr(begin, ',');
|
||||
if (!end)
|
||||
end = begin + strlen(begin);
|
||||
if (!parse_clause(&changed, begin, end))
|
||||
return false;
|
||||
if (*end == '\0')
|
||||
int state = CHMOD_STATE_1ST_HALF;
|
||||
unsigned where = 0;
|
||||
int what = 0, op = 0, topbits = 0, topoct = 0, flags = 0;
|
||||
const char* p = spec;
|
||||
while (state != CHMOD_STATE_ERROR) {
|
||||
if (*p == '\0' || *p == ',') {
|
||||
int bits;
|
||||
if (!op) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
if (where)
|
||||
bits = (int)(where * (unsigned)what);
|
||||
else {
|
||||
where = 0111;
|
||||
bits = (int)((where * (unsigned)what) & ~(unsigned)file_process_umask());
|
||||
}
|
||||
int mode_and, mode_or;
|
||||
switch (op) {
|
||||
case CHMOD_OP_ADD:
|
||||
mode_and = CHMOD_BITS;
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
case CHMOD_OP_SUB:
|
||||
mode_and = CHMOD_BITS - bits - topoct;
|
||||
mode_or = 0;
|
||||
break;
|
||||
case CHMOD_OP_EQ:
|
||||
mode_and = CHMOD_BITS - (int)(where * 7U) - (topoct ? topbits : 0);
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
default:
|
||||
mode_and = 0;
|
||||
mode_or = bits;
|
||||
break;
|
||||
}
|
||||
bool is_dir = S_ISDIR(nonperm);
|
||||
if (!((flags & CHMOD_FLAG_DIRS_ONLY) && !is_dir) &&
|
||||
!((flags & CHMOD_FLAG_FILES_ONLY) && is_dir)) {
|
||||
changed &= (mode_t)mode_and;
|
||||
if ((flags & CHMOD_FLAG_X_KEEP) && !initially_executable && !is_dir)
|
||||
changed |= (mode_t)(mode_or & ~0111);
|
||||
else
|
||||
changed |= (mode_t)mode_or;
|
||||
}
|
||||
if (*p == '\0')
|
||||
break;
|
||||
p++;
|
||||
state = CHMOD_STATE_1ST_HALF;
|
||||
where = 0;
|
||||
what = op = topoct = topbits = flags = 0;
|
||||
continue;
|
||||
}
|
||||
switch (state) {
|
||||
case CHMOD_STATE_1ST_HALF:
|
||||
switch (*p) {
|
||||
case 'D':
|
||||
if (flags & CHMOD_FLAG_FILES_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_DIRS_ONLY;
|
||||
break;
|
||||
case 'F':
|
||||
if (flags & CHMOD_FLAG_DIRS_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_FILES_ONLY;
|
||||
break;
|
||||
case 'u':
|
||||
where |= 0100;
|
||||
topbits |= 04000;
|
||||
break;
|
||||
case 'g':
|
||||
where |= 0010;
|
||||
topbits |= 02000;
|
||||
break;
|
||||
case 'o':
|
||||
where |= 0001;
|
||||
break;
|
||||
case 'a':
|
||||
where |= 0111;
|
||||
break;
|
||||
case '+':
|
||||
op = CHMOD_OP_ADD;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '-':
|
||||
op = CHMOD_OP_SUB;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '=':
|
||||
op = CHMOD_OP_EQ;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7' && !where) {
|
||||
op = CHMOD_OP_SET;
|
||||
state = CHMOD_STATE_OCTAL;
|
||||
where = 1;
|
||||
what = *p - '0';
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
begin = end + 1;
|
||||
if (!*begin)
|
||||
return false;
|
||||
case CHMOD_STATE_2ND_HALF:
|
||||
switch (*p) {
|
||||
case 'r':
|
||||
what |= 4;
|
||||
break;
|
||||
case 'w':
|
||||
what |= 2;
|
||||
break;
|
||||
case 'X':
|
||||
flags |= CHMOD_FLAG_X_KEEP;
|
||||
/* fall through */
|
||||
case 'x':
|
||||
what |= 1;
|
||||
break;
|
||||
case 's':
|
||||
if (topbits)
|
||||
topoct |= topbits;
|
||||
else
|
||||
topoct = 04000;
|
||||
break;
|
||||
case 't':
|
||||
topoct |= 01000;
|
||||
break;
|
||||
default:
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7') {
|
||||
what = what * 8 + (*p - '0');
|
||||
if (what > CHMOD_BITS)
|
||||
state = CHMOD_STATE_ERROR;
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
p++;
|
||||
}
|
||||
*result = changed;
|
||||
if (state == CHMOD_STATE_ERROR)
|
||||
return false;
|
||||
*result = (changed & (mode_t)CHMOD_BITS) | nonperm;
|
||||
return true;
|
||||
}
|
||||
+4
-1
@@ -4,7 +4,10 @@
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Apply the supported rsync --chmod syntax to a permission mode. */
|
||||
/* Apply rsync's --chmod syntax to a permission mode, including the D/F/X
|
||||
* selectors and the s/t special bits. `mode` should carry the file type bits
|
||||
* (S_IFDIR/S_IFREG) so D/F/X can be evaluated; the type bits are preserved in
|
||||
* `result`. A spec may contain comma-separated clauses, which accumulate. */
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result);
|
||||
|
||||
#endif
|
||||
+51
-4
@@ -1,6 +1,7 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -16,10 +17,38 @@
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
|
||||
/* Maximum individual file data size within a chunk (64 MB) */
|
||||
#define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024)
|
||||
/* Maximum individual file data size within a chunk (64 MB). Distinct from the
|
||||
* receiver's whole-file MAX_FILE_DATA_SIZE (256 MB) in file_receive.c. */
|
||||
#define MAX_CHUNK_FILE_DATA_SIZE (64ULL * 1024 * 1024)
|
||||
#define MAX_FILES_PER_CHUNK 65536U
|
||||
|
||||
/* Reserve `charge` against `session`'s connection budget. This mirrors the
|
||||
static protocol_reserve_memory() in protocol.c: the receive-side call sites
|
||||
only have the Data.owner pointer (a ProtocolSession*), and protocol.c is out
|
||||
of scope for this fix, so the same atomic CAS accounting is reproduced here.
|
||||
The matching release always goes through data_destroy()'s Data.owner path. */
|
||||
static bool chunk_session_reserve(ProtocolSession* session, size_t charge) {
|
||||
unsigned long long allocated = atomic_load(&session->total_allocated_bytes);
|
||||
while (true) {
|
||||
if (allocated > MAX_CONNECTION_MEMORY ||
|
||||
(unsigned long long)charge > MAX_CONNECTION_MEMORY - allocated)
|
||||
return false;
|
||||
if (atomic_compare_exchange_weak(&session->total_allocated_bytes, &allocated,
|
||||
allocated + (unsigned long long)charge))
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge) {
|
||||
if (!data || charge == 0 || session == NULL)
|
||||
return true;
|
||||
if (!chunk_session_reserve(session, charge))
|
||||
return false;
|
||||
data->owner = session;
|
||||
data->protocol_charge = charge;
|
||||
return true;
|
||||
}
|
||||
|
||||
Chunk* chunk_create(File** items, int element_count) {
|
||||
if (element_count < 0 || (element_count > 0 && items == NULL))
|
||||
return NULL;
|
||||
@@ -354,9 +383,9 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
}
|
||||
|
||||
// Reject individual file data larger than the maximum allowed size.
|
||||
if (file_data_size > MAX_FILE_DATA_SIZE) {
|
||||
if (file_data_size > MAX_CHUNK_FILE_DATA_SIZE) {
|
||||
log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size,
|
||||
(unsigned long long)MAX_FILE_DATA_SIZE);
|
||||
(unsigned long long)MAX_CHUNK_FILE_DATA_SIZE);
|
||||
goto error;
|
||||
}
|
||||
|
||||
@@ -370,6 +399,15 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
Data* replacement = data_create(file_data, file_data_size);
|
||||
if (replacement == NULL)
|
||||
goto error;
|
||||
/* Charge the retained per-file copy to the connection budget (when the
|
||||
inbound chunk carries an owning session) so the queued copies are not
|
||||
held outside MAX_CONNECTION_MEMORY (B6). A NULL owner (e.g. a local
|
||||
batch apply) leaves the copy uncharged. */
|
||||
if (!data_charge_session(replacement, data->owner, allocation_size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for chunk file data");
|
||||
data_destroy(replacement);
|
||||
goto error;
|
||||
}
|
||||
data_destroy(file->data);
|
||||
file->data = replacement;
|
||||
data_pointer += file_data_size;
|
||||
@@ -466,12 +504,21 @@ Chunk* receive_chunk_data(int fd, const Config* config) {
|
||||
}
|
||||
Data* data_to_process = chunk_data;
|
||||
if (config->use_compression) {
|
||||
/* Preserve the inbound session across decompression so the (larger)
|
||||
decompressed chunk is charged to the same connection budget; the
|
||||
compressed buffer's own charge is released by data_destroy below. */
|
||||
ProtocolSession* owner = chunk_data->owner;
|
||||
data_to_process = data_decompress_limited(chunk_data, MAX_CHUNK_SIZE);
|
||||
data_destroy(chunk_data);
|
||||
if (data_to_process == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to decompress chunk");
|
||||
return NULL;
|
||||
}
|
||||
if (!data_charge_session(data_to_process, owner, data_to_process->size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for decompressed chunk");
|
||||
data_destroy(data_to_process);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
// Reject chunks larger than the maximum allowed size to prevent OOM.
|
||||
|
||||
@@ -23,4 +23,15 @@ Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_
|
||||
int compression_threads);
|
||||
Chunk* receive_chunk_data(int fd, const Config* config);
|
||||
|
||||
/* Charge `charge` retained bytes of `data` against `session`'s per-connection
|
||||
* budget (MAX_CONNECTION_MEMORY), mirroring the protocol layer's accounting, and
|
||||
* record them on `data` so data_destroy() returns the charge through the
|
||||
* Data.owner path. Returns false (leaving `data` uncharged) when the ceiling
|
||||
* would be exceeded. A NULL/zero-size charge or a NULL session is a no-op
|
||||
* success. The receive-side decompression and chunk-copy paths know the owning
|
||||
* session only through the Data.owner of the buffer they are processing, so
|
||||
* this is the entry point that lets them participate in the connection budget
|
||||
* without a session handle (B6). */
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
+794
-40
@@ -2,40 +2,206 @@
|
||||
#include "data.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <lz4.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <zlib.h>
|
||||
#include <zstd.h>
|
||||
|
||||
#define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024)
|
||||
#define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */
|
||||
|
||||
static char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv",
|
||||
".zip", ".gz", ".xz", ".zst", NULL};
|
||||
/* Hard ceiling for a single decompression. The sender compresses whole files
|
||||
* up to the protocol's whole-file receive bound, so the decompressor must
|
||||
* accept payloads that large; referencing the protocol constant keeps the two
|
||||
* bounds from drifting apart (they previously did: a 100 MB ceiling rejected
|
||||
* 100-256 MB files). This remains a real bomb guard -- every allocation in the
|
||||
* paths below is clamped to it -- so it must not exceed the protocol bound. */
|
||||
#define MAX_DECOMPRESSED_SIZE MAX_RECEIVE_WHOLE_FILE_SIZE
|
||||
|
||||
/* rsync 3.4.1's built-in skip-compress suffix list (the `--skip-compress`
|
||||
* defaults, in the man page's order). rsync stores it as space-separated
|
||||
* "*.suffix" globs; FastSync matches the plain suffix after the final dot, so
|
||||
* the leading "*." is omitted here. A user --skip-compress list replaces this
|
||||
* default entirely (matching rsync). */
|
||||
#define DEFAULT_SKIP_COMPRESS_SUFFIXES \
|
||||
"3g2 3gp 7z aac ace apk avi bz2 deb dmg ear f4v flac flv gpg gz iso jar jpeg jpg lrz lz lz4 " \
|
||||
"lzma " \
|
||||
"lzo m1a m1v m2a m2ts m2v m4a m4b m4p m4r m4v mka mkv mov mp1 mp2 mp3 mp4 mpa mpeg mpg mpv mts " \
|
||||
"odb odf odg odi odm odp ods odt oga ogg ogm ogv ogx opus otg oth otp ots ott oxt png qt rar " \
|
||||
"rpm " \
|
||||
"rz rzip spx squashfs sxc sxd sxg sxm sxw sz tbz tbz2 tgz tlz ts txz tzo vob war webm webp xz " \
|
||||
"z " \
|
||||
"zip zst"
|
||||
|
||||
/* Self-describing compressed frames: the first byte is the CompressionAlgo id.
|
||||
* zlib/lz4 store the uncompressed size as a little-endian uint32 after the
|
||||
* codec byte so decompression can be exactly pre-sized and bounded. */
|
||||
#define LZ4_SIZE_PREFIX_LEN 4
|
||||
|
||||
static _Atomic int g_compression_algo = COMPRESSION_ALGO_ZSTD;
|
||||
|
||||
/* Case-insensitive match of a bare suffix (no leading dot) against a
|
||||
* space-separated suffix list. */
|
||||
static bool suffix_in_list(const char* name, const char* list) {
|
||||
size_t name_len = strlen(name);
|
||||
while (*list) {
|
||||
while (*list == ' ')
|
||||
list++;
|
||||
const char* start = list;
|
||||
while (*list && *list != ' ')
|
||||
list++;
|
||||
size_t len = (size_t)(list - start);
|
||||
if (len == name_len && strncasecmp(name, start, len) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count) {
|
||||
if (!path)
|
||||
return false;
|
||||
const char* dot = strrchr(path, '.');
|
||||
if (!dot)
|
||||
if (!dot || dot[1] == '\0')
|
||||
return false;
|
||||
if (count < 0) {
|
||||
suffixes = SKIP_COMPRESSION_EXTENSIONS;
|
||||
count = 0;
|
||||
while (SKIP_COMPRESSION_EXTENSIONS[count])
|
||||
count++;
|
||||
}
|
||||
const char* name = dot + 1;
|
||||
/* count < 0 (the user gave no --skip-compress) selects rsync's built-in
|
||||
* default list; a non-negative count is the user's explicit list. */
|
||||
if (count < 0)
|
||||
return suffix_in_list(name, DEFAULT_SKIP_COMPRESS_SUFFIXES);
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (strcasecmp(dot, suffixes[i]) == 0)
|
||||
const char* suffix = suffixes[i];
|
||||
if (suffix[0] == '.')
|
||||
suffix++;
|
||||
if (strcasecmp(name, suffix) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
int compression_algo_from_name(const char* name) {
|
||||
if (!name)
|
||||
return -1;
|
||||
if (strcasecmp(name, "zstd") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZSTD;
|
||||
if (strcasecmp(name, "lz4") == 0)
|
||||
return (int)COMPRESSION_ALGO_LZ4;
|
||||
if (strcasecmp(name, "zlib") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZLIB;
|
||||
if (strcasecmp(name, "zlibx") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZLIBX;
|
||||
if (strcasecmp(name, "none") == 0)
|
||||
return (int)COMPRESSION_ALGO_NONE;
|
||||
return -1;
|
||||
}
|
||||
|
||||
const char* compression_algo_name(CompressionAlgo algo) {
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return "none";
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return "zstd";
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return "lz4";
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
return "zlib";
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return "zlibx";
|
||||
}
|
||||
return "<unknown>";
|
||||
}
|
||||
|
||||
bool compression_algo_valid(int algo) {
|
||||
return algo == (int)COMPRESSION_ALGO_NONE || algo == (int)COMPRESSION_ALGO_ZSTD ||
|
||||
algo == (int)COMPRESSION_ALGO_LZ4 || algo == (int)COMPRESSION_ALGO_ZLIB ||
|
||||
algo == (int)COMPRESSION_ALGO_ZLIBX;
|
||||
}
|
||||
|
||||
bool compression_algo_enabled(CompressionAlgo algo) {
|
||||
return algo != COMPRESSION_ALGO_NONE;
|
||||
}
|
||||
|
||||
static CompressionAlgo compiled_preference_first(void) {
|
||||
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
|
||||
* resolves to zstd. */
|
||||
static const CompressionAlgo preference[] = {
|
||||
COMPRESSION_ALGO_ZSTD, COMPRESSION_ALGO_LZ4, COMPRESSION_ALGO_ZLIBX,
|
||||
COMPRESSION_ALGO_ZLIB, COMPRESSION_ALGO_NONE,
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(preference) / sizeof(preference[0]); i++) {
|
||||
if (compression_algo_valid((int)preference[i]))
|
||||
return preference[i];
|
||||
}
|
||||
return COMPRESSION_ALGO_ZSTD;
|
||||
}
|
||||
|
||||
int compression_choice_resolve(void) {
|
||||
bool specified = false;
|
||||
int env = env_choice_first("RSYNC_COMPRESS_LIST", compression_algo_from_name, &specified);
|
||||
if (specified)
|
||||
return env; /* -1 = the list named no supported codec */
|
||||
return (int)compiled_preference_first();
|
||||
}
|
||||
|
||||
CompressionAlgo compression_negotiate_default(void) {
|
||||
int resolved = compression_choice_resolve();
|
||||
return resolved >= 0 ? (CompressionAlgo)resolved : compiled_preference_first();
|
||||
}
|
||||
|
||||
int compression_default_level(CompressionAlgo algo) {
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return ZSTD_CLEVEL_DEFAULT;
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return 6; /* rsync resolves zlib's Z_DEFAULT_COMPRESSION (-1) to 6 */
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return 1; /* rsync lz4 level is 0/ignored; positive keeps the gate on */
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
int compression_clamp_level(CompressionAlgo algo, int level) {
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
if (level < 1)
|
||||
return 1;
|
||||
if (level > 22)
|
||||
return 22;
|
||||
return level;
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
if (level < 1)
|
||||
return 1;
|
||||
if (level > 9)
|
||||
return 9;
|
||||
return level;
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return 1; /* ignored by lz4_compress; keeps the "compress" gate on */
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return level;
|
||||
}
|
||||
|
||||
void compression_set_algo(CompressionAlgo algo) {
|
||||
if (compression_algo_valid((int)algo))
|
||||
atomic_store(&g_compression_algo, (int)algo);
|
||||
}
|
||||
|
||||
CompressionAlgo compression_get_algo(void) {
|
||||
return (CompressionAlgo)atomic_load(&g_compression_algo);
|
||||
}
|
||||
|
||||
/* Per-thread cache of zstd contexts plus the grow-only compression scratch
|
||||
* buffer. zstd contexts are stateful and not safe to share between threads,
|
||||
* so each thread keeps its own (see compression_get_thread_ctx). The cache is
|
||||
@@ -126,17 +292,25 @@ static void compression_ctx_put(CompressionThreadCtx* ctx) {
|
||||
compression_ctx_free(ctx);
|
||||
}
|
||||
|
||||
Data* data_compress(Data* data_to_compress, int compression_level) {
|
||||
return data_compress_with_threads(data_to_compress, compression_level, 0);
|
||||
/* Build a frame consisting of a copy of `src` prefixed by `codec`. */
|
||||
static Data* frame_with_codec(const void* src, size_t size, CompressionAlgo codec) {
|
||||
if (size > SIZE_MAX - 1)
|
||||
return NULL;
|
||||
Data* out = data_create_empty(size + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
((uint8_t*)out->data)[0] = (uint8_t)codec;
|
||||
if (size > 0)
|
||||
memcpy((uint8_t*)out->data + 1, src, size);
|
||||
out->size = size + 1;
|
||||
return out;
|
||||
}
|
||||
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads) {
|
||||
if (!data_to_compress || (!data_to_compress->data && data_to_compress->size != 0) ||
|
||||
compression_threads < 0 || compression_threads > COMPRESSION_MAX_THREADS)
|
||||
static Data* zstd_compress(Data* in, int compression_level, int compression_threads) {
|
||||
size_t dst_size = ZSTD_compressBound(in->size);
|
||||
if (dst_size > SIZE_MAX - 1)
|
||||
return NULL;
|
||||
log_message(LOG_LEVEL_DEBUG, "Starting to compress data");
|
||||
size_t dst_size = ZSTD_compressBound(data_to_compress->size);
|
||||
dst_size += 1; /* codec prefix */
|
||||
|
||||
CompressionThreadCtx* ctx = compression_get_thread_ctx();
|
||||
if (ctx == NULL) {
|
||||
@@ -187,7 +361,7 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
|
||||
if (available_threads > 0) {
|
||||
/* Streaming compression needs the source size before threaded mode can end a frame. */
|
||||
size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, data_to_compress->size);
|
||||
size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, in->size);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
@@ -205,8 +379,8 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
ctx->out_cap = dst_size;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0};
|
||||
ZSTD_outBuffer output = {ctx->out_buf, dst_size, 0};
|
||||
ZSTD_inBuffer input = {in->data, in->size, 0};
|
||||
ZSTD_outBuffer output = {(uint8_t*)ctx->out_buf + 1, dst_size - 1, 0};
|
||||
|
||||
size_t ret;
|
||||
do {
|
||||
@@ -219,42 +393,208 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
|
||||
/* Hand off an exactly-sized copy; the scratch buffer stays cached so the next
|
||||
* call does not reallocate a ZSTD_compressBound-sized block. */
|
||||
compressed_data = data_create_empty(output.pos);
|
||||
compressed_data = data_create_empty(output.pos + 1);
|
||||
if (compressed_data == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate compressed data");
|
||||
goto cleanup;
|
||||
}
|
||||
((uint8_t*)compressed_data->data)[0] = (uint8_t)COMPRESSION_ALGO_ZSTD;
|
||||
if (output.pos > 0)
|
||||
memcpy(compressed_data->data, ctx->out_buf, output.pos);
|
||||
compressed_data->size = output.pos;
|
||||
memcpy((uint8_t*)compressed_data->data + 1, (uint8_t*)ctx->out_buf + 1, output.pos);
|
||||
compressed_data->size = output.pos + 1;
|
||||
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu",
|
||||
data_to_compress->size, compressed_data->size);
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu", in->size,
|
||||
compressed_data->size);
|
||||
|
||||
cleanup:
|
||||
compression_ctx_put(ctx);
|
||||
return compressed_data;
|
||||
}
|
||||
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) ||
|
||||
maximum_size == 0)
|
||||
static Data* lz4_compress(Data* in) {
|
||||
int bound = LZ4_compressBound((int)in->size);
|
||||
if (bound < 0 || in->size > (size_t)INT_MAX)
|
||||
return NULL;
|
||||
Data* out = data_create_empty((size_t)bound + 1 + LZ4_SIZE_PREFIX_LEN);
|
||||
if (!out)
|
||||
return NULL;
|
||||
uint32_t raw_size = (uint32_t)in->size;
|
||||
uint8_t* p = (uint8_t*)out->data;
|
||||
p[0] = (uint8_t)COMPRESSION_ALGO_LZ4;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
p[1 + i] = (uint8_t)((raw_size >> (8 * i)) & 0xff);
|
||||
int written = 0;
|
||||
if (in->size > 0) {
|
||||
written = LZ4_compress_default((const char*)in->data, (char*)p + 1 + LZ4_SIZE_PREFIX_LEN,
|
||||
(int)in->size, bound);
|
||||
if (written <= 0) {
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
out->size = (size_t)written + 1 + LZ4_SIZE_PREFIX_LEN;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zlib_compress(Data* in, CompressionAlgo algo, int compression_level) {
|
||||
int level = compression_level;
|
||||
if (level < 1)
|
||||
level = Z_DEFAULT_COMPRESSION;
|
||||
if (level > 9)
|
||||
level = 9;
|
||||
uLong bound = compressBound((uLong)in->size);
|
||||
if (in->size > (size_t)ULONG_MAX)
|
||||
return NULL;
|
||||
Data* out = data_create_empty((size_t)bound + 1 + LZ4_SIZE_PREFIX_LEN);
|
||||
if (!out)
|
||||
return NULL;
|
||||
uint32_t raw_size = (uint32_t)in->size;
|
||||
uint8_t* p = (uint8_t*)out->data;
|
||||
p[0] = (uint8_t)algo;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
p[1 + i] = (uint8_t)((raw_size >> (8 * i)) & 0xff);
|
||||
uLongf dest_len = bound;
|
||||
int rc = compress2(p + 1 + LZ4_SIZE_PREFIX_LEN, &dest_len, (const Bytef*)in->data,
|
||||
(uLong)in->size, level);
|
||||
if (rc != Z_OK) {
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = (size_t)dest_len + 1 + LZ4_SIZE_PREFIX_LEN;
|
||||
return out;
|
||||
}
|
||||
|
||||
Data* data_compress_codec(Data* data_to_compress, CompressionAlgo algo, int compression_level,
|
||||
int compression_threads) {
|
||||
if (!data_to_compress || (!data_to_compress->data && data_to_compress->size != 0) ||
|
||||
compression_threads < 0 || compression_threads > COMPRESSION_MAX_THREADS)
|
||||
return NULL;
|
||||
if (!compression_algo_valid((int)algo))
|
||||
return NULL;
|
||||
log_message(LOG_LEVEL_DEBUG, "Starting to compress data");
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return frame_with_codec(data_to_compress->data, data_to_compress->size, COMPRESSION_ALGO_NONE);
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return zstd_compress(data_to_compress, compression_level, compression_threads);
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return lz4_compress(data_to_compress);
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return zlib_compress(data_to_compress, algo, compression_level);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads) {
|
||||
return data_compress_codec(data_to_compress, compression_get_algo(), compression_level,
|
||||
compression_threads);
|
||||
}
|
||||
|
||||
Data* data_compress(Data* data_to_compress, int compression_level) {
|
||||
return data_compress_codec(data_to_compress, compression_get_algo(), compression_level, 0);
|
||||
}
|
||||
|
||||
static Data* decompress_none(const Data* compressed_data, size_t maximum_size) {
|
||||
size_t size = compressed_data->size - 1;
|
||||
if (size > maximum_size)
|
||||
return NULL;
|
||||
Data* out = data_create_empty(size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (size > 0)
|
||||
memcpy(out->data, (const uint8_t*)compressed_data->data + 1, size);
|
||||
out->size = size;
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Read the 4-byte little-endian raw size stored after the codec byte. */
|
||||
static bool read_raw_size(const Data* in, uint32_t* raw_size) {
|
||||
if (in->size < 1 + LZ4_SIZE_PREFIX_LEN)
|
||||
return false;
|
||||
const uint8_t* p = (const uint8_t*)in->data;
|
||||
uint32_t v = 0;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
v |= (uint32_t)p[1 + i] << (8 * i);
|
||||
*raw_size = v;
|
||||
return true;
|
||||
}
|
||||
|
||||
static Data* lz4_decompress(Data* compressed_data, size_t maximum_size, size_t hard_limit) {
|
||||
uint32_t raw_size = 0;
|
||||
if (!read_raw_size(compressed_data, &raw_size))
|
||||
return NULL;
|
||||
if (raw_size > hard_limit || raw_size > maximum_size)
|
||||
return NULL;
|
||||
size_t comp_size = compressed_data->size - 1 - LZ4_SIZE_PREFIX_LEN;
|
||||
Data* out = data_create_empty(raw_size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (raw_size == 0) {
|
||||
out->size = 0;
|
||||
return out;
|
||||
}
|
||||
int rc = LZ4_decompress_safe((const char*)compressed_data->data + 1 + LZ4_SIZE_PREFIX_LEN,
|
||||
(char*)out->data, (int)comp_size, (int)raw_size);
|
||||
if (rc < 0 || (uint32_t)rc != raw_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "LZ4 decompression failed");
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = raw_size;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zlib_decompress(Data* compressed_data, size_t maximum_size, size_t hard_limit) {
|
||||
uint32_t raw_size = 0;
|
||||
if (!read_raw_size(compressed_data, &raw_size))
|
||||
return NULL;
|
||||
if (raw_size > hard_limit || raw_size > maximum_size)
|
||||
return NULL;
|
||||
size_t comp_size = compressed_data->size - 1 - LZ4_SIZE_PREFIX_LEN;
|
||||
Data* out = data_create_empty(raw_size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (raw_size == 0) {
|
||||
out->size = 0;
|
||||
return out;
|
||||
}
|
||||
uLongf dest_len = raw_size;
|
||||
int rc =
|
||||
uncompress((Bytef*)out->data, &dest_len,
|
||||
(const Bytef*)compressed_data->data + 1 + LZ4_SIZE_PREFIX_LEN, (uLong)comp_size);
|
||||
if (rc != Z_OK || dest_len != raw_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "zlib decompression failed");
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = raw_size;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zstd_decompress(Data* compressed_data, size_t maximum_size) {
|
||||
/* The zstd frame starts after the codec byte. */
|
||||
const void* frame = (const uint8_t*)compressed_data->data + 1;
|
||||
size_t frame_size = compressed_data->size - 1;
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Start to decompress data");
|
||||
unsigned long long dst_size =
|
||||
ZSTD_getFrameContentSize(compressed_data->data, compressed_data->size);
|
||||
if (ZSTD_isError(dst_size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: %s",
|
||||
ZSTD_getErrorName(dst_size));
|
||||
unsigned long long dst_size = ZSTD_getFrameContentSize(frame, frame_size);
|
||||
/* ZSTD_isError() is also true for ZSTD_CONTENTSIZE_ERROR and
|
||||
* ZSTD_CONTENTSIZE_UNKNOWN (both are encoded near (size_t)-1), so test the
|
||||
* sentinels explicitly instead of blanket-rejecting every error-ish value:
|
||||
* only CONTENTSIZE_ERROR means an unreadable header, while CONTENTSIZE_UNKNOWN
|
||||
* must reach the estimate fallback below. */
|
||||
if (dst_size == ZSTD_CONTENTSIZE_ERROR) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: invalid zstd frame");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
// ZSTD_CONTENTSIZE_UNKNOWN (~2^64) can cause massive allocation;
|
||||
// fall back to a conservative estimate (3x compressed size) when unknown.
|
||||
if (dst_size == ZSTD_CONTENTSIZE_UNKNOWN) {
|
||||
if (compressed_data->size > ULLONG_MAX / 3)
|
||||
if (frame_size > ULLONG_MAX / 3)
|
||||
return NULL;
|
||||
dst_size = compressed_data->size * 3;
|
||||
dst_size = frame_size * 3;
|
||||
if (dst_size < INITIAL_DECOMPRESS_BUF_SIZE)
|
||||
dst_size = INITIAL_DECOMPRESS_BUF_SIZE;
|
||||
}
|
||||
@@ -291,7 +631,7 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
goto cleanup;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {compressed_data->data, compressed_data->size, 0};
|
||||
ZSTD_inBuffer input = {frame, frame_size, 0};
|
||||
ZSTD_outBuffer output = {uncompressed_data->data, buf_size, 0};
|
||||
|
||||
size_t ret;
|
||||
@@ -305,8 +645,7 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
}
|
||||
if (ret > 0 && output.pos == output.size) {
|
||||
if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) {
|
||||
log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes",
|
||||
(unsigned long long)MAX_DECOMPRESSED_SIZE);
|
||||
log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes", hard_limit);
|
||||
data_destroy(uncompressed_data);
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
@@ -324,6 +663,20 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
uncompressed_data->data = new_data;
|
||||
output.dst = new_data;
|
||||
output.size = buf_size;
|
||||
/* Re-attempt with the larger output buffer; the truncated-frame check
|
||||
* below must not reject a complete frame that merely filled the previous
|
||||
* buffer exactly. */
|
||||
continue;
|
||||
}
|
||||
/* A positive hint with all input consumed means the frame is incomplete: a
|
||||
* truncated stream would otherwise spin here forever (ZSTD_decompressStream
|
||||
* keeps returning the same hint). Fail instead of burning CPU. */
|
||||
if (ret != 0 && input.pos == input.size) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"Truncated zstd frame: input exhausted with %zu bytes still expected", ret);
|
||||
data_destroy(uncompressed_data);
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
} while (ret > 0);
|
||||
|
||||
@@ -336,6 +689,407 @@ cleanup:
|
||||
return uncompressed_data;
|
||||
}
|
||||
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) ||
|
||||
maximum_size == 0)
|
||||
return NULL;
|
||||
if (compressed_data->size < 1)
|
||||
return NULL;
|
||||
unsigned long long hard_limit =
|
||||
maximum_size < MAX_DECOMPRESSED_SIZE ? maximum_size : MAX_DECOMPRESSED_SIZE;
|
||||
uint8_t codec = ((const uint8_t*)compressed_data->data)[0];
|
||||
if (!compression_algo_valid(codec))
|
||||
return NULL;
|
||||
switch ((CompressionAlgo)codec) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return decompress_none(compressed_data, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return zstd_decompress(compressed_data, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return lz4_decompress(compressed_data, maximum_size, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return zlib_decompress(compressed_data, maximum_size, (size_t)hard_limit);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* data_decompress(Data* compressed_data) {
|
||||
return data_decompress_limited(compressed_data, MAX_DECOMPRESSED_SIZE);
|
||||
}
|
||||
|
||||
/* ---- streaming decompression ---- */
|
||||
|
||||
#define STREAM_DECOMPRESS_OUT_CHUNK (256 * 1024)
|
||||
|
||||
struct CompressionStreamDecompressor {
|
||||
CompressionAlgo algo;
|
||||
unsigned long long expected_out;
|
||||
unsigned long long total;
|
||||
int out_fd;
|
||||
unsigned char* out_buf;
|
||||
ZSTD_DCtx* dctx;
|
||||
z_stream zs;
|
||||
bool zs_initialized;
|
||||
bool failed;
|
||||
};
|
||||
|
||||
static bool stream_write_all(int fd, const void* data, size_t size) {
|
||||
const unsigned char* p = data;
|
||||
size_t done = 0;
|
||||
while (done < size) {
|
||||
ssize_t n = write(fd, p + done, size - done);
|
||||
if (n < 0 && errno == EINTR)
|
||||
continue;
|
||||
if (n <= 0)
|
||||
return false;
|
||||
done += (size_t)n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
CompressionStreamDecompressor*
|
||||
compression_stream_decompressor_create(CompressionAlgo algo, unsigned long long expected_out) {
|
||||
CompressionStreamDecompressor* d = calloc(1, sizeof(*d));
|
||||
if (!d)
|
||||
return NULL;
|
||||
d->algo = algo;
|
||||
d->expected_out = expected_out;
|
||||
d->out_fd = -1;
|
||||
d->out_buf = malloc(STREAM_DECOMPRESS_OUT_CHUNK);
|
||||
if (!d->out_buf) {
|
||||
free(d);
|
||||
return NULL;
|
||||
}
|
||||
if (algo == COMPRESSION_ALGO_ZSTD) {
|
||||
d->dctx = ZSTD_createDCtx();
|
||||
if (!d->dctx) {
|
||||
free(d->out_buf);
|
||||
free(d);
|
||||
return NULL;
|
||||
}
|
||||
} else if (algo == COMPRESSION_ALGO_ZLIB || algo == COMPRESSION_ALGO_ZLIBX) {
|
||||
if (inflateInit(&d->zs) != Z_OK) {
|
||||
free(d->out_buf);
|
||||
free(d);
|
||||
return NULL;
|
||||
}
|
||||
d->zs_initialized = true;
|
||||
} else if (algo != COMPRESSION_ALGO_NONE) {
|
||||
/* lz4's block format cannot be decompressed incrementally. */
|
||||
free(d->out_buf);
|
||||
free(d);
|
||||
return NULL;
|
||||
}
|
||||
return d;
|
||||
}
|
||||
|
||||
static bool stream_emit(CompressionStreamDecompressor* d, const void* buf, size_t len) {
|
||||
if (len == 0)
|
||||
return true;
|
||||
if (d->expected_out != 0 && (d->total > d->expected_out || len > d->expected_out - d->total)) {
|
||||
d->failed = true;
|
||||
return false;
|
||||
}
|
||||
if (!stream_write_all(d->out_fd, buf, len)) {
|
||||
d->failed = true;
|
||||
return false;
|
||||
}
|
||||
d->total += len;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool stream_feed_none(CompressionStreamDecompressor* d, const void* in, size_t in_len,
|
||||
bool* done) {
|
||||
if (!stream_emit(d, in, in_len))
|
||||
return false;
|
||||
/* NONE has no end marker; the caller knows the frame length. */
|
||||
*done = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool stream_feed_zstd(CompressionStreamDecompressor* d, const void* in, size_t in_len,
|
||||
bool* done) {
|
||||
ZSTD_inBuffer input = {in, in_len, 0};
|
||||
while (input.pos < input.size) {
|
||||
ZSTD_outBuffer output = {d->out_buf, STREAM_DECOMPRESS_OUT_CHUNK, 0};
|
||||
size_t ret = ZSTD_decompressStream(d->dctx, &output, &input);
|
||||
if (ZSTD_isError(ret)) {
|
||||
d->failed = true;
|
||||
return false;
|
||||
}
|
||||
if (!stream_emit(d, d->out_buf, output.pos))
|
||||
return false;
|
||||
if (ret == 0) {
|
||||
*done = true;
|
||||
/* Trailing bytes after a complete frame are malformed; stop consuming. */
|
||||
if (input.pos < input.size) {
|
||||
d->failed = true;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool stream_feed_zlib(CompressionStreamDecompressor* d, const void* in, size_t in_len,
|
||||
bool* done) {
|
||||
d->zs.next_in = (Bytef*)in;
|
||||
d->zs.avail_in = (uInt)in_len;
|
||||
while (d->zs.avail_in > 0) {
|
||||
d->zs.next_out = d->out_buf;
|
||||
d->zs.avail_out = STREAM_DECOMPRESS_OUT_CHUNK;
|
||||
int rc = inflate(&d->zs, Z_NO_FLUSH);
|
||||
if (rc != Z_OK && rc != Z_STREAM_END && rc != Z_BUF_ERROR) {
|
||||
d->failed = true;
|
||||
return false;
|
||||
}
|
||||
size_t produced = STREAM_DECOMPRESS_OUT_CHUNK - d->zs.avail_out;
|
||||
if (!stream_emit(d, d->out_buf, produced))
|
||||
return false;
|
||||
if (rc == Z_STREAM_END) {
|
||||
*done = true;
|
||||
return d->zs.avail_in == 0;
|
||||
}
|
||||
if (rc == Z_BUF_ERROR && produced == 0) {
|
||||
/* Need more input. */
|
||||
break;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool compression_stream_decompressor_feed(CompressionStreamDecompressor* d, const void* in,
|
||||
size_t in_len, int out_fd, bool* done) {
|
||||
if (!d || d->failed)
|
||||
return false;
|
||||
d->out_fd = out_fd;
|
||||
if (done)
|
||||
*done = false;
|
||||
switch (d->algo) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return stream_feed_none(d, in, in_len, done);
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return stream_feed_zstd(d, in, in_len, done);
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return stream_feed_zlib(d, in, in_len, done);
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
break;
|
||||
}
|
||||
d->failed = true;
|
||||
return false;
|
||||
}
|
||||
|
||||
unsigned long long compression_stream_decompressor_total(const CompressionStreamDecompressor* d) {
|
||||
return d ? d->total : 0;
|
||||
}
|
||||
|
||||
void compression_stream_decompressor_destroy(CompressionStreamDecompressor* d) {
|
||||
if (!d)
|
||||
return;
|
||||
if (d->dctx)
|
||||
ZSTD_freeDCtx(d->dctx);
|
||||
if (d->zs_initialized)
|
||||
inflateEnd(&d->zs);
|
||||
free(d->out_buf);
|
||||
free(d);
|
||||
}
|
||||
|
||||
/* ---- streaming compression ---- */
|
||||
|
||||
struct CompressionStreamCompressor {
|
||||
CompressionAlgo algo;
|
||||
int level;
|
||||
ZSTD_CCtx* cctx;
|
||||
z_stream zs;
|
||||
bool zs_initialized;
|
||||
unsigned char* out_buf;
|
||||
bool failed;
|
||||
};
|
||||
|
||||
bool compression_stream_compress_supported(CompressionAlgo algo) {
|
||||
return algo == COMPRESSION_ALGO_ZSTD || algo == COMPRESSION_ALGO_ZLIB ||
|
||||
algo == COMPRESSION_ALGO_ZLIBX;
|
||||
}
|
||||
|
||||
CompressionStreamCompressor* compression_stream_compressor_create(CompressionAlgo algo, int level,
|
||||
int threads) {
|
||||
(void)threads;
|
||||
if (!compression_algo_valid((int)algo) || algo == COMPRESSION_ALGO_LZ4)
|
||||
return NULL;
|
||||
CompressionStreamCompressor* c = calloc(1, sizeof(*c));
|
||||
if (!c)
|
||||
return NULL;
|
||||
c->algo = algo;
|
||||
c->level = level;
|
||||
c->out_buf = malloc(STREAM_DECOMPRESS_OUT_CHUNK);
|
||||
if (!c->out_buf) {
|
||||
free(c);
|
||||
return NULL;
|
||||
}
|
||||
if (algo == COMPRESSION_ALGO_ZSTD) {
|
||||
c->cctx = ZSTD_createCCtx();
|
||||
if (!c->cctx) {
|
||||
free(c->out_buf);
|
||||
free(c);
|
||||
return NULL;
|
||||
}
|
||||
} else if (algo == COMPRESSION_ALGO_ZLIB || algo == COMPRESSION_ALGO_ZLIBX) {
|
||||
if (deflateInit(&c->zs, level < 1 ? Z_DEFAULT_COMPRESSION : level) != Z_OK) {
|
||||
free(c->out_buf);
|
||||
free(c);
|
||||
return NULL;
|
||||
}
|
||||
c->zs_initialized = true;
|
||||
}
|
||||
return c;
|
||||
}
|
||||
|
||||
bool compression_stream_compressor_begin(CompressionStreamCompressor* c,
|
||||
unsigned long long raw_size, int out_fd) {
|
||||
if (!c || c->failed)
|
||||
return false;
|
||||
unsigned char hdr[1 + 4];
|
||||
size_t hdr_len = 1;
|
||||
hdr[0] = (unsigned char)c->algo;
|
||||
if (c->algo == COMPRESSION_ALGO_ZLIB || c->algo == COMPRESSION_ALGO_ZLIBX) {
|
||||
uint32_t size32 = raw_size > UINT32_MAX ? UINT32_MAX : (uint32_t)raw_size;
|
||||
for (int i = 0; i < 4; i++)
|
||||
hdr[1 + i] = (uint8_t)((size32 >> (8 * i)) & 0xff);
|
||||
hdr_len = 5;
|
||||
}
|
||||
if (c->algo == COMPRESSION_ALGO_ZSTD) {
|
||||
/* Pledge the source size and force the frame content-size field so the
|
||||
receiver can decide whether to stream from the frame header alone. */
|
||||
if (ZSTD_isError(ZSTD_CCtx_setPledgedSrcSize(c->cctx, raw_size)) ||
|
||||
ZSTD_isError(ZSTD_CCtx_setParameter(c->cctx, ZSTD_c_compressionLevel, c->level)) ||
|
||||
ZSTD_isError(ZSTD_CCtx_setParameter(c->cctx, ZSTD_c_contentSizeFlag, 1))) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
if (ZSTD_isError(ZSTD_CCtx_setParameter(c->cctx, ZSTD_c_checksumFlag, 0))) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!stream_write_all(out_fd, hdr, hdr_len)) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool stream_compress_zlib(CompressionStreamCompressor* c, const void* in, size_t in_len,
|
||||
int out_fd, int flush) {
|
||||
c->zs.next_in = (Bytef*)in;
|
||||
c->zs.avail_in = (uInt)in_len;
|
||||
do {
|
||||
c->zs.next_out = c->out_buf;
|
||||
c->zs.avail_out = STREAM_DECOMPRESS_OUT_CHUNK;
|
||||
int rc = deflate(&c->zs, flush);
|
||||
if (rc != Z_OK && rc != Z_STREAM_END && rc != Z_BUF_ERROR) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
size_t produced = STREAM_DECOMPRESS_OUT_CHUNK - c->zs.avail_out;
|
||||
if (!stream_write_all(out_fd, c->out_buf, produced)) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
if (rc == Z_STREAM_END)
|
||||
return true;
|
||||
if (rc == Z_BUF_ERROR && produced == 0)
|
||||
break;
|
||||
} while (c->zs.avail_in > 0 || flush == Z_FINISH);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool compression_stream_compressor_feed(CompressionStreamCompressor* c, const void* in,
|
||||
size_t in_len, int out_fd) {
|
||||
if (!c || c->failed)
|
||||
return false;
|
||||
if (c->algo == COMPRESSION_ALGO_NONE)
|
||||
return stream_write_all(out_fd, in, in_len);
|
||||
if (c->algo == COMPRESSION_ALGO_ZSTD) {
|
||||
ZSTD_inBuffer input = {in, in_len, 0};
|
||||
while (input.pos < input.size) {
|
||||
ZSTD_outBuffer output = {c->out_buf, STREAM_DECOMPRESS_OUT_CHUNK, 0};
|
||||
size_t ret = ZSTD_compressStream2(c->cctx, &output, &input, ZSTD_e_continue);
|
||||
if (ZSTD_isError(ret)) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
if (!stream_write_all(out_fd, c->out_buf, output.pos)) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
if (output.pos == 0 && input.pos < input.size)
|
||||
break; /* avoid spinning; zstd buffers the rest internally */
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return stream_compress_zlib(c, in, in_len, out_fd, Z_NO_FLUSH);
|
||||
}
|
||||
|
||||
bool compression_stream_compressor_finish(CompressionStreamCompressor* c, int out_fd) {
|
||||
if (!c || c->failed)
|
||||
return false;
|
||||
if (c->algo == COMPRESSION_ALGO_NONE)
|
||||
return true;
|
||||
if (c->algo == COMPRESSION_ALGO_ZSTD) {
|
||||
size_t ret;
|
||||
do {
|
||||
ZSTD_inBuffer input = {NULL, 0, 0};
|
||||
ZSTD_outBuffer output = {c->out_buf, STREAM_DECOMPRESS_OUT_CHUNK, 0};
|
||||
ret = ZSTD_compressStream2(c->cctx, &output, &input, ZSTD_e_end);
|
||||
if (ZSTD_isError(ret)) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
if (!stream_write_all(out_fd, c->out_buf, output.pos)) {
|
||||
c->failed = true;
|
||||
return false;
|
||||
}
|
||||
} while (ret > 0);
|
||||
return true;
|
||||
}
|
||||
return stream_compress_zlib(c, NULL, 0, out_fd, Z_FINISH);
|
||||
}
|
||||
|
||||
void compression_stream_compressor_destroy(CompressionStreamCompressor* c) {
|
||||
if (!c)
|
||||
return;
|
||||
if (c->cctx)
|
||||
ZSTD_freeCCtx(c->cctx);
|
||||
if (c->zs_initialized)
|
||||
deflateEnd(&c->zs);
|
||||
free(c->out_buf);
|
||||
free(c);
|
||||
}
|
||||
|
||||
unsigned long long compression_peek_frame_content_size(const void* buf, size_t len) {
|
||||
if (!buf || len < 1)
|
||||
return 0;
|
||||
const uint8_t* p = buf;
|
||||
uint8_t codec = p[0];
|
||||
if (!compression_algo_valid(codec))
|
||||
return 0;
|
||||
if (codec == (uint8_t)COMPRESSION_ALGO_NONE)
|
||||
return len - 1;
|
||||
if (codec == (uint8_t)COMPRESSION_ALGO_ZSTD) {
|
||||
if (len < 2)
|
||||
return 0;
|
||||
unsigned long long size = ZSTD_getFrameContentSize(p + 1, len - 1);
|
||||
if (size == ZSTD_CONTENTSIZE_ERROR || size == ZSTD_CONTENTSIZE_UNKNOWN)
|
||||
return 0;
|
||||
return size;
|
||||
}
|
||||
if (len < 1 + LZ4_SIZE_PREFIX_LEN)
|
||||
return 0;
|
||||
uint32_t raw = 0;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
raw |= (uint32_t)p[1 + i] << (8 * i);
|
||||
return raw;
|
||||
}
|
||||
+117
-1
@@ -6,13 +6,129 @@
|
||||
|
||||
#define COMPRESSION_MAX_THREADS 64
|
||||
|
||||
/* Compression algorithms selectable with --compress-choice / -z. The ids are
|
||||
* the values placed on the wire (Config->compression_algo), so they must be
|
||||
* kept stable. NONE is "no compression"; ZSTD is the historical FastSync
|
||||
* default and the negotiated "auto" choice. ZLIBX is rsync's zlib-without-
|
||||
* matched-data variant: FastSync compresses only the delta/token bytes (it does
|
||||
* not put matched file data in the compression stream), so its zlib codec is
|
||||
* already the "x" form and zlib/zlibx share the same implementation, recorded
|
||||
* under distinct ids. */
|
||||
typedef enum {
|
||||
COMPRESSION_ALGO_NONE = 0,
|
||||
COMPRESSION_ALGO_ZSTD = 1,
|
||||
COMPRESSION_ALGO_LZ4 = 2,
|
||||
COMPRESSION_ALGO_ZLIB = 3,
|
||||
COMPRESSION_ALGO_ZLIBX = 4
|
||||
} CompressionAlgo;
|
||||
|
||||
/* Resolve a --compress-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "zstd", "lz4", "zlib", "zlibx", "none". "auto" is not an algorithm
|
||||
* here; the caller resolves it to the negotiated default. Returns -1 for any
|
||||
* unrecognized name. */
|
||||
int compression_algo_from_name(const char* name);
|
||||
const char* compression_algo_name(CompressionAlgo algo);
|
||||
bool compression_algo_valid(int algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list
|
||||
* (rsync 3.4.1's `--version` order: zstd lz4 zlibx zlib none). Resolves
|
||||
* "auto". */
|
||||
CompressionAlgo compression_negotiate_default(void);
|
||||
|
||||
/* Resolve "auto" the way rsync does: the first supported name in
|
||||
* RSYNC_COMPRESS_LIST (whitespace-separated, client half ends at '&'), then the
|
||||
* compiled-in preference order when the variable is unset/blank. Returns -1
|
||||
* when the variable is set but names no supported codec (rsync's failed
|
||||
* negotiation), otherwise a valid CompressionAlgo id. */
|
||||
int compression_choice_resolve(void);
|
||||
|
||||
/* rsync 3.4.1's per-codec default level, applied when the user did not pass
|
||||
* --compress-level/--zl. zstd uses ZSTD_CLEVEL_DEFAULT (3) and zlib/zlibx the
|
||||
* resolved Z_DEFAULT_COMPRESSION (6). lz4 has no tunable level in rsync
|
||||
* (always the default acceleration); FastSync returns a positive placeholder so
|
||||
* its "level > 0" compression gate stays engaged, and lz4_compress ignores the
|
||||
* value, so the output is identical to rsync's. none is 0. */
|
||||
int compression_default_level(CompressionAlgo algo);
|
||||
|
||||
/* Clamp an explicit --compress-level to the codec's accepted range the way
|
||||
* rsync's init_compression_level() does: zstd 1..22, zlib/zlibx 1..9, lz4
|
||||
* ignored (fixed positive placeholder), none 0. */
|
||||
int compression_clamp_level(CompressionAlgo algo, int level);
|
||||
|
||||
/* True when the algorithm actually compresses (i.e. is not NONE). */
|
||||
bool compression_algo_enabled(CompressionAlgo algo);
|
||||
|
||||
/* Select the process-wide codec used by the legacy wrappers below. Each
|
||||
* process serves exactly one transfer config (the server forks per connection,
|
||||
* the client configures itself before spawning transfer threads), so a
|
||||
* process-global default is sufficient and constant for the lifetime of a
|
||||
* transfer. Defaults to ZSTD when never set. Thread-safe. */
|
||||
void compression_set_algo(CompressionAlgo algo);
|
||||
CompressionAlgo compression_get_algo(void);
|
||||
|
||||
/* Codec-aware primitives. The compressed buffer is self-describing: its first
|
||||
* byte is the CompressionAlgo id, so decompression never needs the codec passed
|
||||
* separately (this keeps every existing Decompress call site source-compatible).
|
||||
* `data_compress_codec` returns NULL on invalid input or an unsupported codec. */
|
||||
Data* data_compress_codec(Data* data_to_compress, CompressionAlgo algo, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
|
||||
/* Legacy zstd-default wrappers retained for existing callers/tests. */
|
||||
Data* data_compress(Data* data_to_compress, int compression_level);
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress(Data* compressed_data);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count);
|
||||
|
||||
/* Streaming decompression for a payload too large to hold in memory. The
|
||||
* caller consumes the frame's leading codec byte (and, for lz4/zlib/zlibx, the
|
||||
* 4-byte little-endian raw-size prefix) and then feeds the remaining frame
|
||||
* bytes in bounded chunks; decompressed output is written straight to `out_fd`
|
||||
* so neither the compressed nor the decompressed image is ever materialized.
|
||||
* Only zstd (the default), zlib/zlibx and none support streaming; lz4's block
|
||||
* format is one-shot, so its stream decompressor reports failure and the caller
|
||||
* falls back (the whole-buffer path keeps its existing bound). */
|
||||
typedef struct CompressionStreamDecompressor CompressionStreamDecompressor;
|
||||
|
||||
CompressionStreamDecompressor*
|
||||
compression_stream_decompressor_create(CompressionAlgo algo, unsigned long long expected_out);
|
||||
/* Feed one chunk. Returns false on a malformed frame, an I/O error, or when the
|
||||
* total output would exceed `expected_out` (when non-zero). *done is set once
|
||||
* the frame end has been reached. */
|
||||
bool compression_stream_decompressor_feed(CompressionStreamDecompressor* d, const void* in,
|
||||
size_t in_len, int out_fd, bool* done);
|
||||
unsigned long long compression_stream_decompressor_total(const CompressionStreamDecompressor* d);
|
||||
void compression_stream_decompressor_destroy(CompressionStreamDecompressor* d);
|
||||
|
||||
/* Peek the logical (decompressed) size from the leading bytes of a compressed
|
||||
* frame (codec byte + header), returning 0 when it cannot be determined from
|
||||
* the supplied prefix. Used to decide whether a frame must take the streaming
|
||||
* path before its body is read. */
|
||||
unsigned long long compression_peek_frame_content_size(const void* buf, size_t len);
|
||||
|
||||
/* Streaming compression (sender side). Compresses a source in bounded chunks
|
||||
* into `out_fd` as one self-describing frame (codec byte, the lz4/zlib raw-size
|
||||
* prefix, then the codec stream), so a whole file can be compressed without
|
||||
* materializing it in memory. zstd/zlib/zlibx/none are supported; lz4's block
|
||||
* format is one-shot, so its create() returns NULL and the caller keeps the
|
||||
* buffered path. `raw_size` is the known source length (used for the zlib
|
||||
* prefix and, for zstd, the frame content-size field). */
|
||||
typedef struct CompressionStreamCompressor CompressionStreamCompressor;
|
||||
|
||||
/* True when `algo` can be stream-compressed (zstd/zlib/zlibx; lz4's block format
|
||||
* is one-shot). Used by the sender to decide whether an over-threshold source
|
||||
* may stay unloaded. */
|
||||
bool compression_stream_compress_supported(CompressionAlgo algo);
|
||||
CompressionStreamCompressor* compression_stream_compressor_create(CompressionAlgo algo, int level,
|
||||
int threads);
|
||||
bool compression_stream_compressor_begin(CompressionStreamCompressor* c,
|
||||
unsigned long long raw_size, int out_fd);
|
||||
bool compression_stream_compressor_feed(CompressionStreamCompressor* c, const void* in,
|
||||
size_t in_len, int out_fd);
|
||||
bool compression_stream_compressor_finish(CompressionStreamCompressor* c, int out_fd);
|
||||
void compression_stream_compressor_destroy(CompressionStreamCompressor* c);
|
||||
|
||||
/* Release the calling thread's cached zstd contexts (compressor, decompressor
|
||||
* and scratch buffer). The cache is thread-local and is also released
|
||||
* automatically when a worker thread exits (via a C11 tss destructor) and for
|
||||
|
||||
+749
-649
File diff suppressed because it is too large.
Load diff
+777
-299
File diff suppressed because it is too large.
Load diff
+206
-20
@@ -8,12 +8,13 @@
|
||||
#include <openssl/evp.h>
|
||||
#include <openssl/params.h>
|
||||
#include <openssl/rand.h>
|
||||
#include <stdarg.h>
|
||||
#include <poll.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* One store entry: a username and its salted PBKDF2 verifier. The plaintext
|
||||
@@ -58,19 +59,37 @@ struct CredentialStore {
|
||||
static const uint8_t k_dummy_stored_key[CREDENTIAL_KEY_LEN] = {0};
|
||||
static const uint8_t k_dummy_server_key[CREDENTIAL_KEY_LEN] = {0};
|
||||
|
||||
static void set_error(char* err, size_t err_size, const char* fmt, ...) {
|
||||
if (!err || err_size == 0)
|
||||
return;
|
||||
va_list args;
|
||||
va_start(args, fmt);
|
||||
vsnprintf(err, err_size, fmt, args);
|
||||
va_end(args);
|
||||
}
|
||||
#define set_error utils_set_error
|
||||
|
||||
static bool is_comment_char(char c) {
|
||||
return c == '#' || c == ';';
|
||||
}
|
||||
|
||||
/* True for a literal fd-backed store path: exactly "/dev/fd/<digits>" or
|
||||
* "/proc/self/fd/<digits>", with no trailing component and no "..". These name
|
||||
* the calling process's own open descriptors (e.g. a bash process substitution
|
||||
* `<(...)`, which passes /dev/fd/N), and both prefixes are symlinks by
|
||||
* construction. */
|
||||
static bool is_fd_backed_path(const char* path) {
|
||||
static const char* const prefixes[] = {"/dev/fd/", "/proc/self/fd/"};
|
||||
if (!path)
|
||||
return false;
|
||||
for (size_t i = 0; i < sizeof(prefixes) / sizeof(prefixes[0]); i++) {
|
||||
const char* prefix = prefixes[i];
|
||||
size_t prefix_len = strlen(prefix);
|
||||
if (strncmp(path, prefix, prefix_len) != 0)
|
||||
continue;
|
||||
const char* digits = path + prefix_len;
|
||||
if (*digits < '0' || *digits > '9')
|
||||
return false;
|
||||
const char* p = digits;
|
||||
while (*p >= '0' && *p <= '9')
|
||||
p++;
|
||||
return *p == '\0';
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Open a --password-file / --early-input after verifying the EXACT inode we
|
||||
* will read: it must be owned by the effective user and grant no group/other
|
||||
* permission bit (so 0600 and stricter modes such as 0400 are accepted),
|
||||
@@ -80,12 +99,31 @@ static bool is_comment_char(char c) {
|
||||
* path and then fstat the resulting fd (rather than stat()ing the path first
|
||||
* and reopening it), so the permission decision is made on the same inode that
|
||||
* is read and cannot be raced by swapping the path between check and open.
|
||||
* The path may be a process-substitution pipe (`<(...)` -> /dev/fd/N), so
|
||||
* regular files and FIFOs are accepted when the ownership/mode checks pass.
|
||||
* O_NOFOLLOW refuses a symlinked path outright (ELOOP fails closed) instead of
|
||||
* following it before the owner/mode gate can run. The one exception is a
|
||||
* literal fd-backed path (/dev/fd/N or /proc/self/fd/N, see
|
||||
* is_fd_backed_path): those entries are symlinks to the CALLING process's own
|
||||
* descriptors, so following them is not the untrusted-symlink hazard
|
||||
* O_NOFOLLOW guards against, and requiring O_NOFOLLOW would break the
|
||||
* documented process-substitution/FIFO usage. For them only, O_NOFOLLOW is
|
||||
* omitted; the same fstat owner/mode gate still applies to the resolved inode.
|
||||
* O_NONBLOCK keeps the OPEN itself from
|
||||
* blocking forever on a writer-less FIFO (a blocking O_RDONLY open would wait
|
||||
* for a writer). The fd is left nonblocking for FIFOs so a read never blocks
|
||||
* either; the read loop (secret_read_line) absorbs the resulting EAGAIN by
|
||||
* waiting, under a bounded deadline, for the writer -- this is what makes a
|
||||
* slow process substitution (`--password-file <(sleep 1; ...)`) work while a
|
||||
* writer-less FIFO still fails after the deadline instead of hanging. Only
|
||||
* regular files and FIFOs pass the ownership/mode checks; O_NONBLOCK is
|
||||
* cleared for regular files, where it is a no-op anyway and no EAGAIN can
|
||||
* occur, so their stdio read path is byte-for-byte unchanged.
|
||||
*
|
||||
* Returns a FILE* the caller must fclose, or NULL with `err` filled. */
|
||||
static FILE* secret_file_open(const char* path, char* err, size_t err_size) {
|
||||
int fd = open(path, O_RDONLY | O_CLOEXEC);
|
||||
int flags = O_RDONLY | O_NONBLOCK | O_CLOEXEC;
|
||||
if (!is_fd_backed_path(path))
|
||||
flags |= O_NOFOLLOW;
|
||||
int fd = open(path, flags);
|
||||
if (fd < 0) {
|
||||
set_error(err, err_size, "cannot open secret file '%s': %s", path, strerror(errno));
|
||||
return NULL;
|
||||
@@ -105,6 +143,15 @@ static FILE* secret_file_open(const char* path, char* err, size_t err_size) {
|
||||
close(fd);
|
||||
return NULL;
|
||||
}
|
||||
/* O_NONBLOCK is only meaningful for the FIFO allowance. Restore blocking
|
||||
* mode on a regular file so its read path is exactly as before; a no-op on
|
||||
* most systems, but explicit. Failures here are ignored: O_NONBLOCK on a
|
||||
* regular file does not affect reads either way. */
|
||||
if (S_ISREG(st.st_mode)) {
|
||||
int status_flags = fcntl(fd, F_GETFL);
|
||||
if (status_flags >= 0)
|
||||
(void)fcntl(fd, F_SETFL, status_flags & ~O_NONBLOCK);
|
||||
}
|
||||
FILE* fp = fdopen(fd, "r");
|
||||
if (!fp) {
|
||||
set_error(err, err_size, "cannot read secret file '%s': %s", path, strerror(errno));
|
||||
@@ -114,6 +161,123 @@ static FILE* secret_file_open(const char* path, char* err, size_t err_size) {
|
||||
return fp;
|
||||
}
|
||||
|
||||
/* Overall bound on how long the reader waits for a process-substitution/FIFO
|
||||
* writer to produce data before giving up. It must comfortably exceed a
|
||||
* producer's startup delay (e.g. `--password-file <(sleep 1; ...)`) while still
|
||||
* bounding a writer-less FIFO, so a stray or hostile FIFO cannot stall the
|
||||
* daemon or client indefinitely. */
|
||||
#define CREDENTIAL_FIFO_READ_TIMEOUT_MS 3000
|
||||
|
||||
/* Monotonic milliseconds, used only for the read deadline (wall-clock changes
|
||||
* must not extend or shorten the wait). */
|
||||
static int64_t credential_monotonic_ms(void) {
|
||||
struct timespec ts;
|
||||
if (clock_gettime(CLOCK_MONOTONIC, &ts) != 0)
|
||||
return 0;
|
||||
return (int64_t)ts.tv_sec * 1000 + (int64_t)(ts.tv_nsec / 1000000);
|
||||
}
|
||||
|
||||
/* Wait until `fd` is readable or the deadline passes. Returns true when it is
|
||||
* readable, false on timeout or a poll error (err filled). EINTR is retried
|
||||
* against the same deadline, so signals cannot extend the wait. */
|
||||
static bool credential_wait_readable(int fd, int64_t deadline, const char* label, const char* path,
|
||||
char* err, size_t err_size) {
|
||||
for (;;) {
|
||||
int64_t remaining = deadline - credential_monotonic_ms();
|
||||
if (remaining <= 0)
|
||||
break;
|
||||
if (remaining > INT_MAX)
|
||||
remaining = INT_MAX;
|
||||
struct pollfd pfd = {.fd = fd, .events = POLLIN, .revents = 0};
|
||||
int rc = poll(&pfd, 1, (int)remaining);
|
||||
if (rc > 0)
|
||||
return true;
|
||||
if (rc == 0)
|
||||
break;
|
||||
if (errno != EINTR) {
|
||||
set_error(err, err_size, "error waiting for %s '%s': %s", label, path, strerror(errno));
|
||||
return false;
|
||||
}
|
||||
}
|
||||
set_error(err, err_size, "timed out after %d ms waiting for %s '%s'",
|
||||
CREDENTIAL_FIFO_READ_TIMEOUT_MS, label, path);
|
||||
return false;
|
||||
}
|
||||
|
||||
typedef enum {
|
||||
SECRET_READ_LINE,
|
||||
SECRET_READ_EOF,
|
||||
SECRET_READ_ERROR,
|
||||
} SecretReadResult;
|
||||
|
||||
/* Read one complete line from `fp` into `line` (capacity `cap`), including the
|
||||
* trailing newline when present and always NUL-terminating. `*out_len`
|
||||
* receives strlen(line).
|
||||
*
|
||||
* A regular file is read exactly as before: secret_file_open leaves it
|
||||
* blocking, so fgets never sees EAGAIN. A FIFO stays nonblocking, so fgets
|
||||
* returns NULL (or a partial line) with EAGAIN while the writer is still
|
||||
* starting up; instead of treating that as a fatal error the loop clearerr()s
|
||||
* and polls for readability against one overall deadline. The `used`
|
||||
* accumulator reassembles a line that arrived in several write()s into a single
|
||||
* line, so a split write is not misparsed as two entries.
|
||||
*
|
||||
* Returns SECRET_READ_LINE, SECRET_READ_EOF, or SECRET_READ_ERROR (err filled)
|
||||
* on timeout or a genuine read error. */
|
||||
static SecretReadResult secret_read_line(char* line, size_t cap, FILE* fp, const char* label,
|
||||
const char* path, size_t* out_len, char* err,
|
||||
size_t err_size) {
|
||||
int fd = fileno(fp);
|
||||
int64_t deadline = credential_monotonic_ms() + CREDENTIAL_FIFO_READ_TIMEOUT_MS;
|
||||
size_t used = 0;
|
||||
line[0] = '\0';
|
||||
for (;;) {
|
||||
errno = 0;
|
||||
if (fgets(line + used, (int)(cap - used), fp)) {
|
||||
used += strlen(line + used);
|
||||
if (used > 0 && line[used - 1] == '\n') {
|
||||
*out_len = used;
|
||||
return SECRET_READ_LINE;
|
||||
}
|
||||
if (feof(fp)) {
|
||||
*out_len = used; /* final unterminated line */
|
||||
return SECRET_READ_LINE;
|
||||
}
|
||||
/* No newline and not EOF. A full buffer is the caller's over-long-line
|
||||
* case; otherwise the line is only partially available (a nonblocking
|
||||
* FIFO under a slow writer), so any genuine read error fails and anything
|
||||
* else waits for the rest. */
|
||||
if (used >= cap - 1) {
|
||||
*out_len = used;
|
||||
return SECRET_READ_LINE;
|
||||
}
|
||||
int e = ferror(fp) ? errno : 0;
|
||||
if (e != 0 && e != EAGAIN && e != EWOULDBLOCK) {
|
||||
set_error(err, err_size, "error reading %s '%s': %s", label, path, strerror(e));
|
||||
return SECRET_READ_ERROR;
|
||||
}
|
||||
clearerr(fp);
|
||||
if (!credential_wait_readable(fd, deadline, label, path, err, err_size))
|
||||
return SECRET_READ_ERROR;
|
||||
continue;
|
||||
}
|
||||
/* fgets returned NULL: EOF, a not-yet-readable FIFO, or a real error. */
|
||||
if (feof(fp)) {
|
||||
*out_len = used;
|
||||
return used > 0 ? SECRET_READ_LINE : SECRET_READ_EOF;
|
||||
}
|
||||
if (errno == EAGAIN || errno == EWOULDBLOCK) {
|
||||
clearerr(fp);
|
||||
if (!credential_wait_readable(fd, deadline, label, path, err, err_size))
|
||||
return SECRET_READ_ERROR;
|
||||
continue;
|
||||
}
|
||||
set_error(err, err_size, "error reading %s '%s': %s", label, path,
|
||||
errno != 0 ? strerror(errno) : "read failed");
|
||||
return SECRET_READ_ERROR;
|
||||
}
|
||||
}
|
||||
|
||||
/* Trim leading/trailing ASCII space and tab in place; returns the new start. */
|
||||
static char* trim_space(char* s) {
|
||||
while (*s == ' ' || *s == '\t')
|
||||
@@ -158,13 +322,13 @@ static int hex_value(char c) {
|
||||
* digits. Such a line is refused loudly (and never accepted) so an operator
|
||||
* cannot keep a replayable bearer digest in place after the protocol bump. */
|
||||
static bool secret_is_legacy_hex(const char* s) {
|
||||
if (!s)
|
||||
if (!s || strlen(s) != 64)
|
||||
return false;
|
||||
for (int i = 0; i < 64; i++) {
|
||||
if (hex_value(s[i]) < 0)
|
||||
return false;
|
||||
}
|
||||
return s[64] == '\0';
|
||||
return true;
|
||||
}
|
||||
|
||||
bool credentials_b64_encode(const uint8_t* in, size_t n, char* out, size_t out_sz) {
|
||||
@@ -509,9 +673,17 @@ static CredentialStore* load_store_file(const char* path, char* err, size_t err_
|
||||
char line[CREDENTIAL_MAX_LINE + 2];
|
||||
bool ok = true;
|
||||
|
||||
while (fgets(line, sizeof(line), fp)) {
|
||||
for (;;) {
|
||||
size_t len = 0;
|
||||
SecretReadResult rr =
|
||||
secret_read_line(line, sizeof(line), fp, "credential file", path, &len, err, err_size);
|
||||
if (rr == SECRET_READ_EOF)
|
||||
break;
|
||||
if (rr == SECRET_READ_ERROR) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
line_no++;
|
||||
size_t len = strlen(line);
|
||||
if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) {
|
||||
set_error(err, err_size, "credential file '%s' line %d exceeds the %d-byte limit", path,
|
||||
line_no, CREDENTIAL_MAX_LINE);
|
||||
@@ -1148,9 +1320,17 @@ int credentials_hash_file(const char* path, uint32_t iters, FILE* out, char* err
|
||||
int line_no = 0;
|
||||
int result = 0;
|
||||
char line[CREDENTIAL_MAX_LINE + 2];
|
||||
while (fgets(line, sizeof(line), fp)) {
|
||||
for (;;) {
|
||||
size_t len = 0;
|
||||
SecretReadResult rr =
|
||||
secret_read_line(line, sizeof(line), fp, "plaintext file", path, &len, err, err_size);
|
||||
if (rr == SECRET_READ_EOF)
|
||||
break;
|
||||
if (rr == SECRET_READ_ERROR) {
|
||||
result = -1;
|
||||
break;
|
||||
}
|
||||
line_no++;
|
||||
size_t len = strlen(line);
|
||||
if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) {
|
||||
set_error(err, err_size, "plaintext file '%s' line %d exceeds the %d-byte limit", path,
|
||||
line_no, CREDENTIAL_MAX_LINE);
|
||||
@@ -1228,9 +1408,15 @@ int credentials_read_secret_file(const char* path, char** user_out, char** passw
|
||||
char line[CREDENTIAL_MAX_LINE + 2];
|
||||
int result = -1;
|
||||
|
||||
while (fgets(line, sizeof(line), fp)) {
|
||||
for (;;) {
|
||||
size_t len = 0;
|
||||
SecretReadResult rr =
|
||||
secret_read_line(line, sizeof(line), fp, "password file", path, &len, err, err_size);
|
||||
if (rr == SECRET_READ_EOF)
|
||||
break;
|
||||
if (rr == SECRET_READ_ERROR)
|
||||
goto done;
|
||||
line_no++;
|
||||
size_t len = strlen(line);
|
||||
if (len == CREDENTIAL_MAX_LINE + 1 && line[len - 1] != '\n' && !feof(fp)) {
|
||||
set_error(err, err_size, "password file '%s' line %d exceeds the %d-byte limit", path,
|
||||
line_no, CREDENTIAL_MAX_LINE);
|
||||
|
||||
+283
-13
@@ -1,12 +1,12 @@
|
||||
#include "daemon_conf.h"
|
||||
#include "credentials.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <ctype.h>
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <netinet/in.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
@@ -17,14 +17,7 @@
|
||||
/* helpers */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
static void set_error(char* err, size_t err_size, const char* fmt, ...) {
|
||||
if (!err || err_size == 0)
|
||||
return;
|
||||
va_list args;
|
||||
va_start(args, fmt);
|
||||
vsnprintf(err, err_size, fmt, args);
|
||||
va_end(args);
|
||||
}
|
||||
#define set_error utils_set_error
|
||||
|
||||
/* Trim leading and trailing ASCII space/tab in place; returns the new start. */
|
||||
static char* trim_ws(char* s) {
|
||||
@@ -41,6 +34,158 @@ static bool key_equals(const char* key, const char* canonical) {
|
||||
return strcasecmp(key, canonical) == 0;
|
||||
}
|
||||
|
||||
/* True when `key` matches one of the NUL-terminated names in `list`. */
|
||||
static bool key_in_list(const char* key, const char* const* list, size_t count) {
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
if (strcasecmp(key, list[i]) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* rsync 3.4.1 rsyncd.conf GLOBAL keys accepted in the pre-module section that
|
||||
* have no FastSync equivalent. They are recognized and documented as inert:
|
||||
* accepting a real rsync config must not fail on a logging/process key, but a
|
||||
* silently-reinterpreted key is never invented. `pidfile`/`logfile` are the
|
||||
* compact --dparam spellings rsync documents. The same list is used by the
|
||||
* `--dparam` dispatch (apply_global_key), so there is a single impl. */
|
||||
static const char* const kRsyncInertGlobalKeys[] = {
|
||||
"pid file",
|
||||
"pidfile",
|
||||
"log file",
|
||||
"logfile",
|
||||
"socket options",
|
||||
"sockopts",
|
||||
"listen backlog",
|
||||
"syslog facility",
|
||||
"syslog tag",
|
||||
"log format",
|
||||
"use chroot",
|
||||
"uid",
|
||||
"gid",
|
||||
"timeout",
|
||||
"max verbosity",
|
||||
"min verbosity",
|
||||
"lock file",
|
||||
"transfer logging",
|
||||
"strict modes",
|
||||
"reverse lookup",
|
||||
"forward lookup",
|
||||
"ignore errors",
|
||||
"ignore nonreadable",
|
||||
"dont compress",
|
||||
};
|
||||
|
||||
/* rsync 3.4.1 rsyncd.conf MODULE keys accepted in a [module] section that have
|
||||
* no FastSync equivalent (accepted-and-documented inert). Keys with a FastSync
|
||||
* meaning (`path`, `read only`, `write only`, `auth users`, `max connections`,
|
||||
* `hosts allow`/`hosts deny`, `client owner`) are handled by apply_module_key
|
||||
* before this list is consulted. Security-relevant keys (`exclude`, `filter`,
|
||||
* `secrets file`, `refuse options`, ...) are inert, so a daemon-side filter or
|
||||
* rsync secrets file is NOT enforced: each is loudly warned about at load time
|
||||
* (see kRsyncUnenforcedModuleSecurityKeys) and documented as a residual in
|
||||
* RSYNC_COMPAT.md. */
|
||||
static const char* const kRsyncInertModuleKeys[] = {
|
||||
"comment",
|
||||
"use chroot",
|
||||
"daemon chroot",
|
||||
"uid",
|
||||
"gid",
|
||||
"daemon uid",
|
||||
"daemon gid",
|
||||
"exclude",
|
||||
"include",
|
||||
"exclude from",
|
||||
"include from",
|
||||
"filter",
|
||||
"max verbosity",
|
||||
"min verbosity",
|
||||
"lock file",
|
||||
"transfer logging",
|
||||
"log file",
|
||||
"log format",
|
||||
"syslog facility",
|
||||
"syslog tag",
|
||||
"timeout",
|
||||
"secrets file",
|
||||
"auth digest",
|
||||
"strict modes",
|
||||
"numeric ids",
|
||||
"fake super",
|
||||
"munge symlinks",
|
||||
"list",
|
||||
"dont compress",
|
||||
"charset",
|
||||
"refuse options",
|
||||
"incoming chmod",
|
||||
"outgoing chmod",
|
||||
"open noatime",
|
||||
"max size",
|
||||
"min size",
|
||||
"temp dir",
|
||||
"pre-xfer exec",
|
||||
"post-xfer exec",
|
||||
"name converter",
|
||||
"proxy protocol",
|
||||
"proxy protocol hosts",
|
||||
"reverse lookup",
|
||||
"forward lookup",
|
||||
"ignore errors",
|
||||
"ignore nonreadable",
|
||||
};
|
||||
|
||||
/* Subset of the inert rsync keys whose intent is access control (data
|
||||
* visibility, credential source, transfer hooks, daemon privilege), plus the
|
||||
* global keys that shape the daemon's privilege/identity. These load for
|
||||
* rsync-config compatibility, but because FastSync ignores them an operator
|
||||
* migrating a hardened rsyncd.conf must not believe the restriction applies.
|
||||
* The loader emits one LOG_LEVEL_WARNING per occurrence naming the key (and the
|
||||
* module, for a module key). `write only` is deliberately absent: it is mapped
|
||||
* onto writability instead (FastSync is push-only, so a write-only module is
|
||||
* simply writable). */
|
||||
static const char* const kRsyncUnenforcedModuleSecurityKeys[] = {
|
||||
"secrets file",
|
||||
"auth digest",
|
||||
"refuse options",
|
||||
"exclude",
|
||||
"include",
|
||||
"exclude from",
|
||||
"include from",
|
||||
"filter",
|
||||
"max size",
|
||||
"min size",
|
||||
"pre-xfer exec",
|
||||
"post-xfer exec",
|
||||
"incoming chmod",
|
||||
"outgoing chmod",
|
||||
"name converter",
|
||||
"use chroot",
|
||||
"daemon chroot",
|
||||
"uid",
|
||||
"gid",
|
||||
"daemon uid",
|
||||
"daemon gid",
|
||||
"munge symlinks",
|
||||
"fake super",
|
||||
"strict modes",
|
||||
"proxy protocol",
|
||||
"proxy protocol hosts",
|
||||
};
|
||||
|
||||
static const char* const kRsyncUnenforcedGlobalSecurityKeys[] = {
|
||||
"use chroot",
|
||||
"uid",
|
||||
"gid",
|
||||
"strict modes",
|
||||
};
|
||||
|
||||
#define kRsyncInertGlobalCount (sizeof(kRsyncInertGlobalKeys) / sizeof(kRsyncInertGlobalKeys[0]))
|
||||
#define kRsyncInertModuleCount (sizeof(kRsyncInertModuleKeys) / sizeof(kRsyncInertModuleKeys[0]))
|
||||
#define kRsyncUnenforcedModuleSecurityCount \
|
||||
(sizeof(kRsyncUnenforcedModuleSecurityKeys) / sizeof(kRsyncUnenforcedModuleSecurityKeys[0]))
|
||||
#define kRsyncUnenforcedGlobalSecurityCount \
|
||||
(sizeof(kRsyncUnenforcedGlobalSecurityKeys) / sizeof(kRsyncUnenforcedGlobalSecurityKeys[0]))
|
||||
|
||||
static bool parse_bool_value(const char* value, bool* out) {
|
||||
if (strcasecmp(value, "yes") == 0 || strcasecmp(value, "true") == 0 || strcmp(value, "1") == 0) {
|
||||
*out = true;
|
||||
@@ -133,6 +278,7 @@ static bool store_host_list(char*** list, int* count, const char* value, const c
|
||||
return false;
|
||||
}
|
||||
char* save = NULL;
|
||||
int added = 0;
|
||||
for (char* token = strtok_r(copy, ", \t", &save); token; token = strtok_r(NULL, ", \t", &save)) {
|
||||
if (!host_pattern_valid(token)) {
|
||||
if (module_name)
|
||||
@@ -163,8 +309,20 @@ static bool store_host_list(char*** list, int* count, const char* value, const c
|
||||
return false;
|
||||
}
|
||||
(*list)[(*count)++] = dup;
|
||||
added++;
|
||||
}
|
||||
free(copy);
|
||||
/* A present key with an empty (or separator-only) value would otherwise
|
||||
* install a zero-length list, i.e. no ACL at all: a strict-parse config must
|
||||
* never silently turn a restrictive directive into "allow everyone". */
|
||||
if (added == 0) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': '%s' must list at least one host pattern", module_name,
|
||||
key);
|
||||
else
|
||||
set_error(err, err_size, "'%s' must list at least one host pattern", key);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -190,6 +348,26 @@ static bool store_max_connections(int* slot, const char* value, const char* modu
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse a non-negative concurrency cap where 0 means unlimited/disabled
|
||||
* (per-module `max connections`, `max connections per host`,
|
||||
* `auth lockout threshold`). Negative/garbage/oversized values are rejected. */
|
||||
static bool store_optional_cap(int* slot, const char* value, int max_value, const char* key,
|
||||
const char* module_name, char* err, size_t err_size) {
|
||||
char* end = NULL;
|
||||
errno = 0;
|
||||
long n = strtol(value, &end, 10);
|
||||
if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 || n > max_value) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': invalid '%s' '%s' (must be 0-%d)", module_name, key,
|
||||
value, max_value);
|
||||
else
|
||||
set_error(err, err_size, "invalid '%s' '%s' (must be 0-%d)", key, value, max_value);
|
||||
return false;
|
||||
}
|
||||
*slot = (int)n;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse an `auth failure delay` value: 0 (disabled) through the configured cap. */
|
||||
static bool store_auth_failure_delay(int* slot, const char* value, char* err, size_t err_size) {
|
||||
char* end = NULL;
|
||||
@@ -225,8 +403,15 @@ DaemonConf* daemon_conf_create(void) {
|
||||
if (!conf)
|
||||
return NULL;
|
||||
conf->global.port = DAEMON_CONF_DEFAULT_PORT;
|
||||
/* rsync modules are READ-ONLY unless `read only = no` (or `write only = yes`)
|
||||
* is set, so FastSync must default the same way: a migrated rsyncd.conf that
|
||||
* omits `read only` is served read-only, never writable. */
|
||||
conf->global.read_only_default = true;
|
||||
conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS;
|
||||
conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS;
|
||||
conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST;
|
||||
conf->global.auth_lockout_threshold = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD;
|
||||
conf->global.auth_lockout_duration_sec = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC;
|
||||
return conf;
|
||||
}
|
||||
|
||||
@@ -296,7 +481,7 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo
|
||||
char* err, size_t err_size) {
|
||||
if (key_equals(key, "port"))
|
||||
return store_port(&conf->global.port, value, err, err_size);
|
||||
if (key_equals(key, "motd file")) {
|
||||
if (key_equals(key, "motd file") || key_equals(key, "motdfile")) {
|
||||
if (!store_string(&conf->global.motd_file, value)) {
|
||||
set_error(err, err_size, "out of memory parsing 'motd file'");
|
||||
return false;
|
||||
@@ -310,16 +495,56 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, boo
|
||||
}
|
||||
return true;
|
||||
}
|
||||
/* rsync allows the `read only` module key in the global section as the
|
||||
* default for modules defined after it. Map it to that default (a later
|
||||
* --dparam re-applies it to modules that did not set their own value) so a
|
||||
* global `read only = yes` cannot be silently dropped into a writable
|
||||
* default. */
|
||||
if (key_equals(key, "read only")) {
|
||||
bool parsed;
|
||||
if (!parse_bool_value(value, &parsed)) {
|
||||
set_error(err, err_size, "global 'read only' must be yes/no (or true/false/1/0), got '%s'",
|
||||
value);
|
||||
return false;
|
||||
}
|
||||
conf->global.read_only_default = parsed;
|
||||
for (int i = 0; i < conf->module_count; i++) {
|
||||
if (!conf->modules[i].read_only_explicit)
|
||||
conf->modules[i].read_only = parsed;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (key_equals(key, "max connections"))
|
||||
return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size);
|
||||
if (key_equals(key, "max connections per host"))
|
||||
return store_optional_cap(&conf->global.max_connections_per_host, value,
|
||||
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "max connections per host", NULL,
|
||||
err, err_size);
|
||||
if (key_equals(key, "auth failure delay"))
|
||||
return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size);
|
||||
if (key_equals(key, "auth lockout threshold"))
|
||||
return store_optional_cap(&conf->global.auth_lockout_threshold, value,
|
||||
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "auth lockout threshold", NULL,
|
||||
err, err_size);
|
||||
if (key_equals(key, "auth lockout duration"))
|
||||
return store_optional_cap(&conf->global.auth_lockout_duration_sec, value,
|
||||
DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC, "auth lockout duration",
|
||||
NULL, err, err_size);
|
||||
if (key_equals(key, "hosts allow"))
|
||||
return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value,
|
||||
"hosts allow", NULL, replace_hosts, err, err_size);
|
||||
if (key_equals(key, "hosts deny"))
|
||||
return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value,
|
||||
"hosts deny", NULL, replace_hosts, err, err_size);
|
||||
/* A recognized rsync global key with no FastSync equivalent loads inert. */
|
||||
if (key_in_list(key, kRsyncInertGlobalKeys, kRsyncInertGlobalCount)) {
|
||||
if (key_in_list(key, kRsyncUnenforcedGlobalSecurityKeys, kRsyncUnenforcedGlobalSecurityCount))
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"daemon config: global key '%s' is accepted for rsync compatibility but is NOT "
|
||||
"enforced by FastSync; the restriction it expresses will not be applied",
|
||||
key);
|
||||
return true;
|
||||
}
|
||||
set_error(err, err_size, "unknown global key '%s'", key);
|
||||
return false;
|
||||
}
|
||||
@@ -348,6 +573,26 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char*
|
||||
return false;
|
||||
}
|
||||
module->read_only = parsed;
|
||||
module->read_only_explicit = true;
|
||||
return true;
|
||||
}
|
||||
/* rsync's `write only = yes` makes the module client-writable. FastSync has
|
||||
* no read/pull path, so mapping it to writability is the exact
|
||||
* security-relevant effect; set `read_only_explicit` so a global default
|
||||
* cannot override the module's explicit choice. `write only = no` is the
|
||||
* rsync default and leaves the module's read-only state untouched. */
|
||||
if (key_equals(key, "write only")) {
|
||||
bool parsed;
|
||||
if (!parse_bool_value(value, &parsed)) {
|
||||
set_error(err, err_size,
|
||||
"module '%s': 'write only' must be yes/no (or true/false/1/0), got '%s'",
|
||||
module->name, value);
|
||||
return false;
|
||||
}
|
||||
if (parsed) {
|
||||
module->read_only = false;
|
||||
module->read_only_explicit = true;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (key_equals(key, "client owner")) {
|
||||
@@ -368,6 +613,7 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char*
|
||||
return false;
|
||||
}
|
||||
char* save = NULL;
|
||||
int added = 0;
|
||||
for (char* token = strtok_r(list, ",", &save); token; token = strtok_r(NULL, ",", &save)) {
|
||||
const char* user = trim_ws(token);
|
||||
if (*user == '\0')
|
||||
@@ -395,18 +641,36 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char*
|
||||
return false;
|
||||
}
|
||||
module->auth_users[module->auth_user_count++] = dup;
|
||||
added++;
|
||||
}
|
||||
free(list);
|
||||
/* An empty/separator-only value must not silently disable authentication:
|
||||
* the key's presence is an explicit request for an allow-list. */
|
||||
if (added == 0) {
|
||||
set_error(err, err_size, "module '%s': 'auth users' must list at least one user",
|
||||
module->name);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (key_equals(key, "max connections"))
|
||||
return store_max_connections(&module->max_connections, value, module->name, err, err_size);
|
||||
return store_optional_cap(&module->max_connections, value, DAEMON_CONF_MAX_CONCURRENCY_LIMIT,
|
||||
"max connections", module->name, err, err_size);
|
||||
if (key_equals(key, "hosts allow"))
|
||||
return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow",
|
||||
false, module->name, err, err_size);
|
||||
module->name, false, err, err_size);
|
||||
if (key_equals(key, "hosts deny"))
|
||||
return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny",
|
||||
false, module->name, err, err_size);
|
||||
module->name, false, err, err_size);
|
||||
/* A recognized rsync module key with no FastSync equivalent loads inert. */
|
||||
if (key_in_list(key, kRsyncInertModuleKeys, kRsyncInertModuleCount)) {
|
||||
if (key_in_list(key, kRsyncUnenforcedModuleSecurityKeys, kRsyncUnenforcedModuleSecurityCount))
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"daemon config: module '%s' key '%s' is accepted for rsync compatibility but is "
|
||||
"NOT enforced by FastSync; the restriction it expresses will not be applied",
|
||||
module->name, key);
|
||||
return true;
|
||||
}
|
||||
set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name);
|
||||
return false;
|
||||
}
|
||||
@@ -444,6 +708,11 @@ static int open_module(DaemonConf* conf, int* current_module, const char* name,
|
||||
set_error(err, err_size, "duplicate module '%s'", name);
|
||||
return -1;
|
||||
}
|
||||
if (conf->module_count >= DAEMON_CONF_MAX_MODULES) {
|
||||
set_error(err, err_size, "too many modules (limit %d); module '%s' rejected",
|
||||
DAEMON_CONF_MAX_MODULES, name);
|
||||
return -1;
|
||||
}
|
||||
DaemonModule* grown =
|
||||
realloc(conf->modules, (size_t)(conf->module_count + 1) * sizeof(DaemonModule));
|
||||
if (!grown) {
|
||||
@@ -452,6 +721,7 @@ static int open_module(DaemonConf* conf, int* current_module, const char* name,
|
||||
}
|
||||
conf->modules = grown;
|
||||
memset(&conf->modules[conf->module_count], 0, sizeof(DaemonModule));
|
||||
conf->modules[conf->module_count].read_only = conf->global.read_only_default;
|
||||
conf->modules[conf->module_count].name = str_dup(name);
|
||||
if (!conf->modules[conf->module_count].name) {
|
||||
set_error(err, err_size, "out of memory adding module '%s'", name);
|
||||
|
||||
+79
-27
@@ -17,7 +17,20 @@
|
||||
* DAEMON_CONF_MAX_LINE all fail the whole load with a clear, line-numbered
|
||||
* error instead of being silently ignored. This keeps a typo from silently
|
||||
* changing what a module serves.
|
||||
*/
|
||||
*
|
||||
* rsync compatibility: to reduce the divergence from rsync 3.4.1's rsyncd.conf
|
||||
* grammar, the parser also ACCEPTS the common rsync GLOBAL and MODULE keys.
|
||||
* Keys with a FastSync equivalent are mapped onto it (the native spellings are
|
||||
* unchanged; `read only` defaults to yes like rsync, and `write only = yes`
|
||||
* opts a module into writability). Keys with no FastSync equivalent are
|
||||
* accepted and documented as inert (they load successfully but have no effect)
|
||||
* rather than failing the whole config; the accepted inert set is listed in
|
||||
* kRsyncInertGlobalKeys / kRsyncInertModuleKeys in daemon_conf.c and in
|
||||
* RSYNC_COMPAT.md. Every inert key whose intent is access control is loudly
|
||||
* warned about at load time (kRsyncUnenforced*SecurityKeys) so an operator
|
||||
* migrating a hardened rsyncd.conf is never misled into believing the
|
||||
* restriction is enforced. A key outside both the FastSync-native grammar and
|
||||
* the recognized rsync subset is still rejected as unknown. */
|
||||
|
||||
/* A daemon module's configured root is used exactly like the standalone
|
||||
* server's --destination-root: the daemon confines every connection that
|
||||
@@ -42,21 +55,26 @@
|
||||
* store refuses (fail closed) rather than falling open; see server.c. Auth is
|
||||
* never bypassed by ignoring the list. */
|
||||
typedef struct DaemonModule {
|
||||
char* name; /* module name, as the client requests it */
|
||||
char* path; /* module root (daemon-side authorized root) */
|
||||
bool read_only; /* `read only = yes/no`; default no */
|
||||
bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in
|
||||
that lets this module's clients choose ownership
|
||||
(--numeric-ids/--chown/--usermap/--groupmap/--fake-super/
|
||||
--copy-as) and request explicit --super super-user
|
||||
activities. Without it the daemon refuses all of them. */
|
||||
char** auth_users; /* `auth users = a,b`; Wave B credential list */
|
||||
char* name; /* module name, as the client requests it */
|
||||
char* path; /* module root (daemon-side authorized root) */
|
||||
bool read_only; /* `read only = yes/no`; defaults to the global `read only`
|
||||
default (rsync allows it in the global section), which is
|
||||
itself default YES (rsync modules are read-only unless
|
||||
`read only = no` / `write only = yes` opts in) */
|
||||
bool read_only_explicit; /* set when this module set its own `read only` or
|
||||
`write only = yes`, so a later global default (from a
|
||||
`--dparam read only=`) does not override it */
|
||||
bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in
|
||||
that lets this module's clients choose ownership
|
||||
(--numeric-ids/--chown/--usermap/--groupmap/--fake-super/
|
||||
--copy-as) and request explicit --super super-user
|
||||
activities. Without it the daemon refuses all of them. */
|
||||
char** auth_users; /* `auth users = a,b`; Wave B credential list */
|
||||
int auth_user_count;
|
||||
/* `max connections = N` (optional per-module cap). 0 means "not set"
|
||||
* (inherit the global cap). Parsed, stored, and validated, but NOT enforced
|
||||
* per-module: connections are counted in the accept-loop parent before the
|
||||
* client's module is known, so only the global cap is enforced (see
|
||||
* transport_tcp.c and the Daemon Mode notes in RSYNC_COMPAT.md). */
|
||||
/* `max connections = N` (optional per-module cap). 0 means unlimited. The
|
||||
* per-connection child records the selected module in the shared registry
|
||||
* (daemon_limits.c) once the config frame names it, so the cap is enforced
|
||||
* across all forked children; the parent reclaims the slot on SIGCHLD. */
|
||||
int max_connections;
|
||||
char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
@@ -67,14 +85,29 @@ typedef struct DaemonModule {
|
||||
/* Global (pre-module) scalar keys. `motd file` is parsed and stored but has
|
||||
* no wire effect yet (MOTD display is Wave C). */
|
||||
typedef struct DaemonConfGlobals {
|
||||
int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */
|
||||
char* motd_file; /* `motd file`, may be NULL */
|
||||
char* address; /* `address` (optional bind address), may be NULL */
|
||||
int max_connections; /* `max connections`, default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */
|
||||
int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */
|
||||
char** hosts_allow; /* `hosts allow`; global host access allow patterns */
|
||||
int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */
|
||||
char* motd_file; /* `motd file`, may be NULL */
|
||||
char* address; /* `address` (optional bind address), may be NULL */
|
||||
bool read_only_default; /* global `read only` default for modules defined
|
||||
after it (rsync allows the module key in the
|
||||
global section); default YES to match rsync's
|
||||
read-only modules */
|
||||
int max_connections; /* `max connections`, default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */
|
||||
int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */
|
||||
int max_connections_per_host; /* `max connections per host`, concurrent cap per
|
||||
source IP; default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST (0 =
|
||||
unlimited) */
|
||||
int auth_lockout_threshold; /* `auth lockout threshold`, failed attempts from
|
||||
one source before lockout; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD (0
|
||||
disables) */
|
||||
int auth_lockout_duration_sec; /* `auth lockout duration`, seconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC
|
||||
(0 disables) */
|
||||
char** hosts_allow; /* `hosts allow`; global host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
char** hosts_deny; /* `hosts deny`; global host access deny patterns */
|
||||
int hosts_deny_count;
|
||||
@@ -92,11 +125,26 @@ typedef struct DaemonConf {
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100
|
||||
/* Default `auth failure delay` in milliseconds (0 disables the throttle). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500
|
||||
/* Default `max connections per host` (0 = unlimited). */
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST 0
|
||||
/* Default cross-process auth lockout: 10 failed attempts from one source lock
|
||||
* it out for 300 s (0 disables either knob). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD 10
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC 300
|
||||
/* Upper bound on a `max connections per host` or `auth lockout threshold`
|
||||
* value, so a typo cannot size the shared registry absurdly. */
|
||||
#define DAEMON_CONF_MAX_CONCURRENCY_LIMIT 1000000
|
||||
/* Upper bound on `auth lockout duration` (7 days). */
|
||||
#define DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC 604800
|
||||
/* Largest accepted `auth failure delay`, so a typo cannot pin a connection
|
||||
* child in nanosleep for an absurd time. */
|
||||
/* Bounded well below the socket I/O timeout so a failed-auth child cannot hold
|
||||
* a connection slot for long enough to amplify connection-cap exhaustion. */
|
||||
#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000
|
||||
/* Upper bound on the number of [module] sections, so the shared registry's
|
||||
* per-module counter array stays fixed-size. The parser rejects the next
|
||||
* section past this bound. */
|
||||
#define DAEMON_CONF_MAX_MODULES 256
|
||||
/* Longest accepted config line (excluding the trailing newline). Longer lines
|
||||
* are rejected rather than buffered unboundedly. */
|
||||
#define DAEMON_CONF_MAX_LINE 4096
|
||||
@@ -127,10 +175,14 @@ const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char*
|
||||
bool daemon_module_name_valid(const char* name);
|
||||
|
||||
/* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and
|
||||
* apply it to the global keys only. Keys are case-insensitive and limited to
|
||||
* the global keys defined by the grammar (port, motd file, address,
|
||||
* max connections, auth failure delay, hosts allow, hosts deny). Returns 0 on
|
||||
* success, -1 on error (err filled). */
|
||||
* apply it to the global keys only. Keys are case-insensitive and cover the
|
||||
* global keys defined by the grammar (port, motd file, address, read only,
|
||||
* max connections, max connections per host, auth failure delay,
|
||||
* auth lockout threshold, auth lockout duration, hosts allow, hosts deny) plus
|
||||
* the recognized inert rsync global keys and the compact rsync spellings
|
||||
* (`motdfile`, `pidfile`, `logfile`). Applying `read only` sets the global
|
||||
* default and re-applies it to every module that did not set its own value.
|
||||
* Returns 0 on success, -1 on error (err filled). */
|
||||
int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size);
|
||||
|
||||
/* Host access-control matching (pure; no I/O). `daemon_host_pattern_match`
|
||||
|
||||
@@ -0,0 +1,494 @@
|
||||
#include "daemon_limits.h"
|
||||
#include "daemon_conf.h"
|
||||
#include "log.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <netinet/in.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <time.h>
|
||||
|
||||
/* The two module-count bounds must agree: the daemon config parser never
|
||||
* produces more than DAEMON_CONF_MAX_MODULES modules, so the shared registry's
|
||||
* per-module counter array is sized from the same bound. */
|
||||
_Static_assert(DAEMON_LIMITS_MAX_MODULES == DAEMON_CONF_MAX_MODULES,
|
||||
"daemon_limits module bound must match daemon_conf");
|
||||
|
||||
/* Slot lifecycle states (stored in slot_state). */
|
||||
enum {
|
||||
SLOT_FREE = 0,
|
||||
SLOT_CLAIMED = 1,
|
||||
SLOT_REGISTERED = 2,
|
||||
};
|
||||
|
||||
/* The registry header lives at the base of the shared mapping; the pointer
|
||||
* fields point at the arrays carved out of the same mapping. Absolute pointers
|
||||
* remain valid in a forked child because fork() clones the address space and
|
||||
* mapping, so parent and child observe the same virtual addresses. */
|
||||
struct DaemonLimitRegistry {
|
||||
int max_slots;
|
||||
int module_count;
|
||||
int host_slots; /* power of two; 1 when no per-source tracking is needed */
|
||||
int per_host_cap;
|
||||
int lockout_threshold;
|
||||
int lockout_duration_sec;
|
||||
size_t map_size;
|
||||
_Atomic long long host_full_warn; /* last "table full" warning epoch */
|
||||
_Atomic int* slot_state;
|
||||
_Atomic int* slot_pid;
|
||||
_Atomic int* slot_module;
|
||||
_Atomic int* slot_host; /* per-source table bucket, or -1 */
|
||||
_Atomic int* module_active;
|
||||
_Atomic uint64_t* host_key; /* 0 == empty bucket */
|
||||
_Atomic int* host_active;
|
||||
_Atomic int* host_fail;
|
||||
_Atomic long long* host_until; /* epoch seconds the lockout expires */
|
||||
_Atomic long long* host_last_use; /* epoch seconds the bucket was last touched */
|
||||
};
|
||||
|
||||
static size_t round_up(size_t n, size_t align) {
|
||||
return (n + align - 1) & ~(align - 1);
|
||||
}
|
||||
|
||||
static size_t next_pow2(size_t n) {
|
||||
size_t p = 1;
|
||||
while (p < n)
|
||||
p <<= 1;
|
||||
return p;
|
||||
}
|
||||
|
||||
/* Parse a numeric IPv4/IPv6 peer string into family + raw bytes. */
|
||||
static bool parse_peer_ip(const char* peer_ip, int* family, unsigned char* bytes) {
|
||||
if (!peer_ip || *peer_ip == '\0')
|
||||
return false;
|
||||
struct in_addr v4;
|
||||
if (inet_pton(AF_INET, peer_ip, &v4) == 1) {
|
||||
memcpy(bytes, &v4, sizeof(v4));
|
||||
*family = AF_INET;
|
||||
return true;
|
||||
}
|
||||
struct in6_addr v6;
|
||||
if (inet_pton(AF_INET6, peer_ip, &v6) == 1) {
|
||||
memcpy(bytes, &v6, sizeof(v6));
|
||||
*family = AF_INET6;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) {
|
||||
if (ok)
|
||||
*ok = false;
|
||||
unsigned char bytes[16];
|
||||
int family = AF_UNSPEC;
|
||||
if (!parse_peer_ip(peer_ip, &family, bytes))
|
||||
return 0;
|
||||
uint64_t hash = 14695981039346656037ULL ^ (uint64_t)(uint32_t)family;
|
||||
size_t length = family == AF_INET ? 4 : 16;
|
||||
for (size_t i = 0; i < length; i++) {
|
||||
hash ^= bytes[i];
|
||||
hash *= 1099511628211ULL;
|
||||
}
|
||||
if (hash == 0)
|
||||
hash = 0x9e3779b97f4a7c15ULL;
|
||||
if (ok)
|
||||
*ok = true;
|
||||
return hash;
|
||||
}
|
||||
|
||||
/* True when the registry must maintain per-source buckets: either the per-host
|
||||
* cap is configured, or the auth lockout is (threshold AND duration > 0). A
|
||||
* lockout threshold without a duration is a no-op, so it must not size or intern
|
||||
* the table. create(), register() and the lockout paths all agree on this. */
|
||||
static bool registry_tracks_hosts(const DaemonLimitRegistry* registry) {
|
||||
return registry->per_host_cap > 0 ||
|
||||
(registry->lockout_threshold > 0 && registry->lockout_duration_sec > 0);
|
||||
}
|
||||
|
||||
/* Find the bucket holding `peer_ip`, or -1 when it has no entry. Finding a
|
||||
* bucket refreshes its last-use time so the eviction policy sees it as live. */
|
||||
static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], (long long)time(NULL),
|
||||
memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0)
|
||||
return -1; /* no tombstones: an empty bucket ends the probe chain */
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* A bucket with no live connection may be repurposed: immediately when its
|
||||
* lockout deadline has already passed (the review's "expired" case), or after an
|
||||
* idle window when it holds no pending lockout. A bucket with a future lockout
|
||||
* deadline is retained so the lockout actually lasts its configured duration. */
|
||||
static bool host_bucket_reclaimable(DaemonLimitRegistry* registry, size_t idx, long long now) {
|
||||
if (atomic_load_explicit(®istry->host_active[idx], memory_order_relaxed) != 0)
|
||||
return false;
|
||||
long long until = atomic_load_explicit(®istry->host_until[idx], memory_order_relaxed);
|
||||
if (until != 0)
|
||||
return until <= now;
|
||||
long long last_use = atomic_load_explicit(®istry->host_last_use[idx], memory_order_relaxed);
|
||||
/* A bucket whose key is published but whose last_use has not yet been stamped
|
||||
* (last_use == 0) must be treated as live: reclaiming it here would steal a
|
||||
* bucket a racing child just claimed. The claim path also stamps last_use
|
||||
* before publishing the key, so this window cannot persist. */
|
||||
return last_use != 0 && now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC;
|
||||
}
|
||||
|
||||
/* Emit at most one "per-source table full" warning per
|
||||
* DAEMON_LIMITS_HOST_FULL_WARN_SEC across all forked children. Called from a
|
||||
* normal (non-signal) child path, so logging is safe here. */
|
||||
static void host_warn_table_full(DaemonLimitRegistry* registry, long long now) {
|
||||
long long last = atomic_load_explicit(®istry->host_full_warn, memory_order_relaxed);
|
||||
if (last != 0 && now - last < DAEMON_LIMITS_HOST_FULL_WARN_SEC)
|
||||
return;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_full_warn, &last, now,
|
||||
memory_order_relaxed, memory_order_relaxed)) {
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"daemon: per-source registry is full (%d slots) and no bucket can be reclaimed; "
|
||||
"'max connections per host' and the auth lockout are temporarily not enforced for "
|
||||
"new sources (the per-module cap and host ACLs still apply)",
|
||||
registry->host_slots);
|
||||
}
|
||||
}
|
||||
|
||||
/* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two
|
||||
* forked children racing on the same source converge on one bucket.
|
||||
*
|
||||
* When the probe finds no empty bucket it reclaims, via a key CAS, the first
|
||||
* bucket that is reclaimable (expired lockout or idle, and no active
|
||||
* connection) and resets its counters. This bounds the table's lifetime so it
|
||||
* cannot fill permanently and stay fail-open. Returns -1 only when the address
|
||||
* is unparseable or the table is genuinely full of live/locked buckets
|
||||
* (callers fail open: the global/module caps and ACLs still apply). */
|
||||
static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
long long now = (long long)time(NULL);
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
/* A couple of passes bound the work: the first normally claims/seeds a bucket;
|
||||
* a lost eviction CAS retries once against the freshly observed table. */
|
||||
for (int pass = 0; pass < 2; pass++) {
|
||||
int evict = -1;
|
||||
uint64_t evict_key = 0;
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0) {
|
||||
/* Stamp last_use *before* publishing the key so a reclaimer racing the
|
||||
* claim can never observe a claimed bucket with last_use == 0 and
|
||||
* evict it. A pre-stamp is harmless if the CAS loses: the bucket is
|
||||
* either still empty (never inspected for reclaim) or has just been
|
||||
* taken by another source that wants a fresh timestamp anyway. */
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
uint64_t expected = 0;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
return (int)idx;
|
||||
}
|
||||
if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) {
|
||||
return (int)idx;
|
||||
}
|
||||
continue; /* another child won this empty bucket; keep probing */
|
||||
}
|
||||
if (evict < 0 && host_bucket_reclaimable(registry, idx, now)) {
|
||||
evict = (int)idx;
|
||||
evict_key = current;
|
||||
}
|
||||
}
|
||||
if (evict >= 0) {
|
||||
/* Refresh the timestamp before the key changes hands so the reused bucket
|
||||
* is not seen as immediately idle by a racing reclaimer. */
|
||||
atomic_store_explicit(®istry->host_last_use[evict], now, memory_order_relaxed);
|
||||
uint64_t expected = evict_key;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[evict], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
/* The bucket now belongs to the new source; clear the evicted source's
|
||||
* stale lockout/failure state. */
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_fail[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_until[evict], 0, memory_order_relaxed);
|
||||
/* Two children can race to intern the same brand-new key into different
|
||||
* eviction targets, leaving the table with duplicate buckets for `key`.
|
||||
* Re-scan for the first (canonical) bucket holding `key`; when it
|
||||
* precedes `evict`, drop our duplicate's occupancy and hand back the
|
||||
* canonical bucket so per-source counts are not orphaned on the
|
||||
* duplicate. The duplicate keeps its key, so no tombstone hole is
|
||||
* created and probe chains stay intact; it ages out normally. */
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t candidate = (start + i) & mask;
|
||||
uint64_t found =
|
||||
atomic_load_explicit(®istry->host_key[candidate], memory_order_acquire);
|
||||
if (found == key) {
|
||||
if (candidate != (size_t)evict) {
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
return (int)candidate;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (found == 0)
|
||||
break; /* the key is present at `evict`, so this cannot happen first */
|
||||
}
|
||||
return evict;
|
||||
}
|
||||
continue; /* lost the race; re-probe with fresh observations */
|
||||
}
|
||||
break; /* no free and no reclaimable bucket: genuinely full */
|
||||
}
|
||||
host_warn_table_full(registry, now);
|
||||
return -1;
|
||||
}
|
||||
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec) {
|
||||
if (max_slots < DAEMON_LIMITS_MIN_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MIN_SLOTS;
|
||||
if (max_slots > DAEMON_LIMITS_MAX_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MAX_SLOTS;
|
||||
if (module_count < 1)
|
||||
module_count = 1;
|
||||
if (module_count > DAEMON_LIMITS_MAX_MODULES)
|
||||
module_count = DAEMON_LIMITS_MAX_MODULES;
|
||||
if (per_host_cap < 0)
|
||||
per_host_cap = 0;
|
||||
if (lockout_threshold < 0)
|
||||
lockout_threshold = 0;
|
||||
if (lockout_duration_sec < 0)
|
||||
lockout_duration_sec = 0;
|
||||
|
||||
bool need_hosts = per_host_cap > 0 || (lockout_threshold > 0 && lockout_duration_sec > 0);
|
||||
int host_slots = 1;
|
||||
if (need_hosts) {
|
||||
size_t want = (size_t)max_slots * 4;
|
||||
if (want < 64)
|
||||
want = 64;
|
||||
if (want > DAEMON_LIMITS_MAX_HOST_SLOTS)
|
||||
want = DAEMON_LIMITS_MAX_HOST_SLOTS;
|
||||
host_slots = (int)next_pow2(want);
|
||||
}
|
||||
|
||||
size_t header = round_up(sizeof(DaemonLimitRegistry), 16);
|
||||
size_t slot_bytes =
|
||||
round_up((size_t)max_slots * sizeof(_Atomic int), 16) * 4; /* state,pid,module,host */
|
||||
size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16);
|
||||
size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16);
|
||||
size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2;
|
||||
size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16) * 2;
|
||||
size_t total =
|
||||
header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16;
|
||||
|
||||
void* map = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0);
|
||||
if (map == MAP_FAILED)
|
||||
return NULL;
|
||||
memset(map, 0, total);
|
||||
|
||||
DaemonLimitRegistry* registry = (DaemonLimitRegistry*)map;
|
||||
registry->max_slots = max_slots;
|
||||
registry->module_count = module_count;
|
||||
registry->host_slots = host_slots;
|
||||
registry->per_host_cap = per_host_cap;
|
||||
registry->lockout_threshold = lockout_threshold;
|
||||
registry->lockout_duration_sec = lockout_duration_sec;
|
||||
registry->map_size = total;
|
||||
|
||||
unsigned char* cursor = (unsigned char*)map + header;
|
||||
registry->slot_state = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_pid = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_module = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_host = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->module_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)module_count * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_key = (_Atomic uint64_t*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic uint64_t);
|
||||
registry->host_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
registry->host_fail = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_until = (atomic_llong*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic long long);
|
||||
registry->host_last_use = (atomic_llong*)cursor;
|
||||
|
||||
for (int i = 0; i < max_slots; i++) {
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
}
|
||||
return registry;
|
||||
}
|
||||
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
munmap(registry, registry->map_size);
|
||||
}
|
||||
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
int expected = SLOT_FREE;
|
||||
if (atomic_compare_exchange_strong(®istry->slot_state[i], &expected, SLOT_CLAIMED)) {
|
||||
atomic_store(®istry->slot_pid[i], 0);
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
}
|
||||
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_store(®istry->slot_pid[slot], (int)pid);
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel);
|
||||
atomic_store_explicit(®istry->slot_pid[slot], 0, memory_order_relaxed);
|
||||
/* The module/host occupancy arrays are derived from the slot table; do not
|
||||
* decrement here or a SIGKILL between a child's increment and its REGISTERED
|
||||
* publish would leak a count. Callers that need the derived counts call
|
||||
* daemon_limits_recompute. */
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) {
|
||||
if (!registry || pid <= 0)
|
||||
return;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load(®istry->slot_state[i]) == SLOT_FREE)
|
||||
continue;
|
||||
if (atomic_load(®istry->slot_pid[i]) == (int)pid) {
|
||||
daemon_limits_reclaim_slot(registry, i);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
/* Zero the derived arrays, then re-derive solely from the REGISTERED slots.
|
||||
* A child that was SIGKILLed after incrementing a counter but before
|
||||
* publishing REGISTERED is not counted, and its leaked increment is erased by
|
||||
* the zeroing, so the leak cannot persist. */
|
||||
for (int m = 0; m < registry->module_count; m++)
|
||||
atomic_store_explicit(®istry->module_active[m], 0, memory_order_relaxed);
|
||||
for (int h = 0; h < registry->host_slots; h++)
|
||||
atomic_store_explicit(®istry->host_active[h], 0, memory_order_relaxed);
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load_explicit(®istry->slot_state[i], memory_order_acquire) != SLOT_REGISTERED)
|
||||
continue;
|
||||
int module = atomic_load_explicit(®istry->slot_module[i], memory_order_relaxed);
|
||||
if (module >= 0 && module < registry->module_count)
|
||||
atomic_fetch_add_explicit(®istry->module_active[module], 1, memory_order_relaxed);
|
||||
int host = atomic_load_explicit(®istry->slot_host[i], memory_order_relaxed);
|
||||
if (host >= 0 && host < registry->host_slots)
|
||||
atomic_fetch_add_explicit(®istry->host_active[host], 1, memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (module_index < 0 || module_index >= registry->module_count)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (atomic_load_explicit(®istry->slot_state[slot], memory_order_acquire) != SLOT_CLAIMED)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
|
||||
int host = -1;
|
||||
if (registry_tracks_hosts(registry))
|
||||
host = host_intern(registry, peer_ip);
|
||||
|
||||
int module_count = atomic_fetch_add(®istry->module_active[module_index], 1) + 1;
|
||||
if (module_cap > 0 && module_count > module_cap) {
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_MODULE_FULL;
|
||||
}
|
||||
if (host >= 0) {
|
||||
int host_count = atomic_fetch_add(®istry->host_active[host], 1) + 1;
|
||||
if (registry->per_host_cap > 0 && host_count > registry->per_host_cap) {
|
||||
atomic_fetch_sub(®istry->host_active[host], 1);
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_HOST_FULL;
|
||||
}
|
||||
}
|
||||
atomic_store(®istry->slot_module[slot], module_index);
|
||||
atomic_store(®istry->slot_host[slot], host);
|
||||
atomic_store_explicit(®istry->slot_state[slot], SLOT_REGISTERED, memory_order_release);
|
||||
return DAEMON_LIMIT_OK;
|
||||
}
|
||||
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return false;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return false;
|
||||
long long until = atomic_load(®istry->host_until[bucket]);
|
||||
long long now = (long long)time(NULL);
|
||||
if (until > now) {
|
||||
if (seconds_remaining)
|
||||
*seconds_remaining = (int)(until - now);
|
||||
return true;
|
||||
}
|
||||
if (until != 0) {
|
||||
/* The previous lockout has expired: clear the stale counter so the source
|
||||
* gets a fresh allowance. */
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return;
|
||||
int bucket = host_intern(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
int failures = atomic_fetch_add(®istry->host_fail[bucket], 1) + 1;
|
||||
if (failures >= registry->lockout_threshold) {
|
||||
long long now = (long long)time(NULL);
|
||||
atomic_store(®istry->host_until[bucket], now + (long long)registry->lockout_duration_sec);
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry)
|
||||
return;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
#ifndef DAEMON_LIMITS_H
|
||||
#define DAEMON_LIMITS_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* Cross-process daemon connection registry.
|
||||
*
|
||||
* The daemon listener forks ONE child per accepted connection, so any
|
||||
* per-module / per-source accounting must live in state shared across the
|
||||
* forked children. This module owns a fixed-size registry carved out of an
|
||||
* anonymous shared mapping (mmap(MAP_SHARED | MAP_ANONYMOUS)) created by the
|
||||
* accept-loop PARENT before it forks; every child inherits the mapping (and the
|
||||
* pointer to it) across fork().
|
||||
*
|
||||
* Rules:
|
||||
* - ONLY C11 atomics (atomic_*); never mtx_t/pthread locks, which can deadlock
|
||||
* in a forked child if another thread held them at fork time.
|
||||
* - No heap allocation after fork: the mapping is fixed-size and all access is
|
||||
* atomic load/store/CAS over preallocated arrays.
|
||||
*
|
||||
* Slot lifecycle (the parent reclaims even when a child is SIGKILLed):
|
||||
* FREE --(parent claim_slot)--> CLAIMED
|
||||
* CLAIMED --(child register)--> REGISTERED
|
||||
* any --(parent reclaim)--> FREE
|
||||
* The child records its module index and per-source bucket into the slot before
|
||||
* publishing REGISTERED; the parent's SIGCHLD handler matches the reaped pid to
|
||||
* the slot and, when REGISTERED, decrements the module/per-source counters.
|
||||
* A child killed before registering holds no counts, so reclaiming a CLAIMED
|
||||
* slot only frees the slot.
|
||||
*
|
||||
* Per-source identity is the normalized numeric peer IP (IPv4-mapped IPv6 is
|
||||
* already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an
|
||||
* open-addressed, linear-probing table keyed by a 64-bit hash. The same table
|
||||
* also carries the cross-process auth-failure counter and lockout deadline.
|
||||
*
|
||||
* Per-source table lifetime: a bucket's key is never cleared back to empty (that
|
||||
* would break every later probe chain that passed through it). Instead the
|
||||
* table has a bounded-lifetime eviction policy: when no empty bucket exists, the
|
||||
* first bucket that is reclaimable -- no active connection AND (its lockout
|
||||
* deadline has passed OR it has been idle for
|
||||
* DAEMON_LIMITS_HOST_EVICT_IDLE_SEC) -- is atomically repurposed for the new
|
||||
* source via a CAS of its key, and its counters are reset. The table therefore
|
||||
* cannot fill permanently, and a full table degrades to fail-open for the
|
||||
* per-source cap/lockout of new sources (the per-module cap and host ACLs still
|
||||
* apply) instead of staying fail-open forever. A rate-limited warning is logged
|
||||
* on the fail-open path. The eviction race with a concurrent
|
||||
* registration/reclaim on the same bucket is benign: it can at worst lose one
|
||||
* source's counter (fail-open), never corrupt memory or the module caps.
|
||||
*/
|
||||
|
||||
typedef struct DaemonLimitRegistry DaemonLimitRegistry;
|
||||
|
||||
/* Result of a per-connection admission check. */
|
||||
typedef enum {
|
||||
DAEMON_LIMIT_OK = 0, /* admitted; slot is now REGISTERED */
|
||||
DAEMON_LIMIT_MODULE_FULL, /* module's `max connections` cap reached */
|
||||
DAEMON_LIMIT_HOST_FULL, /* global `max connections per host` cap reached */
|
||||
DAEMON_LIMIT_UNAVAILABLE, /* registry/slot unusable (caller fails open) */
|
||||
} DaemonLimitResult;
|
||||
|
||||
/* Bounds for registry sizing. A slot is one concurrently live child. */
|
||||
#define DAEMON_LIMITS_MIN_SLOTS 16
|
||||
#define DAEMON_LIMITS_MAX_SLOTS 65536
|
||||
#define DAEMON_LIMITS_MAX_HOST_SLOTS 65536
|
||||
#define DAEMON_LIMITS_NO_SLOT (-1)
|
||||
/* Upper bound on `module_count`, matching daemon_conf.h's DAEMON_CONF_MAX_MODULES
|
||||
* (asserted in daemon_limits.c) so a caller can never size the per-module counter
|
||||
* array larger than the config parser can produce. */
|
||||
#define DAEMON_LIMITS_MAX_MODULES 256
|
||||
|
||||
/* Per-source table lifetime: a bucket with no active connection and no pending
|
||||
* lockout is reclaimable once it has been idle this long, so a flood of distinct
|
||||
* sources cannot pin the table full forever. A bucket whose lockout deadline
|
||||
* has passed is reclaimable immediately (independent of this idle window). */
|
||||
#define DAEMON_LIMITS_HOST_EVICT_IDLE_SEC 300
|
||||
/* Minimum spacing between "per-source table is full" warnings, so a table-full
|
||||
* attack cannot flood the log. */
|
||||
#define DAEMON_LIMITS_HOST_FULL_WARN_SEC 60
|
||||
|
||||
/* Create the shared registry in the calling (parent) process. `max_slots` is
|
||||
* the number of concurrently live children to track (clamped to
|
||||
* [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the
|
||||
* number of daemon modules (clamped to
|
||||
* [1, DAEMON_LIMITS_MAX_MODULES]); `per_host_cap` and the lockout pair come
|
||||
* from the daemon config (0 disables). Returns NULL on failure (e.g. mmap
|
||||
* allocation); callers must degrade gracefully (global cap + ACLs still
|
||||
* apply). */
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec);
|
||||
|
||||
/* Unmap the registry. Only the creating process may call this. */
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Parent side: reserve a slot for the next fork. Returns the slot index or
|
||||
* DAEMON_LIMITS_NO_SLOT when every slot is in use. */
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry);
|
||||
/* Parent side: record the forked child's pid in a claimed slot. */
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid);
|
||||
/* Parent side: release a slot. The slot becomes FREE; the module/per-source
|
||||
* occupancy arrays are DERIVED state and are only refreshed by
|
||||
* daemon_limits_recompute, which callers must invoke afterwards when they rely
|
||||
* on the derived counts (the SIGCHLD handler batches one recompute for the whole
|
||||
* reap). Idempotent. */
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot);
|
||||
/* Parent SIGCHLD side: release the slot owned by `pid` (no-op when not found).
|
||||
* Like reclaim_slot this does not touch the derived occupancy arrays; call
|
||||
* daemon_limits_recompute after a batch of releases. */
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid);
|
||||
|
||||
/* Parent side (async-signal-safe; atomics only, no malloc/log): rebuild
|
||||
* module_active[] / host_active[] from scratch by scanning the REGISTERED slots.
|
||||
* The slot table is the single source of truth, so this self-heals any
|
||||
* count leaked by a child that was SIGKILLed mid-registration (it zeroes the
|
||||
* arrays and re-derives them). Bounded by max_slots + host_slots. A
|
||||
* registration racing this call can be transiently undercounted until the next
|
||||
* recompute, which can only relax a cap briefly -- never corrupt memory. */
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Child side: admit the connection for `module_index` from `peer_ip`. Always
|
||||
* tracks the module/per-source occupancy (so the parent's reclaim is
|
||||
* symmetric); when `module_cap` > 0 it additionally enforces the per-module
|
||||
* cap. A NULL/empty or non-numeric `peer_ip` skips the per-source track (the
|
||||
* callers use that to exempt a trusted loopback peer from the per-host cap; the
|
||||
* per-module cap still applies). Returns DAEMON_LIMIT_OK and publishes the
|
||||
* slot, or a refusal reason. */
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap);
|
||||
|
||||
/* Child side: true when `peer_ip` is currently locked out after too many failed
|
||||
* authentications. `seconds_remaining` may be NULL. */
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining);
|
||||
/* Child side: count one failed authentication for `peer_ip`; once the threshold
|
||||
* is reached the source is locked out for the configured duration. */
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
/* Child side: clear the failure counter/lockout for a source that authenticated
|
||||
* successfully (no-op when the source has no table entry). */
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
|
||||
/* Pure helper: 64-bit FNV-1a hash of a numeric peer IP plus its family, used to
|
||||
* index the per-source table. *ok is set false (and 0 returned) for a NULL or
|
||||
* non-numeric address. Exposed for unit testing. */
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok);
|
||||
|
||||
#endif
|
||||
+8
-2
@@ -23,6 +23,7 @@ Data* data_create_reserve(size_t size) {
|
||||
d->data = NULL;
|
||||
d->size = size;
|
||||
d->protocol_charge = 0;
|
||||
d->owner = NULL;
|
||||
return d;
|
||||
}
|
||||
|
||||
@@ -36,14 +37,19 @@ Data* data_create(void* data, size_t data_size) {
|
||||
new_data->data = data;
|
||||
new_data->size = data_size;
|
||||
new_data->protocol_charge = 0;
|
||||
new_data->owner = NULL;
|
||||
return new_data;
|
||||
}
|
||||
|
||||
void data_destroy(Data* data) {
|
||||
if (data == NULL)
|
||||
return;
|
||||
if (data->protocol_charge != 0)
|
||||
protocol_release_memory(data->protocol_charge);
|
||||
if (data->protocol_charge != 0) {
|
||||
if (data->owner != NULL)
|
||||
protocol_release_memory_for_session(data->owner, data->protocol_charge);
|
||||
else
|
||||
protocol_release_memory(data->protocol_charge);
|
||||
}
|
||||
free(data->data);
|
||||
free(data);
|
||||
}
|
||||
@@ -3,11 +3,25 @@
|
||||
|
||||
#include <stdlib.h>
|
||||
|
||||
/* Forward declaration for the connection budget a received Data is charged
|
||||
* against; defined in protocol.h (which includes this header). */
|
||||
typedef struct ProtocolSession ProtocolSession;
|
||||
|
||||
typedef struct {
|
||||
void* data;
|
||||
size_t size;
|
||||
/* Non-zero only for a buffer charged to the protocol connection budget. */
|
||||
size_t protocol_charge;
|
||||
/* Session whose budget `protocol_charge` was reserved from. When non-NULL,
|
||||
* the charge is returned to this session directly, regardless of which
|
||||
* session (if any) is bound to the destroying thread. owner is not
|
||||
* guaranteed to be set whenever protocol_charge is non-zero: it is NULL for
|
||||
* uncharged Data and for Data that has no recorded owner, in which case any
|
||||
* charge falls back to the session bound at destroy time.
|
||||
*
|
||||
* Lifetime contract: a Data with a non-NULL owner must not outlive that
|
||||
* ProtocolSession -- data_destroy dereferences owner to return the charge. */
|
||||
ProtocolSession* owner;
|
||||
} Data;
|
||||
|
||||
Data* data_create_empty(size_t data_size);
|
||||
@@ -15,5 +29,9 @@ Data* data_create_reserve(size_t size);
|
||||
Data* data_create(void* data, size_t data_size);
|
||||
void data_destroy(Data* data);
|
||||
void protocol_release_memory(size_t charge);
|
||||
/* Release `charge` against `session` directly instead of the thread-local bound
|
||||
* session. Used by data_destroy to honor Data.owner; `session` must outlive
|
||||
* the Data whose charge is being returned. A NULL session is a no-op. */
|
||||
void protocol_release_memory_for_session(ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
+126
-30
@@ -8,13 +8,49 @@
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/file.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* Process-wide counter so two staging contexts created in the same process (or
|
||||
within the same clock tick) can never pick the same name. */
|
||||
static unsigned long long delay_updates_next_sequence(void) {
|
||||
static atomic_ullong sequence;
|
||||
return atomic_fetch_add_explicit(&sequence, 1, memory_order_relaxed);
|
||||
}
|
||||
|
||||
/* Build the per-run staging directory basename: the reserved prefix plus the
|
||||
pid and an entropy token. A fixed name could collide with a genuine
|
||||
destination entry; the token makes such a collision vanishingly unlikely and,
|
||||
if it ever happens, prepare() refuses to touch the existing directory. */
|
||||
static char* delay_updates_make_staging_name(void) {
|
||||
unsigned long long entropy = 0;
|
||||
int fd = open("/dev/urandom", O_RDONLY | O_CLOEXEC);
|
||||
if (fd >= 0) {
|
||||
ssize_t got = read(fd, &entropy, sizeof(entropy));
|
||||
close(fd);
|
||||
if (got != (ssize_t)sizeof(entropy))
|
||||
entropy = 0;
|
||||
}
|
||||
if (entropy == 0)
|
||||
entropy = ((unsigned long long)time(NULL) << 20) ^ ((unsigned long long)getpid() << 8) ^
|
||||
delay_updates_next_sequence();
|
||||
int length = snprintf(NULL, 0, DELAY_UPDATES_STAGING_DIR ".%ld.%llx", (long)getpid(), entropy);
|
||||
if (length < 0)
|
||||
return NULL;
|
||||
char* name = malloc((size_t)length + 1);
|
||||
if (!name)
|
||||
return NULL;
|
||||
snprintf(name, (size_t)length + 1, DELAY_UPDATES_STAGING_DIR ".%ld.%llx", (long)getpid(),
|
||||
entropy);
|
||||
return name;
|
||||
}
|
||||
|
||||
DelayUpdatesContext* delay_updates_context_create(const char* root_directory) {
|
||||
if (!root_directory)
|
||||
return NULL;
|
||||
@@ -26,8 +62,15 @@ DelayUpdatesContext* delay_updates_context_create(const char* root_directory) {
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
context->staging_root = path_cat(root_directory, DELAY_UPDATES_STAGING_DIR);
|
||||
context->staging_name = delay_updates_make_staging_name();
|
||||
if (!context->staging_name) {
|
||||
free(context->root_directory);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
context->staging_root = path_cat(root_directory, context->staging_name);
|
||||
if (!context->staging_root) {
|
||||
free(context->staging_name);
|
||||
free(context->root_directory);
|
||||
free(context);
|
||||
return NULL;
|
||||
@@ -39,6 +82,7 @@ DelayUpdatesContext* delay_updates_context_create(const char* root_directory) {
|
||||
context->lock_fd = -1;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success) {
|
||||
free(context->staging_root);
|
||||
free(context->staging_name);
|
||||
free(context->root_directory);
|
||||
free(context);
|
||||
return NULL;
|
||||
@@ -54,6 +98,7 @@ void delay_updates_context_destroy(DelayUpdatesContext* context) {
|
||||
close(context->lock_fd);
|
||||
context->lock_fd = -1;
|
||||
free(context->staging_root);
|
||||
free(context->staging_name);
|
||||
free(context->root_directory);
|
||||
for (size_t i = 0; i < context->count; i++) {
|
||||
free(context->entries[i].staged_path);
|
||||
@@ -125,48 +170,81 @@ bool delay_updates_prepare(DelayUpdatesContext* context) {
|
||||
return false;
|
||||
if (context->prepared)
|
||||
return true;
|
||||
int fd = file_open_private_dir(context->staging_root);
|
||||
if (fd < 0) {
|
||||
/* Create the per-run staging directory with O_EXCL semantics. The name is
|
||||
unique to this transfer, so if the path already exists it is NOT ours:
|
||||
either a genuine destination entry that happens to share the name or a
|
||||
leftover from another session. Refuse rather than wipe it -- the old
|
||||
fixed-name design could destroy a real destination entry. A crash
|
||||
leftover is never reused (the next run picks a fresh name). */
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(context->staging_root, &leaf, true);
|
||||
if (parent_fd < 0) {
|
||||
int saved_errno = errno;
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not create --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
/* Hold an exclusive advisory lock on the staging directory for the whole
|
||||
transfer. The staging directory name is fixed, so two simultaneous
|
||||
delayed transfers to the same destination root would otherwise share it
|
||||
and destroy each other's staged files. The lock makes the second session
|
||||
fail cleanly instead of corrupting the first. The lock is released when
|
||||
the context (and its file descriptor) is destroyed. */
|
||||
int fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (fd >= 0) {
|
||||
close(fd);
|
||||
close(parent_fd);
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--delay-updates staging directory '%s' already exists and is not owned by this "
|
||||
"transfer; refusing to overwrite it",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
if (errno != ENOENT) {
|
||||
int saved_errno = errno;
|
||||
close(parent_fd);
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
if (mkdirat(parent_fd, leaf, 0700) != 0) {
|
||||
int saved_errno = errno;
|
||||
close(parent_fd);
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not create --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
if (fd < 0) {
|
||||
int saved_errno = errno;
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
/* Keep the exclusive advisory lock as defense in depth: the unique name
|
||||
already prevents two sessions from sharing a staging directory, but the
|
||||
lock also catches an improbable same-name collision that raced between the
|
||||
existence check above and the open. */
|
||||
if (flock(fd, LOCK_EX | LOCK_NB) != 0) {
|
||||
int saved_errno = errno;
|
||||
close(fd);
|
||||
if (saved_errno == EWOULDBLOCK || saved_errno == EAGAIN) {
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"another --delay-updates transfer to '%s' is already in progress; refusing to "
|
||||
"share the staging directory",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
} else {
|
||||
log_message(LOG_LEVEL_ERROR, "could not lock --delay-updates staging directory '%s': %s",
|
||||
context->staging_root, strerror(saved_errno));
|
||||
}
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not lock --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
context->lock_fd = fd;
|
||||
/* Only now, with exclusive ownership, wipe leftovers from an interrupted
|
||||
earlier transfer; this can never race with a live session. */
|
||||
bool ok = delay_wipe_dir_fd(fd);
|
||||
if (!ok) {
|
||||
log_message(LOG_LEVEL_ERROR, "could not clear stale --delay-updates staging files under '%s'",
|
||||
context->staging_root);
|
||||
close(context->lock_fd);
|
||||
context->lock_fd = -1;
|
||||
return false;
|
||||
}
|
||||
context->prepared = true;
|
||||
return true;
|
||||
}
|
||||
@@ -264,6 +342,24 @@ static bool delay_publish_entry(DelayUpdatesContext* context, const Config* conf
|
||||
const StagedFileEntry* entry) {
|
||||
if (!delay_publish_backup(context, config, entry))
|
||||
return false;
|
||||
/* An incoming regular file/symlink may replace a destination DIRECTORY that
|
||||
blocks it. rsync removes the blocker recursively when --delete or --force
|
||||
is active (its generator's "make way" deletion), and a --delay-updates run
|
||||
stages elsewhere so it only discovers the blocker here. FastSync's
|
||||
immediate-install path clears it too; without --delete/--force a non-empty
|
||||
blocker fails the run (rsync's "could not make way for new regular file").
|
||||
use_delete is gated by the server --allow-delete policy, so a client can
|
||||
never use this to bypass deletion authorization. */
|
||||
if (config && (config->force_delete || config->use_delete) &&
|
||||
file_directory_exists_secure(entry->final_path)) {
|
||||
if (!file_remove_tree_secure(entry->final_path)) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not remove destination directory blocking '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!file_rename_secure(entry->staged_path, entry->final_path)) {
|
||||
if (errno == EXDEV) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
|
||||
@@ -24,6 +24,7 @@ typedef struct {
|
||||
shared with the publish/cleanup phase that runs after the threads join. */
|
||||
typedef struct DelayUpdatesContext {
|
||||
char* root_directory; /* receive root the staging dir lives under */
|
||||
char* staging_name; /* per-run unique staging dir basename */
|
||||
char* staging_root; /* root_directory/<staging dir name> */
|
||||
mtx_t mutex;
|
||||
StagedFileEntry* entries;
|
||||
@@ -33,7 +34,11 @@ typedef struct DelayUpdatesContext {
|
||||
int lock_fd; /* advisory exclusive flock held on the staging dir, or -1 */
|
||||
} DelayUpdatesContext;
|
||||
|
||||
/* Name of the private staging subdirectory created under the receive root. */
|
||||
/* Reserved prefix for the private staging subdirectory created under the
|
||||
receive root. The actual directory name is per-run unique (the prefix plus a
|
||||
pid/entropy token) so it can never clobber a genuine destination entry that
|
||||
happens to share the name; the bare prefix is still what a --backup-dir must
|
||||
not collide with. */
|
||||
#define DELAY_UPDATES_STAGING_DIR ".fastsync-stage"
|
||||
|
||||
/* True when `dir` (ignoring a trailing "/") is the reserved staging directory
|
||||
|
||||
@@ -0,0 +1,777 @@
|
||||
#include "delete.h"
|
||||
|
||||
#include "delay_updates.h"
|
||||
#include "filter.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* Build the keep-set index from the exact manifest entries only. A lookup of
|
||||
`rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor
|
||||
directory of kept content (the old is_dir_in_manifest predicate); the sorted
|
||||
view answers "is an ancestor of kept content" without materializing any
|
||||
per-component prefix copy, so the index is O(manifest size) memory. */
|
||||
static bool build_keep_index(const ArrayList* manifest, PathIndex* index) {
|
||||
if (!manifest || manifest->size <= 0)
|
||||
return path_index_build(index, NULL, 0);
|
||||
return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size);
|
||||
}
|
||||
|
||||
static bool keep_is_dir(const PathIndex* index, const char* rel_path) {
|
||||
return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path);
|
||||
}
|
||||
|
||||
static bool keep_is_file(const PathIndex* index, const char* rel_path) {
|
||||
return path_index_contains(index, rel_path);
|
||||
}
|
||||
|
||||
/* rsync's receiver-side verdict for one candidate extra: the per-directory
|
||||
* chain first (deepest directory before ancestors), then the command-line base
|
||||
* rules. Either rule set may be absent. */
|
||||
FilterAction delete_protect_verdict(const DeleteProtectRules* protect, const char* rel_path,
|
||||
const char* leaf, bool is_dir) {
|
||||
if (!protect)
|
||||
return FILTER_ACTION_NONE;
|
||||
/* rsync protects its own --backup files from the delete pass: a name ending
|
||||
in the backup suffix is never an extra. Checked before the filter rules so
|
||||
an explicit exclude cannot be bypassed (the suffix is always a shield). */
|
||||
if (protect->backup_suffix && protect->backup_suffix[0] != '\0') {
|
||||
size_t name_len = strlen(leaf);
|
||||
size_t suffix_len = strlen(protect->backup_suffix);
|
||||
if (name_len > suffix_len &&
|
||||
strcmp(leaf + (name_len - suffix_len), protect->backup_suffix) == 0)
|
||||
return FILTER_ACTION_PROTECT;
|
||||
}
|
||||
FilterAction action = filter_dir_rules_apply_side(protect->dir_rules, rel_path, leaf, is_dir);
|
||||
if (action != FILTER_ACTION_NONE)
|
||||
return action;
|
||||
return filter_rules_apply_side(protect->base_rules, rel_path, leaf, is_dir, FILTER_SIDE_RECEIVER);
|
||||
}
|
||||
|
||||
const char* delete_backup_suffix(const Config* config) {
|
||||
if (!config || !config->backup || config->ignore_existing)
|
||||
return NULL;
|
||||
const char* suffix = config->suffix ? config->suffix : "~";
|
||||
if (!suffix[0] || strchr(suffix, '/'))
|
||||
return NULL;
|
||||
return suffix;
|
||||
}
|
||||
|
||||
/* Classify a removed entry from its st_mode for the per-type delete counters. */
|
||||
DeleteEntryType delete_entry_type_of_mode(mode_t mode) {
|
||||
if (S_ISDIR(mode))
|
||||
return DELETE_ENTRY_DIR;
|
||||
if (S_ISLNK(mode))
|
||||
return DELETE_ENTRY_LINK;
|
||||
if (S_ISREG(mode))
|
||||
return DELETE_ENTRY_REG;
|
||||
return DELETE_ENTRY_SPECIAL;
|
||||
}
|
||||
|
||||
/* True when child_rel is, or lies below, a protected entry. A prefix "a"
|
||||
therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only
|
||||
set only protect DIRECT children of the receive root (at_root); nested
|
||||
directories that share such a name stay ordinary destination content. */
|
||||
bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips,
|
||||
int skip_count) {
|
||||
for (int i = 0; i < skip_count; i++) {
|
||||
if (skips[i].top_level_only && !at_root)
|
||||
continue;
|
||||
size_t prefix_len = strlen(skips[i].prefix);
|
||||
if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 &&
|
||||
(child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/'))
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Per-run deletion budget and tallies. `max_delete` is the cap on the number
|
||||
of entries the walker may remove (SIZE_MAX = unlimited); once it is reached
|
||||
the remaining extras are counted in `skipped` and left in place, matching
|
||||
rsync's partial --max-delete behavior. */
|
||||
typedef struct {
|
||||
size_t max_delete;
|
||||
size_t deleted;
|
||||
size_t skipped;
|
||||
bool limit_hit;
|
||||
} DeleteBudget;
|
||||
|
||||
/* True when direct children of the directory named by `rel` may be removed.
|
||||
With no synchronization info (dirs == NULL) the whole tree is deletable; when
|
||||
a dirs index is supplied only its exact entries are (the receive root is the
|
||||
"." sentinel). */
|
||||
static bool is_synced_dir(const PathIndex* dirs, const char* rel) {
|
||||
if (!dirs)
|
||||
return true;
|
||||
return path_index_contains(dirs, rel[0] == '\0' ? "." : rel);
|
||||
}
|
||||
|
||||
/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed
|
||||
strcmp would order bytes >= 0x80 differently). */
|
||||
static int delete_name_cmp(const char* a, const char* b) {
|
||||
const unsigned char* pa = (const unsigned char*)a;
|
||||
const unsigned char* pb = (const unsigned char*)b;
|
||||
while (*pa != '\0' && *pa == *pb) {
|
||||
pa++;
|
||||
pb++;
|
||||
}
|
||||
return (int)*pa - (int)*pb;
|
||||
}
|
||||
|
||||
bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count,
|
||||
bool* operation_ok) {
|
||||
*out = NULL;
|
||||
*count = 0;
|
||||
if (operation_ok)
|
||||
*operation_ok = true;
|
||||
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
DIR* dir = fdopendir(scanfd);
|
||||
if (!dir) {
|
||||
close(scanfd);
|
||||
return false;
|
||||
}
|
||||
DeleteDirEntry* entries = NULL;
|
||||
size_t used = 0;
|
||||
size_t capacity = 0;
|
||||
bool ok = true;
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT && operation_ok)
|
||||
*operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (used == capacity) {
|
||||
size_t next = capacity == 0 ? 16 : capacity * 2;
|
||||
DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown));
|
||||
if (!grown) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
entries = grown;
|
||||
capacity = next;
|
||||
}
|
||||
entries[used].name = str_dup(entry->d_name);
|
||||
if (!entries[used].name) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
entries[used].is_dir = S_ISDIR(st.st_mode);
|
||||
entries[used].mode = st.st_mode;
|
||||
used++;
|
||||
}
|
||||
closedir(dir);
|
||||
if (!ok) {
|
||||
delete_dir_entries_free(entries, used);
|
||||
return false;
|
||||
}
|
||||
*out = entries;
|
||||
*count = used;
|
||||
return true;
|
||||
}
|
||||
|
||||
void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) {
|
||||
if (!entries)
|
||||
return;
|
||||
for (size_t i = 0; i < count; i++)
|
||||
free(entries[i].name);
|
||||
free(entries);
|
||||
}
|
||||
|
||||
/* rsync's extraneous-entry order: subdirectories before files, each group in
|
||||
descending name order. */
|
||||
int delete_dir_entry_cmp_desc(const void* a, const void* b) {
|
||||
const DeleteDirEntry* ea = a;
|
||||
const DeleteDirEntry* eb = b;
|
||||
if (ea->is_dir != eb->is_dir)
|
||||
return ea->is_dir ? -1 : 1;
|
||||
return -delete_name_cmp(ea->name, eb->name);
|
||||
}
|
||||
|
||||
/* rsync's kept-subdirectory order: plain ascending name. */
|
||||
int delete_dir_entry_cmp_asc(const void* a, const void* b) {
|
||||
const DeleteDirEntry* ea = a;
|
||||
const DeleteDirEntry* eb = b;
|
||||
return delete_name_cmp(ea->name, eb->name);
|
||||
}
|
||||
|
||||
/* How the shared classification/descent walk disposes of an extra it has
|
||||
identified. LIST records the destination-relative path without touching disk
|
||||
(the -n/--dry-run would-delete enumeration); DELETE unlinks/rmdirs it, charges
|
||||
the shared --max-delete budget and notifies the observer. Both modes classify
|
||||
and traverse identically, so the dry-run enumeration and the real deletion
|
||||
cannot drift. */
|
||||
typedef enum { DELETE_WALK_MODE_DELETE, DELETE_WALK_MODE_LIST } DeleteWalkMode;
|
||||
|
||||
typedef struct {
|
||||
DeleteWalkMode mode;
|
||||
DeleteBudget* budget; /* DELETE mode */
|
||||
ArrayList* out; /* LIST mode: receives strdup'd relative paths */
|
||||
size_t* recorded; /* LIST mode */
|
||||
DeletePathObserver observer; /* DELETE mode */
|
||||
void* observer_context; /* DELETE mode */
|
||||
} DeleteWalkState;
|
||||
|
||||
/* The per-walk invariants threaded unchanged through every recursive descent:
|
||||
the keep/synchronized-dir indexes, the destination mode and the protection
|
||||
rules. Bundling them keeps the recursive helpers below to a handful of
|
||||
positional arguments. */
|
||||
typedef struct {
|
||||
const PathIndex* keep;
|
||||
const PathIndex* dirs;
|
||||
DeleteWalkState* state;
|
||||
const DeleteSkipEntry* skips;
|
||||
int skip_count;
|
||||
const DeleteProtectRules* protect;
|
||||
} DeleteWalkContext;
|
||||
|
||||
/* Duplicate `path` with rsync's trailing-slash convention, used to report a
|
||||
removed (or would-be-removed) directory. Returns NULL on allocation
|
||||
failure. */
|
||||
static char* with_trailing_slash(const char* path) {
|
||||
size_t len = strlen(path);
|
||||
char* copy = malloc(len + 2);
|
||||
if (!copy)
|
||||
return NULL;
|
||||
memcpy(copy, path, len);
|
||||
copy[len] = '/';
|
||||
copy[len + 1] = '\0';
|
||||
return copy;
|
||||
}
|
||||
|
||||
/* Forward declaration: the ordered passes below recurse through the driver. */
|
||||
static bool delete_walk_fd(int dirfd, const char* rel_path, const DeleteWalkContext* ctx,
|
||||
bool parent_deletable, bool* all_removed);
|
||||
|
||||
/* Descend into the child directory `name` of `dirfd`, walking it as part of the
|
||||
current operation. Returns false on a genuine open/walk failure; on success
|
||||
*child_all_removed reports whether the child removed everything it held (so
|
||||
the caller may rmdir it). */
|
||||
static bool delete_walk_child(int dirfd, const char* name, const char* child_rel,
|
||||
const DeleteWalkContext* ctx, bool deletable,
|
||||
bool* child_all_removed) {
|
||||
*child_all_removed = false;
|
||||
int childfd = openat(dirfd, name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (childfd < 0)
|
||||
return errno == ENOENT;
|
||||
bool ok = delete_walk_fd(childfd, child_rel, ctx, deletable, child_all_removed);
|
||||
close(childfd);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Classify every entry up front (the verdict does not depend on processing
|
||||
order) so the ordered passes below can act on it. Sets shielded[]/is_extra[]
|
||||
and reports through *local_survives whether anything in this directory stays
|
||||
in place. Returns false on a path-construction failure. */
|
||||
static bool delete_walk_classify(const char* rel_path, const DeleteDirEntry* entries, size_t count,
|
||||
const DeleteWalkContext* ctx, bool deletable, bool at_root,
|
||||
bool* shielded, bool* is_extra, bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
/* A --delay-updates run keeps its staging directory as a direct child of
|
||||
the receive root, and basis-dir snapshots live below it too. Their
|
||||
contents are not manifest entries, so descending into them would delete
|
||||
every staged / basis file as an "extra". Only the staging name (a
|
||||
top-level-only prefix) and the basis prefixes are protected: a nested
|
||||
destination directory that happens to be called .fastsync-stage is
|
||||
ordinary content. */
|
||||
if (path_under_skip_prefix(child_rel, at_root, ctx->skips, ctx->skip_count)) {
|
||||
shielded[i] = true;
|
||||
*local_survives = true;
|
||||
} else if (delete_protect_verdict(ctx->protect, child_rel, entries[i].name,
|
||||
entries[i].is_dir) == FILTER_ACTION_PROTECT) {
|
||||
/* A first-match protect rule shields the extra; for a directory the whole
|
||||
subtree is shielded (rsync prunes an excluded directory), so do not
|
||||
descend. */
|
||||
shielded[i] = true;
|
||||
*local_survives = true;
|
||||
} else if (entries[i].is_dir) {
|
||||
bool child_synced = ctx->dirs && path_index_contains(ctx->dirs, child_rel);
|
||||
is_extra[i] = deletable && !child_synced && !keep_is_dir(ctx->keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
*local_survives = true;
|
||||
} else {
|
||||
is_extra[i] = deletable && !keep_is_file(ctx->keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
*local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Pass 1: extraneous subdirectories, descending. Recurses into each and, when
|
||||
the child removed everything it held, records or removes it and charges the
|
||||
budget. */
|
||||
static bool delete_walk_extra_dirs(int dirfd, const char* rel_path, const DeleteDirEntry* entries,
|
||||
size_t dir_count, const DeleteWalkContext* ctx, bool deletable,
|
||||
const bool* is_extra, bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < dir_count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
bool child_all_removed = false;
|
||||
if (!delete_walk_child(dirfd, entries[i].name, child_rel, ctx, deletable, &child_all_removed))
|
||||
ok = false;
|
||||
if (child_all_removed && deletable) {
|
||||
if (ctx->state->mode == DELETE_WALK_MODE_LIST) {
|
||||
/* Record the directory with rsync's trailing slash. */
|
||||
char* copy = with_trailing_slash(child_rel);
|
||||
if (!copy) {
|
||||
ok = false;
|
||||
} else if (!array_list_add(ctx->state->out, copy)) {
|
||||
free(copy);
|
||||
ok = false;
|
||||
} else {
|
||||
(*ctx->state->recorded)++;
|
||||
}
|
||||
} else if (ctx->state->budget->deleted >= ctx->state->budget->max_delete) {
|
||||
ctx->state->budget->limit_hit = true;
|
||||
ctx->state->budget->skipped++;
|
||||
*local_survives = true;
|
||||
} else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) {
|
||||
/* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still
|
||||
holds entries the walker leaves in place (a protected excluded
|
||||
prefix, a kept file the manifest protects, a symlink); rsync leaves
|
||||
such a directory behind, so this is not an error. Only genuine I/O
|
||||
failures abort the deletion. */
|
||||
if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST)
|
||||
ok = false;
|
||||
*local_survives = true;
|
||||
} else {
|
||||
ctx->state->budget->deleted++;
|
||||
/* rsync reports a removed directory with a trailing slash. */
|
||||
if (ctx->state->observer) {
|
||||
char* with_slash = with_trailing_slash(child_rel);
|
||||
if (with_slash) {
|
||||
ctx->state->observer(ctx->state->observer_context, with_slash, DELETE_ENTRY_DIR);
|
||||
free(with_slash);
|
||||
} else {
|
||||
ctx->state->observer(ctx->state->observer_context, child_rel, DELETE_ENTRY_DIR);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
*local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Pass 2: extraneous files, descending. */
|
||||
static bool delete_walk_extra_files(int dirfd, const char* rel_path, const DeleteDirEntry* entries,
|
||||
size_t dir_count, size_t count, const DeleteWalkContext* ctx,
|
||||
const bool* is_extra, bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = dir_count; i < count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
if (ctx->state->mode == DELETE_WALK_MODE_LIST) {
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
char* copy = str_dup(child_rel);
|
||||
if (!copy || !array_list_add(ctx->state->out, copy)) {
|
||||
free(copy);
|
||||
ok = false;
|
||||
} else {
|
||||
(*ctx->state->recorded)++;
|
||||
}
|
||||
free(child_rel);
|
||||
} else if (ctx->state->budget->deleted >= ctx->state->budget->max_delete) {
|
||||
ctx->state->budget->limit_hit = true;
|
||||
ctx->state->budget->skipped++;
|
||||
*local_survives = true;
|
||||
} else if (unlinkat(dirfd, entries[i].name, 0) != 0) {
|
||||
if (errno != ENOENT)
|
||||
ok = false;
|
||||
*local_survives = true;
|
||||
} else {
|
||||
ctx->state->budget->deleted++;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (child_rel) {
|
||||
if (ctx->state->observer)
|
||||
ctx->state->observer(ctx->state->observer_context, child_rel,
|
||||
delete_entry_type_of_mode(entries[i].mode));
|
||||
char* escaped_path = output_escape(child_rel, log_get_8_bit_output());
|
||||
fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : "<allocation failed>");
|
||||
free(escaped_path);
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Pass 3: kept subdirectories, ascending (rsync descends into these only after
|
||||
the parent's own extras have been handled). */
|
||||
static bool delete_walk_kept_dirs(int dirfd, const char* rel_path, const DeleteDirEntry* entries,
|
||||
size_t dir_count, const DeleteWalkContext* ctx, bool deletable,
|
||||
const bool* is_extra, const bool* shielded,
|
||||
bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = dir_count; i-- > 0;) {
|
||||
if (is_extra[i] || shielded[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
bool child_all_removed = false;
|
||||
if (!delete_walk_child(dirfd, entries[i].name, child_rel, ctx, deletable, &child_all_removed))
|
||||
ok = false;
|
||||
/* A kept/synchronized directory is never removed. */
|
||||
*local_survives = true;
|
||||
free(child_rel);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Remove the extras directly inside the directory open on `dirfd` (DELETE mode)
|
||||
or record the paths that WOULD be removed (LIST mode), recursing into every
|
||||
child directory so kept content below a synchronized prefix is reached.
|
||||
`all_removed` reports whether every child entry was removed (so the caller may
|
||||
rmdir this directory). A child directory is never removed when it is itself a
|
||||
synchronized directory or holds kept content; with a dirs index supplied,
|
||||
direct children of a non-synchronized directory are never extras at all (they
|
||||
are left in place but still descended into). Symlinks are unlinked like any
|
||||
other non-directory extra (never followed).
|
||||
|
||||
Entries are processed in rsync's order (extraneous subdirectories in
|
||||
descending name order, then extraneous files, then kept subdirectories in
|
||||
ascending order) rather than readdir() order, so `--max-delete` leaves the
|
||||
same survivors and the `--info=del`/dry-run line order matches rsync. */
|
||||
static bool delete_walk_fd(int dirfd, const char* rel_path, const DeleteWalkContext* ctx,
|
||||
bool parent_deletable, bool* all_removed) {
|
||||
DeleteDirEntry* entries = NULL;
|
||||
size_t count = 0;
|
||||
bool collect_ok = true;
|
||||
if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok))
|
||||
return false;
|
||||
bool operation_ok = collect_ok;
|
||||
bool local_survives = false;
|
||||
bool* shielded = calloc(count ? count : 1, sizeof(bool));
|
||||
bool* is_extra = calloc(count ? count : 1, sizeof(bool));
|
||||
if (!shielded || !is_extra) {
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
return false;
|
||||
}
|
||||
/* A directory is deletable when it or ANY ancestor is synchronized; the
|
||||
`parent_deletable` flag carries that down the recursion so dest-only
|
||||
directories below a synchronized root are removed wholesale. */
|
||||
bool deletable = parent_deletable || is_synced_dir(ctx->dirs, rel_path);
|
||||
bool at_root = rel_path[0] == '\0';
|
||||
|
||||
/* Reproduce rsync's traversal order: extraneous subdirectories in descending
|
||||
name order, then extraneous files in descending name order, and kept
|
||||
subdirectories only afterwards (ascending). Sorting up front also fixes the
|
||||
identity of the survivors under a partial --max-delete. */
|
||||
if (count > 1)
|
||||
qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc);
|
||||
size_t dir_count = 0;
|
||||
while (dir_count < count && entries[dir_count].is_dir)
|
||||
dir_count++;
|
||||
|
||||
if (!delete_walk_classify(rel_path, entries, count, ctx, deletable, at_root, shielded, is_extra,
|
||||
&local_survives))
|
||||
operation_ok = false;
|
||||
if (!delete_walk_extra_dirs(dirfd, rel_path, entries, dir_count, ctx, deletable, is_extra,
|
||||
&local_survives))
|
||||
operation_ok = false;
|
||||
if (!delete_walk_extra_files(dirfd, rel_path, entries, dir_count, count, ctx, is_extra,
|
||||
&local_survives))
|
||||
operation_ok = false;
|
||||
if (!delete_walk_kept_dirs(dirfd, rel_path, entries, dir_count, ctx, deletable, is_extra,
|
||||
shielded, &local_survives))
|
||||
operation_ok = false;
|
||||
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
*all_removed = !local_survives;
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
/* Open the receive root following the same authorized-root confinement the
|
||||
walker uses, or dest_root directly when no authorized root is installed. */
|
||||
static int open_destination_root(const char* dest_root) {
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
if (root_fd >= 0) {
|
||||
if (utils_get_authorized_root_path())
|
||||
return utils_open_authorized_destination(dest_root);
|
||||
if (dest_root == NULL)
|
||||
return dup(root_fd);
|
||||
return -1;
|
||||
}
|
||||
return open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
}
|
||||
|
||||
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, ArrayList* out, size_t* count_out) {
|
||||
if (count_out)
|
||||
*count_out = 0;
|
||||
if (!manifest || !out)
|
||||
return false;
|
||||
PathIndex keep;
|
||||
if (!build_keep_index(manifest, &keep))
|
||||
return false;
|
||||
PathIndex dirs;
|
||||
bool have_dirs = synced_dirs != NULL;
|
||||
if (have_dirs &&
|
||||
!path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) {
|
||||
path_index_free(&keep);
|
||||
return false;
|
||||
}
|
||||
int rootfd = open_destination_root(dest_root);
|
||||
if (rootfd < 0) {
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
return false;
|
||||
}
|
||||
bool all_removed = false;
|
||||
size_t recorded = 0;
|
||||
DeleteWalkState state = {.mode = DELETE_WALK_MODE_LIST,
|
||||
.budget = NULL,
|
||||
.out = out,
|
||||
.recorded = &recorded,
|
||||
.observer = NULL,
|
||||
.observer_context = NULL};
|
||||
DeleteWalkContext ctx = {.keep = &keep,
|
||||
.dirs = have_dirs ? &dirs : NULL,
|
||||
.state = &state,
|
||||
.skips = skips,
|
||||
.skip_count = skip_count,
|
||||
.protect = protect};
|
||||
bool ok = delete_walk_fd(rootfd, "", &ctx, false, &all_removed);
|
||||
if (close(rootfd) != 0)
|
||||
ok = false;
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
if (count_out)
|
||||
*count_out = recorded;
|
||||
return ok;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect,
|
||||
size_t* deleted_out, size_t* skipped_out,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (deleted_out)
|
||||
*deleted_out = 0;
|
||||
if (skipped_out)
|
||||
*skipped_out = 0;
|
||||
if (!manifest)
|
||||
return DELETE_WALK_ERROR;
|
||||
/* Index the keep-set (and the synchronized-dir set, when supplied) once so
|
||||
membership is answered in O(path length) instead of scanning every entry
|
||||
for every destination entry. */
|
||||
PathIndex keep;
|
||||
if (!build_keep_index(manifest, &keep))
|
||||
return DELETE_WALK_ERROR;
|
||||
PathIndex dirs;
|
||||
bool have_dirs = synced_dirs != NULL;
|
||||
if (have_dirs &&
|
||||
!path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) {
|
||||
path_index_free(&keep);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
int rootfd = open_destination_root(dest_root);
|
||||
if (rootfd < 0) {
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
bool all_removed = false;
|
||||
DeleteWalkState state = {.mode = DELETE_WALK_MODE_DELETE,
|
||||
.budget = &budget,
|
||||
.out = NULL,
|
||||
.recorded = NULL,
|
||||
.observer = observer,
|
||||
.observer_context = observer_context};
|
||||
DeleteWalkContext ctx = {.keep = &keep,
|
||||
.dirs = have_dirs ? &dirs : NULL,
|
||||
.state = &state,
|
||||
.skips = skips,
|
||||
.skip_count = skip_count,
|
||||
.protect = protect};
|
||||
bool ok = delete_walk_fd(rootfd, "", &ctx, false, &all_removed);
|
||||
if (close(rootfd) != 0)
|
||||
ok = false;
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
if (deleted_out)
|
||||
*deleted_out = budget.deleted;
|
||||
if (skipped_out)
|
||||
*skipped_out = budget.skipped;
|
||||
if (!ok)
|
||||
return DELETE_WALK_ERROR;
|
||||
return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, size_t* deleted_out,
|
||||
size_t* skipped_out) {
|
||||
return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips,
|
||||
skip_count, protect, deleted_out, skipped_out, NULL, NULL);
|
||||
}
|
||||
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest) {
|
||||
return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) ==
|
||||
DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
/* Build the delete-walk protection prefix for one basis directory. The walker
|
||||
compares paths relative to the receive root, so a relative entry is already
|
||||
in the right form; an absolute entry that lies below the root is converted to
|
||||
its root-relative form, and one outside the root returns NULL (the walk
|
||||
cannot reach it, and it is not protected data beneath the root). Exposed so
|
||||
tests can exercise the root-of-"/" child mapping directly. */
|
||||
char* delete_basis_relative(const Config* config, const char* path) {
|
||||
if (!path)
|
||||
return NULL;
|
||||
if (path[0] != '/')
|
||||
return str_dup(path);
|
||||
const char* root = config->receive_root_directory;
|
||||
if (!root || root[0] != '/')
|
||||
return NULL;
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(path, root, root_len) != 0)
|
||||
return NULL;
|
||||
if (root_len == 1) {
|
||||
/* `root` is "/" (the only single-character absolute root): every absolute
|
||||
path is below it, and the child relative form is everything after the
|
||||
leading '/'. */
|
||||
if (path[1] == '\0')
|
||||
return NULL; /* identical to the root, not a child */
|
||||
return str_dup(path + 1);
|
||||
}
|
||||
if (path[root_len] != '/')
|
||||
return NULL; /* identical or a sibling sharing a name prefix */
|
||||
return str_dup(path + root_len + 1);
|
||||
}
|
||||
|
||||
bool delete_skips_build(const Config* config, const ArrayList* protected_paths,
|
||||
const ArrayList* size_skipped, bool basis_root_relative,
|
||||
DeleteSkipSet* out) {
|
||||
if (!out)
|
||||
return false;
|
||||
out->entries = NULL;
|
||||
out->owned_prefixes = NULL;
|
||||
out->count = 0;
|
||||
out->owned_count = 0;
|
||||
if (!config)
|
||||
return false;
|
||||
int protected_count = protected_paths ? protected_paths->size : 0;
|
||||
int size_skipped_count = size_skipped ? size_skipped->size : 0;
|
||||
int count =
|
||||
(config->delay_updates ? 1 : 0) + config->basis_count + protected_count + size_skipped_count;
|
||||
if (count == 0)
|
||||
return true;
|
||||
out->entries = calloc((size_t)count, sizeof(DeleteSkipEntry));
|
||||
if (!out->entries)
|
||||
return false;
|
||||
if (basis_root_relative && config->basis_count > 0) {
|
||||
out->owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*));
|
||||
if (!out->owned_prefixes) {
|
||||
free(out->entries);
|
||||
out->entries = NULL;
|
||||
return false;
|
||||
}
|
||||
out->owned_count = config->basis_count;
|
||||
}
|
||||
int idx = 0;
|
||||
if (config->delay_updates) {
|
||||
/* Protect this transfer's actual (per-run unique) staging directory. The
|
||||
runtime name is only known to the receiver-side context; fall back to the
|
||||
reserved prefix for a context that was never created (e.g. a dry run). */
|
||||
const char* staging_name = (config->delay_context && config->delay_context->staging_name)
|
||||
? config->delay_context->staging_name
|
||||
: DELAY_UPDATES_STAGING_DIR;
|
||||
out->entries[idx].prefix = staging_name;
|
||||
out->entries[idx].top_level_only = true;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < config->basis_count; i++) {
|
||||
const char* prefix = config->basis_dirs[i].path;
|
||||
if (basis_root_relative) {
|
||||
/* An absolute basis outside the receive root is unreachable by this walk,
|
||||
so it contributes no protection prefix (and no slot). */
|
||||
char* relative = delete_basis_relative(config, config->basis_dirs[i].path);
|
||||
if (!relative)
|
||||
continue;
|
||||
out->owned_prefixes[i] = relative;
|
||||
prefix = relative;
|
||||
}
|
||||
out->entries[idx].prefix = prefix;
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < protected_count; i++) {
|
||||
out->entries[idx].prefix = (const char*)protected_paths->items[i];
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < size_skipped_count; i++) {
|
||||
out->entries[idx].prefix = (const char*)size_skipped->items[i];
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
out->count = idx;
|
||||
return true;
|
||||
}
|
||||
|
||||
void delete_skips_free(DeleteSkipSet* set) {
|
||||
if (!set)
|
||||
return;
|
||||
if (set->owned_prefixes) {
|
||||
for (int i = 0; i < set->owned_count; i++)
|
||||
free(set->owned_prefixes[i]);
|
||||
}
|
||||
free(set->owned_prefixes);
|
||||
free(set->entries);
|
||||
set->entries = NULL;
|
||||
set->owned_prefixes = NULL;
|
||||
set->count = 0;
|
||||
set->owned_count = 0;
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
#ifndef DELETE_H
|
||||
#define DELETE_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "filter.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Delete engine.
|
||||
*
|
||||
* This module owns destination-relative delete traversal: the ordered directory
|
||||
* walker that reproduces rsync's extraneous-entry order, the skip-prefix
|
||||
* protection set shared by every delete pass, and the read-only enumeration
|
||||
* that mirrors the walker for -n/--dry-run. The budgeted manifest commit
|
||||
* (delete_commit.c) and the per-directory delete plans (delete_plan.c) are
|
||||
* built on the primitives exported here. */
|
||||
|
||||
/* Result of a bounded extra-file deletion run. */
|
||||
typedef enum {
|
||||
/* Every extra entry was removed (or there were none). */
|
||||
DELETE_WALK_OK = 0,
|
||||
/* The numeric cap for this run was reached before every extra was removed.
|
||||
The walker removed exactly the entries the cap allowed and skipped (without
|
||||
removing) the rest, matching rsync's partial --max-delete behavior. */
|
||||
DELETE_WALK_LIMIT_REACHED,
|
||||
/* A traversal or unlink failure aborted the deletion (partial removal is
|
||||
possible, mirroring the delete pass). */
|
||||
DELETE_WALK_ERROR
|
||||
} DeleteWalkResult;
|
||||
|
||||
/* Receiver-side delete-protection rules for one walk. `base_rules` is the
|
||||
* command-line rule set the config frame carried (owner "" rules); `dir_rules`
|
||||
* is the received per-directory rule set (rules carrying their owner directory
|
||||
* and no-inherit flag). Either may be NULL. */
|
||||
typedef struct {
|
||||
const FilterRuleList* base_rules;
|
||||
const FilterRuleList* dir_rules;
|
||||
/* When non-NULL and non-empty, a destination entry whose name ends with this
|
||||
suffix is protected from deletion. rsync never treats a --backup file as
|
||||
an extra, so a backup created at --delay-updates publication (or a
|
||||
pre-existing one) survives the delete-after pass. */
|
||||
const char* backup_suffix;
|
||||
} DeleteProtectRules;
|
||||
|
||||
/* rsync's first-match-wins receiver verdict for one candidate extra: the
|
||||
* per-directory chain is evaluated first (the containing directory's rules,
|
||||
* then each ancestor's, then the receive root's), then the base rules. Returns
|
||||
* FILTER_ACTION_PROTECT when the entry is shielded by a receiver-side exclude,
|
||||
* FILTER_ACTION_RISK when an include explicitly leaves it at risk, or
|
||||
* FILTER_ACTION_NONE when no rule matched. */
|
||||
FilterAction delete_protect_verdict(const DeleteProtectRules* protect, const char* rel_path,
|
||||
const char* leaf, bool is_dir);
|
||||
|
||||
/* The backup suffix the delete walker must shield from deletion, or NULL when
|
||||
--backup is inactive or the configured suffix is unusable (empty, or holding
|
||||
a path separator). Matches the suffix file_save uses for backups. */
|
||||
const char* delete_backup_suffix(const Config* config);
|
||||
|
||||
/* One protected entry for the delete walker. When top_level_only is true the
|
||||
prefix is skipped only as a DIRECT child of dest_root (the --delay-updates
|
||||
staging directory, which must not hide genuine extras inside a nested
|
||||
destination directory that happens to share the staging name); otherwise the
|
||||
prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest
|
||||
basis trees, and the sender-side protected filter-excluded prefixes, which
|
||||
are never destination content). */
|
||||
typedef struct {
|
||||
const char* prefix;
|
||||
bool top_level_only;
|
||||
} DeleteSkipEntry;
|
||||
|
||||
/* A built skip-prefix set. `entries`/`count` are what path_under_skip_prefix()
|
||||
consumes. `owned_prefixes` holds any prefix strings the builder had to
|
||||
allocate (root-relative basis-dir conversions); it is NULL when every prefix
|
||||
is borrowed from the config or the caller's lists. Release with
|
||||
delete_skips_free(). */
|
||||
typedef struct {
|
||||
DeleteSkipEntry* entries;
|
||||
char** owned_prefixes;
|
||||
int count;
|
||||
int owned_count;
|
||||
} DeleteSkipSet;
|
||||
|
||||
/* True when child_rel is, or lies below, one of the protected entries (a prefix
|
||||
"a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect
|
||||
only DIRECT children of the destination root, i.e. child_rel has no '/'). */
|
||||
bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips,
|
||||
int skip_count);
|
||||
|
||||
/* One destination-directory entry collected up front so the delete walkers can
|
||||
reproduce rsync's traversal order instead of readdir() order. rsync processes
|
||||
a directory's extraneous subdirectories first (descending name, depth-first),
|
||||
then its extraneous files (descending name), and only afterwards descends into
|
||||
its kept subdirectories (ascending name). */
|
||||
typedef struct {
|
||||
char* name;
|
||||
bool is_dir;
|
||||
/* The entry's full st_mode from the AT_SYMLINK_NOFOLLOW stat, so a delete
|
||||
observer can classify a removed non-directory as reg/link/special. */
|
||||
mode_t mode;
|
||||
} DeleteDirEntry;
|
||||
/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."),
|
||||
stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of
|
||||
*count entries whose names the caller frees with delete_dir_entries_free().
|
||||
Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is
|
||||
skipped, any other stat failure is reported through *operation_ok while the
|
||||
walk continues. */
|
||||
bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok);
|
||||
void delete_dir_entries_free(DeleteDirEntry* entries, size_t count);
|
||||
/* Sort comparators: `_desc` orders subdirectories before files and each group by
|
||||
descending name (rsync's extraneous-entry order); `_asc` orders plain ascending
|
||||
name (rsync's kept-subdirectory order). */
|
||||
int delete_dir_entry_cmp_desc(const void* a, const void* b);
|
||||
int delete_dir_entry_cmp_asc(const void* a, const void* b);
|
||||
|
||||
/* Remove files/dirs/symlinks under dest_root that are not listed in manifest
|
||||
without ever descending into a protected prefix (see DeleteSkipEntry). When
|
||||
`synced_dirs` is non-NULL, extras are only removed directly inside a directory
|
||||
whose destination-relative path is an exact entry in that list (the receive
|
||||
root is the "." sentinel); directories outside the synchronized set are still
|
||||
descended into so kept content below a listed directory is preserved, but
|
||||
nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of
|
||||
treating the whole destination tree as deletable. `max_delete` caps the
|
||||
number of removed entries (SIZE_MAX = unlimited): the walker removes up to the
|
||||
cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained.
|
||||
`deleted_out`/`skipped_out` optionally receive the number of entries removed
|
||||
and the number skipped because of the cap. */
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, size_t* deleted_out,
|
||||
size_t* skipped_out);
|
||||
|
||||
/* Entry kind of a removed path, reported to the delete observer so the receiver
|
||||
can build rsync's `--stats` `Number of deleted files` per-type breakdown. The
|
||||
four categories are a strict partition of every removed entry. */
|
||||
typedef enum {
|
||||
DELETE_ENTRY_REG = 0,
|
||||
DELETE_ENTRY_DIR,
|
||||
DELETE_ENTRY_LINK,
|
||||
DELETE_ENTRY_SPECIAL
|
||||
} DeleteEntryType;
|
||||
|
||||
/* Optional per-deletion observer: called for each destination-relative path
|
||||
actually removed (a file, symlink, or directory) with its entry kind, in
|
||||
removal order, so the receiver can stream rsync's `--info=del`/`--info=remove`
|
||||
lines and tally the per-type `--stats` counters. */
|
||||
typedef void (*DeletePathObserver)(void* context, const char* rel_path, DeleteEntryType type);
|
||||
|
||||
/* Classify a removed entry from its st_mode for the per-type delete counters. */
|
||||
DeleteEntryType delete_entry_type_of_mode(mode_t mode);
|
||||
|
||||
/* `delete_extras_limited_observed` is delete_extras_limited with an optional
|
||||
* observer; the observer is invoked only for entries truly removed. When
|
||||
* `protect` is non-NULL its receiver-side verdict is evaluated for every
|
||||
* candidate extra: a first-match PROTECT leaves the entry (and, for a
|
||||
* directory, its whole subtree) in place, while RISK/NONE fall through to the
|
||||
* ordinary skip-prefix/keep-set logic. */
|
||||
DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect,
|
||||
size_t* deleted_out, size_t* skipped_out,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context);
|
||||
/* Read-only companion to delete_extras_limited: walk the destination exactly as
|
||||
the delete pass would and APPEND (strdup'd) destination-relative paths that
|
||||
WOULD be removed, without touching disk. Used for -n/--dry-run --delete
|
||||
would-delete reporting. Returns true on a clean walk; the caller owns the
|
||||
strings appended to `out` and receives their count in *count_out. */
|
||||
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, ArrayList* out, size_t* count_out);
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest);
|
||||
|
||||
/* Build the delete walk's skip-prefix set from the config's --delay-updates
|
||||
staging directory, its --compare-dest/--copy-dest/--link-dest basis dirs, and
|
||||
the caller-supplied protection lists, in that order. `protected_paths` and
|
||||
`size_skipped` are borrowed (may be NULL); every entry in them is protected at
|
||||
any depth. The staging directory is protected only as a DIRECT child of the
|
||||
receive root. `basis_root_relative` selects how a basis path becomes a
|
||||
prefix: true converts an absolute path under the receive root to its
|
||||
root-relative form (the whole-tree commit walk; an unreachable path
|
||||
contributes no slot), false keeps the configured path verbatim (the
|
||||
per-directory plan walk). On success the caller releases `*out` with
|
||||
delete_skips_free(); returns false on allocation failure. */
|
||||
bool delete_skips_build(const Config* config, const ArrayList* protected_paths,
|
||||
const ArrayList* size_skipped, bool basis_root_relative,
|
||||
DeleteSkipSet* out);
|
||||
void delete_skips_free(DeleteSkipSet* set);
|
||||
|
||||
/* Convert one basis-directory path to the receive-root-relative protection
|
||||
prefix the delete walker uses (NULL when it lies outside the root). Exposed
|
||||
for unit tests of the root-of-"/" and normalization edge cases. */
|
||||
char* delete_basis_relative(const Config* config, const char* path);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,530 @@
|
||||
#include <errno.h>
|
||||
#include <ctype.h>
|
||||
#include <dirent.h>
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/sysmacros.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "array_list.h"
|
||||
#include "charset.h"
|
||||
#include "chmod.h"
|
||||
#include "chunk.h"
|
||||
#include "compression.h"
|
||||
#include "config.h"
|
||||
#include "data.h"
|
||||
#include "delay_updates.h"
|
||||
#include "delete_commit.h"
|
||||
#include "delete_plan.h"
|
||||
#include "delta.h"
|
||||
#include "file.h"
|
||||
#include "format.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include "metadata.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include "xattr.h"
|
||||
|
||||
#define MAX_SERVER_DELETE_COUNT 100000U
|
||||
/* Retained cost of one delete-manifest entry beyond its path bytes: the
|
||||
ArrayList pointer slot plus an approximate malloc header/rounding for the
|
||||
heap copy. Charged against MAX_MANIFEST_BYTES so a frame full of tiny paths
|
||||
cannot retain far more than the byte budget (B5). */
|
||||
#define MANIFEST_ENTRY_OVERHEAD (sizeof(char*) + 16)
|
||||
|
||||
/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already
|
||||
been consumed): a keep-set entry count followed by that many
|
||||
destination-relative paths, then a protected-prefix count followed by that
|
||||
many destination-relative prefixes, then a missing-args count followed by that
|
||||
many destination-relative delete paths, then (protocol 2.23.0) a
|
||||
synchronized-directory count followed by that many destination-relative
|
||||
directory paths (the receive root is the "." sentinel). The frame is
|
||||
self-delimiting (the counts are authoritative), so the caller decides what to
|
||||
do next and continues reading the following STATUS_* frame. Every section is
|
||||
validated identically: an entry must be non-empty, relative and traversal-free
|
||||
and the aggregate length across ALL sections is capped by MAX_MANIFEST_BYTES
|
||||
(so the missing-args deletion requests are confined like the rest of the
|
||||
manifest). Returns an owned DeleteManifest, or NULL after sending STATUS_ERROR
|
||||
when the frame is malformed (bad count, empty/absolute path, path traversal,
|
||||
or an aggregate size beyond MAX_MANIFEST_BYTES). */
|
||||
static bool receive_manifest_section(int fd, ArrayList* list, size_t* manifest_bytes,
|
||||
size_t* manifest_entries) {
|
||||
int count;
|
||||
if (!receive_int(fd, &count)) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
if (count < 0 || count > MAX_MANIFEST_ENTRIES ||
|
||||
(size_t)count > MAX_MANIFEST_ENTRIES - *manifest_entries) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < count; i++) {
|
||||
char* s = receive_wire_str(fd);
|
||||
size_t entry_size = s ? strlen(s) + MANIFEST_ENTRY_OVERHEAD : 0;
|
||||
if (!s || s[0] == '\0' || s[0] == '/' || has_path_traversal(s) ||
|
||||
entry_size > MAX_MANIFEST_BYTES - *manifest_bytes ||
|
||||
(*manifest_bytes += entry_size) > MAX_MANIFEST_BYTES || !array_list_add(list, s)) {
|
||||
free(s);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
*manifest_entries += (size_t)count;
|
||||
return true;
|
||||
}
|
||||
|
||||
DeleteManifest* receive_manifest_entries(int fd) {
|
||||
DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest));
|
||||
if (!manifest) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return NULL;
|
||||
}
|
||||
manifest->keeps = array_list_create(free);
|
||||
manifest->protected = array_list_create(free);
|
||||
manifest->missing = array_list_create(free);
|
||||
manifest->dirs = array_list_create(free);
|
||||
if (!manifest->keeps || !manifest->protected || !manifest->missing || !manifest->dirs) {
|
||||
delete_manifest_free(manifest);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return NULL;
|
||||
}
|
||||
size_t manifest_bytes = 0;
|
||||
size_t manifest_entries = 0;
|
||||
if (!receive_manifest_section(fd, manifest->keeps, &manifest_bytes, &manifest_entries) ||
|
||||
!receive_manifest_section(fd, manifest->protected, &manifest_bytes, &manifest_entries) ||
|
||||
!receive_manifest_section(fd, manifest->missing, &manifest_bytes, &manifest_entries) ||
|
||||
!receive_manifest_section(fd, manifest->dirs, &manifest_bytes, &manifest_entries)) {
|
||||
delete_manifest_free(manifest);
|
||||
return NULL;
|
||||
}
|
||||
/* Per-directory filter rules (protocol 2.30.0) follow the manifest sections
|
||||
* with their own bounded self-describing format. */
|
||||
if (!delete_filter_dir_rules_receive(fd, &manifest->per_dir_rules)) {
|
||||
delete_manifest_free(manifest);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return NULL;
|
||||
}
|
||||
return manifest;
|
||||
}
|
||||
|
||||
void delete_manifest_free(DeleteManifest* manifest) {
|
||||
if (!manifest)
|
||||
return;
|
||||
array_list_delete(manifest->keeps);
|
||||
array_list_delete(manifest->protected);
|
||||
array_list_delete(manifest->missing);
|
||||
array_list_delete(manifest->dirs);
|
||||
filter_rule_list_free(manifest->per_dir_rules);
|
||||
free(manifest);
|
||||
}
|
||||
|
||||
/* Shared --max-delete budget for one receiver-side deletion commit. Both the
|
||||
--delete-missing-args exact-path removals and the ordinary extras walk draw
|
||||
from the same tally, matching rsync (whose --max-delete counts every deleted
|
||||
file or directory). `max_delete` is SIZE_MAX for an unlimited budget. */
|
||||
typedef struct {
|
||||
size_t max_delete;
|
||||
size_t deleted;
|
||||
size_t skipped;
|
||||
bool limit_hit;
|
||||
} DeleteBudgetState;
|
||||
|
||||
/* Remove every destination entry under the receive root that is not in the
|
||||
keep-set, bounded by the shared budget (a smaller client --max-delete=NUM
|
||||
replaces the server hard bound; rsync deletes up to the bound and skips the
|
||||
rest). With --delay-updates the not-yet-published staging directory is a
|
||||
direct child of the receive root and must not be treated as a set of extras;
|
||||
the manifest's protected prefixes (paths excluded on the source), the
|
||||
size-pruned prefixes (--max-size/--min-size, always protected) and the
|
||||
alternate basis directories are never destination content and are skipped at
|
||||
any depth. Returns true unless a traversal/unlink error aborted the walk;
|
||||
the budget's limit_hit/skipped fields report a cap-stopped run. */
|
||||
static bool delete_extras_budgeted_observed(const Config* config, const DeleteManifest* manifest,
|
||||
DeleteBudgetState* budget, DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (!config || !manifest || !manifest->keeps)
|
||||
return false;
|
||||
fprintf(stderr, "Deleting files not in manifest...\n");
|
||||
/* Protected entries: the --delay-updates staging name (only as a DIRECT child
|
||||
of the receive root), the alternate basis directories and the sender-side
|
||||
protected prefixes (filter-excluded and size-pruned source mirrors), all at
|
||||
any depth. See delete_skips_build(). */
|
||||
DeleteSkipSet skips;
|
||||
if (!delete_skips_build(config, manifest->protected, NULL, true, &skips))
|
||||
return false;
|
||||
/* Clamp rather than subtract: an accounting bug where deleted already exceeds
|
||||
max_delete must never underflow into an effectively unlimited budget. */
|
||||
size_t remaining;
|
||||
if (budget->max_delete == SIZE_MAX)
|
||||
remaining = SIZE_MAX;
|
||||
else if (budget->deleted >= budget->max_delete)
|
||||
remaining = 0;
|
||||
else
|
||||
remaining = budget->max_delete - budget->deleted;
|
||||
size_t deleted = 0;
|
||||
size_t skipped = 0;
|
||||
DeleteProtectRules protect = {.base_rules = config->protect_rules,
|
||||
.dir_rules = manifest->per_dir_rules,
|
||||
.backup_suffix = delete_backup_suffix(config)};
|
||||
DeleteWalkResult result = delete_extras_limited_observed(
|
||||
config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips.entries,
|
||||
skips.count, &protect, &deleted, &skipped, observer, observer_context);
|
||||
delete_skips_free(&skips);
|
||||
budget->deleted += deleted;
|
||||
budget->skipped += skipped;
|
||||
if (result == DELETE_WALK_LIMIT_REACHED) {
|
||||
budget->limit_hit = true;
|
||||
return true;
|
||||
}
|
||||
if (result != DELETE_WALK_OK) {
|
||||
log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool delete_extras_budgeted(const Config* config, const DeleteManifest* manifest,
|
||||
DeleteBudgetState* budget) {
|
||||
return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL);
|
||||
}
|
||||
|
||||
/* Prefixes every observed path with a fixed subtree root, so a nested walk
|
||||
(a recursively removed missing-arg directory) reports receive-root-relative
|
||||
names like the rest of the delete output. */
|
||||
typedef struct {
|
||||
DeletePathObserver inner;
|
||||
void* inner_context;
|
||||
const char* prefix;
|
||||
} PrefixedDeleteObserver;
|
||||
|
||||
static void prefixed_delete_observer(void* context, const char* rel, DeleteEntryType type) {
|
||||
PrefixedDeleteObserver* prefixed = context;
|
||||
if (!prefixed->inner || !rel)
|
||||
return;
|
||||
char* joined = path_cat((char*)prefixed->prefix, rel);
|
||||
if (joined) {
|
||||
prefixed->inner(prefixed->inner_context, joined, type);
|
||||
free(joined);
|
||||
}
|
||||
}
|
||||
|
||||
/* --delete-missing-args exact-path deletions: each destination mirror in
|
||||
manifest->missing is an explicit user request, so it is removed even when the
|
||||
ordinary extras walk (with its protected prefixes) would leave it alone. The
|
||||
--delay-updates staging directory and basis snapshots are receiver artifacts
|
||||
and stay protected exactly as in the extras walker. A regular file or
|
||||
symlink is unlinked, an empty directory removed, and a NON-empty directory is
|
||||
removed recursively only when --delete or --force is in effect (rsync parity:
|
||||
the man page says a non-empty directory mirror is only deleted with --force
|
||||
or --delete); otherwise it is left with a warning and the run continues. A
|
||||
mirror that does not exist is a no-op. Each removal draws from the shared
|
||||
--max-delete budget: once it is exhausted the remaining requests are skipped
|
||||
and counted. Returns false only on a genuine error (a confinement failure on
|
||||
a validated path or an I/O error), which fails the run. */
|
||||
|
||||
/* How one missing-args request leaves the driver loop. The original walker
|
||||
`continue`s past an invalid/protected/absent/budget-skipped request (without
|
||||
breaking) but stops after a request that ran to completion while an error is
|
||||
pending; NEXT/STOP preserve that control flow exactly. */
|
||||
typedef enum { MISSING_ARG_NEXT, MISSING_ARG_STOP } MissingArgStep;
|
||||
|
||||
/* Remove a NON-empty missing-args directory recursively (--delete/--force in
|
||||
effect): walk its contents through the budgeted extras walker so every removed
|
||||
file/dir counts toward --max-delete (rsync parity), then remove the now-empty
|
||||
directory itself, which costs one more budget unit. A run that hits the cap
|
||||
leaves the remaining entries in place. The observer is wrapped so the nested
|
||||
walk reports receive-root-relative paths. Sets the *removed and *ok outputs. */
|
||||
static void delete_nonempty_missing_dir(const char* full, const char* rel,
|
||||
DeleteBudgetState* budget, DeletePathObserver observer,
|
||||
void* observer_context, bool* removed, bool* ok) {
|
||||
ArrayList* no_keeps = array_list_create(free);
|
||||
/* Never let an accounting slip (deleted > max_delete) underflow the remaining
|
||||
budget into SIZE_MAX, which would grant unlimited deletions. */
|
||||
size_t remaining =
|
||||
budget->deleted >= budget->max_delete ? 0 : budget->max_delete - budget->deleted;
|
||||
size_t contents_deleted = 0;
|
||||
size_t contents_skipped = 0;
|
||||
PrefixedDeleteObserver nested = {observer, observer_context, rel};
|
||||
DeleteWalkResult walk =
|
||||
no_keeps ? delete_extras_limited_observed(full, no_keeps, NULL, remaining, NULL, 0, NULL,
|
||||
&contents_deleted, &contents_skipped,
|
||||
observer ? prefixed_delete_observer : NULL,
|
||||
observer ? &nested : NULL)
|
||||
: DELETE_WALK_ERROR;
|
||||
if (no_keeps)
|
||||
array_list_delete(no_keeps);
|
||||
budget->deleted += contents_deleted;
|
||||
budget->skipped += contents_skipped;
|
||||
if (walk == DELETE_WALK_LIMIT_REACHED) {
|
||||
budget->limit_hit = true;
|
||||
} else if (walk != DELETE_WALK_OK) {
|
||||
*ok = false;
|
||||
} else if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
} else if (file_remove_tree_secure(full)) {
|
||||
/* The shared `if (removed)` tail charges this directory exactly once;
|
||||
counting it here too would consume two budget units. */
|
||||
*removed = true;
|
||||
} else {
|
||||
*ok = false;
|
||||
}
|
||||
}
|
||||
|
||||
/* Remove one missing-args destination mirror. `skips` holds the receiver
|
||||
artifacts (staging directory, basis snapshots) that stay protected. Returns
|
||||
MISSING_ARG_STOP when the driver loop must stop (a completed removal left a
|
||||
genuine error pending) and MISSING_ARG_NEXT otherwise; *ok accumulates the
|
||||
overall success across the whole run. */
|
||||
static MissingArgStep delete_one_missing_arg(const Config* config, const char* rel,
|
||||
const DeleteSkipSet* skips, DeleteBudgetState* budget,
|
||||
DeletePathObserver observer, void* observer_context,
|
||||
bool* ok) {
|
||||
if (!rel || *rel == '\0' || *rel == '/' || has_path_traversal(rel)) {
|
||||
/* Defensive only: receive_manifest_entries already validated every
|
||||
section identically, so a controlled peer never reaches this branch. */
|
||||
log_message(LOG_LEVEL_ERROR, "invalid missing-args delete path");
|
||||
*ok = false;
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
bool at_root = strchr(rel, '/') == NULL;
|
||||
if (path_under_skip_prefix(rel, at_root, skips->entries, skips->count)) {
|
||||
char* escaped = output_escape(rel, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"missing-args path '%s' is protected (staging directory or basis snapshot); "
|
||||
"not deleting",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
char* full = path_cat(config->receive_root_directory, rel);
|
||||
if (!full) {
|
||||
*ok = false;
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(full, &leaf, false);
|
||||
if (parent_fd < 0) {
|
||||
/* The mirror's parent directory may itself not exist on the destination
|
||||
(a deeper missing entry whose leading directories were never created).
|
||||
That is a no-op -- there is nothing to delete -- matching
|
||||
file_remove_tree_secure's absent-path handling; only a genuine I/O
|
||||
error (EACCES, a symlink loop, ...) fails the run. */
|
||||
bool absent = errno == ENOENT || errno == ENOTDIR;
|
||||
free(full);
|
||||
free(leaf);
|
||||
if (!absent)
|
||||
*ok = false;
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
/* Already absent: nothing to delete (a no-op, not a deletion). */
|
||||
if (errno != ENOENT)
|
||||
*ok = false;
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
free(full);
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
/* An entry that exists is one deletion: skip it (and count it) when the
|
||||
shared --max-delete budget is already exhausted. */
|
||||
if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
free(full);
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
bool removed = false;
|
||||
if (S_ISDIR(st.st_mode)) {
|
||||
if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) {
|
||||
removed = true;
|
||||
} else if (errno == ENOTEMPTY || errno == EEXIST) {
|
||||
close(parent_fd);
|
||||
parent_fd = -1;
|
||||
free(leaf);
|
||||
leaf = NULL;
|
||||
if (config->use_delete || config->force_delete) {
|
||||
delete_nonempty_missing_dir(full, rel, budget, observer, observer_context, &removed, ok);
|
||||
} else {
|
||||
char* escaped = output_escape(rel, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"missing-args destination '%s' is a non-empty directory; use --force or "
|
||||
"--delete to remove it",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
}
|
||||
} else if (errno != ENOENT) {
|
||||
*ok = false;
|
||||
}
|
||||
} else {
|
||||
if (unlinkat(parent_fd, leaf, 0) == 0) {
|
||||
removed = true;
|
||||
} else if (errno != ENOENT) {
|
||||
*ok = false;
|
||||
}
|
||||
}
|
||||
if (removed) {
|
||||
budget->deleted++;
|
||||
if (observer)
|
||||
observer(observer_context, rel, delete_entry_type_of_mode(st.st_mode));
|
||||
char* escaped = output_escape(rel, log_get_8_bit_output());
|
||||
fprintf(stderr, " Deleted: %s\n", escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
}
|
||||
if (parent_fd >= 0)
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
free(full);
|
||||
return *ok ? MISSING_ARG_NEXT : MISSING_ARG_STOP;
|
||||
}
|
||||
|
||||
static bool delete_missing_args_budgeted_observed(const Config* config,
|
||||
const DeleteManifest* manifest,
|
||||
DeleteBudgetState* budget,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (!config || !manifest)
|
||||
return false;
|
||||
if (!manifest->missing || manifest->missing->size == 0)
|
||||
return true;
|
||||
fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n");
|
||||
/* The staging directory and basis snapshots stay protected exactly as in the
|
||||
extras walker (the missing-args path overrides the ordinary protected
|
||||
prefixes, so those are not passed here). */
|
||||
DeleteSkipSet skips;
|
||||
if (!delete_skips_build(config, NULL, NULL, true, &skips))
|
||||
return false;
|
||||
bool ok = true;
|
||||
for (int i = 0; i < manifest->missing->size; i++) {
|
||||
const char* rel = (const char*)manifest->missing->items[i];
|
||||
if (delete_one_missing_arg(config, rel, &skips, budget, observer, observer_context, &ok) ==
|
||||
MISSING_ARG_STOP)
|
||||
break;
|
||||
}
|
||||
delete_skips_free(&skips);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Public wrappers used outside the commit path (and by unit tests): no
|
||||
--max-delete budget. */
|
||||
bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest,
|
||||
ArrayList* out, size_t* count_out) {
|
||||
if (count_out)
|
||||
*count_out = 0;
|
||||
if (!config || !manifest || !manifest->keeps || !out)
|
||||
return false;
|
||||
DeleteSkipSet skips;
|
||||
if (!delete_skips_build(config, manifest->protected, NULL, true, &skips))
|
||||
return false;
|
||||
DeleteProtectRules protect = {.base_rules = config->protect_rules,
|
||||
.dir_rules = manifest->per_dir_rules,
|
||||
.backup_suffix = delete_backup_suffix(config)};
|
||||
bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs,
|
||||
skips.entries, skips.count, &protect, out, count_out);
|
||||
delete_skips_free(&skips);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest) {
|
||||
DeleteBudgetState budget = {
|
||||
.max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
return delete_extras_budgeted(config, manifest, &budget);
|
||||
}
|
||||
|
||||
bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest) {
|
||||
DeleteBudgetState budget = {
|
||||
.max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL);
|
||||
}
|
||||
|
||||
bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest,
|
||||
size_t max_delete, size_t* deleted, size_t* skipped,
|
||||
bool* limit_hit) {
|
||||
return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted,
|
||||
skipped, limit_hit, NULL, NULL);
|
||||
}
|
||||
|
||||
bool manifest_delete_missing_args_limited_observed(
|
||||
const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted,
|
||||
size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context) {
|
||||
DeleteBudgetState budget = {
|
||||
.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
bool ok =
|
||||
delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context);
|
||||
if (deleted)
|
||||
*deleted = budget.deleted;
|
||||
if (skipped)
|
||||
*skipped = budget.skipped;
|
||||
if (limit_hit)
|
||||
*limit_hit = budget.limit_hit;
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Commit every deletion family the manifest carries. The --delete-missing-args
|
||||
exact-path deletions run FIRST: they are explicit user requests and must not
|
||||
be blocked by the extras walker's filter-exclusion protection (a protected
|
||||
leftover inside a missing-argument directory must not make that user-requested
|
||||
removal fail). The ordinary extras walk then runs when --delete is active.
|
||||
Both draw from one --max-delete budget; the result reports a cap-stopped
|
||||
(partial) commit distinctly so the client can exit 25 like rsync. */
|
||||
DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest) {
|
||||
return manifest_delete_all_counted(config, manifest, NULL);
|
||||
}
|
||||
|
||||
DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest,
|
||||
size_t* deleted) {
|
||||
return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL);
|
||||
}
|
||||
|
||||
DeleteCommitResult manifest_delete_all_observed(const Config* config,
|
||||
const DeleteManifest* manifest, size_t* deleted,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (deleted)
|
||||
*deleted = 0;
|
||||
if (!config || !manifest)
|
||||
return DELETE_COMMIT_ERROR;
|
||||
/* Central no-mutation guard: a dry-run never deletes. No manifest is sent on
|
||||
the dry-run path, but a hostile/buggy peer could; treat it as a no-op so
|
||||
the receiver can never remove anything. */
|
||||
if (config->dry_run)
|
||||
return DELETE_COMMIT_OK;
|
||||
/* A client --max-delete=NUM smaller than the server's hard bound replaces it
|
||||
for this run; both still bound the commit. */
|
||||
bool user_limited =
|
||||
config->max_delete >= 0 && (size_t)config->max_delete < MAX_SERVER_DELETE_COUNT;
|
||||
DeleteBudgetState budget = {.max_delete = user_limited ? (size_t)config->max_delete
|
||||
: MAX_SERVER_DELETE_COUNT,
|
||||
.deleted = 0,
|
||||
.skipped = 0,
|
||||
.limit_hit = false};
|
||||
if (config->delete_missing_args &&
|
||||
!delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context))
|
||||
return DELETE_COMMIT_ERROR;
|
||||
if (config->use_delete &&
|
||||
!delete_extras_budgeted_observed(config, manifest, &budget, observer, observer_context))
|
||||
return DELETE_COMMIT_ERROR;
|
||||
if (deleted)
|
||||
*deleted = budget.deleted;
|
||||
if (budget.limit_hit) {
|
||||
if (user_limited) {
|
||||
log_message(LOG_LEVEL_ERROR, "Deletions stopped due to --max-delete limit (%zu skipped)",
|
||||
budget.skipped);
|
||||
} else {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"Deletions stopped due to the server deletion limit of %u (%zu skipped)",
|
||||
(unsigned)MAX_SERVER_DELETE_COUNT, budget.skipped);
|
||||
}
|
||||
return DELETE_COMMIT_LIMIT_REACHED;
|
||||
}
|
||||
return DELETE_COMMIT_OK;
|
||||
}
|
||||
@@ -0,0 +1,113 @@
|
||||
#ifndef DELETE_COMMIT_H
|
||||
#define DELETE_COMMIT_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "delete.h"
|
||||
#include "filter.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Delete-commit module: delete-manifest receive plus the budgeted extras and
|
||||
* --delete-missing-args walkers. These declarations are re-exported by the
|
||||
* file_receive.h facade. */
|
||||
|
||||
/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative
|
||||
paths the sender transferred/keeps) plus `protected`, destination-relative
|
||||
prefixes the sender asks the receiver never to delete (paths excluded on the
|
||||
source, protected at any depth). When --delete-excluded is given the sender
|
||||
transmits an empty protected list so excluded destination mirrors are treated
|
||||
as ordinary extras. With --delete-missing-args a third section (`missing`)
|
||||
carries the destination mirrors of explicitly-listed source entries that do
|
||||
not exist: each is an exact deletion request, independent of the ordinary
|
||||
extras walk (never blocked by the protected prefixes) and processed when the
|
||||
manifest is committed. */
|
||||
typedef struct DeleteManifest {
|
||||
ArrayList* keeps;
|
||||
ArrayList* protected;
|
||||
ArrayList* missing;
|
||||
/* Destination-relative paths of the directories the sender synchronized for
|
||||
this run. The extras walker only removes entries directly inside one of
|
||||
these (the receive root is the "." sentinel); `--files-from` runs therefore
|
||||
leave untransmitted directories and the unlisted parts of listed ones
|
||||
alone, matching rsync's "delete only in synchronized directories". */
|
||||
ArrayList* dirs;
|
||||
/* Per-directory filter rules the sender compiled while scanning (protocol
|
||||
2.30.0), each carrying its owner directory and no-inherit flag. The
|
||||
receiver evaluates them (deepest before ancestors, then the command-line
|
||||
base rules) against every candidate extra so a destination-only entry that
|
||||
matches ONLY a per-directory `.rsync-filter`/dir-merge rule is shielded.
|
||||
NULL when the sender transmitted none. */
|
||||
FilterRuleList* per_dir_rules;
|
||||
} DeleteManifest;
|
||||
|
||||
void delete_manifest_free(DeleteManifest* manifest);
|
||||
/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then
|
||||
protected count + protected prefixes, then missing count + missing paths,
|
||||
then synchronized-directory count + directory paths (self-delimiting; the
|
||||
leading STATUS_MANIFEST code has been consumed). Returns an owned
|
||||
DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */
|
||||
DeleteManifest* receive_manifest_entries(int fd);
|
||||
/* Remove destination entries under config->receive_root_directory that are not
|
||||
in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and
|
||||
protected-prefix skips). `--max-delete` and `--force` are honored here. The
|
||||
caller decides WHEN to run it based on the negotiated delete timing. Returns
|
||||
false (and the transfer fails) when the deletion cannot be committed. */
|
||||
bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest);
|
||||
/* --delete-missing-args exact-path deletions: remove each destination mirror
|
||||
in `manifest->missing` (never blocked by the protected prefixes, staging dir
|
||||
and basis dirs excluded). A regular file/symlink is unlinked; an empty
|
||||
directory is removed; a NON-empty directory is removed recursively only when
|
||||
--delete or --force is in effect, otherwise it is left with a warning (rsync
|
||||
parity). A missing path is a no-op. Returns false only on a genuine
|
||||
confinement or I/O error (the run then fails); tolerated per-path cases are
|
||||
reported and skipped. */
|
||||
bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest);
|
||||
/* Budgeted form of manifest_delete_missing_args for the per-directory delete
|
||||
session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited)
|
||||
and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set
|
||||
when the budget stopped the pass with entries left over. Returns false only
|
||||
on a genuine deletion error. */
|
||||
bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest,
|
||||
size_t max_delete, size_t* deleted, size_t* skipped,
|
||||
bool* limit_hit);
|
||||
/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may
|
||||
be NULL) is invoked for every destination-relative path truly removed. */
|
||||
bool manifest_delete_missing_args_limited_observed(
|
||||
const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted,
|
||||
size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context);
|
||||
/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's
|
||||
partial --max-delete result: the budget allowed some deletions and the rest
|
||||
were skipped (the run still stores all file data but the client exits 25). */
|
||||
typedef enum {
|
||||
DELETE_COMMIT_OK = 0,
|
||||
DELETE_COMMIT_LIMIT_REACHED,
|
||||
DELETE_COMMIT_ERROR
|
||||
} DeleteCommitResult;
|
||||
|
||||
/* Run every deletion family the manifest carries: the --delete-missing-args
|
||||
exact-path deletions first (user requests are not blocked by exclusion
|
||||
protection), then the ordinary extras walk when --delete is active. Both
|
||||
share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to
|
||||
do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget
|
||||
stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */
|
||||
DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest);
|
||||
/* Like manifest_delete_all, but reports how many destination entries the commit
|
||||
removed (for the end-of-transfer wire stats). `deleted` may be NULL. */
|
||||
DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest,
|
||||
size_t* deleted);
|
||||
/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL)
|
||||
is invoked for every destination-relative path truly removed. */
|
||||
DeleteCommitResult manifest_delete_all_observed(const Config* config,
|
||||
const DeleteManifest* manifest, size_t* deleted,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context);
|
||||
|
||||
/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as
|
||||
the delete pass would and append (strdup'd) destination-relative paths that
|
||||
WOULD be removed to `out`, without touching disk. Uses the same staging-dir,
|
||||
basis-dir and protected-prefix skips as the real commit. Returns true on a
|
||||
clean walk; `*count_out` receives the number of paths appended. */
|
||||
bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest,
|
||||
ArrayList* out, size_t* count_out);
|
||||
|
||||
#endif
|
||||
File diff suppressed because it is too large.
Load diff
@@ -0,0 +1,120 @@
|
||||
#ifndef DELETE_PLAN_H
|
||||
#define DELETE_PLAN_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "delete.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Per-directory delete plans (protocol 2.24.0).
|
||||
*
|
||||
* rsync's --delete-during removes a directory's extras while the generator
|
||||
* processes that directory, and --delete-delay records the deletion list during
|
||||
* the scan but applies it only after a fully-successful transfer. FastSync has
|
||||
* no per-directory generator pass; instead the sender streams one plan per
|
||||
* source directory, in directory order, and the receiver applies it when it
|
||||
* arrives (during) or snapshots its extras and commits them at the end (delay).
|
||||
*
|
||||
* The sender side builds a plan set from the path-only pre-scan (it needs every
|
||||
* directory's complete direct-child list before the first data byte of that
|
||||
* directory). The receiver side is a session that carries the global protected
|
||||
* prefixes (filter-excluded and size-skipped source mirrors), the
|
||||
* --delete-missing-args exact deletions, the shared --max-delete budget and,
|
||||
* for --delete-delay, the snapshotted extras. */
|
||||
|
||||
/* Per-directory filter-rule block (protocol 2.30.0). The sender compiles the
|
||||
* source's per-directory merge rules as it scans and streams them so the
|
||||
* receiver can re-derive the receiver-side protect/risk verdicts for
|
||||
* destination-only entries. The wire format is a group count, then for each
|
||||
* directory group its relative owner path followed by that directory's rule
|
||||
* records (action, sides, anchored, dir-only, negate, no-inherit, pattern).
|
||||
* All bounds (MAX_FILTER_RULES, MAX_FILTER_BYTES, MAX_PROTECT_PATTERN_LEN) are
|
||||
* enforced on both sides; a malformed receive frame signals STATUS_ERROR and
|
||||
* returns false. Receive yields a flat FilterRuleList whose rules carry their
|
||||
* owner, or NULL when no rules were sent. */
|
||||
bool delete_filter_dir_rules_send(int fd, const FilterRuleList* rules);
|
||||
bool delete_filter_dir_rules_receive(int fd, FilterRuleList** out);
|
||||
|
||||
/* ---- Sender: plan builder ---- */
|
||||
|
||||
typedef struct DeletePlanSender DeletePlanSender;
|
||||
|
||||
DeletePlanSender* delete_plan_sender_create(void);
|
||||
void delete_plan_sender_destroy(DeletePlanSender* sender);
|
||||
/* Record one transmitted entry. `path` is the destination-relative wire path;
|
||||
* is_dir marks an explicit directory entry (--dirs, a -x mount point). */
|
||||
bool delete_plan_sender_add(DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Drop plans for directories outside `synced_dirs` (the --files-from
|
||||
* synchronization scope; pass NULL when a full recursive transfer synchronized
|
||||
* every directory). The receive root is the "." sentinel.
|
||||
*
|
||||
* `walk_root` scopes a general -R transfer: when non-NULL it is the
|
||||
* reconstructed destination prefix the run actually transferred, and only the
|
||||
* plan for that prefix (and directories below it) is ever transmitted, so the
|
||||
* prefix's parent-directory siblings are never walked. Pass NULL for a plain
|
||||
* recursive transfer and for --files-from. */
|
||||
void delete_plan_sender_finalize(DeletePlanSender* sender, const ArrayList* synced_dirs,
|
||||
const char* walk_root);
|
||||
/* True when no transmitted FILE entry was recorded (an ambiguous empty scan).
|
||||
Directory keep entries do not count, so an I/O error that hid every file
|
||||
still refuses to delete. */
|
||||
bool delete_plan_sender_empty(const DeletePlanSender* sender);
|
||||
/* Attach the global config sections advertised on the first plan frame. The
|
||||
* block is always transmitted by delete_plan_send_root(), on a config-only
|
||||
* carrier frame when the scope allows no directory plan. */
|
||||
void delete_plan_sender_set_config(DeletePlanSender* sender, const ArrayList* protected_prefixes,
|
||||
const ArrayList* size_skipped, const ArrayList* missing_args,
|
||||
const FilterRuleList* per_dir_rules);
|
||||
/* Send the root plan (even before any data, so root extras are handled like
|
||||
* rsync's first generator directory), after transmitting the per-run config
|
||||
* block on its own carrier frame. Returns -1 on I/O error. */
|
||||
int delete_plan_send_root(int fd, DeletePlanSender* sender);
|
||||
/* Send the plans for every ancestor of `path` (root-first) and, when is_dir,
|
||||
* for `path` itself; already-sent plans are skipped. */
|
||||
int delete_plan_send_for_path(int fd, DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Send the plan for every directory in `dirs` that has not been transmitted
|
||||
* yet. */
|
||||
int delete_plan_send_remaining(int fd, DeletePlanSender* sender, const ArrayList* dirs);
|
||||
/* Transmit the COMPLETE per-directory plan set in one pass, before any data
|
||||
* frame: the root plan (with the one-shot per-run config block on its carrier
|
||||
* frame) followed by every directory in `dirs`. Because the whole plan set is
|
||||
* known from the path-only pre-scan, sending it all up front means a
|
||||
* mid-transfer abort has already applied every planned removal, matching
|
||||
* rsync's generator (which runs ahead of its throttled sender). A completed
|
||||
* run is unaffected. `dirs` is the set of directories whose direct children
|
||||
* were enumerated (the scanner's plan_dirs sink), so a merely listed but
|
||||
* untraversed directory never gets a plan and its mirror is left intact.
|
||||
* Returns -1 on I/O error. */
|
||||
int delete_plan_send_all(int fd, DeletePlanSender* sender, const ArrayList* dirs);
|
||||
|
||||
/* ---- Receiver: delete session ---- */
|
||||
|
||||
typedef struct DeletePlanSession DeletePlanSession;
|
||||
|
||||
DeletePlanSession* delete_plan_session_create(const Config* config);
|
||||
void delete_plan_session_destroy(DeletePlanSession* session);
|
||||
/* Read one STATUS_DELETE_PLAN frame (the leading status already consumed) and
|
||||
* act on it. Returns 0 on success (including a dry-run/disabled no-op) and -1
|
||||
* after signalling STATUS_ERROR on a malformed frame or a deletion failure. */
|
||||
int delete_plan_session_receive(DeletePlanSession* session, const Config* config, int fd);
|
||||
/* Apply the deferred snapshot (--delete-delay) and the missing-args deletions.
|
||||
* Safe to call once; returns the commit outcome. */
|
||||
DeleteCommitResult delete_plan_session_commit(DeletePlanSession* session, const Config* config);
|
||||
/* True once the shared --max-delete budget stopped part of a deletion. */
|
||||
bool delete_plan_session_limit_reached(const DeletePlanSession* session);
|
||||
/* Number of destination entries the session actually removed, for the
|
||||
end-of-transfer stats. For --delete-delay this excludes a snapshotted entry
|
||||
that survived (e.g. a refilled directory that failed ENOTEMPTY), even though
|
||||
that entry already consumed --max-delete budget at snapshot time. */
|
||||
size_t delete_plan_session_deleted(const DeletePlanSession* session);
|
||||
/* Install an observer invoked for every destination-relative path the session
|
||||
truly removes (including the deferred --delete-delay commit), so the receiver
|
||||
can report rsync's `deleting PATH` lines through the terminal STATUS_STATS
|
||||
record. Pass NULL/0 to clear. */
|
||||
void delete_plan_session_set_delete_observer(DeletePlanSession* session,
|
||||
DeletePathObserver observer, void* context);
|
||||
|
||||
#endif
|
||||
@@ -1,10 +1,12 @@
|
||||
#include "delta.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include <errno.h>
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define XXH_STATIC_LINKING_ONLY
|
||||
#define XXH_IMPLEMENTATION
|
||||
@@ -82,6 +84,64 @@ DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_
|
||||
return sig;
|
||||
}
|
||||
|
||||
/* Bounded read of exactly `len` bytes at `off`; retries on EINTR. */
|
||||
static bool pread_all(int fd, void* buf, size_t len, uint64_t off) {
|
||||
uint8_t* p = buf;
|
||||
size_t done = 0;
|
||||
while (done < len) {
|
||||
ssize_t n = pread(fd, p + done, len - done, (off_t)(off + done));
|
||||
if (n < 0 && errno == EINTR)
|
||||
continue;
|
||||
if (n <= 0)
|
||||
return false;
|
||||
done += (size_t)n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
DeltaSignature* delta_signature_create_fd_seeded(int fd, uint64_t old_file_size,
|
||||
uint32_t block_size, uint32_t seed) {
|
||||
if (fd < 0 || old_file_size == 0 || block_size == 0 || block_size > DELTA_BLOCK_SIZE_MAX ||
|
||||
old_file_size > UINT32_MAX * (uint64_t)block_size)
|
||||
return NULL;
|
||||
uint32_t block_count = (uint32_t)((old_file_size + block_size - 1) / block_size);
|
||||
/* Bound the signature's own memory (block_count * sizeof(DeltaBlockSig)). */
|
||||
if (block_count == 0 || block_count > MAX_DELTA_BLOCKS)
|
||||
return NULL;
|
||||
DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature));
|
||||
if (!sig)
|
||||
return NULL;
|
||||
sig->file_size = old_file_size;
|
||||
sig->block_size = block_size;
|
||||
sig->block_count = block_count;
|
||||
sig->blocks = protocol_alloc((size_t)block_count * sizeof(DeltaBlockSig));
|
||||
if (!sig->blocks) {
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
uint8_t* block = malloc(block_size);
|
||||
if (!block) {
|
||||
free(sig->blocks);
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
for (uint32_t i = 0; i < block_count; i++) {
|
||||
uint64_t offset = (uint64_t)i * block_size;
|
||||
uint32_t len =
|
||||
(uint32_t)((old_file_size - offset < block_size) ? (old_file_size - offset) : block_size);
|
||||
if (!pread_all(fd, block, len, offset)) {
|
||||
free(block);
|
||||
free(sig->blocks);
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
sig->blocks[i].adler32 = delta_adler32(block, len);
|
||||
sig->blocks[i].xxhash = delta_xxhash32_seeded(block, len, seed);
|
||||
}
|
||||
free(block);
|
||||
return sig;
|
||||
}
|
||||
|
||||
Data* delta_signature_serialize(const DeltaSignature* sig) {
|
||||
if (!sig)
|
||||
return NULL;
|
||||
@@ -687,6 +747,130 @@ void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta,
|
||||
return output;
|
||||
}
|
||||
|
||||
void* delta_apply_fd(int src_fd, uint64_t old_size, const Delta* delta, uint32_t block_size) {
|
||||
if (!delta || block_size == 0 || block_size > DELTA_BLOCK_SIZE_MAX ||
|
||||
(delta->instruction_count > 0 && !delta->instructions) || delta->new_file_size == 0 ||
|
||||
delta->new_file_size > SIZE_MAX)
|
||||
return NULL;
|
||||
|
||||
void* output = protocol_alloc((size_t)delta->new_file_size);
|
||||
if (!output)
|
||||
return NULL;
|
||||
|
||||
uint8_t* out = (uint8_t*)output;
|
||||
uint64_t out_pos = 0;
|
||||
|
||||
for (uint32_t i = 0; i < delta->instruction_count; i++) {
|
||||
if (delta->instructions[i].type == DELTA_INSTR_BLOCK_MATCH) {
|
||||
uint64_t src_offset = (uint64_t)delta->instructions[i].match.block_index * block_size;
|
||||
if (src_offset > UINT64_MAX - delta->instructions[i].match.block_offset) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
src_offset += delta->instructions[i].match.block_offset;
|
||||
uint32_t len = delta->instructions[i].match.length;
|
||||
|
||||
if (src_offset > old_size || (uint64_t)len > old_size - src_offset ||
|
||||
out_pos > delta->new_file_size || (uint64_t)len > delta->new_file_size - out_pos) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
if (!pread_all(src_fd, out + out_pos, len, src_offset)) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
out_pos += len;
|
||||
} else if (delta->instructions[i].type == DELTA_INSTR_LITERAL) {
|
||||
uint32_t len = delta->instructions[i].literal.length;
|
||||
if (out_pos > delta->new_file_size || (uint64_t)len > delta->new_file_size - out_pos) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(out + out_pos, delta->instructions[i].literal.data, len);
|
||||
out_pos += len;
|
||||
} else {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
if (out_pos != delta->new_file_size) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
return output;
|
||||
}
|
||||
|
||||
bool delta_apply_to_fd(const void* old_data, int src_fd, uint64_t old_size, const Delta* delta,
|
||||
uint32_t block_size, int dst_fd) {
|
||||
if (!delta || (old_data == NULL && src_fd < 0) || block_size == 0 ||
|
||||
block_size > DELTA_BLOCK_SIZE_MAX || (delta->instruction_count > 0 && !delta->instructions))
|
||||
return false;
|
||||
const int chunk = 1 << 20;
|
||||
uint8_t* buf = malloc((size_t)chunk);
|
||||
if (!buf)
|
||||
return false;
|
||||
uint64_t out_pos = 0;
|
||||
bool ok = true;
|
||||
for (uint32_t i = 0; i < delta->instruction_count && ok; i++) {
|
||||
uint64_t src_offset = 0;
|
||||
uint64_t len = 0;
|
||||
const uint8_t* lit = NULL;
|
||||
if (delta->instructions[i].type == DELTA_INSTR_BLOCK_MATCH) {
|
||||
src_offset = (uint64_t)delta->instructions[i].match.block_index * block_size;
|
||||
if (src_offset > UINT64_MAX - delta->instructions[i].match.block_offset ||
|
||||
src_offset + delta->instructions[i].match.block_offset > old_size) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
src_offset += delta->instructions[i].match.block_offset;
|
||||
len = delta->instructions[i].match.length;
|
||||
if (len > old_size - src_offset) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
} else if (delta->instructions[i].type == DELTA_INSTR_LITERAL) {
|
||||
lit = delta->instructions[i].literal.data;
|
||||
len = delta->instructions[i].literal.length;
|
||||
} else {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (out_pos > delta->new_file_size || len > delta->new_file_size - out_pos) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
uint64_t done = 0;
|
||||
while (ok && done < len) {
|
||||
size_t want = (len - done) < (uint64_t)chunk ? (size_t)(len - done) : (size_t)chunk;
|
||||
if (lit) {
|
||||
memcpy(buf, lit + done, want);
|
||||
} else if (old_data) {
|
||||
memcpy(buf, (const uint8_t*)old_data + src_offset + done, want);
|
||||
} else if (!pread_all(src_fd, buf, want, src_offset + done)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
const uint8_t* p = buf;
|
||||
size_t written = 0;
|
||||
while (written < want) {
|
||||
ssize_t n = write(dst_fd, p + written, want - written);
|
||||
if (n < 0 && errno == EINTR)
|
||||
continue;
|
||||
if (n <= 0) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
written += (size_t)n;
|
||||
}
|
||||
done += want;
|
||||
}
|
||||
out_pos += len;
|
||||
}
|
||||
free(buf);
|
||||
return ok && out_pos == delta->new_file_size;
|
||||
}
|
||||
|
||||
void delta_destroy(Delta* delta) {
|
||||
if (!delta)
|
||||
return;
|
||||
|
||||
@@ -61,6 +61,13 @@ DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_f
|
||||
* identical to the unseeded function. */
|
||||
DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_t old_file_size,
|
||||
uint32_t block_size, uint32_t seed);
|
||||
/* Streaming equivalent of delta_signature_create_seeded: reads the basis blocks
|
||||
* from `fd` in bounded chunks, so an arbitrarily large basis can be signed
|
||||
* without materializing it. The signature itself is bounded (MAX_DELTA_BLOCKS
|
||||
* entries); an over-large basis returns NULL and the caller falls back to a
|
||||
* whole-file transfer. */
|
||||
DeltaSignature* delta_signature_create_fd_seeded(int fd, uint64_t old_file_size,
|
||||
uint32_t block_size, uint32_t seed);
|
||||
Data* delta_signature_serialize(const DeltaSignature* sig);
|
||||
DeltaSignature* delta_signature_deserialize(const Data* data);
|
||||
void delta_signature_destroy(DeltaSignature* sig);
|
||||
@@ -75,6 +82,15 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
Data* delta_serialize(const Delta* delta);
|
||||
Delta* delta_deserialize(const Data* data);
|
||||
void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta, uint32_t block_size);
|
||||
/* Streaming equivalent of delta_apply: matched blocks are read from `src_fd` as
|
||||
* they are emitted, so the basis never has to be resident. */
|
||||
void* delta_apply_fd(int src_fd, uint64_t old_size, const Delta* delta, uint32_t block_size);
|
||||
/* Fully streamed reconstruction: matched blocks come from `old_data` (when
|
||||
* non-NULL) or are read from `src_fd`, and the reconstructed bytes are written
|
||||
* straight to `dst_fd` in bounded chunks, so a reconstructed file larger than
|
||||
* memory is never materialized. */
|
||||
bool delta_apply_to_fd(const void* old_data, int src_fd, uint64_t old_size, const Delta* delta,
|
||||
uint32_t block_size, int dst_fd);
|
||||
void delta_destroy(Delta* delta);
|
||||
|
||||
bool delta_should_attempt(uint64_t old_size, uint64_t new_size, uint64_t max_file_size);
|
||||
|
||||
+1004
-150
File diff suppressed because it is too large.
Load diff
+108
-31
@@ -28,17 +28,36 @@ void file_metadata_destroy(void* metadata);
|
||||
/* --open-noatime process-wide sender policy; see file.c. */
|
||||
void file_set_open_noatime(bool enable);
|
||||
bool file_get_open_noatime(void);
|
||||
/* Capture the process umask ONCE, before any threads are created. Call this at
|
||||
* the very top of main() in both entry points so the cached value is read while
|
||||
* the process is still single-threaded: reading the umask needs a get+set round
|
||||
* trip (umask(0); umask(old)), which would race against receiver threads
|
||||
* creating files if it happened during the first write. Idempotent and safe to
|
||||
* call more than once. */
|
||||
void file_umask_capture(void);
|
||||
/* Process-wide umask, captured once (thread-safe). Used to derive the mode of
|
||||
* a brand-new destination like rsync: source_mode & 0777 & ~umask. Falls back
|
||||
* to file_umask_capture() (behind pthread_once) if capture was never called. */
|
||||
unsigned file_process_umask(void);
|
||||
/* Open `path` read-only for transfer, honouring --open-noatime when set. */
|
||||
int file_open_for_read(const char* path);
|
||||
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse);
|
||||
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave). --munge-links
|
||||
* sender-side marker: every transmitted symlink target is prefixed with this
|
||||
* while the flag is on; the receiver strips it to restore the real target. */
|
||||
#define SYMLINK_MUNGE_PREFIX "#SYMLINK/"
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave; rsync parity).
|
||||
* --munge-links is a RECEIVER-side rewrite: rsync prefixes every stored symlink
|
||||
* target with this marker, making the link unusable while the referenced
|
||||
* directory does not exist. A SENDER receiving a munged source strips it back
|
||||
* off before transmitting (so a munged tree round-trips through the receiver's
|
||||
* re-munging). */
|
||||
#define SYMLINK_MUNGE_PREFIX "/rsyncd-munged/"
|
||||
|
||||
char* file_symlink_munge(const char* target);
|
||||
/* rsync 3.4.1 unsafe_symlink(): true when `target` escapes the transfer tree
|
||||
* rooted at `link_path` (the symlink's transfer-relative path incl. its name).
|
||||
* Absolute/empty targets and targets climbing above the transfer root (via
|
||||
* "..") are unsafe, as are internal "/../" components and trailing "/..". */
|
||||
bool file_symlink_unsafe(const char* target, const char* link_path);
|
||||
/* True when a lexical target is relative and contains no ".." component, so it
|
||||
* can never escape the receive root once created beneath it. */
|
||||
bool file_symlink_target_contained(const char* target);
|
||||
@@ -46,8 +65,9 @@ bool file_symlink_target_contained(const char* target);
|
||||
* returns true when a marker was removed. */
|
||||
bool file_symlink_unmunge(char* target);
|
||||
/* Create a symlink at `path` -> `target`, confined below the authorized root
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). Returns
|
||||
* false when a directory already occupies `path`. */
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). The link
|
||||
* value is copied verbatim (rsync -l); only the placement path is confined.
|
||||
* Returns false when a directory already occupies `path`. */
|
||||
bool file_symlink_at_secure(const char* path, const char* target);
|
||||
/* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing
|
||||
* symlink-to-directory to be followed as a directory. */
|
||||
@@ -62,14 +82,16 @@ void file_set_keep_dirlinks(bool enable);
|
||||
void file_set_trust_sender(bool enable);
|
||||
bool file_get_trust_sender(void);
|
||||
|
||||
/* A configured fd without a canonical identity deliberately rejects paths. */
|
||||
bool file_set_authorized_root(int fd, const char* canonical_path);
|
||||
|
||||
/* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */
|
||||
bool file_path_exists_secure(const char* path);
|
||||
bool file_stat_secure(const char* path, struct stat* st);
|
||||
bool file_destination_is_newer_secure(const char* path, const FileMetadata* metadata);
|
||||
int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs);
|
||||
/* Protocol 2.28.0 variant: also increments *dirs_created for every missing
|
||||
* parent directory this walk creates that lies strictly below `count_floor`
|
||||
* (a receive-root-relative path, or NULL to count all of them). */
|
||||
int file_open_secure_parent_counted(const char* path, char** leaf_out, bool create_dirs,
|
||||
unsigned* dirs_created, const char* count_floor);
|
||||
bool file_ensure_directory_secure(const char* path);
|
||||
bool file_directory_exists_secure(const char* path);
|
||||
bool file_rename_secure(const char* old_path, const char* new_path);
|
||||
@@ -78,37 +100,44 @@ bool file_rename_secure(const char* old_path, const char* new_path);
|
||||
regular file. See the .c for the exact success semantics. */
|
||||
bool file_remove_tree_secure(const char* path);
|
||||
/* Open a private 0700 directory (creating it on demand) that must live below
|
||||
the authorized root. Used for the --temp-dir scratch directory and the
|
||||
--delay-updates staging directory. */
|
||||
the authorized root. Used for the --delay-updates staging directory. */
|
||||
int file_open_private_dir(const char* dir_path);
|
||||
|
||||
/* Open an existing --temp-dir scratch directory (relative or absolute; no
|
||||
creation). When an authorized receive root is configured the directory's
|
||||
REAL path (symlinks resolved) must lie within it, so a client-planted
|
||||
symlink cannot redirect receiver scratch files outside the sandbox; an
|
||||
in-root symlink to another filesystem is still allowed for rsync's EXDEV
|
||||
fallback. */
|
||||
int file_open_temp_dir(const char* dir_path);
|
||||
|
||||
/* The file_to_disk_secure* variants write a temporary copy in the destination
|
||||
directory and atomically rename it over `path`. temp_dir is an absolute,
|
||||
root-confined scratch directory (already validated by the caller): when it
|
||||
is non-NULL the temporary copy is instead created there (with a name unique
|
||||
across the whole scratch directory) and atomically renamed into the
|
||||
destination directory once fully written and fsynced. A rename across
|
||||
filesystems (EXDEV) fails the write with an error; the file is never
|
||||
silently copied into place. Pass NULL for the historical same-directory
|
||||
behavior. --inplace writes never use temp_dir. */
|
||||
directory and atomically rename it over `path`. temp_dir is a scratch
|
||||
directory (an absolute path, or one the caller already resolved against the
|
||||
destination root): when it is non-NULL the temporary copy is instead created
|
||||
there (with a name unique across the whole scratch directory) and atomically
|
||||
renamed into the destination directory once fully written and fsynced. When
|
||||
that rename/link fails with EXDEV (the scratch dir is on another filesystem)
|
||||
the write falls back to a non-atomic copy directly in the destination
|
||||
directory, matching rsync. Pass NULL for the same-directory behavior.
|
||||
--inplace writes never use temp_dir. */
|
||||
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, const char* temp_dir);
|
||||
FileAttrPolicy policy, const char* temp_dir);
|
||||
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
FileAttrPolicy policy, bool use_fsync, const char* temp_dir);
|
||||
/* With update enabled, an existing newer destination is left untouched. The
|
||||
check is descriptor-based for inplace writes; atomic replacement still has
|
||||
an unavoidable final rename race without filesystem locking. */
|
||||
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
unsigned long long data_size, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
/* Receiver write-path variant that also applies per-file xattrs (-X/-A) and the
|
||||
* --fake-super stat xattr fd-relative before the final rename. `update` /
|
||||
@@ -116,10 +145,9 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
* enables --partial best-effort retention of a failed write's temp. */
|
||||
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool update, bool no_replace, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super, bool keep_partial,
|
||||
const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
|
||||
bool fake_super, bool keep_partial, const char* temp_dir);
|
||||
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
|
||||
(via a temp name + rename); fall back to a byte-identical local copy from
|
||||
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
|
||||
@@ -128,16 +156,65 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
|
||||
never re-allocated). */
|
||||
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool use_fsync, const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
/* Like file_to_disk_secure_link, but the byte-copy fallback also applies the
|
||||
* per-file xattrs (-X/-A) and --fake-super stat xattr (fd-relative). On a
|
||||
* successful hard link no attributes are applied (the shared inode already
|
||||
* carries the basis's). */
|
||||
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
/* Streaming --copy-dest install: atomically materialize `path` by copying the
|
||||
* bytes of `basis_path` through a bounded buffer (no whole-file buffering, so
|
||||
* an arbitrarily large basis works), applying the SOURCE metadata and the
|
||||
* per-file xattrs / --fake-super record. `update` honors a newer destination;
|
||||
* a --temp-dir scratch location falls back to a direct write on EXDEV. */
|
||||
bool file_copy_basis_stream_attrs(const char* path, const char* basis_path,
|
||||
unsigned long long expected_size, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
/* Receive one length-prefixed whole-file data frame, streaming the payload
|
||||
* through a bounded buffer when it (or its known logical size) exceeds
|
||||
* `stream_limit`. On success exactly one of *out_buffer / *out_spool is set:
|
||||
* - *out_buffer: the historical charged whole-buffer Data (caller destroys);
|
||||
* - *out_spool: an owned temp path holding the payload, installed through the
|
||||
* File's basis_copy field with File.data_spool set so file_destroy unlinks
|
||||
* it. The destination policy/metadata/atomic-store handling is then the
|
||||
* existing bounded-buffer basis install (file_copy_basis_stream_attrs).
|
||||
* `expected_size` (0 = unknown) is the logical size from the check frame;
|
||||
* `dest_path` locates the spool next to the destination; `compress` selects
|
||||
* incremental decompression. Returns false on any framing/I/O/size error. */
|
||||
bool file_receive_payload(int fd, bool compress, unsigned long long expected_size,
|
||||
const char* dest_path, unsigned long long stream_limit, Data** out_buffer,
|
||||
char** out_spool, unsigned long long* out_size);
|
||||
/* Create a confined spool temp file next to `dest_path`; returns an open write
|
||||
* fd and an owned absolute path (to be installed via File.basis_copy with
|
||||
* File.data_spool set, and unlinked by file_destroy). */
|
||||
int file_spool_for_payload(const char* dest_path, char** out_spool_path);
|
||||
/* Protocol 2.28.0 receiver-stat variants: like the two above but additionally
|
||||
* report through `dirs_created` (when non-NULL) how many parent directories the
|
||||
* confined secure walk had to create that lie strictly below `count_floor` (a
|
||||
* receive-root-relative prefix, or NULL for all). Used to reproduce rsync's
|
||||
* `Number of created files` directory count on a fresh destination. */
|
||||
bool file_to_disk_secure_attrs_counted(
|
||||
const char* path, const void* data, unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
bool keep_partial, const char* temp_dir, unsigned* dirs_created, const char* count_floor,
|
||||
uint32_t fake_super_rdev_major, uint32_t fake_super_rdev_minor);
|
||||
bool file_to_disk_secure_link_attrs_counted(const char* path, const char* basis_path,
|
||||
const void* data, unsigned long long data_size,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
FileAttrPolicy policy, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir, unsigned* dirs_created,
|
||||
const char* count_floor);
|
||||
/* The logical transfer root expressed receive-root-relative, or NULL when the
|
||||
* wire paths carry no mirror scaffolding above it. Caller frees non-NULL. */
|
||||
char* file_transfer_root_floor(const Config* config);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,44 @@
|
||||
#ifndef FILE_ATTR_H
|
||||
#define FILE_ATTR_H
|
||||
|
||||
#include "config.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/*
|
||||
* Per-attribute receiver policy for applying a transmitted FileMetadata. This
|
||||
* is the split-out replacement for the former single use_metadata bundle: each
|
||||
* flag is applied independently, matching rsync's -p/-t/-o/-g/-E/-U semantics.
|
||||
* `use_metadata` remains the transport/presence gate (whether the metadata frame
|
||||
* travelled at all); this struct decides which attributes are ACTUALLY applied.
|
||||
*
|
||||
* It lives in its own header (rather than metadata.h) because xattr.h's
|
||||
* fake_super_restore_fd() takes one and metadata.h <-> file_types.h form an
|
||||
* include cycle that must not be entered from xattr.h.
|
||||
*
|
||||
* The mode leg is: perms wins over executability; an exec-bits-only change is
|
||||
* made only when perms is off; when neither is set the receiver deliberately
|
||||
* sets no source mode. file.c then substitutes the pre-existing destination
|
||||
* mode for a brand-new destination with metadata it uses the sanitized
|
||||
* source-mode-&-umask base (S_IWGRP|S_IWOTH cleared), and the fixed 0644
|
||||
* default only when no metadata is available at all, so a no--p overwrite
|
||||
* does not lose the destination's perms.
|
||||
*/
|
||||
typedef struct FileAttrPolicy {
|
||||
bool perms; /* config->preserve_perms: apply the source mode bits */
|
||||
bool times; /* config->preserve_times: apply the source mtime */
|
||||
bool atimes; /* config->preserve_atimes (-U): apply the source atime */
|
||||
bool executability; /* config->use_executability (-E): exec-bits-only mode */
|
||||
/* privilege_super_mode_permitted(): when false (SUPER_MODE_OFF / --no-super,
|
||||
or a daemon that did not grant `client owner = yes`), the setuid/setgid/
|
||||
sticky bits are stripped from every applied mode (source mode and any
|
||||
--chmod result) even under --perms. When true, rsync's exact semantics are
|
||||
preserved: -p copies the special bits and the kernel decides. */
|
||||
bool super_permitted;
|
||||
} FileAttrPolicy;
|
||||
|
||||
/* Build the per-attribute policy from a connection's Config. A NULL config
|
||||
* yields the all-off policy (no attribute application). */
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config);
|
||||
|
||||
#endif
|
||||
+48
-4
@@ -23,6 +23,8 @@ static void string_list_destroy(StringList* list) {
|
||||
|
||||
static bool string_list_add(StringList* list, const char* text) {
|
||||
if (list->count == list->capacity) {
|
||||
if (list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_cap = list->capacity > 0 ? list->capacity * 2 : 16;
|
||||
char** grown = realloc(list->items, (size_t)new_cap * sizeof(char*));
|
||||
if (!grown)
|
||||
@@ -56,8 +58,14 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings,
|
||||
snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", print_len, raw);
|
||||
return -1;
|
||||
}
|
||||
/* Reject NUL bytes inside a token defensively (NUL-delimited mode splits on
|
||||
* them, so this only guards against embedded garbage). */
|
||||
/* Reject NUL bytes inside a token defensively. In NUL-delimited mode the
|
||||
* delimiter itself is the final byte and is expected; in line mode any NUL is
|
||||
* embedded garbage (strlen-based parsing would otherwise silently truncate). */
|
||||
size_t scan_len = strip_line_endings ? len : len - 1;
|
||||
if (memchr(raw, '\0', scan_len)) {
|
||||
snprintf(err, err_size, "entry contains an embedded NUL byte");
|
||||
return -1;
|
||||
}
|
||||
char* dup = malloc(len + 1);
|
||||
if (!dup) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
@@ -158,10 +166,20 @@ FileListSet* file_list_load(const char* path, bool null_separated, char* err, si
|
||||
StringList raw = {0};
|
||||
char* line = NULL;
|
||||
size_t line_cap = 0;
|
||||
ssize_t n;
|
||||
bool ok = true;
|
||||
char delim = null_separated ? '\0' : '\n';
|
||||
while (ok && (n = getdelim(&line, &line_cap, delim, fp)) != -1) {
|
||||
while (ok) {
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, delim, UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
if (errno == EFBIG)
|
||||
snprintf(err, err_size, "entry in file list exceeds %d bytes", (int)UTILS_MAX_LINE_LEN);
|
||||
else
|
||||
snprintf(err, err_size, "error reading file list: %s", strerror(errno));
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (n == 0)
|
||||
break;
|
||||
int r = normalize_entry(line, (size_t)n, !null_separated, &raw, err, err_size);
|
||||
if (r < 0) {
|
||||
ok = false;
|
||||
@@ -222,3 +240,29 @@ bool file_list_affects(const FileListSet* set, const char* rel) {
|
||||
entry (binary search for the first entry at or after `rel` + '/'). */
|
||||
return path_index_has_descendant(&set->index, rel);
|
||||
}
|
||||
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel) {
|
||||
if (!set || set->whole_tree)
|
||||
return true;
|
||||
if (!rel || rel[0] == '\0')
|
||||
return false;
|
||||
/* `rel` itself is listed, or one of its ancestor prefixes is an exact listed
|
||||
directory (a listed prefix of a directory path is necessarily a
|
||||
directory). */
|
||||
size_t len = strlen(rel);
|
||||
while (len > 0) {
|
||||
const char* slash = NULL;
|
||||
for (size_t i = len; i-- > 0;) {
|
||||
if (rel[i] == '/') {
|
||||
slash = rel + i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!slash)
|
||||
break;
|
||||
len = (size_t)(slash - rel);
|
||||
if (path_index_contains_n(&set->index, rel, len))
|
||||
return true;
|
||||
}
|
||||
return path_index_contains(&set->index, rel);
|
||||
}
|
||||
@@ -40,4 +40,14 @@ void file_list_destroy(FileListSet* set);
|
||||
* this returns true, files are transferred only when it returns true. */
|
||||
bool file_list_affects(const FileListSet* set, const char* rel);
|
||||
|
||||
/* True when the DIRECTORY `rel` (path relative to the source root) is inside a
|
||||
* listed directory subtree: `rel` itself is a listed entry, or one of `rel`'s
|
||||
* ancestor directory prefixes is an exact listed entry. Unlike
|
||||
* file_list_affects this does NOT treat an ancestor of a listed entry as
|
||||
* affected, so an implied parent directory of a listed file is not synchronized
|
||||
* (rsync deletes nothing in it). With no set or a whole-tree set every
|
||||
* directory is in scope. This is the delete-walker's "synchronized directory"
|
||||
* predicate. */
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel);
|
||||
|
||||
#endif
|
||||
+201
-2506
File diff suppressed because it is too large.
Load diff
Loaded 100 of 232 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user