Compare commits
149
Commits
v2.19.0
...
1116da9f64
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1116da9f64 | ||
|
|
684153350a | ||
|
|
ec206b02d0 | ||
|
|
88bdfeeb58 | ||
|
|
3f5b0250f4 | ||
|
|
c41bfb2cdb | ||
|
|
1b2632f968 | ||
|
|
7dbca70a4b | ||
|
|
1a053f06e5 | ||
|
|
58b3a33e82 | ||
|
|
17b0632098 | ||
|
|
376e6500ab | ||
|
|
82a1d5e240 | ||
|
|
3eec5a4cc3 | ||
|
|
6144c7fc7f | ||
|
|
ea4ab661b4 | ||
|
|
84827ca617 | ||
|
|
23552e823d | ||
|
|
d1a567f7e3 | ||
|
|
93c1fc3c1f | ||
|
|
4815b1b281 | ||
|
|
ad7bc3348b | ||
|
|
34970b961c | ||
|
|
b3f7cad4db | ||
|
|
cd8a84c0a2 | ||
|
|
09c384d7d0 | ||
|
|
81ad313ee5 | ||
|
|
919a729206 | ||
|
|
8cd2b550d9 | ||
|
|
99c0fd8016 | ||
|
|
a2200f039a | ||
|
|
1437c6dc6b | ||
|
|
38356ecc1e | ||
|
|
7badac7f97 | ||
|
|
6ccf16b650 | ||
|
|
1174993d6b | ||
|
|
d5fcfa2c5c | ||
|
|
e79d2b47b0 | ||
|
|
7c24a365cf | ||
|
|
5893de4a34 | ||
|
|
825ba69753 | ||
|
|
34abaadb9a | ||
|
|
10c4ffebdf | ||
|
|
dfa2a42028 | ||
|
|
5b0ec5880d | ||
|
|
1a26bde2d4 | ||
|
|
58a28334b8 | ||
|
|
e48f19ee2b | ||
|
|
a2370433b2 | ||
|
|
9da5a0a9ed | ||
|
|
80c1ff321c | ||
|
|
551c187005 | ||
|
|
0d6c1f784f | ||
|
|
f75a69f96a | ||
|
|
df887c73b1 | ||
|
|
5d39619a8a | ||
|
|
6269ae54e5 | ||
|
|
07f7555c1d | ||
|
|
5b8aca5799 | ||
|
|
a1eaa93357 | ||
|
|
99df0a8a6d | ||
|
|
334fc5b3e8 | ||
|
|
88aee6ce94 | ||
|
|
f6e8b6ddc4 | ||
|
|
6f974eff19 | ||
|
|
72cccaa256 | ||
|
|
eb71b29d1c | ||
|
|
4d5befedfe | ||
|
|
f00844cf9a | ||
|
|
2ec17e821c | ||
|
|
6d47d93fd7 | ||
|
|
6ea966781f | ||
|
|
0a7f5faea6 | ||
|
|
25909110ac | ||
|
|
bd43448af2 | ||
|
|
264964411c | ||
|
|
18d1b84246 | ||
|
|
0f95f48899 | ||
|
|
c78a21de57 | ||
|
|
5c8970c64f | ||
|
|
4e918a1b69 | ||
|
|
87f6cb0243 | ||
|
|
e1f8f75e7c | ||
|
|
4c17122b00 | ||
|
|
0abaa62193 | ||
|
|
5334397b81 | ||
|
|
6968ff6734 | ||
|
|
5aca91ab22 | ||
|
|
3260a39ab4 | ||
|
|
5d3c43305e | ||
|
|
9242e86772 | ||
|
|
0155902d95 | ||
|
|
3499baf80b | ||
|
|
83eacf3151 | ||
|
|
c2df0347ef | ||
|
|
dd44537b44 | ||
|
|
2854a9d149 | ||
|
|
1fa2fbd266 | ||
|
|
10b18ab2d2 | ||
|
|
dcc78c14c5 | ||
|
|
5a829adb85 | ||
|
|
42c72030fb | ||
|
|
99f8045105 | ||
|
|
eea66a7848 | ||
|
|
c944787e03 | ||
|
|
57ce6d04f0 | ||
|
|
3cf2e2c91f | ||
|
|
301cb0dbaf | ||
|
|
a4f4110397 | ||
|
|
24fe8c5583 | ||
|
|
d274bdff4e | ||
|
|
6c636a19e6 | ||
|
|
1a83e284c4 | ||
|
|
317d5d081a | ||
|
|
69fe7f3c9f | ||
|
|
ddc71a7df5 | ||
|
|
ffa1d24625 | ||
|
|
b16349b81e | ||
|
|
4bc84fe954 | ||
|
|
fc560246c1 | ||
|
|
dff6609976 | ||
|
|
1acb66628d | ||
|
|
ba1c7a369f | ||
|
|
d28489d83c | ||
|
|
312ed05170 | ||
|
|
fecbe2c90c | ||
|
|
c8f5d80fcb | ||
|
|
8147ff7b50 | ||
|
|
b7fbb56289 | ||
|
|
08063b6d73 | ||
|
|
ea2f76cd7a | ||
|
|
76eeba1773 | ||
|
|
b72ab298ab | ||
|
|
446a714ef8 | ||
|
|
a90e234eb3 | ||
|
|
f8252cf3e7 | ||
|
|
4557924972 | ||
|
|
59ce174d22 | ||
|
|
2a8941ee5c | ||
|
|
37037a6ee7 | ||
|
|
061e9ad43f | ||
|
|
f928879755 | ||
|
|
84b7cb0de3 | ||
|
|
921472b8b3 | ||
|
|
082ac2645d | ||
|
|
eefbd1e849 | ||
|
|
1fd462cca8 | ||
|
|
4ac37c4d8a | ||
|
|
08af945bd6 |
No files matched your search
+12
-12
@@ -9,10 +9,10 @@ on:
|
||||
jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: clang-format check
|
||||
run: find src/ tests/ -name '*.c' -o -name '*.h' | xargs clang-format --dry-run --Werror
|
||||
@@ -26,11 +26,11 @@ jobs:
|
||||
# suite) run on merge to dev/main, so PR CI stays well under ~3 minutes.
|
||||
build-and-test:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
@@ -51,7 +51,7 @@ jobs:
|
||||
|
||||
sanitizers:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
strategy:
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
sanitizer: [address, undefined]
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build-${{ matrix.sanitizer }} -S . -DSANITIZER=${{ matrix.sanitizer }}
|
||||
@@ -72,12 +72,12 @@ jobs:
|
||||
|
||||
fuzz-build:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure (clang + fuzz)
|
||||
run: CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
|
||||
@@ -94,12 +94,12 @@ jobs:
|
||||
|
||||
coverage:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DENABLE_COVERAGE=ON
|
||||
@@ -118,12 +118,12 @@ jobs:
|
||||
|
||||
valgrind:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
@@ -8,3 +8,7 @@ build-*/
|
||||
build2/
|
||||
build3/
|
||||
build_docker2/
|
||||
|
||||
# Test/run artifacts
|
||||
root/
|
||||
test_partial_install_tmp/
|
||||
@@ -128,7 +128,7 @@ Do not wait for the user to tell you CI failed — check proactively. The user s
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -92,7 +92,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,16 +27,19 @@ FetchContent_Declare(xxhash GIT_REPOSITORY https://github.com/Cyan4973/xxHash GI
|
||||
FetchContent_MakeAvailable(xxhash)
|
||||
|
||||
# Sanitizer option
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread none)
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, undefined, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread undefined none)
|
||||
if(SANITIZER STREQUAL "address")
|
||||
add_compile_options(-fsanitize=address -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=address)
|
||||
elseif(SANITIZER STREQUAL "thread")
|
||||
add_compile_options(-fsanitize=thread -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=thread)
|
||||
elseif(SANITIZER STREQUAL "undefined")
|
||||
add_compile_options(-fsanitize=undefined -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=undefined)
|
||||
elseif(NOT SANITIZER STREQUAL "none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, undefined, none")
|
||||
endif()
|
||||
|
||||
option(STRICT_WARNINGS "Enable strict warnings" OFF)
|
||||
@@ -84,7 +87,7 @@ tests/integration/ — Python pytest integration tests
|
||||
### Dependencies
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)`
|
||||
- **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport)
|
||||
- **xxHash** — fetched via `FetchContent` from GitHub (delta transfer hashing, v0.8.3)
|
||||
- **xxHash** — fetched via `FetchContent` from the upstream repository (delta transfer hashing, v0.8.3)
|
||||
- **pthreads** — found via `find_package(Threads REQUIRED)`
|
||||
- **C11 standard** — required
|
||||
- **CMake 3.22+** — minimum version
|
||||
@@ -94,7 +97,7 @@ tests/integration/ — Python pytest integration tests
|
||||
- Use `file(GLOB ...)` for source collection (existing pattern).
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- Include directories: `src/shared`, `src/server`, `src/client`, `tests` (for test target).
|
||||
- Sanitizer support: pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (live option in CMakeLists.txt).
|
||||
- Sanitizer support: pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (live option in CMakeLists.txt).
|
||||
- Build with `cmake -B build -S . && cmake --build build -j$(nproc)`.
|
||||
- For CI, dependencies are provided by the project's custom Docker image (repo-root `Dockerfile`, same image CI uses). For local development, use `nix-shell`. Never add `apt-get install` / `pip install` to CI workflows. See `AGENTS.md`.
|
||||
|
||||
@@ -105,7 +108,7 @@ tests/integration/ — Python pytest integration tests
|
||||
3. Add new dependencies with `find_package` or `find_library`.
|
||||
4. When adding a new executable target, follow the pattern of existing targets.
|
||||
5. When adding a new library (static/shared), use `add_library` and follow the project's naming.
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (matching CI's matrix strategy).
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (matching CI's matrix strategy).
|
||||
7. Always verify the build compiles after changes.
|
||||
|
||||
## Sanitizer Configurations
|
||||
@@ -119,11 +122,9 @@ cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (race conditions)
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
For UndefinedBehaviorSanitizer (no `-DSANITIZER=undefined` option in CMakeLists.txt yet), use the manual flag approach:
|
||||
UndefinedBehaviorSanitizer uses the same built-in option:
|
||||
```bash
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=undefined -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=undefined"
|
||||
cmake -B build -S . -DSANITIZER=undefined
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
@@ -159,7 +160,7 @@ cmake -B build -S . -DCMAKE_BUILD_TYPE=RelWithDebInfo
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/server
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
./build/client
|
||||
./build/tests
|
||||
```
|
||||
@@ -187,7 +188,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ FastSync is a file synchronization tool (like rsync, but faster). It transfers f
|
||||
cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Server (TCP mode)
|
||||
./build/server
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
|
||||
# Client (TCP mode)
|
||||
./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk
|
||||
@@ -37,13 +37,13 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Run tests
|
||||
./build/tests # unit tests
|
||||
python3 test.py # integration tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # integration tests
|
||||
```
|
||||
|
||||
## Code Walkthrough
|
||||
|
||||
### Client Entry Point (`src/client/client_cli.c`)
|
||||
- Parses CLI arguments using `getopt_long`
|
||||
- Parses CLI arguments using a custom option-table parser (`OPTION_TABLE` in `src/client/client_cli.c`); there is no `getopt*` usage
|
||||
- Creates `Config` struct with all options
|
||||
- Detects SSH destinations (contains `:`)
|
||||
- Calls into `client_send.c` for the actual transfer
|
||||
@@ -109,7 +109,7 @@ Collection of files for batch transfer. Serialized with file count, then per-fil
|
||||
zstd streaming compression via `ZSTD_compressStream2`/`ZSTD_decompressStream`. Compression happens per-chunk in the sender stage. Level 1-22 (default 5). Streaming means memory usage stays bounded regardless of file size.
|
||||
|
||||
### "How does sendfile() work?"
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `-f` flag. Only works with TCP (not SSH, not compression).
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `--sendfile` (long form only). Only works with TCP (not SSH, not compression).
|
||||
|
||||
### "How does incremental sync work?"
|
||||
Client sends file metadata (path, size, mtime) to server. Server checks if destination file has same size+mtime. If match, server responds `STATUS_OK` (skip). If mismatch, server responds `STATUS_NEXT` (send).
|
||||
@@ -138,7 +138,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -316,7 +316,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -14,10 +14,9 @@ Diagnose crashes, memory errors, hangs, and logic bugs. You use structured debug
|
||||
### Memory Errors
|
||||
```bash
|
||||
# AddressSanitizer (fast, recommended first)
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/client # or ./build/server
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/client # or ./build-asan/server -p 8080 --allow-unauthenticated
|
||||
|
||||
# Valgrind (slower, more thorough)
|
||||
valgrind --leak-check=full --show-leak-kinds=all --track-origins=yes \
|
||||
@@ -32,10 +31,9 @@ valgrind --tool=drd ./build/client ...
|
||||
|
||||
### Thread Sanitizer
|
||||
```bash
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=thread" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
|
||||
### GDB
|
||||
@@ -143,7 +141,7 @@ gprof ./build/client gmon.out
|
||||
|
||||
### Step 5: Verify
|
||||
- Run `./build/tests` (unit tests)
|
||||
- Run `python3 test.py` (integration tests)
|
||||
- Run `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` (integration tests)
|
||||
- Run under valgrind again to confirm clean
|
||||
- Test under ASan again
|
||||
|
||||
@@ -162,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -96,7 +96,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -16,12 +16,18 @@ Scan the codebase for patterns that suggest new feature opportunities. You ident
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_cli.c Entry point, OPTION_TABLE parser, config setup
|
||||
usage.c Usage/help text (authoritative CLI flag list)
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
change_list.c File change-list bookkeeping
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
server_cli.c Server option-table CLI parsing
|
||||
receiver.c Receiver-side file handling
|
||||
receiver_pipeline.c Receiver worker pipeline
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
@@ -32,40 +38,63 @@ src/shared/ Shared libraries (used by both client and server)
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_send.c/h Sender-side file transfer
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_list.c/h File list model
|
||||
file_store.c/h Destination file store
|
||||
array_list.c/h Dynamic array
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
batch.c/h Batch files (--write-batch/--read-batch)
|
||||
charset.c/h Filename charset conversion (--iconv)
|
||||
chmod.c/h Permission modification (--chmod)
|
||||
xattr.c/h Extended attributes
|
||||
hardlink.c/h Hard-link handling
|
||||
identity.c/h uid/gid mapping (--usermap/--groupmap/--chown)
|
||||
credentials.c/h Daemon credentials
|
||||
daemon_conf.c/h Daemon module configuration
|
||||
motd.c/h Daemon MOTD
|
||||
delay_updates.c/h Delayed update staging
|
||||
stop_condition.c/h Stop-after/stop-at handling
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
file_types.h Shared file type definitions
|
||||
```
|
||||
|
||||
### Existing CLI Flags (from client_cli.c)
|
||||
### Existing CLI Flags (authoritative source: `src/client/usage.c`)
|
||||
```
|
||||
--source-dir <dir> Source directory to sync (required)
|
||||
--dest-dir <dir> Destination directory on server (required)
|
||||
--host <host> Server hostname/IP (required)
|
||||
--port <port> Server TCP port
|
||||
--server-mode Listen as server
|
||||
--use-compression, -c Enable zstd compression
|
||||
--use-multithreading, -m Enable multithreaded transfer
|
||||
--use-sendfile, -s Use sendfile() zero-copy TCP
|
||||
--use-ssh, -S Use SSH transport
|
||||
--use-tls, -T Enable TLS encryption
|
||||
--cert <file> TLS certificate file
|
||||
--key <file> TLS key file
|
||||
--ca <file> TLS CA certificate file
|
||||
--insecure Skip TLS verification
|
||||
--bwlimit <bytes/s> Bandwidth limit
|
||||
--delete Delete files not in source
|
||||
--include <pattern> Include filter pattern
|
||||
--exclude <pattern> Exclude filter pattern
|
||||
--dry-run Print what would be transferred
|
||||
--save-to-disk Save transferred files to disk (for server tests)
|
||||
--source-dir <dir> Source directory
|
||||
--dest-dir <dir> Destination directory on server
|
||||
--server-host <ip> Server IP address (default: 127.0.0.1)
|
||||
--server-port <n> Server port (default: 8080); --port is an alias
|
||||
-c, --checksum Verify content by checksum instead of size+mtime
|
||||
-z, --compress [level] Enable compression (level 1-22, default 5)
|
||||
-j, --threads[=N] Enable multithreaded scanner/loader/sender pipeline
|
||||
--chunk-serialization Enable chunk serialization (long form only)
|
||||
--sendfile sendfile() zero-copy (TCP only; long form only)
|
||||
-s, --secluded-args Protect-args compatibility option (no effect)
|
||||
--tls Enable TLS encryption; --cert/--key/--ca give PEMs
|
||||
--bwlimit <KB/s> Bandwidth limit in kilobytes per second
|
||||
--delete Delete files on receiver not in source
|
||||
--incremental Skip files unchanged since last transfer
|
||||
--delta Delta transfer for changed files (needs --incremental)
|
||||
-f, --filter=RULE rsync-style filter rule (+/- include/exclude)
|
||||
--exclude <pattern> Exclude files matching pattern
|
||||
--include <pattern> Only include files matching pattern
|
||||
-m, --prune-empty-dirs Do not transfer empty directory entries
|
||||
-n, --dry-run Show what would be transferred
|
||||
--save-to-disk Write received files to disk
|
||||
--version Print version and exit
|
||||
--help Print help
|
||||
--help Show help
|
||||
```
|
||||
> Always confirm the current flags with `./build/client --help`; the table above
|
||||
> is a representative subset. `src/client/usage.c` is the authoritative list and
|
||||
> `OPTION_TABLE` in `src/client/client_cli.c` is the parser (there is no `getopt*`).
|
||||
|
||||
## Feature Scout Checklist
|
||||
|
||||
@@ -288,7 +317,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ Design integration tests that verify the full transfer pipeline works end-to-end
|
||||
- Multiple configurations (TCP, SSH, TLS, compression, multithreading)
|
||||
- Network shaping (LAN, WAN profiles)
|
||||
- Feature tests (dry run, archive, exclude, delete, incremental, bandwidth limit)
|
||||
- Run: `python3 -m pytest tests/ -v --tb=short`
|
||||
- Run: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
|
||||
### 3. New: Focused Integration Tests
|
||||
When adding new features or fixing bugs, write targeted integration tests.
|
||||
@@ -35,13 +35,14 @@ mkdir -p /tmp/fastsync_test/src
|
||||
echo "test content" > /tmp/fastsync_test/src/file.txt
|
||||
|
||||
# Start server
|
||||
./build/server &
|
||||
./build/server -p 8080 --allow-unauthenticated &
|
||||
SERVER_PID=$!
|
||||
sleep 0.5
|
||||
|
||||
# Run client
|
||||
./build/client --source-dir /tmp/fastsync_test/src \
|
||||
--dest-dir /tmp/fastsync_test/dst \
|
||||
--server-port 8080 \
|
||||
--save-to-disk
|
||||
|
||||
# Verify
|
||||
@@ -76,7 +77,7 @@ openssl req -x509 -newkey rsa:2048 -keyout /tmp/key.pem -out /tmp/cert.pem \
|
||||
### Pattern 4: Incremental Sync
|
||||
```bash
|
||||
# First sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
|
||||
# Modify source
|
||||
echo "updated" >> /tmp/src/file.txt
|
||||
@@ -89,14 +90,14 @@ echo "updated" >> /tmp/src/file.txt
|
||||
### Pattern 5: Delete Verification
|
||||
```bash
|
||||
# Initial sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
|
||||
# Add extra file to dest
|
||||
echo "extra" > /tmp/dst/.../extra.txt
|
||||
|
||||
# Sync with --delete
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst \
|
||||
--save-to-disk --delete -M
|
||||
--save-to-disk --delete
|
||||
|
||||
# Verify extra.txt is gone
|
||||
test ! -f /tmp/dst/.../extra.txt
|
||||
@@ -115,7 +116,7 @@ The project uses Gitea Actions. Key jobs:
|
||||
jobs:
|
||||
new-job:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v7
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Configure
|
||||
@@ -127,7 +128,7 @@ jobs:
|
||||
- name: Unit Tests
|
||||
run: ./build-${{ matrix.sanitizer }}/tests
|
||||
- name: Integration Tests
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/ -v --tb=short
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
The symlink step is required because `tests/conftest.py` expects `./build` to exist.
|
||||
|
||||
@@ -135,7 +136,7 @@ The symlink step is required because `tests/conftest.py` expects `./build` to ex
|
||||
|
||||
After any code change:
|
||||
- [ ] Unit tests pass: `./build/tests`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/ -v --tb=short`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
- [ ] Build clean: no warnings with `-Wall`
|
||||
- [ ] No memory errors: ASan clean
|
||||
- [ ] No thread errors: TSan clean (if threading involved)
|
||||
@@ -156,7 +157,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
---
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates GitHub issues from their findings.
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates Gitea issues from their findings.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
@@ -12,7 +12,7 @@ You are the primary orchestrator agent. Your job is to:
|
||||
2. Decide which specialized sub-agents to dispatch for analysis
|
||||
3. Delegate analysis work using the task tool
|
||||
4. Receive structured findings from sub-agents
|
||||
5. Create GitHub issues from those findings using `gh issue create`
|
||||
5. Create Gitea issues from those findings using `tea issues create`
|
||||
6. Coordinate the overall analysis workflow end-to-end
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
@@ -97,7 +97,7 @@ First, read the repository structure to understand what exists:
|
||||
### Phase 2: Determine Analysis Scope
|
||||
Based on what the user requests or what needs attention:
|
||||
- **New features wanted?** → Dispatch `feature-scout` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-screener` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-auditor` sub-agent
|
||||
- **Code quality review?** → Dispatch `code-quality-guardian` sub-agent
|
||||
- **All of the above?** → Run all three in parallel
|
||||
|
||||
@@ -110,7 +110,7 @@ Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
```
|
||||
Task: Ask the security-screener agent to analyze the codebase.
|
||||
Task: Ask the security-auditor agent to analyze the codebase.
|
||||
Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
@@ -138,14 +138,14 @@ Each sub-agent returns findings in this structured format:
|
||||
- **Labels**: comma-separated labels for the issue
|
||||
```
|
||||
|
||||
### Phase 5: Create GitHub Issues
|
||||
For each finding, create a GitHub issue:
|
||||
### Phase 5: Create Gitea Issues
|
||||
For each finding, create a Gitea issue:
|
||||
|
||||
```bash
|
||||
gh issue create \
|
||||
tea issues create --repo TapTap/FastSync \
|
||||
--title "<Finding Title>" \
|
||||
--label "<labels>" \
|
||||
--body "## Description
|
||||
--labels "<labels>" \
|
||||
--description "## Description
|
||||
<description>
|
||||
|
||||
## Location
|
||||
@@ -175,11 +175,13 @@ _This issue was automatically generated by the issue-creator agent._"
|
||||
|
||||
### Duplicate Detection
|
||||
Before creating an issue:
|
||||
1. Check existing open issues: `gh issue list --state open --label "<label>"`
|
||||
2. Search for similar titles using `gh issue list --search "<keywords>"`
|
||||
1. Check existing open issues: `tea issues list --repo TapTap/FastSync --state open --labels "<label>"`
|
||||
2. Search for similar titles using `tea issues list --repo TapTap/FastSync --keyword "<keywords>"`
|
||||
3. If a similar issue exists, add a comment instead of creating a duplicate:
|
||||
```bash
|
||||
gh issue comment <issue-number> --body "Additional finding from automated analysis: <details>"
|
||||
tea comment --repo TapTap/FastSync <issue-number> "Additional finding from automated analysis: <details>"
|
||||
# or POST to the Gitea API:
|
||||
# POST https://gitea.tap-tap.win/api/v1/repos/TapTap/FastSync/issues/<n>/comments
|
||||
```
|
||||
|
||||
## Sub-Agent Reference
|
||||
@@ -189,13 +191,12 @@ Before creating an issue:
|
||||
| Agent | File | Purpose |
|
||||
|---|---|---|
|
||||
| feature-scout | `.opencode/agents/feature-scout.md` | Scans for feature opportunities |
|
||||
| security-screener | `.opencode/agents/security-screener.md` | Scans for security vulnerabilities |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits and vulnerability scans |
|
||||
| code-quality-guardian | `.opencode/agents/code-quality-guardian.md` | Scans for code quality improvements |
|
||||
| architect | `.opencode/agents/architect.md` | Architecture reviews |
|
||||
| c-reviewer | `.opencode/agents/c-reviewer.md` | C code correctness reviews |
|
||||
| debugger | `.opencode/agents/debugger.md` | Bug diagnosis |
|
||||
| refactorer | `.opencode/agents/refactorer.md` | Code refactoring |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits |
|
||||
| test-writer | `.opencode/agents/test-writer.md` | Test development |
|
||||
| perf-analyst | `.opencode/agents/perf-analyst.md` | Performance analysis |
|
||||
| protocol-designer | `.opencode/agents/protocol-designer.md` | Protocol design |
|
||||
@@ -242,7 +243,7 @@ tests/test_file.c — File tests
|
||||
tests/test_transport_tcp.c — TCP transport tests
|
||||
tests/test_transport_tls.c — TLS transport tests
|
||||
tests/test_array_list.c — Array list tests
|
||||
tests/pytest/ — Python integration tests
|
||||
tests/integration/ — Python pytest integration tests
|
||||
```
|
||||
|
||||
### Build & Config Files
|
||||
@@ -259,7 +260,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -56,9 +56,11 @@ DirectoryScanner → Queue(Scanner→Loader) → ChunkBuilder → Queue(Loader
|
||||
|
||||
### Benchmark Context
|
||||
|
||||
From README benchmarks (25MB mixed files, localhost):
|
||||
- Best config: `-m -c` (multithread + compression) → 0.20s, 11.2× faster than rsync
|
||||
- `sendfile()` bypasses userspace → ~2× faster on localhost
|
||||
Use the maintained benchmark tool — do not cite stale README numbers:
|
||||
- `python3 benchmark/bench.py` runs the repeatable throughput benchmark.
|
||||
- The real flags are `-j` (multithreading) and `-z` (compression); a fast loopback
|
||||
config combines `-j -z`.
|
||||
- `sendfile()` (via `--sendfile`) bypasses userspace → ~2× faster on localhost
|
||||
- Compression reduces wire data enough that transfer becomes latency-bound on WAN
|
||||
|
||||
## Output Format
|
||||
@@ -120,6 +122,7 @@ time ./build/client [args...]
|
||||
|
||||
# High precision
|
||||
perf stat -e task-clock ./build/client [args...]
|
||||
```
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
@@ -127,9 +130,8 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
```
|
||||
@@ -91,7 +91,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -160,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -3,25 +3,64 @@ description: Audits FastSync for security vulnerabilities — TLS config, input
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are a security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
You are the security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport. This is the single canonical security agent.
|
||||
|
||||
## Your Role
|
||||
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices.
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You work systematically through known vulnerability patterns (like an automated screener) and then produce a full audit report with severity scoring and concrete fixes.
|
||||
|
||||
## Attack Surface
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
|
||||
### Network Input Points
|
||||
1. **TCP server** (`src/server/server.c`) — accepts connections from any client
|
||||
2. **SSH transport** (`src/shared/transport_ssh.c`) — receives data via stdio pipe
|
||||
3. **Protocol parsing** (`src/shared/protocol.c`) — deserializes all incoming data
|
||||
4. **Config deserialization** (`src/shared/config.c`) — receives remote config
|
||||
5. **Chunk deserialization** (`src/shared/chunk.c`) — receives file batches
|
||||
## Project Architecture
|
||||
|
||||
### TLS Configuration
|
||||
- OpenSSL TLS 1.2+ via `src/shared/transport_tls.c`
|
||||
- Certificate/key loading, CA verification
|
||||
- SSL context setup, cipher suite selection
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
receiver.c Receiver-side file handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_store.c/h Destination file store
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
xattr.c/h Extended attributes
|
||||
identity.c/h uid/gid mapping
|
||||
credentials.c/h Daemon credentials
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` / `receiver.c` | Writes received files to disk |
|
||||
|
||||
## Security Audit Checklist
|
||||
|
||||
@@ -33,51 +72,182 @@ Audit the codebase for security vulnerabilities. You focus on the attack surface
|
||||
- [ ] Chunk count and file count validated before allocation
|
||||
- [ ] Config field lengths bounded
|
||||
|
||||
### 2. Buffer Safety
|
||||
- [ ] No `strcpy` — use `snprintf` or `strncpy` with null termination
|
||||
- [ ] `malloc` size calculations don't overflow (e.g., `count * sizeof(...)`)
|
||||
- [ ] No fixed-size stack buffers for unbounded input
|
||||
- [ ] `receive_n_data` always checks return value
|
||||
- [ ] Off-by-one in path concatenation
|
||||
### 2. Buffer Overflow Risks
|
||||
|
||||
### 3. Memory Safety in Error Paths
|
||||
- [ ] All error paths free allocated resources
|
||||
- [ ] No use-after-free on error paths
|
||||
- [ ] No double-free on error paths
|
||||
- [ ] Partial reads handled (don't use incomplete data)
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
|
||||
### 4. TLS/SSL Security
|
||||
- [ ] TLS 1.2 minimum enforced (no SSLv3, TLS 1.0, TLS 1.1)
|
||||
- [ ] Certificate verification enabled when CA provided
|
||||
- [ ] Certificate verification disabled only with explicit warning
|
||||
- [ ] Private key file permissions checked
|
||||
- [ ] No hardcoded certificates or keys
|
||||
- [ ] Cipher suites restricted to strong algorithms
|
||||
- [ ] SSL error codes checked after `SSL_read`/`SSL_write`
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
|
||||
### 5. Authentication & Authorization
|
||||
### 3. Path Traversal in File Operations
|
||||
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 4. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 5. TLS / SSL Security
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--ca`/warning
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
- [ ] **No hardcoded certificates or keys**
|
||||
- [ ] **SSL error codes checked after `SSL_read`/`SSL_write`**
|
||||
|
||||
### 6. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
- [ ] **All error paths free allocated resources** (no leaks / UAF / double-free)
|
||||
- [ ] **Partial reads handled** (don't use incomplete data)
|
||||
|
||||
### 7. Integer Overflow in Allocation
|
||||
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 8. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 9. Authentication & Authorization
|
||||
- [ ] SSH transport relies on SSH authentication (not custom auth)
|
||||
- [ ] No password/credential storage in plaintext
|
||||
- [ ] Server doesn't trust client-supplied paths blindly
|
||||
- [ ] Destination directory validated before writing
|
||||
|
||||
### 6. Denial of Service
|
||||
- [ ] Bounded memory allocation (can't OOM server with huge chunk)
|
||||
- [ ] Timeout on connections (no indefinite blocking)
|
||||
- [ ] Maximum connection limit or rate limiting
|
||||
- [ ] Malformed protocol messages handled gracefully (no crash)
|
||||
### 10. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
|
||||
### 7. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
### 11. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
|
||||
### 8. File System Security
|
||||
### 12. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 13. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 14. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
### 15. File System Security
|
||||
- [ ] Received file permissions validated (no SUID/SGID injection)
|
||||
- [ ] Symlink attack prevention (don't follow symlinks in destination)
|
||||
- [ ] Race conditions in file creation (TOCTOU)
|
||||
- [ ] Temporary file security (if any)
|
||||
|
||||
### 16. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
|
||||
## Common Vulnerability Patterns
|
||||
|
||||
### Format String Bugs
|
||||
@@ -118,9 +288,59 @@ receive_n_data(fd, buffer, expected_size);
|
||||
if (!receive_n_data(fd, buffer, expected_size)) { /* handle error */ }
|
||||
```
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
For each vulnerability found:
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Detailed Finding Fields
|
||||
|
||||
For each vulnerability found, also be prepared to report:
|
||||
1. **Location** — file:line
|
||||
2. **Severity** — critical / high / medium / low / informational
|
||||
3. **Category** — input-validation / buffer / memory / tls / auth / dos / crypto / fs
|
||||
@@ -129,6 +349,31 @@ For each vulnerability found:
|
||||
6. **Fix** — concrete code change
|
||||
7. **CVSS estimate** — rough severity score if exploitable
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### Audit Summary
|
||||
|
||||
Also provide a summary:
|
||||
```
|
||||
=== SECURITY AUDIT SUMMARY ===
|
||||
@@ -140,13 +385,30 @@ Low: <count>
|
||||
Informational: <count>
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -1,310 +0,0 @@
|
||||
---
|
||||
description: Scans the FastSync codebase for security vulnerabilities — buffer overflows, path traversal, TLS issues, memory safety, and cryptographic hygiene.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are a security screener for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
|
||||
## Your Role
|
||||
|
||||
Scan the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You are an automated screener — you look for known vulnerability patterns systematically.
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
|
||||
## Project Architecture
|
||||
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
array_list.c/h Dynamic array
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` | Writes received files to disk |
|
||||
|
||||
## Security Screener Checklist
|
||||
|
||||
### 1. Buffer Overflow Risks
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
|
||||
### 2. Path Traversal in File Operations
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 3. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 4. TLS / SSL Misconfiguration
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--insecure` flag
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
|
||||
### 5. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
|
||||
### 6. Integer Overflow in Allocation
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 7. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 8. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
|
||||
### 9. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
|
||||
### 10. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 11. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 12. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
@@ -138,11 +138,9 @@ int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
|
||||
|
||||
Build for fuzzing:
|
||||
```bash
|
||||
cmake -B build-fuzz -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=fuzzer,address,undefined -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=fuzzer,address,undefined"
|
||||
CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
|
||||
cmake --build build-fuzz -j$(nproc)
|
||||
./build-fuzz/tests/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
./build-fuzz/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
```
|
||||
|
||||
### AFL++ Harness
|
||||
@@ -216,7 +214,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -38,16 +38,20 @@ dd if=/dev/urandom of=/tmp/fastsync_bench/src/large.bin bs=1M count=10 2>/dev/nu
|
||||
Test each configuration 3 times, record median:
|
||||
|
||||
```bash
|
||||
# Real FastSync flags: -z=compression, -j=multithreading,
|
||||
# --chunk-serialization, --sendfile (long form only). The old rsync-style
|
||||
# spellings -c/-m/-s/-f are NOT the same options (-c=--checksum,
|
||||
# -m=--prune-empty-dirs, -s=--secluded-args, -f=--filter) and must not be used.
|
||||
CONFIGS=(
|
||||
"Standard|"
|
||||
"Compression|-c"
|
||||
"Multithreading|-m"
|
||||
"MT+Compression|-m -c"
|
||||
"Chunk Serialization|-s"
|
||||
"MT+Compression+Chunk|-m -c -s"
|
||||
"Sendfile|-f"
|
||||
"Compression|-z"
|
||||
"Multithreading|-j"
|
||||
"MT+Compression|-j -z"
|
||||
"Chunk Serialization|-j -z --chunk-serialization"
|
||||
"Sendfile|--sendfile"
|
||||
)
|
||||
|
||||
PORT=18080
|
||||
for config in "${CONFIGS[@]}"; do
|
||||
IFS='|' read -r name flags <<< "$config"
|
||||
echo "=== $name ==="
|
||||
@@ -55,13 +59,14 @@ for config in "${CONFIGS[@]}"; do
|
||||
rm -rf /tmp/fastsync_bench/dst
|
||||
mkdir -p /tmp/fastsync_bench/dst
|
||||
|
||||
./build/server &
|
||||
./build/server -p "$PORT" --allow-unauthenticated &
|
||||
SERVER_PID=$!
|
||||
sleep 0.5
|
||||
|
||||
START=$(date +%s%N)
|
||||
./build/client --source-dir /tmp/fastsync_bench/src \
|
||||
--dest-dir /tmp/fastsync_bench/dst \
|
||||
--server-port "$PORT" \
|
||||
--save-to-disk $flags
|
||||
END=$(date +%s%N)
|
||||
|
||||
@@ -74,14 +79,21 @@ for config in "${CONFIGS[@]}"; do
|
||||
done
|
||||
```
|
||||
|
||||
### Step 4: Full Integration Benchmark (Optional)
|
||||
### Step 4: Full Benchmark Tool (Preferred)
|
||||
|
||||
The maintained benchmark tool is `benchmark/bench.py`. It handles building,
|
||||
data generation, network shaping (LAN/WAN profiles or custom `--delay`/`--jitter`/
|
||||
`--throughput`/`--loss`), rsync comparison, and JSON/table reporting:
|
||||
|
||||
For comprehensive benchmarking with network shaping:
|
||||
```bash
|
||||
python3 test.py --full
|
||||
python3 benchmark/bench.py --help
|
||||
python3 benchmark/bench.py --runs 5 --profiles unlimited
|
||||
python3 benchmark/bench.py --size-mb 100 --random-ratio 0.5 --output json
|
||||
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
|
||||
```
|
||||
|
||||
This tests LAN/WAN profiles, SSH, TLS, and compares against rsync.
|
||||
Network shaping needs root (`tc`/`netem` on `lo`). SSH and TLS coverage lives in
|
||||
the pytest integration suite, not the benchmark tool.
|
||||
|
||||
### Step 5: Report Results
|
||||
|
||||
@@ -93,12 +105,13 @@ Platform: <OS, CPU, network>
|
||||
Configuration | Run 1 | Run 2 | Run 3 | Median
|
||||
-----------------------|---------|---------|---------|--------
|
||||
Standard | 0.12s | 0.11s | 0.12s | 0.12s
|
||||
Compression (-c) | 0.09s | 0.08s | 0.09s | 0.09s
|
||||
Multithreading (-m) | 0.07s | 0.07s | 0.08s | 0.07s
|
||||
MT+Compression (-m -c) | 0.05s | 0.05s | 0.06s | 0.05s
|
||||
Sendfile (-f) | 0.04s | 0.04s | 0.04s | 0.04s
|
||||
Compression (-z) | 0.09s | 0.08s | 0.09s | 0.09s
|
||||
Multithreading (-j) | 0.07s | 0.07s | 0.08s | 0.07s
|
||||
MT+Compression (-j -z) | 0.05s | 0.05s | 0.06s | 0.05s
|
||||
Chunk Serialization (--chunk-serialization) | 0.05s | 0.04s | 0.05s | 0.05s
|
||||
Sendfile (--sendfile) | 0.04s | 0.04s | 0.04s | 0.04s
|
||||
|
||||
Best configuration: MT+Compression (-m -c)
|
||||
Best configuration: Sendfile (--sendfile)
|
||||
Throughput: <X> MB/s
|
||||
```
|
||||
|
||||
|
||||
@@ -32,23 +32,19 @@ Try to reproduce the issue with the exact command the user provides.
|
||||
|
||||
**Memory errors (first priority):**
|
||||
```bash
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
# or run the failing command
|
||||
```
|
||||
|
||||
**Thread errors:**
|
||||
```bash
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
|
||||
**Valgrind (if ASan doesn't find it):**
|
||||
@@ -108,13 +104,12 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
|
||||
# If integration test needed
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
|
||||
# Re-run under sanitizer to confirm fix
|
||||
rm -rf build
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
# reproduce the original failing command
|
||||
```
|
||||
|
||||
|
||||
@@ -19,7 +19,7 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log main..HEAD --oneline
|
||||
git log dev..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Clean build
|
||||
@@ -39,17 +39,13 @@ If the PR touches threading, memory management, or network code, also build with
|
||||
```bash
|
||||
# AddressSanitizer
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
|
||||
# ThreadSanitizer (if threading changes)
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
@@ -91,10 +87,10 @@ If tests fail:
|
||||
### Step 6: Run integration tests (optional)
|
||||
|
||||
```bash
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
This runs the integration + benchmark suite. It takes longer — only run if the user asks or if unit tests pass.
|
||||
This runs the integration suite (benchmarking is `benchmark/bench.py`). It takes longer — only run if the user asks or if unit tests pass.
|
||||
|
||||
### Step 7: Fix and commit
|
||||
|
||||
|
||||
@@ -19,13 +19,13 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log main..HEAD --oneline
|
||||
git log dev..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Get changed files
|
||||
|
||||
```bash
|
||||
git diff main --name-only -- '*.c' '*.h'
|
||||
git diff dev --name-only -- '*.c' '*.h'
|
||||
```
|
||||
|
||||
This gives the list of C source and header files changed in the PR.
|
||||
@@ -125,7 +125,7 @@ STYLE: <count>
|
||||
|
||||
If the user wants to post the review as a PR comment:
|
||||
```bash
|
||||
tea pr comment <number> --comment "<review report>"
|
||||
tea comment --repo TapTap/FastSync <number> "<review report>"
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -16,7 +16,7 @@ Ask the user or determine from context:
|
||||
- **Minor** (x.Y.0) — new features, backward compatible
|
||||
- **Patch** (x.y.Z) — bug fixes, no protocol changes
|
||||
|
||||
Current version: `PROTOCOL_VERSION "1.1.0"` in `src/shared/config.h`
|
||||
Current version: `PROTOCOL_VERSION "2.23.0"` in `src/shared/config.h`
|
||||
|
||||
### Step 2: Check Protocol Version
|
||||
|
||||
@@ -37,7 +37,7 @@ rm -rf build
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
ALL tests must pass before release.
|
||||
@@ -46,12 +46,10 @@ ALL tests must pass before release.
|
||||
|
||||
```bash
|
||||
# ASan
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
```
|
||||
|
||||
### Step 5: Update README (If Needed)
|
||||
@@ -79,12 +77,23 @@ git commit -m "Release vX.Y.Z
|
||||
git tag -a vX.Y.Z -m "Release vX.Y.Z"
|
||||
```
|
||||
|
||||
### Step 8: Push
|
||||
### Step 8: Push and Open dev → main PR
|
||||
|
||||
`main` is protected and only receives changes via `dev` → `main` PRs (see AGENTS.md). Never push directly to `main`.
|
||||
|
||||
```bash
|
||||
git push origin main --tags
|
||||
# Push the release commit and tag to dev
|
||||
git push origin dev
|
||||
git push origin vX.Y.Z
|
||||
|
||||
# Open the dev → main release PR for review + CI
|
||||
tea pr create --repo TapTap/FastSync --head dev --base main \
|
||||
--title "Release vX.Y.Z" \
|
||||
--description "Release vX.Y.Z"
|
||||
```
|
||||
|
||||
Then wait for the full CI to pass and request review before the PR is merged to `main`.
|
||||
|
||||
### Step 9: Report
|
||||
|
||||
```
|
||||
|
||||
@@ -102,9 +102,9 @@ Informational: <count>
|
||||
...
|
||||
|
||||
=== VERDICT ===
|
||||
[PASS] No critical/high issues found
|
||||
[PASS] No critical/high-severity issues found
|
||||
— or —
|
||||
[FAIL] <N> critical/high issues must be fixed
|
||||
[FAIL] <N> critical/high-severity issues must be fixed
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -4,18 +4,19 @@ FastSync is a high-performance file synchronization system written in C11. It su
|
||||
|
||||
## Dependency installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v10`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, and Node.js.
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v11`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, Node.js, plus `rsync` 3.4.1 (with zstd/xxhash/lz4), `acl` and `attr` (setfacl/getfacl, setfattr/getfattr) for drop-in parity tests.
|
||||
|
||||
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
|
||||
|
||||
```bash
|
||||
# Use the prebuilt CI image directly (faster, guaranteed CI parity)
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v10 fastsync-ci:local
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v11 fastsync-ci:local
|
||||
|
||||
# Or build the image from the repo-root Dockerfile
|
||||
# (Note: the prebuilt :v10 image reflects the previous Dockerfile state;
|
||||
# rebuild from source to pick up any newly added packages like lcov/valgrind.)
|
||||
# (Note: the prebuilt :v11 image is built from the current Dockerfile and
|
||||
# includes rsync 3.4.1 plus acl/attr; rebuild from source after changing
|
||||
# the Dockerfile.)
|
||||
docker build -t fastsync-ci:local .
|
||||
|
||||
# Build, run unit tests, and run integration tests inside the container
|
||||
@@ -28,7 +29,7 @@ docker run --rm --user "$(id -u):$(id -g)" -v "$PWD:/workspace" \
|
||||
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load'
|
||||
```
|
||||
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash from GitHub via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
|
||||
If a dependency is missing from the CI image, add it to the `Dockerfile` (and rebuild) rather than adding an install step to the CI workflow.
|
||||
|
||||
@@ -59,28 +60,28 @@ python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset o
|
||||
|
||||
## CI Workflow — Waiting for Results
|
||||
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Use `gh run watch` or similar to monitor CI status, then inspect logs on failure.
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Monitor CI status via the Gitea API (see below) or `tea actions`, then inspect logs on failure.
|
||||
|
||||
## CI Troubleshooting
|
||||
|
||||
### If lint (clang-format) fails
|
||||
Run clang-format in the CI Docker image to match the exact CI version:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i'
|
||||
```
|
||||
|
||||
### If cppcheck fails
|
||||
Fix reported issues locally, then verify with:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/'
|
||||
```
|
||||
|
||||
### If integration tests fail
|
||||
Run locally before pushing:
|
||||
```bash
|
||||
python3 -m pytest tests/ -v --tb=short
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
## Branch Strategy
|
||||
@@ -165,7 +166,7 @@ This can be cron'd locally if desired (e.g., `crontab -e` with `opencode run`).
|
||||
## Is opencode a good option?
|
||||
|
||||
**Yes, for FastSync's needs.** The hybrid model works well:
|
||||
- opencode's 17 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- opencode's 16 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- The assistant orchestrates subagents, merges branches, and iterates on CI
|
||||
- You only review the final output
|
||||
|
||||
|
||||
+248
@@ -4,6 +4,254 @@ All notable changes to FastSync are documented here. Versions match
|
||||
`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must
|
||||
run the same version because the handshake is strict.
|
||||
|
||||
## [2.23.0] - 2026-09-16
|
||||
|
||||
### Added
|
||||
|
||||
- **Rsync-parity wave.** Closed the remaining CLI, filesystem, ownership,
|
||||
deletion, and output gaps against rsync 3.4.1.
|
||||
- Short options `-r` (`--recursive`), `-b` (`--backup`), `-L`
|
||||
(`--copy-links`), and `-B` (`--block-size`/`--delta-block`); rsync
|
||||
short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached/inline values
|
||||
(`--opt=value`, `-B1000`, `-essh`, `-MOPT`). A value that starts with `-`
|
||||
is not mistaken for a cluster.
|
||||
- `-c`/`--checksum` now implies the incremental checksum quick-check (and,
|
||||
like rsync, does not imply `-t`).
|
||||
- `--checksum-choice`/`--cc` accepts `xxh64`/`xxhash`/`xxh3`/`xxh128`/`md5`/
|
||||
`auto` and rejects `md4`/`sha1`/`none` and the two-name form by name;
|
||||
`--checksum-seed=0` (the default) is randomized per transfer and the chosen
|
||||
seed is sent to the receiver.
|
||||
- `--compress-choice`/`--zc` accepts `zstd`/`none`/`auto` and rejects
|
||||
`lz4`/`zlib`/`zlibx` by name; `--skip-compress` defaults to rsync 3.4.1's
|
||||
built-in suffix list; `--no-whole-file` is accepted.
|
||||
- `--timeout` defaults to 0 (disabled) and `--contimeout` to 60 s (both `0`
|
||||
disables), matching rsync; `--max-alloc=0` means no local limit.
|
||||
- `--temp-dir` is confined to the receive root (absolute/`..` rejected by the
|
||||
receiver) and an `EXDEV` install falls back to a non-atomic copy.
|
||||
- `--numeric-ids` is documented as a mapping modifier only;
|
||||
`--usermap`/`--groupmap` support inclusive `LOW-HIGH` ranges, `*`,
|
||||
empty-`FROM` (unnamed ids), and receiver-resolved `TO` names; `--chown`
|
||||
conflicts with a map on the same side are rejected.
|
||||
- `--fake-super` records the *resolved* owner (never a real chown) and replays
|
||||
mode/time; directory ownership and directory xattrs/ACLs are preserved.
|
||||
- `-l`/`--links` stores symlink targets verbatim (absolute and `..`-bearing
|
||||
included), matching rsync; `--safe-links`/`--copy-unsafe-links` are applied
|
||||
sender-side and `--munge-links` uses rsync's `/rsyncd-munged/` marker;
|
||||
`--trust-sender` no longer affects symlink targets.
|
||||
- `--specials` recreates unix sockets with `mknod(S_IFSOCK)` (so `-D` covers
|
||||
the full rsync node set).
|
||||
- Deletion: the manifest carries a synchronized-directory section so
|
||||
`--files-from` subsets no longer delete untransmitted paths;
|
||||
`--delete-excluded` leaves size-pruned mirrors protected; extraneous
|
||||
destination symlinks are unlinked (never followed); `--max-delete=N` is
|
||||
partial (delete up to N, skip the rest, exit 25) and `--delete-missing-args`
|
||||
removals draw from the same budget; `--force` is honored during
|
||||
`--delay-updates` publication.
|
||||
- `-x`/`--one-file-system` emits the mount-point directory entry; the
|
||||
`--include`/`--exclude` layers are an ordered first-match rule list.
|
||||
- `--chmod` is a faithful port of rsync 3.4.1 (numeric/symbolic, `D`/`F`/`X`,
|
||||
`s`/`t`, append semantics, no `-p` implication, no sanitization).
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.22.0 → 2.23.0`: the delete manifest gains a
|
||||
synchronized-directory section and the terminal status gains
|
||||
`STATUS_DELETE_LIMIT` (client exit 25 on a `--max-delete`-capped commit).
|
||||
- **The 2.22.0 mode-masking divergence is removed.** Under `-p` the source mode
|
||||
is copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky;
|
||||
`--chmod` no longer implies `-p`. New files without `-p` still use
|
||||
`source_mode & ~umask` when metadata is present (else `0644`), and new
|
||||
directories without `-p` still use the `0755` creation default.
|
||||
- `--protocol=NUM` accepts only the current `2.23.0` version string.
|
||||
|
||||
### Notes
|
||||
|
||||
- The rsync-compatibility matrix (`RSYNC_COMPAT.md`) now classifies every row
|
||||
as **parity**, **caveat** (works with a documented divergence), or
|
||||
**divergent** (not supported/no-op/impossible), replacing the previous
|
||||
misleading "N implemented / 0 divergence" summary. Durable documented
|
||||
divergences remain: receiver-side symlink target containment is not enforced
|
||||
by default (verbatim storage is rsync parity; use `--safe-links`),
|
||||
`--temp-dir` rejects absolute/foreign-filesystem paths, `--copy-devices`
|
||||
reads a bounded `st_size`, a broken referent under `--copy-links` exits 0,
|
||||
new directories without `-p` use `0755`, `--stats` receiver-only counters are
|
||||
0, and `--password-file`/`--early-input`/`--hash-credentials`/`--iterations`
|
||||
and the batch format are FastSync-native.
|
||||
|
||||
## [2.22.0] - 2026-09-15
|
||||
|
||||
### Added
|
||||
|
||||
- **Per-attribute metadata preservation (protocol 2.22.0).** The former single
|
||||
metadata bundle is split into four independent, rsync-compatible flags:
|
||||
`-p/--perms`, `-t/--times`, `-o/--owner`, and `-g/--group`, each applied
|
||||
independently on the receiver, with negations `--no-perms`/`--no-times`/
|
||||
`--no-owner`/`--no-group` (short `--no-p`/`--no-t`/`--no-o`/`--no-g`) and
|
||||
`--no-preserve` clearing all four. `-a/--archive` is now full rsync
|
||||
`-rlptgoD` (owner and group included; their application stays
|
||||
privilege-gated). `-A/--acls` and `--chmod` imply `-p`, `-X/--xattrs` does
|
||||
not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply
|
||||
`-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the
|
||||
user explicitly negated them.
|
||||
- Receiver applies directory modes under `-p` (at the end of the transfer,
|
||||
alongside the deferred directory times) and symlink mode under `-p`; `-O`
|
||||
suppresses directory times only.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.21.0 → 2.22.0`: the binary config frame gains
|
||||
four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/
|
||||
`preserve_group`) after `omit_link_times`. The fixed-width `FileMetadata`
|
||||
layout is unchanged; the receiver derives the metadata-frame gate
|
||||
(`use_metadata`) from the four attributes.
|
||||
|
||||
### Notes
|
||||
|
||||
- Documented divergences from rsync: a client-supplied mode never grants
|
||||
group/other write (`S_IWGRP|S_IWOTH` are stripped for files, directories,
|
||||
symlinks, and specials; rsync's `-p` preserves them exactly); a brand-new file
|
||||
without `-p` gets `source_mode & ~umask` (sanitized) when metadata is present,
|
||||
else the historical fixed `0644`; `--chmod` implies `-p` (rsync does not);
|
||||
`-o`/`-g` map by name on the receiver with a raw-numeric fallback (only
|
||||
numeric ids cross the wire); and a daemon module without `client owner = yes`
|
||||
does not refuse a plain `-a`/`-o`/`-g` but forces super-user activities off,
|
||||
applies no ownership, and logs a warning (explicit `--chown`/`--usermap`/
|
||||
`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused).
|
||||
|
||||
## [2.21.0] - 2026-09-14
|
||||
|
||||
### Added
|
||||
|
||||
- Optional server→client rejection detail (protocol 2.21.0). A rejected
|
||||
operation may now carry a bounded human-readable reason via
|
||||
`STATUS_ERROR_DETAIL` instead of a bare `STATUS_ERROR`, so the client can
|
||||
report *why* the server refused (daemon module gate, config validation,
|
||||
receiver-side path/node validation). `receive_status()` transparently maps the
|
||||
new status back to `STATUS_ERROR` for every existing call site and captures
|
||||
the reason into a thread-local buffer exposed by `protocol_last_error()`. The
|
||||
detail body is always consumed, so the stream cannot desynchronize, and
|
||||
messages are sliced to `MAX_ERROR_DETAIL_BYTES` (4096) on send.
|
||||
- **Server-contacting `--dry-run` (protocol 2.21.0).** `--dry-run` now performs
|
||||
a real handshake with a remote/daemon receiver and reports exactly what WOULD
|
||||
change based on receiver state (existing destination files, mtimes, checksums,
|
||||
basis dirs). The wire config carries the dry-run intent (`Config.dry_run`) and
|
||||
the receiver answers each per-file check with `STATUS_DRY_RUN_TRANSFER` (would
|
||||
transfer) or `STATUS_OK` (already up to date); the sender prints the
|
||||
would-transfer set and its trailer without sending any file data. The receiver
|
||||
performs the normal read-only incremental decision but mutates nothing: no temp
|
||||
files, writes, renames, deletes, metadata/xattr/chown, or directory creation.
|
||||
A plain local destination (no explicit `--server-port`/remote) keeps the
|
||||
original client-side dry-run. Would-delete reporting for `--delete*` is
|
||||
deferred to a follow-up; dry-run never deletes.
|
||||
- Daemon `max connections per host` (per-source-IP concurrent cap, default 0 =
|
||||
unlimited), `auth lockout threshold` (default 10; 0 disables) and
|
||||
`auth lockout duration` (default 300 s) config keys.
|
||||
- `fastsync-server --allow-super` opt-in for a privileged standalone TCP server;
|
||||
without it a root standalone receiver forces super-user activities off (device
|
||||
nodes, `--write-devices`, ownership). The `--stdio` SSH argv is client-composed,
|
||||
so super activities always stay off there.
|
||||
|
||||
### Changed
|
||||
|
||||
- Config wire fields are now declared once in an X-macro table
|
||||
(`CONFIG_WIRE_FIELDS` in `src/shared/config.h`) that generates the struct
|
||||
members, defaults, and the send/receive sequence, removing the manual
|
||||
six-site field sync. Wire bytes and `PROTOCOL_VERSION` are unchanged.
|
||||
- `receive_incremental_check()` (the per-file `STATUS_CHECK` fast path) is split
|
||||
into small static helpers with a short linear orchestrator. Pure refactor: the
|
||||
wire byte stream and all cleanup are unchanged.
|
||||
- `authorized_root` state has a single owner (`utils.c`) with read accessors; the
|
||||
duplicated statics in `file.c` and the server were removed.
|
||||
- `Data` records its owning `ProtocolSession` so its memory charge is returned to
|
||||
the session that reserved it, regardless of the destroying thread.
|
||||
- The receiver pipeline moved out of `shared` into `server/receiver_pipeline.[ch]`;
|
||||
the build now uses explicit `fastsync_shared` / `fastsync_client_core` /
|
||||
`fastsync_server_core` targets instead of a GLOB, and the client no longer links
|
||||
server code.
|
||||
- The benchmark tool generates the requested random/compressible data mix
|
||||
accurately, verifies each transfer before recording it, computes correct
|
||||
percentiles, adds a MB/s column, handles `tc`/netem without requiring `sudo`
|
||||
when already root, builds into a dedicated `build-bench/` directory, and adds a
|
||||
`--warm` incremental-transfer mode.
|
||||
- The `nix-shell` dev environment provides the full toolchain (clang-format,
|
||||
cppcheck, pytest-xdist, OpenSSH, rsync, iproute2, valgrind, lcov) and no longer
|
||||
builds on entry.
|
||||
|
||||
### Security
|
||||
|
||||
- Enforce the daemon's per-module `max connections` cap (0 = unlimited) and add
|
||||
the shared per-source `max connections per host` cap plus a cross-process
|
||||
`auth lockout`. Because the listener forks one child per connection, the
|
||||
counters live in an anonymous shared mapping created before the accept loop and
|
||||
reclaimed by the parent's `SIGCHLD` handler, so the per-module, per-source and
|
||||
auth-failure state is shared across every child (including after `SIGKILL`). The
|
||||
per-source table has a bounded lifetime (expired/idle entries are reclaimed,
|
||||
with a rate-limited warning when genuinely full), and the occupancy counters are
|
||||
re-derived from the shared slot table on every child exit. Trusted loopback
|
||||
peers are exempt (they share one address); clients behind a shared NAT/proxy
|
||||
share a single per-host budget and lockout, which is documented.
|
||||
- Hardening from a full security audit:
|
||||
- Fail a truncated zstd frame instead of spinning forever (remote DoS).
|
||||
- Open receiver destination/basis/hard-link entries `O_NONBLOCK` so a
|
||||
client-planted FIFO cannot block a worker indefinitely.
|
||||
- Require a regular file before `--inplace` writes, closing a FIFO-hang and a
|
||||
raw-device write that bypassed the `--write-devices` gate.
|
||||
- Reject SSH destinations whose user/host begins with `-` and insert `--` before
|
||||
the host token, closing `-o ProxyCommand=…` argument injection (RCE).
|
||||
- Gate client `--force` recursive removal behind the server `--allow-delete`
|
||||
policy.
|
||||
- Reject empty `hosts allow`/`hosts deny`/`auth users` values instead of
|
||||
silently meaning "unrestricted".
|
||||
- Restrict TLS 1.2 to AEAD suites and set server cipher preference; load the
|
||||
private key TOCTOU-safely from an `O_NOFOLLOW` fd; verify IP literals against
|
||||
IP SANs; guard client-cert CN truncation.
|
||||
- Make `--dry-run` content-blind: it neither reads destination files nor
|
||||
hashes basis files, removing a 1-bit content oracle against `read only`
|
||||
modules.
|
||||
- Bound glob matching (iterative DP, no exponential backtracking) and bound
|
||||
line reads for filter/`--files-from`/pattern files.
|
||||
- Gate `system.posix_acl_*` xattrs on `--acls` and charge decompression/chunk
|
||||
allocations against the per-connection memory budget.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Pre-auth NULL dereference in `config_delete()` when an over-long
|
||||
`basis_count` (and the analogous count fields) was received and then failed
|
||||
validation; received counts are now validated before being published.
|
||||
- Leaked inherited `Data` in the forked compression-truncation unit test
|
||||
(valgrind definite leak).
|
||||
- `receive_status()` no longer loses a captured rejection reason when owed
|
||||
keepalives are drained.
|
||||
|
||||
## [2.20.0] - 2026-09-13
|
||||
|
||||
### Security
|
||||
|
||||
- Cap cumulative `DirTimeList` growth and bound pre-auth config-string memory
|
||||
(remote memory-exhaustion DoS).
|
||||
- Daemon host access control (`hosts allow`/`hosts deny`, IPv4/IPv6/CIDR),
|
||||
configurable global `max connections`, connection audit logging, and a
|
||||
bounded `auth failure delay` throttle. IPv4-mapped peers are normalized and
|
||||
invalid patterns are rejected at parse time (no silent fail-open).
|
||||
- Honor `--timeout` for protocol I/O and bound idle/session time to defeat
|
||||
keepalive slowloris; child-safe signal handling in the forked daemon.
|
||||
- Compiler/linker hardening (`_FORTIFY_SOURCE`, stack protector, PIE, RELRO)
|
||||
and pinned build dependencies.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Use-after-free in the basis-dir oversize preflight.
|
||||
- Placeholder `Data` leaks, `missing_args` leak, scanner chunk leak.
|
||||
- Thread-safe logging; single fd owner and cleanup epilogue in the server
|
||||
handler.
|
||||
|
||||
### Performance
|
||||
|
||||
- Metadata now crosses the wire as one packed frame (protocol 2.20.0).
|
||||
- Delete keep-set and `--files-from` lookups indexed (O(n*m) → O(n)).
|
||||
- Reused per-thread zstd contexts; `TCP_NODELAY` by default.
|
||||
- Byte-bounded sender queues; removed a redundant scanner `stat()`.
|
||||
|
||||
## [2.19.0] - 2026-09-12
|
||||
|
||||
### Security
|
||||
|
||||
+201
-26
@@ -1,6 +1,6 @@
|
||||
cmake_minimum_required(VERSION 3.22)
|
||||
|
||||
project(FastFileTransfer VERSION 2.19.0)
|
||||
project(FastFileTransfer VERSION 2.23.0)
|
||||
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_C_STANDARD 11)
|
||||
@@ -38,11 +38,24 @@ if(ENABLE_COVERAGE)
|
||||
add_link_options(--coverage)
|
||||
endif()
|
||||
|
||||
# --- Build hardening option ---
|
||||
# Production hardening is applied to the shipping server/client binaries only,
|
||||
# and only when no sanitizer or coverage instrumentation is active: sanitizers
|
||||
# carry their own instrumentation, and _FORTIFY_SOURCE requires an optimising
|
||||
# build (never the -O0 used for coverage).
|
||||
option(ENABLE_HARDENING "Enable compiler/linker hardening for production targets" ON)
|
||||
set(HARDENING_ACTIVE OFF)
|
||||
if(ENABLE_HARDENING AND SANITIZER STREQUAL "none" AND NOT ENABLE_COVERAGE)
|
||||
set(HARDENING_ACTIVE ON)
|
||||
endif()
|
||||
|
||||
include(FetchContent)
|
||||
FetchContent_Declare(
|
||||
xxhash
|
||||
GIT_REPOSITORY https://github.com/Cyan4973/xxHash
|
||||
GIT_TAG v0.8.3
|
||||
# v0.8.3 is a lightweight tag pointing at this exact commit (no ^{} peel
|
||||
# entry); pin the commit SHA instead of the mutable tag.
|
||||
GIT_TAG e626a72bc2321cd320e953a0ccf1584cad60f363 # v0.8.3
|
||||
SOURCE_SUBDIR cmake_unofficial
|
||||
)
|
||||
FetchContent_MakeAvailable(xxhash)
|
||||
@@ -57,35 +70,181 @@ endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
file(GLOB SHARED_SRCS "src/shared/*.c")
|
||||
set(FILE_STORE_SRCS "${CMAKE_CURRENT_SOURCE_DIR}/src/shared/file_store.c")
|
||||
list(REMOVE_ITEM SHARED_SRCS ${FILE_STORE_SRCS})
|
||||
file(GLOB SERVER_SRCS "src/server/*.c")
|
||||
set(SERVER_RECEIVER_SRCS src/server/receiver.c)
|
||||
file(GLOB CLIENT_SRCS "src/client/*.c")
|
||||
# --- Explicit source lists ---
|
||||
# The shared library is self-contained: it must never depend on the client or
|
||||
# server modules. In particular, the receiver pipeline (receive_thread /
|
||||
# write_thread) lives under src/server, not here, so the client executable can
|
||||
# link the shared library without pulling in any server code.
|
||||
set(SHARED_SRCS
|
||||
src/shared/array_list.c
|
||||
src/shared/batch.c
|
||||
src/shared/charset.c
|
||||
src/shared/checksum.c
|
||||
src/shared/chmod.c
|
||||
src/shared/chunk.c
|
||||
src/shared/compression.c
|
||||
src/shared/config.c
|
||||
src/shared/credentials.c
|
||||
src/shared/daemon_conf.c
|
||||
src/shared/daemon_limits.c
|
||||
src/shared/data.c
|
||||
src/shared/delay_updates.c
|
||||
src/shared/delta.c
|
||||
src/shared/file.c
|
||||
src/shared/file_list.c
|
||||
src/shared/file_receive.c
|
||||
src/shared/file_send.c
|
||||
src/shared/file_store.c
|
||||
src/shared/filter.c
|
||||
src/shared/format.c
|
||||
src/shared/hardlink.c
|
||||
src/shared/identity.c
|
||||
src/shared/log.c
|
||||
src/shared/metadata.c
|
||||
src/shared/motd.c
|
||||
src/shared/multiprocessing.c
|
||||
src/shared/protocol.c
|
||||
src/shared/queue.c
|
||||
src/shared/stop_condition.c
|
||||
src/shared/transport_ssh.c
|
||||
src/shared/transport_tcp.c
|
||||
src/shared/transport_tls.c
|
||||
src/shared/utils.c
|
||||
src/shared/xattr.c
|
||||
)
|
||||
|
||||
# Server implementation (no main): the receiver read/write pipeline plus the
|
||||
# CLI parser. The server executable adds its own main (server.c).
|
||||
set(SERVER_CORE_SRCS
|
||||
src/server/receiver.c
|
||||
src/server/receiver_pipeline.c
|
||||
src/server/server_cli.c
|
||||
)
|
||||
set(SERVER_MAIN_SRCS src/server/server.c)
|
||||
|
||||
# Client implementation (no main): everything except the CLI entry point.
|
||||
set(CLIENT_CORE_SRCS
|
||||
src/client/change_list.c
|
||||
src/client/client_send.c
|
||||
src/client/client_validation.c
|
||||
src/client/scanner.c
|
||||
src/client/usage.c
|
||||
)
|
||||
set(CLIENT_MAIN_SRCS src/client/client_cli.c)
|
||||
|
||||
# --- Library targets ---
|
||||
add_library(fastsync_shared STATIC ${SHARED_SRCS})
|
||||
target_include_directories(fastsync_shared PUBLIC src/shared)
|
||||
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL
|
||||
OpenSSL::Crypto xxhash)
|
||||
|
||||
add_library(fastsync_client_core STATIC ${CLIENT_CORE_SRCS})
|
||||
target_include_directories(fastsync_client_core PUBLIC src/client)
|
||||
target_link_libraries(fastsync_client_core PUBLIC fastsync_shared)
|
||||
|
||||
add_library(fastsync_server_core STATIC ${SERVER_CORE_SRCS})
|
||||
target_include_directories(fastsync_server_core PUBLIC src/server)
|
||||
target_link_libraries(fastsync_server_core PUBLIC fastsync_shared)
|
||||
|
||||
# --- Main executables ---
|
||||
add_executable(server ${SERVER_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS})
|
||||
target_include_directories(server PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
# The client links only the shared library and its own core; it deliberately
|
||||
# does NOT get src/server on its include path nor compile receiver.c.
|
||||
add_executable(server ${SERVER_MAIN_SRCS})
|
||||
target_link_libraries(server PRIVATE fastsync_server_core)
|
||||
|
||||
add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
|
||||
target_include_directories(client PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
add_executable(client ${CLIENT_MAIN_SRCS})
|
||||
target_link_libraries(client PRIVATE fastsync_client_core)
|
||||
|
||||
# --- Production hardening ---
|
||||
# Each compile flag is probed so a compiler/architecture that lacks it still
|
||||
# configures cleanly. _FORTIFY_SOURCE is guarded separately because it only
|
||||
# works in an optimising build. xxHash is a static archive built by
|
||||
# FetchContent, so it must be position-independent for the -pie link; the same
|
||||
# applies to the first-party static libraries linked into the -pie binaries.
|
||||
if(HARDENING_ACTIVE)
|
||||
set_target_properties(xxhash fastsync_shared fastsync_server_core fastsync_client_core
|
||||
PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
include(CheckCCompilerFlag)
|
||||
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
|
||||
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
|
||||
check_c_compiler_flag("${flag}" ${_harden_var})
|
||||
endforeach()
|
||||
check_c_compiler_flag("-D_FORTIFY_SOURCE=2" HARDEN_FORTIFY_SOURCE)
|
||||
foreach(target fastsync_shared fastsync_server_core fastsync_client_core server client)
|
||||
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
|
||||
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
|
||||
if(${_harden_var})
|
||||
target_compile_options(${target} PRIVATE ${flag})
|
||||
endif()
|
||||
endforeach()
|
||||
if(HARDEN_FORTIFY_SOURCE)
|
||||
target_compile_options(${target} PRIVATE -D_FORTIFY_SOURCE=2)
|
||||
endif()
|
||||
endforeach()
|
||||
foreach(target server client)
|
||||
target_link_options(${target} PRIVATE -pie -Wl,-z,relro -Wl,-z,now -Wl,-z,noexecstack)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
# --- Testing ---
|
||||
enable_testing()
|
||||
|
||||
# Common test libraries
|
||||
set(TEST_LIBS Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
set(TEST_INCLUDES tests src/shared src/server src/client)
|
||||
# --- Unit tests ---
|
||||
# The monolithic test binary exercises both client and server code, so it is
|
||||
# the one place that legitimately sees both include directories and links both
|
||||
# core libraries. client_cli.c is compiled here directly (with the test build
|
||||
# define) rather than linked from fastsync_client_core so its test-only shims
|
||||
# and the absence of main() are preserved.
|
||||
set(TEST_SRCS
|
||||
tests/runner.c
|
||||
tests/test_array_list.c
|
||||
tests/test_batch.c
|
||||
tests/test_change_list.c
|
||||
tests/test_checksum.c
|
||||
tests/test_chunk.c
|
||||
tests/test_client_cli.c
|
||||
tests/test_compression.c
|
||||
tests/test_config.c
|
||||
tests/test_credentials.c
|
||||
tests/test_daemon_conf.c
|
||||
tests/test_daemon_limits.c
|
||||
tests/test_data.c
|
||||
tests/test_delay_updates.c
|
||||
tests/test_delta.c
|
||||
tests/test_file.c
|
||||
tests/test_file_list.c
|
||||
tests/test_file_sendfile.c
|
||||
tests/test_format.c
|
||||
tests/test_fuzz_smoke.c
|
||||
tests/test_glob.c
|
||||
tests/test_hardlink.c
|
||||
tests/test_iconv.c
|
||||
tests/test_log.c
|
||||
tests/test_metadata.c
|
||||
tests/test_motd.c
|
||||
tests/test_multiprocessing.c
|
||||
tests/test_property.c
|
||||
tests/test_protocol.c
|
||||
tests/test_protocol_error.c
|
||||
tests/test_queue.c
|
||||
tests/test_receiver_timeout.c
|
||||
tests/test_robustness.c
|
||||
tests/test_scanner.c
|
||||
tests/test_server.c
|
||||
tests/test_server_cli.c
|
||||
tests/test_shared_utils.c
|
||||
tests/test_stop.c
|
||||
tests/test_stress.c
|
||||
tests/test_transport_ssh.c
|
||||
tests/test_transport_tcp.c
|
||||
tests/test_transport_tls.c
|
||||
tests/test_xattr.c
|
||||
)
|
||||
|
||||
# Monolithic test binary (backward compatible)
|
||||
file(GLOB TEST_SRCS "tests/test_*.c" "tests/runner.c")
|
||||
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS} src/client/scanner.c src/client/change_list.c src/client/client_cli.c src/client/client_validation.c src/client/usage.c src/server/server_cli.c)
|
||||
target_include_directories(tests PRIVATE ${TEST_INCLUDES})
|
||||
add_executable(tests ${TEST_SRCS} src/client/client_cli.c)
|
||||
target_include_directories(tests PRIVATE tests)
|
||||
target_compile_definitions(tests PRIVATE FASTSYNC_TEST_BUILD)
|
||||
target_link_libraries(tests PRIVATE ${TEST_LIBS})
|
||||
target_link_libraries(tests PRIVATE fastsync_server_core fastsync_client_core)
|
||||
add_test(NAME unit_all COMMAND tests)
|
||||
|
||||
# --- Fuzz targets (requires clang) ---
|
||||
@@ -94,13 +253,29 @@ if(ENABLE_FUZZ)
|
||||
if(NOT CMAKE_C_COMPILER_ID MATCHES "Clang")
|
||||
message(FATAL_ERROR "ENABLE_FUZZ requires Clang (compiler is ${CMAKE_C_COMPILER_ID})")
|
||||
endif()
|
||||
file(GLOB FUZZ_SRCS "tests/fuzz/*.c")
|
||||
set(FUZZ_SRCS
|
||||
tests/fuzz/fuzz_chunk_deserialize.c
|
||||
tests/fuzz/fuzz_compress_decompress.c
|
||||
tests/fuzz/fuzz_config_receive.c
|
||||
tests/fuzz/fuzz_delta_deserialize.c
|
||||
tests/fuzz/fuzz_delta_signature_deserialize.c
|
||||
tests/fuzz/fuzz_glob_match.c
|
||||
tests/fuzz/fuzz_identity_parse.c
|
||||
tests/fuzz/fuzz_manifest.c
|
||||
tests/fuzz/fuzz_metadata_from_buf.c
|
||||
tests/fuzz/fuzz_protocol_framing.c
|
||||
tests/fuzz/fuzz_xattr_block.c
|
||||
)
|
||||
# Compile the sources under test directly so libFuzzer's coverage
|
||||
# instrumentation sees them (static libraries would be uninstrumented).
|
||||
set(FUZZ_CORE_SRCS ${SHARED_SRCS} src/server/receiver.c src/server/receiver_pipeline.c)
|
||||
foreach(FUZZ_SRC ${FUZZ_SRCS})
|
||||
get_filename_component(FUZZ_NAME ${FUZZ_SRC} NAME_WE)
|
||||
add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE ${TEST_INCLUDES})
|
||||
add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${FUZZ_CORE_SRCS})
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE tests src/shared src/server)
|
||||
target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer)
|
||||
target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE ${TEST_LIBS})
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL
|
||||
OpenSSL::Crypto xxhash)
|
||||
endforeach()
|
||||
endif()
|
||||
+15
-1
@@ -2,8 +2,22 @@ FROM ubuntu:24.04
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \
|
||||
python3 python3-pip python3-venv openssl openssh-client \
|
||||
lcov valgrind clang libclang-rt-18-dev && \
|
||||
lcov valgrind clang libclang-rt-18-dev \
|
||||
acl attr zlib1g-dev liblz4-dev libxxhash-dev && \
|
||||
pip3 install --break-system-packages pytest pytest-xdist && \
|
||||
curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \
|
||||
apt-get install -y --no-install-recommends nodejs && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# rsync is used as the reference implementation for drop-in parity tests.
|
||||
# Ubuntu 24.04 ships 3.2.7, so build the pinned 3.4.1 reference from source.
|
||||
ARG RSYNC_VERSION=3.4.1
|
||||
ARG RSYNC_SHA256=2924bcb3a1ed8b551fc101f740b9f0fe0a202b115027647cf69850d65fd88c52
|
||||
RUN curl -fsSL "https://download.samba.org/pub/rsync/src/rsync-${RSYNC_VERSION}.tar.gz" -o /tmp/rsync.tar.gz && \
|
||||
echo "${RSYNC_SHA256} /tmp/rsync.tar.gz" | sha256sum -c - && \
|
||||
tar -xzf /tmp/rsync.tar.gz -C /tmp && \
|
||||
cd "/tmp/rsync-${RSYNC_VERSION}" && \
|
||||
./configure --enable-zstd --enable-xxhash --enable-lz4 && \
|
||||
make -j"$(nproc)" && \
|
||||
make install && \
|
||||
rm -rf "/tmp/rsync-${RSYNC_VERSION}" /tmp/rsync.tar.gz
|
||||
+66
@@ -0,0 +1,66 @@
|
||||
# FastSync — Session Handoff (2026-09-14)
|
||||
|
||||
## Current status
|
||||
- **Release `v2.21.0`** tagged (`919a729`, "Release v2.21.0"); full CI green
|
||||
(run 552: lint, build-and-test, ASan, UBSan, fuzz-build, coverage, valgrind).
|
||||
`dev` has the release commit plus later doc-only merges (a README refresh and
|
||||
this handoff).
|
||||
- **Release PR #284 (`dev` -> `main`)** open, CI green (run 553).
|
||||
`main` is protected: it needs review/approval to merge.
|
||||
https://gitea.tap-tap.win/TapTap/FastSync/pulls/284
|
||||
- **`PROTOCOL_VERSION` = `"2.23.0"`** (`src/shared/config.h`); CMake
|
||||
`project(FastFileTransfer VERSION 2.23.0)`.
|
||||
- Working tree clean; no wave worktrees remain.
|
||||
|
||||
## What landed this session
|
||||
1. **Wave 8 (refactors):** Config X-macro wire table; single-owner `authorized_root`;
|
||||
daemon per-module/per-host caps + cross-process auth lockout (`daemon_limits.[ch]`);
|
||||
`Data` charge returns to its owning `ProtocolSession`.
|
||||
2. **Wave 9 (protocol 2.21.0):** optional `STATUS_ERROR_DETAIL` rejection reasons;
|
||||
server-contacting `--dry-run` (`STATUS_DRY_RUN_TRANSFER`, receiver mutates nothing).
|
||||
3. **Security wave:** ran 5 parallel audits (wire parsing; daemon/transport/TLS/auth;
|
||||
receiver confinement; client/CLI/SSH; crypto/memory/limits). Fixed all HIGH and the
|
||||
confirmed MEDIUMs:
|
||||
- SSH `-o ProxyCommand=…` argument injection (RCE) — reject leading `-`, insert `--`.
|
||||
- Truncated zstd frame infinite CPU loop (remote DoS).
|
||||
- FIFO receiver opens lacked `O_NONBLOCK` (indefinite hang).
|
||||
- `--inplace` could write a FIFO/device (bypass of `--write-devices` gate).
|
||||
- `--force` not gated by server `--allow-delete`.
|
||||
- Privileged standalone server defaulted super activities on; added `--allow-super`
|
||||
(never honored with `--stdio`).
|
||||
- `--dry-run` content/hash oracle on `read only`/basis files removed.
|
||||
- Empty `hosts allow`/`deny`/`auth users` now rejected.
|
||||
- TLS: AEAD-only 1.2 + server preference, TOCTOU-safe key load, IP-SAN verify,
|
||||
CN-truncation guard. Glob backtracking bounded; line reads bounded; ACL xattrs
|
||||
gated on `--acls`; decompression/chunk memory charged; pre-auth `basis_count`
|
||||
NULL-deref fixed.
|
||||
4. **Tooling:** benchmark accuracy (data mix, verification, percentiles, `tc`,
|
||||
`build-bench/`, `--warm` mode); `shell.nix` full toolchain and no build-on-entry;
|
||||
docs state push-only / remote-source unsupported.
|
||||
5. **Preserve-attribute split (protocol 2.22.0)** landed on `feat/preserve-attr-split`: per-attribute `-p/-t/-o/-g` + `--no-*` negations, `-a` = `-rlptgoD`, and the 2.21.0 → 2.22.0 wire bump.
|
||||
6. **Rsync-parity wave (protocol 2.23.0)** on `feat/rsync-parity`: rsync short options/clustering/attached values (`-r`/`-b`/`-L`/`-B`, `-av`, `-aAX`, `-B1000`, `-essh`, `-MOPT`), `-c` checksum quick-check, `--checksum-choice`/`--compress-choice` validation and seed randomization, rsync timeout/max-alloc defaults, temp-dir confinement + `EXDEV` fallback, ownership/mapping parity (numeric-ids modifier, map ranges/`*`/empty-FROM, `--chown`+map conflicts, fake-super resolved-owner record), verbatim symlink storage with rsync `--safe-links`/`--munge-links`, socket recreation under `--specials`, `--chmod` 3.4.1 semantics, and delete scoping + `--max-delete` partial/exit-25. Wire: appended delete-manifest synchronized-directory section and `STATUS_DELETE_LIMIT`.
|
||||
|
||||
## Next steps
|
||||
1. **Merge PR #284** (`dev` -> `main`) once reviewed (protected branch).
|
||||
2. **Deferred security items** (documented, not implemented):
|
||||
- Pre-auth config/daemon-auth handshake has no aggregate wall-clock deadline
|
||||
(per-message timeout only) — slowloris holds connection slots.
|
||||
- Per-source registry fails open when the shared table is full (per-module/global
|
||||
caps and host ACLs still apply); consider fail-closed or larger/evicting table.
|
||||
- SCRAM-like daemon auth has no TLS channel binding (and is not RFC 5802).
|
||||
- `cleanup()` signal handler calls non-async-signal-safe teardown; daemon `umask(0)`.
|
||||
- Wire protocol assumes homogeneous word size/endianness (lengths are native
|
||||
`size_t`) — document or move to fixed-width framing.
|
||||
3. **Out of scope / intentional:** pull (remote source) mode is **not** planned —
|
||||
FastSync is push-only; see `RSYNC_COMPAT.md#direction`.
|
||||
|
||||
## Key facts / commands
|
||||
- CI image: `gitea.tap-tap.win/taptap/fastsync-ci:v11` (alias `fastsync-ci:local`).
|
||||
- Build/test: `cmake -B build -S . -DSTRICT_WARNINGS=ON && cmake --build build -j$(nproc) && ./build/tests`
|
||||
then `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`.
|
||||
- Dev shell: `nix-shell` (provides clang-format, cppcheck, pytest-xdist, openssh,
|
||||
rsync, iproute2, valgrind, lcov; does not build on entry).
|
||||
- Gitea API token: supplied out-of-band via the `TOKEN` environment variable; it is
|
||||
intentionally **not** recorded in this file.
|
||||
- CI polling: `GET /api/v1/repos/TapTap/FastSync/actions/runs?limit=N`, match `head_sha`,
|
||||
then `/actions/runs/<id>/jobs`.
|
||||
@@ -7,7 +7,7 @@ multithreading, streaming zstd compression, chunking, zero-copy TCP transfers,
|
||||
and native TCP/TLS transports.
|
||||
|
||||
The release version is FastSync's client/server protocol version (printed by
|
||||
`fastsync --version`); client and server must match. See
|
||||
`./build/client --version`); client and server must match. See
|
||||
[CHANGELOG.md](CHANGELOG.md) for the history.
|
||||
|
||||
The compatibility target is straightforward:
|
||||
@@ -51,8 +51,9 @@ replacement for every rsync feature or protocol mode.
|
||||
- Rsync-style source and destination arguments.
|
||||
- SSH transport using `user@host:destination` paths below the remote authorized root.
|
||||
- TCP client/server transfers.
|
||||
- Dry runs, excludes, includes, size filters, backups, statistics, and
|
||||
bandwidth limiting.
|
||||
- Dry runs (server-contacting since protocol 2.21.0 for server-routed targets),
|
||||
excludes, includes, size filters, backups, statistics, and bandwidth
|
||||
limiting.
|
||||
- Incremental size/mtime checks and optional xxHash64 content checks.
|
||||
- FastSync-native delta transfer for changed files.
|
||||
- Optional mode and timestamp preservation.
|
||||
@@ -60,19 +61,40 @@ replacement for every rsync feature or protocol mode.
|
||||
- Temporary-file writes with atomic rename by default.
|
||||
- Path traversal checks and destination-root confinement.
|
||||
|
||||
### Not yet equivalent to rsync
|
||||
### Boundaries and documented divergences
|
||||
|
||||
The items below summarize FastSync's rsync compatibility status — recently
|
||||
closed gaps and the remaining known divergences. Each row of the detailed
|
||||
matrix is classified as parity, caveat, or divergent in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
- The FastSync wire protocol is not the rsync wire protocol.
|
||||
- SSH mode requires `fastsync-server` on the remote host.
|
||||
- Archive mode does not yet provide all of rsync's `-rlptgoD` behavior.
|
||||
- Symlink transfer is incomplete; link targets are not yet recreated in all
|
||||
modes.
|
||||
- Owner/group, ACL, xattr, and hard-link handling is incomplete or
|
||||
unavailable.
|
||||
- Archive mode covers rsync's `-rlptgoD` behavior — links, permissions, times,
|
||||
owner, group, devices, and special files — and does not imply compression or
|
||||
multithreading (see [Client](#client)). Ownership application is still
|
||||
privilege-gated: a receiver that cannot `chown` logs a warning and skips it.
|
||||
Under `-p` the source mode is copied exactly, including setuid/setgid/sticky
|
||||
and group/other-write bits (strict rsync parity; see
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)).
|
||||
- Symlink transfer stores targets **verbatim** (`-l`/`--links`), including
|
||||
absolute and `..`-bearing targets, matching rsync. The receiver does not
|
||||
enforce a containment predicate by default; `--safe-links` drops unsafe
|
||||
targets on the sender, and `--munge-links` rewrites them with rsync's
|
||||
`/rsyncd-munged/` marker. `--trust-sender` does not affect symlink targets.
|
||||
A destination later consumed by a link-following tool can therefore follow a
|
||||
link outside the receive root — use `--safe-links` for untrusted sources.
|
||||
- Hard links (`-H`/`--hard-links`), extended attributes (`-X`/`--xattrs`), and
|
||||
POSIX ACLs (`-A`/`--acls`) are preserved; owner/group is applied through
|
||||
`-o`/`-g` (or an `-a`/`--archive` transfer), through the opt-in identity flags
|
||||
(`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`), and only when
|
||||
the receiver has permission. See
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md) for the exact semantics and documented
|
||||
divergences.
|
||||
- Device and special-file preservation is implemented with documented
|
||||
divergences: recreated device nodes require `CAP_MKNOD` on the receiver (a
|
||||
non-root receiver skips the entry), and sockets cannot be recreated (FIFOs
|
||||
are).
|
||||
non-root receiver skips the entry), while FIFOs **and unix sockets** are
|
||||
recreated (`--specials`).
|
||||
- Sparse-file hole preservation (`-S`, `--sparse`) is implemented receiver-side:
|
||||
long all-zero runs are written as holes (no wire change; the full file image
|
||||
is already in memory).
|
||||
@@ -80,76 +102,164 @@ replacement for every rsync feature or protocol mode.
|
||||
the write atomic (temp + rename). With `--partial`, a failed/interrupted write
|
||||
now retains the already-written temp at the destination path (best-effort) so
|
||||
a later `--append`/`--append-verify` run can resume it.
|
||||
- `--dirs` is not implemented. Its compatibility aliases `--old-dirs` and
|
||||
`--old-d` are recognized but rejected explicitly rather than silently using
|
||||
FastSync's recursive directory behavior.
|
||||
- `-d`/`--dirs` and its aliases `--old-dirs`/`--old-d` transfer the named
|
||||
directory entries without recursing into their contents.
|
||||
- Short-option names are now rsync-parity (Phase 7 Wave A): FastSync's former
|
||||
collisions were renamed (`-j`/`--threads`, `--preserve`, `--sendfile`,
|
||||
`--chunk-serialization`, `--timeout`, `--ssh-port`), so `-m`, `-M`, `-f`,
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync. See `RSYNC_COMPAT.md`.
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync.
|
||||
- Short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached values
|
||||
(`-B1000`, `-essh`, `-MOPT`, `--opt=value`) are accepted, matching rsync.
|
||||
- `-r`, `-b`, `-L`, and `-B` are parsed with the rsync short names.
|
||||
- `--stats` prints the counters FastSync can observe locally; receiver-only
|
||||
counters (matched data, file-list bytes, deleted count) are reported as 0, and
|
||||
`--progress` is an aggregate line rather than a per-file block.
|
||||
|
||||
The detailed flag matrix is maintained in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented,
|
||||
partial, alternate, and planned behavior.
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It reports each row as **parity**,
|
||||
**caveat** (works with a documented divergence), or **divergent** (not
|
||||
supported), rather than treating "parsed" as parity.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Build
|
||||
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
This produces `./build/client` and `./build/server`. `compile_commands.json` is a symlink to `build/compile_commands.json` and is used by clangd/editor tooling; its target is generated by the build, so it dangles until the first build.
|
||||
|
||||
### Client
|
||||
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| Positional | `<source> <dest>` — automatic SSH detection if dest contains `:` |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime (implies the incremental checksum quick-check) |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh64`/`xxhash` (default), `xxh3`, `xxh128`, `md5`, or `auto`; `md4`/`sha1`/`none` are rejected by name |
|
||||
| `-z, --compress [level]` | Enable streaming zstd compression (level 1–22, default 5) |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials (not compression/multithreading) |
|
||||
| `-j, --threads` | Multithreading mode |
|
||||
| `--compress-choice <alg>` | Compression algorithm: `zstd` (default), `none`, or `auto`; `lz4`/`zlib`/`zlibx` are rejected by name |
|
||||
| `--skip-compress <list>` | Skip compression for suffixes (`/`- or `,`-separated); defaults to rsync 3.4.1's built-in suffix list |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated (not compression/multithreading) |
|
||||
| `-j, --threads[=N]` | Multithreading mode; `N` (1–256) sets the parallel scanner worker count, bare `-j`/`--threads` uses the default |
|
||||
| `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) |
|
||||
| `-r, --recursive` | Recurse into directories (FastSync is always recursive; accepted for rsync compatibility) |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents; aliases `--old-dirs`/`--old-d` |
|
||||
| `-R, --relative` | With `--files-from`, preserve each listed entry's relative path below the destination root |
|
||||
| `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) |
|
||||
| `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) |
|
||||
| `--sendfile` | Sendfile zero-copy. Incompatible with compression / chunk serialization. TCP only. Long form only. |
|
||||
| `--preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) |
|
||||
| `-n, --dry-run` | Scan and print what would be transferred |
|
||||
| `-p, --perms` | Preserve permission bits (part of the metadata bundle) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show real-time transfer speed |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded) |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk) |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`) |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch) |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing; `--no-whole-file` clears it |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`) |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime |
|
||||
| `--preserve` | Preserve mode and mtime (`-p` + `-t`; add `-o`/`-g` for owner/group or `-U`/`--atimes` for atime; `-N`/`--crtimes` captures birth time but cannot apply it) |
|
||||
| `-U, --atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N, --crtimes` | Capture birth time; cannot be applied (documented divergence) |
|
||||
| `-p, --perms` | Preserve permission bits. Strict rsync parity: the source mode is copied exactly, including setuid/setgid/sticky and group/other-write bits |
|
||||
| `-t, --times` | Preserve modification times |
|
||||
| `-o, --owner` | Preserve the source owner (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `-g, --group` | Preserve the source group (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group`, `--no-preserve` | Negate the per-attribute flags (short `--no-p`/`--no-t`/`--no-o`/`--no-g`; `--no-preserve` clears all four) |
|
||||
| `-E, --executability` | Preserve executable permission bits |
|
||||
| `-X, --xattrs` | Preserve user `user.*` extended attributes |
|
||||
| `-A, --acls` | Preserve POSIX ACLs |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax) |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership |
|
||||
| `--numeric-ids` | Apply source numeric uid/gid directly instead of mapping by name |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP] (requires a privileged receiver) |
|
||||
| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes) |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`) |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`) |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--include-from <file>` | Read include patterns from a file |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `-x, --one-file-system` | Do not cross filesystem boundaries; the mount-point directory entry is emitted (empty at the destination) without descending |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G; `0` = no local limit, matching rsync) |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`) |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win) |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded). Scoped to the synchronized directories, so `--files-from` subsets are safe |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) |
|
||||
| `--delete-after` | Explicit delete-after timing (implies `--delete`) |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G) |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds (default: 30) |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 10) |
|
||||
| `--backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing) |
|
||||
| `-h, --human-readable` | Format transfer byte sizes with binary units |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication) |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install; confined to the receive root (relative only), with an `EXDEV` non-atomic copy fallback |
|
||||
| `-n, --dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show a periodic aggregate transfer line (bytes sent, current rate); not rsync's per-file progress block |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing). Receiver-only counters (matched data, file-list bytes, deleted count) are reported as 0 |
|
||||
| `-i, --itemize-changes` | Print an rsync-style per-file change line |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`) |
|
||||
| `--list-only` | List source files instead of transferring |
|
||||
| `--fsync` | Fsync every written file before publication |
|
||||
| `-h, --human-readable` | Format transfer byte/rate counts with rsync's decimal (base-1000) units |
|
||||
| `--max-depth <n>` | Maximum directory depth to recurse (0 = unlimited, default: 0) |
|
||||
| `--log-file <path>` | Write log messages to file instead of stderr |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server) |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server) |
|
||||
| `--source-dir <path>` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) |
|
||||
| `--dest-dir <path>` | Server destination directory (overrides `FASTSYNC_DEST_DIR`) |
|
||||
| `--save-to-disk` | Write received files to disk |
|
||||
| `--server-host <ip>` | Server IP address (default: `127.0.0.1`) |
|
||||
| `--server-port <n>` | Server port (default: `8080`) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-e, --rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments, e.g. `-e "ssh -p 2222"`) |
|
||||
| `-M, --remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable) |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address |
|
||||
| `-4, --ipv4` | Force IPv4 for destination resolution |
|
||||
| `-6, --ipv6` | Force IPv6 for destination resolution |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect (`TCP_NODELAY`, `SO_KEEPALIVE`, `SO_RCVBUF`, `SO_SNDBUF`, `SO_REUSEADDR`) |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline. Default `0` = disabled (matching rsync); `0` disables it. `--no-timeout` is the negation. The value is not sent on the wire; the server side keeps its own safe floor. |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 60, matching rsync); `0` disables it (`--no-contimeout` is the negation) |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes (a positive integer); whatever was already transferred is kept |
|
||||
| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`); an early stop skips the late `--delete` keep-set |
|
||||
| `-b, --backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--tls` | Enable TLS encryption |
|
||||
| `--cert <path>` | TLS certificate file (PEM) |
|
||||
| `--key <path>` | TLS private key file (PEM) |
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--client-cn <name>` | TLS client certificate common name; mandatory with `--tls` (a TLS connection always verifies the client CN) |
|
||||
|
||||
The exhaustive rsync flag matrix is in [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
**Per-message vs. connection timeouts.** `--timeout` bounds each individual protocol
|
||||
send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It
|
||||
does not, by itself, stop a peer that keeps sending well-formed frames forever. The
|
||||
receiver therefore also enforces two wall-clock (`CLOCK_MONOTONIC`) bounds on a
|
||||
connection: a **1 hour** idle limit and a **24 hour** overall session cap. Only
|
||||
frames that move real work (not `STATUS_KEEPALIVE`/`STATUS_ABORT` and not an
|
||||
empty `STATUS_CHECK_BATCH`/`STATUS_DIR_TIMES`) refresh the idle timestamp, so a
|
||||
peer cannot hold a connection slot by emitting cheap empty frames; a peer that
|
||||
fabricates minimal non-empty frames can still occupy a slot until the 24 hour
|
||||
cap, since no bound can require actual payload without risking a legitimate
|
||||
long operation. Both are deliberately generous so a legitimate long-running
|
||||
transfer is never aborted.
|
||||
|
||||
### Server
|
||||
|
||||
@@ -163,6 +273,7 @@ partial, alternate, and planned behavior.
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--destination-root <path>` | Authorized destination root (default: `.`) |
|
||||
| `--allow-delete` | Permit manifest deletion |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. **Rejected with `--stdio`** (the SSH remote argv is client-composed, so a client could otherwise pass it and defeat the secure default; operators exposing `fastsync-server --stdio` over SSH must use a forced command if the default must hold). No effect when not root. |
|
||||
| `--allow-unauthenticated` | Permit plaintext TCP clients. For an `auth users` module this opts in **loopback plaintext only**; remote auth still requires verified TLS, so the flag never permits remote plaintext auth. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `--help` | Show help |
|
||||
@@ -174,60 +285,64 @@ partial, alternate, and planned behavior.
|
||||
| `FASTSYNC_SOURCE_DIR` | — | Source directory fallback |
|
||||
| `FASTSYNC_DEST_DIR` | — | Destination directory fallback |
|
||||
| `FASTSYNC_SAVE_TO_DISK` | `false` | Disk persistence fallback |
|
||||
| `FASTSYNC_SSH_PORT` | `22` | Default SSH port |
|
||||
| `FASTSYNC_SERVER_HOST` | `127.0.0.1` | Default server host |
|
||||
| `FASTSYNC_SERVER_PORT` | `8080` | Default server port |
|
||||
| `FASTSYNC_TLS_CERT` | — | Default TLS certificate path |
|
||||
| `FASTSYNC_TLS_KEY` | — | Default TLS private key path |
|
||||
| `FASTSYNC_TLS_CA` | — | Default TLS CA certificate path |
|
||||
|
||||
## Implementation Details
|
||||
|
||||
### Data Structures
|
||||
1. **Chunk** — collection of files (~10 MB total by default)
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`;
|
||||
uid / gid are advisory wire fields and are never applied by the receiver;
|
||||
atime is unsupported
|
||||
4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`, `queue_size`.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement
|
||||
|
||||
1. **Chunk** — collection of files (~10 MB total by default).
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer.
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec` (plus
|
||||
atime/crtime fields). `uid`/`gid` are applied only through the opt-in
|
||||
identity path; atime is preserved with `-U`/`--atimes`; crtime is captured
|
||||
but cannot be set on the destination.
|
||||
4. **Config** — runtime parameters. Most cross the wire (TLS settings
|
||||
excluded); `backup` and `backup_dir` are in the serialized wire table, while
|
||||
`timeout`, `contimeout`, `quiet`, `stats`, `max_depth`, and `log_file` are
|
||||
client-only.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables.
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include
|
||||
pattern support, max-depth enforcement.
|
||||
|
||||
### Key Algorithms
|
||||
1. **File scanning** — BFS directory traversal;
|
||||
entries matched against exclude and include patterns,
|
||||
max - depth enforced 2. * *Chunking ** — files accumulated until `chunk_size` threshold,
|
||||
then flushed 3. *
|
||||
*Compression ** — streaming zstd
|
||||
via `ZSTD_compressStream2` / `ZSTD_decompressStream` 4. *
|
||||
*Network protocol ** — status -
|
||||
code - driven exchange with metadata packing,
|
||||
keep - alive,
|
||||
and abort support 5. * *Incremental check ** — client sends `STATUS_CHECK` + path + size +
|
||||
mtime and,
|
||||
with `--checksum`, XXH64 content checksum; server compares against destination. Can be batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with `nanosleep` throttling on 64 KB write chunks
|
||||
7. **Metadata restoration** — `chmod()`, `chown()`, `utimensat()` on the receiving side
|
||||
8. **`--delete`** — sender tracks all sent paths;
|
||||
receiver walks destination tree and removes unlisted files / directories 9. *
|
||||
*SSH transport *
|
||||
* — `socketpair()` + `fork()` + `execvp("ssh",
|
||||
...)` with `ControlMaster` and port support
|
||||
10. *
|
||||
*TLS transport ** — OpenSSL `SSL_CTX` with TLS
|
||||
1.2 minimum,
|
||||
mutual CA verification,
|
||||
transparent `SSL_read`/`SSL_write` via `io_set_ssl()` 11. *
|
||||
*Path traversal protection ** — `has_path_traversal()` rejects any file path
|
||||
containing `..` components,
|
||||
preventing directory escape attacks 12. *
|
||||
*Connection limiting ** — server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100)13. *
|
||||
*Keep
|
||||
- alive ** — idle connections receive periodic `STATUS_KEEPALIVE` to detect half
|
||||
- open TCP connections 14. * *Abort handling ** — `SIGINT` sets an abort flag; the next protocol operation sends `STATUS_ABORT` for clean server cleanup
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically renamed via `rename()`, preventing partial files
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir` (or same directory with `~` suffix) preserving the original
|
||||
|
||||
1. **File scanning** — BFS directory traversal; entries matched against exclude
|
||||
and include patterns, with max-depth enforced.
|
||||
2. **Chunking** — files accumulated until the `chunk_size` threshold (default
|
||||
10 MiB) is reached, then flushed.
|
||||
3. **Compression** — streaming zstd via `ZSTD_compressStream2()` /
|
||||
`ZSTD_decompressStream()`.
|
||||
4. **Network protocol** — status-code-driven exchange with metadata packing,
|
||||
keep-alive, and abort support.
|
||||
5. **Incremental check** — the client sends `STATUS_CHECK` + path + size +
|
||||
mtime and, with `--checksum`, a whole-file content checksum (xxHash64 by
|
||||
default, or md5 via `--checksum-choice=md5`/`--cc`, seeded by
|
||||
`--checksum-seed`); the server compares against the destination. Can be
|
||||
batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with sleep throttling on
|
||||
64 KiB write chunks.
|
||||
7. **Metadata restoration** — mode via `chmod()`/`fchmod()`, times via
|
||||
`utimensat()`/`futimens()`, and ownership only with an identity flag via
|
||||
fd-relative `fchown()`/`fchownat()`.
|
||||
8. **`--delete`** — the sender tracks all sent paths; the receiver walks the
|
||||
destination tree and removes unlisted files and directories.
|
||||
9. **SSH transport** — `socketpair()` + `fork()` + `execvp("ssh", ...)` with
|
||||
`ControlMaster` and port support.
|
||||
10. **TLS transport** — OpenSSL `SSL_CTX` with TLS 1.2 minimum, mutual CA
|
||||
verification, and transparent `SSL_read()`/`SSL_write()` via
|
||||
`io_set_ssl()`.
|
||||
11. **Path traversal protection** — `has_path_traversal()` rejects any file
|
||||
path containing `..` components, preventing directory escape attacks.
|
||||
12. **Connection limiting** — the server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100).
|
||||
13. **Keep-alive** — idle connections receive periodic `STATUS_KEEPALIVE` to
|
||||
detect half-open TCP connections.
|
||||
14. **Abort handling** — `SIGINT` sets an abort flag; the next protocol
|
||||
operation sends `STATUS_ABORT` for clean server cleanup.
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically
|
||||
renamed via `rename()`, preventing partial files.
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir`
|
||||
(or the same directory with a `~` suffix), preserving the original.
|
||||
|
||||
## Security Features
|
||||
|
||||
@@ -284,22 +399,34 @@ cmake --build build -j$(nproc)
|
||||
|
||||
### SSH transfer
|
||||
|
||||
The remote host must have `fastsync-server` available in `PATH`, or use
|
||||
The remote host must have `fastsync-server` available in `PATH` (install or
|
||||
copy the built `./build/server` there as `fastsync-server`), or use
|
||||
`--fastsync-server-path`. SSH starts `fastsync-server --stdio` in its remote
|
||||
working directory, so use a destination below that directory unless the
|
||||
remote server is otherwise configured with a matching authorized root.
|
||||
|
||||
The remote `--stdio` server argv is composed by the client, so it must never
|
||||
be trusted to opt a root receiver into super-user activities: `--allow-super`
|
||||
is rejected with `--stdio` and super stays off on that path. Operators
|
||||
exposing `fastsync-server --stdio` over SSH must use a forced command (e.g. an
|
||||
`authorized_keys` `command=` entry) if the default must hold.
|
||||
|
||||
```bash
|
||||
ssh user@host 'mkdir -p destination'
|
||||
./build/client /path/to/source user@host:destination
|
||||
```
|
||||
|
||||
FastSync is **push-only**: the source (first argument) is always a local
|
||||
directory and only the destination may be remote. A remote source such as
|
||||
`client user@host:src ./local` (a "pull") is intentionally not supported; see
|
||||
[RSYNC_COMPAT.md](RSYNC_COMPAT.md#direction).
|
||||
|
||||
### TCP transfer
|
||||
|
||||
Start the FastSync server:
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to -p 8080
|
||||
./build/server --destination-root /path/to -p 8080 --allow-unauthenticated
|
||||
```
|
||||
|
||||
Then run the client:
|
||||
@@ -314,8 +441,13 @@ Plain TCP requires the explicit `--allow-unauthenticated` server option. Use TLS
|
||||
authenticated network connections.
|
||||
|
||||
### TLS transfer
|
||||
|
||||
Server TLS requires `--cert`, `--key`, `--ca`, and `--client-cn`; the client
|
||||
requires `--cert`, `--key`, and `--ca`.
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem -p 8443
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem \
|
||||
--ca ca.pem --client-cn client -p 8443
|
||||
./build/client --tls --cert client.pem --key client-key.pem --ca ca.pem \
|
||||
--server-host example.com --server-port 8443 \
|
||||
--source-dir /path/to/source --dest-dir /path/to/destination \
|
||||
@@ -351,7 +483,7 @@ FastSync-native are optional performance or transport extensions.
|
||||
./build/client --incremental --checksum /source/ user@host:destination/
|
||||
|
||||
# Preserve supported mode and timestamp metadata
|
||||
./build/client -M /source/ user@host:destination/
|
||||
./build/client --preserve /source/ user@host:destination/
|
||||
|
||||
# Keep backups of overwritten destination files
|
||||
./build/client --backup --backup-dir backups \
|
||||
@@ -365,12 +497,12 @@ features without changing the meaning of ordinary compatibility options.
|
||||
|
||||
| Option | Purpose |
|
||||
|---|---|
|
||||
| `-j`, `--threads` | Enable the multithreaded scanner/loader/sender pipeline. |
|
||||
| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming zstd compression, levels 1-22. |
|
||||
| `--compress-level <n>` | Set the zstd compression level. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` and `none`. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd`, `none`, and `auto`; `lz4`/`zlib`/`zlibx` are rejected by name. |
|
||||
| `--zl <n>` | Alias for `--compress-level`. |
|
||||
| `--skip-compress <list>` | Skip compression for comma-separated suffixes; incompatible with `--chunk-serialization`. |
|
||||
| `--skip-compress <list>` | Skip compression for `/`- or `,`-separated suffixes; defaults to rsync 3.4.1's built-in list. Incompatible with `--chunk-serialization`. |
|
||||
| `--compress-threads <n>` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. |
|
||||
| `--chunk-size <bytes>` | Set the transfer chunk size. |
|
||||
| `--chunk-serialization` | Enable FastSync chunk serialization (long form only; `-s` is rsync's `--secluded-args`). |
|
||||
@@ -379,13 +511,13 @@ features without changing the meaning of ordinary compatibility options.
|
||||
| `--delta-block <bytes>` | Set the FastSync delta block size (`--block-size` is an alias). |
|
||||
| `--delta-max <bytes>` | Limit files eligible for FastSync delta transfer. |
|
||||
| `--server-host <host>` | Select the TCP server host. |
|
||||
| `--server-port <port>` | Select the TCP server port. |
|
||||
| `--server-port <port>` | Select the TCP server port (`--port <port>` and `--port=<port>` are rsync-friendly aliases). |
|
||||
| `--tls` | Enable TLS for TCP transport. |
|
||||
| `--bwlimit <KB/s>` | Apply token-bucket bandwidth limiting. |
|
||||
| `--progress` | Show transfer progress and throughput. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `--timeout <seconds>` | Set I/O timeout. |
|
||||
| `--contimeout <seconds>` | Set connection timeout. |
|
||||
| `--progress` | Show a periodic aggregate transfer line (throughput; not a per-file block). |
|
||||
| `--stats` | Print transfer statistics (receiver-only counters are 0). |
|
||||
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout. Default `0` = disabled (matching rsync); `0` disables it. |
|
||||
| `--contimeout <seconds>` | Connection timeout (default 60, matching rsync); `0` disables it. |
|
||||
|
||||
Short-option conflicts with rsync have been resolved for the CLI namespace
|
||||
(Phase 7): `-c` is now rsync's `--checksum`, `-m` is `--prune-empty-dirs`, `-M`
|
||||
@@ -394,7 +526,8 @@ is `--remote-option`, `-f` is `--filter`, `-s` is `--secluded-args`, `-p` is
|
||||
long-form-only or new shorts: multithreading is `-j`/`--threads`, metadata
|
||||
is `--preserve`, sendfile is `--sendfile`, chunk serialization is
|
||||
`--chunk-serialization`, timeout is `--timeout`, and SSH port is `--ssh-port`.
|
||||
`-a`/`--archive` is now real rsync archive (`-rlptgoD`).
|
||||
`-a`/`--archive` is now rsync archive `-rlptgoD` (owner/group implied, but the
|
||||
receiver still needs privilege to apply them).
|
||||
|
||||
`--secluded-args` (and its short form `-s`) is accepted as a compatibility
|
||||
no-op. It does not change FastSync's transport or protocol behavior, because
|
||||
@@ -406,42 +539,93 @@ remote SSH argv is already built injection-safe.
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials. |
|
||||
| `-n`, `--dry-run` | Scan and report without writing files. |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. |
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated. |
|
||||
| `-n`, `--dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer. |
|
||||
| `--incremental` | Skip files matching destination size and mtime. Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `-c, --checksum` | Verify content by checksum (implies the incremental quick-check). Algorithm selectable with `--checksum-choice`. |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh64`/`xxhash` (default), `xxh3`, `xxh128`, `md5`, or `auto`. |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync. |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime. |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match. |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver. |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing (`--no-whole-file` clears it). |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`). |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents (aliases `--old-dirs`/`--old-d`). |
|
||||
| `-R, --relative` | With `--files-from`, preserve each listed entry's relative path below the destination root. |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root). |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`). |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination. |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win). |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk). |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`). |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch). |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. Scoped to the synchronized directories, so `--files-from` subsets are safe. |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). |
|
||||
| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected). |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync. |
|
||||
| `--force` | Allow an incoming file/symlink to replace a destination directory (also during `--delay-updates` publication). |
|
||||
| `--exclude <pattern>` | Exclude matching paths. Repeatable. |
|
||||
| `--include <pattern>` | Include matching paths. Repeatable. |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file. |
|
||||
| `--include-from <file>` | Read include patterns from a file. |
|
||||
| `--max-size <bytes>` | Skip files larger than the limit. |
|
||||
| `--min-size <bytes>` | Skip files smaller than the limit. |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth;
|
||||
zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.|
|
||||
| `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` |
|
||||
Back up overwritten files.| | `--backup - dir<dir>` | Store backups under a separate directory.|
|
||||
| `--suffix<suffix>` | Set the backup filename suffix.| | `--partial` |
|
||||
Select partial - transfer handling. On failed/interrupted writes the
|
||||
already-written temp file is retained (best-effort) for resumption.|
|
||||
With `--partial --partial-dir <dir>`, completed files are written under the
|
||||
partial directory and installed atomically. | | `--partial - dir<dir>` |
|
||||
Set a relative partial - transfer directory below the server destination root.
|
||||
Use with `--partial`. |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units; default 1G; `0` = no local limit). |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth; zero means unlimited. |
|
||||
| `-b, --backup` | Back up overwritten files. |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install (confined to the receive root; `EXDEV` falls back to a non-atomic copy). |
|
||||
| `--backup-dir <dir>` | Store backups under a separate directory (requires `--backup`). |
|
||||
| `--suffix <suffix>` | Set the backup filename suffix (default: `~`). |
|
||||
| `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir <dir>`, completed files are written under the partial directory and installed atomically. |
|
||||
| `--partial-dir <dir>` | Set a relative partial-transfer directory below the server destination root. Use with `--partial`. |
|
||||
| `--inplace` | Write directly to the destination instead of using a temporary file. |
|
||||
| `--fsync` | Fsync every written file before publication. |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree. |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server). |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server). |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes; whatever was already transferred is kept. |
|
||||
| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`). An early stop skips the late `--delete` keep-set. |
|
||||
|
||||
### Metadata and links
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--preserve` | Preserve supported file metadata, currently mode and modification time (long form only). |
|
||||
| `-l`, `--links` | Request symlink preservation;
|
||||
link-target transfer remains incomplete. |
|
||||
| `--copy-links` | Copy symlink referents. |
|
||||
| `--safe-links` | Skip symlinks that point outside the transfer tree. |
|
||||
| `--preserve` | Preserve mode and mtime (long form only; equivalent to `-p` + `-t`). Add `-o`/`-g` for owner/group, `-U`/`--atimes` for atime, or an identity flag (`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`) for mapped ownership. |
|
||||
| `-U`, `--atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N`, `--crtimes` | Capture birth time and transmit it; it cannot be applied because no portable filesystem call can set a birth time (documented divergence). |
|
||||
| `-p`, `--perms` | Preserve permission bits. One of the four per-attribute preserve flags (with `-t`/`-o`/`-g`); under `-p` the source mode is copied exactly (setuid/setgid/sticky and group/other-write included), matching rsync. |
|
||||
| `-t`, `--times` | Preserve modification times. Independent of the other attributes; `-O`/`--omit-dir-times` suppresses directories only. |
|
||||
| `-o`, `--owner` | Preserve the source owner (uid). Mapped by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); application is privilege-gated. |
|
||||
| `-g`, `--group` | Preserve the source group (gid). Same name-mapping/numeric-fallback and privilege gating as `-o`. |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group` | Negate each per-attribute flag (also `--no-p`/`--no-t`/`--no-o`/`--no-g`); `--no-preserve` clears all four. |
|
||||
| `-E`, `--executability` | Preserve executable permission bits. |
|
||||
| `-X`, `--xattrs` | Preserve user `user.*` extended attributes. |
|
||||
| `-A`, `--acls` | Preserve POSIX ACLs. |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax, including `D`/`F`/`X` selectors and `s`/`t`); does not imply `-p`. |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files (`USER:GROUP`, `USER`, or `:GROUP`); conflicts with `--usermap`/`--groupmap` on the same side. |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership (`FROM:TO` rules; names, ids, `LOW-HIGH` ranges, `*`, empty-`FROM`). |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership (same syntax as `--usermap`). |
|
||||
| `--numeric-ids` | Mapping modifier: apply the source numeric uid/gid directly instead of mapping by name (combine with `-o`/`-g`, `-a`, or a map). |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP]; requires a privileged receiver. |
|
||||
| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown. |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes). |
|
||||
| `--no-super` | Forbid those super-user activities even when the receiver is root. |
|
||||
| `-l`, `--links` | Copy symlinks as symlinks; the target is stored verbatim (absolute and `..`-bearing targets included), matching rsync. |
|
||||
| `-L`, `--copy-links` | Copy symlink referents (a broken referent exits 0). |
|
||||
| `--safe-links` | Skip symlinks whose target points outside the transfer tree (applied on the sender). |
|
||||
| `--copy-unsafe-links` | Copy unsafe symlink referents. |
|
||||
| `--munge-links` | Rewrite stored symlink targets with rsync's `/rsyncd-munged/` marker. |
|
||||
| `-k`, `--copy-dirlinks` | Treat a symlink to a directory as a real directory on the sender. |
|
||||
| `-K`, `--keep-dirlinks` | Follow an existing destination symlink-to-directory (confined to the receive root). |
|
||||
| `-H`, `--hard-links` | Preserve hard-link relationships across the transfer. |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`). |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`). |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets. |
|
||||
| `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). |
|
||||
|
||||
### Output and logging
|
||||
@@ -449,8 +633,12 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--progress` | Show live transfer progress. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `-q`, `--quiet` | Suppress non-error output. |
|
||||
| `--progress` | Show a periodic aggregate transfer line (not a per-file block). |
|
||||
| `--stats` | Print transfer statistics (receiver-only counters are reported as 0). |
|
||||
| `-i`, `--itemize-changes` | Print an rsync-style per-file change line. |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`). |
|
||||
| `--list-only` | List source files instead of transferring. |
|
||||
| `--log-file <path>` | Write log output to a file. |
|
||||
| `-V`, `--version` | Print the FastSync protocol version. |
|
||||
| `--help` | Print command usage. |
|
||||
@@ -460,33 +648,116 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--ssh-port <port>` | SSH port for the SSH transport (default: 22). Note the short `-p` is now rsync's `--perms`. |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode. |
|
||||
| `-e`, `--rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments). |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode (client-only; never crosses the wire). |
|
||||
| `--rsync-path <path>` | Alias for `--fastsync-server-path`. |
|
||||
| `-M`, `--remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable; rejected for daemon/TCP destinations). |
|
||||
| `--trust-sender` | Receiver-local: trust the remote sender's file list and skip path re-validation (does not affect symlink targets). |
|
||||
| `--timeout <sec>` | Socket + per-message I/O timeout; default `0` = disabled. |
|
||||
| `--contimeout <sec>` | Connection timeout; default 60; `0` disables. |
|
||||
| `--source-dir <path>` | Set the source directory explicitly. |
|
||||
| `--dest-dir <path>` | Set the destination directory explicitly. |
|
||||
| `--save-to-disk` | Enable server-side disk persistence. |
|
||||
| `--server-host <host>` | TCP server address. |
|
||||
| `--server-port <port>` | TCP server port. |
|
||||
| `--tls` | Enable TLS. Requires `--cert` and `--key`. |
|
||||
| `--server-port <port>` | TCP server port. `--port <port>` / `--port=<port>` is an alias. |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address. |
|
||||
| `-4`, `--ipv4` | Force IPv4 for destination resolution. |
|
||||
| `-6`, `--ipv6` | Force IPv6 for destination resolution. |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect. |
|
||||
| `--tls` | Enable TLS. Requires `--cert`, `--key`, and `--ca`. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--ca <path>` | CA file for peer verification (always required with `--tls`). |
|
||||
|
||||
## Server Options
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--stdio` | Serve one SSH connection over standard input/output. |
|
||||
| `-p <port>` | TCP listen port. |
|
||||
| `--daemon` | Run as a persistent daemon listener using a module config file; the daemon default port is 873 (unlike `-p`, which defaults to 8080). |
|
||||
| `--config=FILE` | Daemon config file (default: `~/.config/fastsync/fastsyncd.conf`, else `/etc/fastsyncd.conf`). Requires `--daemon`. |
|
||||
| `--dparam=KEY=VALUE` | Override one global config key on the command line. Requires `--daemon`. |
|
||||
| `--no-detach` | Stay in the foreground (default detaches to the background when running `--daemon`). |
|
||||
| `-p, --port <port>` | TCP listen port (default: 8080, range: 1–65535). |
|
||||
| `--tls` | Enable TLS. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root;
|
||||
defaults to the current directory. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. |
|
||||
| `--cert <path>` | TLS certificate file (PEM). |
|
||||
| `--key <path>` | TLS private key file (PEM). |
|
||||
| `--ca <path>` | CA file for peer verification (PEM). |
|
||||
| `--client-cn <name>` | TLS client certificate CN; mandatory with `--tls` (the server verifies the client CN). |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root; defaults to the current directory. |
|
||||
| `--address <addr>` | Bind the listening socket to this address. |
|
||||
| `-4`, `--ipv4` | Bind an IPv4 socket (default). |
|
||||
| `-6`, `--ipv6` | Bind an IPv6 socket. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. This also gates `--force` (which can recursively replace/remove a destination directory tree). |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. Rejected with `--stdio` (the SSH remote argv is client-composed; use a forced command if the default must hold). No effect when not root. Daemon modules opt in per module with `client owner = yes`. |
|
||||
| `--trust-sender` | Trust the remote sender's file list: skip the receiver's up-front path-traversal re-validation (fewer checks, faster, potentially unsafe; off by default). It does not affect symlink targets, which are stored verbatim either way. |
|
||||
| `--no-super` | Operator veto: never attempt super-user activities (ownership, device nodes) even as root, and refuse any client `--copy-as`/`--super` request. |
|
||||
| `--allow-unauthenticated` | Permit plaintext/anonymous network clients; an auth-required module still accepts only opted-in loopback plaintext. |
|
||||
| `--iconv=LOCAL[,REMOTE]` | Declare this server's LOCAL charset for file-name conversion. |
|
||||
| `--password-file=FILE` | Credential store for modules that declare `auth users`. Requires `--daemon`. |
|
||||
| `--early-input=FILE` | Second credential store layered over `--password-file`. Requires `--daemon`. |
|
||||
| `--hash-credentials <file>` | Read `<file>`'s `user:password` lines and print PBKDF2 credential-store lines to stdout, then exit. Cannot be combined with `--daemon` or `--stdio`. |
|
||||
| `--iterations N` | PBKDF2 iteration count for `--hash-credentials` (default 600000, range 100000–10000000). Requires `--hash-credentials`. |
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--help` | Print server usage. |
|
||||
|
||||
### Daemon configuration
|
||||
|
||||
`fastsync-server --daemon --config FILE` reads a line-based module config (an
|
||||
implicit global section, then `[module]` sections). Besides `port`, `motd file`,
|
||||
and `address`, the global section accepts:
|
||||
|
||||
- `max connections = N` — global cap on concurrent connections, default 100. The
|
||||
listener enforces it; `0`, negative, and non-numeric values are parse errors.
|
||||
- `max connections per host = N` — cap on concurrent connections from a single
|
||||
source IP, default 0 (unlimited). Enforced across all forked connection
|
||||
children through a shared registry.
|
||||
- `auth failure delay = MS` — milliseconds to sleep after a failed
|
||||
authentication, default 500. `0` disables it and the value is capped at 5000,
|
||||
so online password guessing is rate-limited per connection. Successful auths
|
||||
are never delayed.
|
||||
- `auth lockout threshold = N` — number of failed authentications from one source
|
||||
IP before that source is locked out, default 10; `0` disables the lockout. The
|
||||
failure counter is shared across every connection child, so the lockout holds
|
||||
even when the next attempt is handled by a different forked child.
|
||||
- `auth lockout duration = SECONDS` — how long a locked-out source is refused
|
||||
(default 300). A locked-out client is refused before any SCRAM challenge is
|
||||
sent; a successful authentication clears the counter.
|
||||
- `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access
|
||||
patterns.
|
||||
|
||||
A `[module]` requires `path`, and may also set `read only`, `client owner`,
|
||||
`auth users`, `max connections` (0 = unlimited; enforced per module across all
|
||||
connection children), and its own `hosts allow`/`hosts deny`.
|
||||
|
||||
The per-host cap and the shared auth lockout identify a source by its numeric
|
||||
peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local
|
||||
client shares that one address, so counting or locking them out would let one
|
||||
local process deny service to all the others. The per-module and global
|
||||
`max connections` caps still apply to loopback. Because the key is the peer IP,
|
||||
`max connections per host` and `auth lockout` also cannot distinguish clients
|
||||
behind the same NAT, proxy, or reverse-proxy address — they share one budget and
|
||||
one lockout counter, so an over-aggressive lockout can affect unrelated users
|
||||
behind that address. Prefer TLS client certificates (`--client-cn`) plus
|
||||
`hosts allow`/`hosts deny` for per-client policy when clients share an address,
|
||||
and size `auth lockout threshold` accordingly.
|
||||
|
||||
The shared per-source table has a bounded lifetime: an entry with no live
|
||||
connection is reclaimed once its lockout has expired, or after it has been idle
|
||||
(300 s). If every entry is still live or locked, a new source is admitted without
|
||||
per-host accounting (fail open) and a rate-limited warning is logged; the
|
||||
per-module cap and host ACLs still apply. The occupancy counters are re-derived
|
||||
from the shared slot table after every child exit, so a child killed mid-transfer
|
||||
(or mid-registration) cannot leak a slot or an occupancy count.
|
||||
|
||||
Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR
|
||||
(`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs
|
||||
are rejected at parse time rather than silently never matching. A matching
|
||||
`hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of
|
||||
them is rejected; deny takes precedence over allow. The global list is checked
|
||||
before the module list, before authentication, and the connecting peer address
|
||||
(IPv4 or IPv6) appears in the connection and authentication audit log lines.
|
||||
|
||||
## Architecture
|
||||
|
||||
### Client
|
||||
@@ -511,7 +782,7 @@ defaults to the current directory. |
|
||||
|
||||
## Protocol and Security
|
||||
|
||||
FastSync protocol version `2.19.0` is shared by the client and server. The
|
||||
FastSync protocol version `2.23.0` is shared by the client and server. The
|
||||
current protocol is sender-driven and includes configuration negotiation,
|
||||
including the maximum allocation limit, incremental checks, checksums,
|
||||
manifests, keep-alives, abort handling, per-file remove-source results, and
|
||||
@@ -563,10 +834,9 @@ mandates `--client-cn`, so a TLS connection to an auth-required module always
|
||||
has its client CN verified (`--client-cn` matches the certificate's CN only, not
|
||||
a subjectAltName, which is acceptable for a private CA).
|
||||
|
||||
TLS provides encrypted TCP transport. Supplying `--ca` enables certificate
|
||||
verification; without it, traffic is encrypted but peer identity is not
|
||||
verified. Use certificate verification for deployments where authentication
|
||||
matters. The default TCP transport is not encrypted.
|
||||
TLS provides encrypted TCP transport. Both the client and the server require
|
||||
`--ca` together with `--tls`, so peer certificates are always verified
|
||||
(`SSL_VERIFY_PEER`, depth 4). The default TCP transport is not encrypted.
|
||||
|
||||
The receiver protects its destination root with path validation, `openat()`
|
||||
directory traversal, `O_NOFOLLOW`, temporary files, and atomic renames. Delete
|
||||
@@ -577,18 +847,24 @@ operations require the server's explicit `--allow-delete` policy.
|
||||
The project will reach the drop-in replacement goal in stages:
|
||||
|
||||
1. Correct rsync option meanings, including short options, combined options,
|
||||
and `--option=value` syntax.
|
||||
and `--option=value` syntax — **done** in the rsync-parity wave: `-r`/`-b`/
|
||||
`-L`/`-B`, short-option clustering (`-av`, `-aAX`, `-rlpt`), and attached
|
||||
values (`-B1000`, `-essh`, `-MOPT`) all parse.
|
||||
2. Add differential tests that compare FastSync and rsync contents, metadata,
|
||||
links, deletes, filters, dry runs, and exit codes.
|
||||
3. Make `-a` implement the expected recursive, links, permissions, times,
|
||||
owner/group, and supported special-file behavior.
|
||||
4. Complete symlink, sparse-file, metadata, delete-policy, and resumable-write
|
||||
semantics.
|
||||
3. `-a` implements full rsync `-rlptgoD`; under `-p` the source mode is copied
|
||||
exactly (no masking). Ownership application stays privilege-gated, as in
|
||||
rsync.
|
||||
4. Symlink (verbatim storage), sparse-file, metadata, delete-policy (including
|
||||
`--max-delete` partial + exit 25), and resumable-write semantics are
|
||||
implemented; remaining work is the documented edge cases, which the
|
||||
**Rsync-Parity Wave** section of `RSYNC_COMPAT.md` enumerates honestly.
|
||||
5. Add rsync remote-shell and daemon protocol interoperability.
|
||||
6. Keep FastSync performance options as negotiated, optional extensions.
|
||||
|
||||
The exhaustive implementation matrix and compatibility notes are in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md); each row is classified as parity, caveat,
|
||||
or divergent.
|
||||
|
||||
## Testing
|
||||
|
||||
@@ -601,7 +877,7 @@ Run the unit test binary:
|
||||
Run the Python integration suite:
|
||||
|
||||
```bash
|
||||
python3 -m pytest tests/
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
For stricter local validation:
|
||||
@@ -625,10 +901,13 @@ rsync protocol or filesystem-semantic compatibility.
|
||||
|
||||
## Performance Guidance
|
||||
|
||||
- Use `-m` for workloads with many files or enough CPU parallelism.
|
||||
- Use `-c` or `-z` when network bandwidth is more constrained than CPU.
|
||||
- Use `-j`/`--threads` for workloads with many files or enough CPU parallelism
|
||||
(`-m` is `--prune-empty-dirs`).
|
||||
- Use `-z` when network bandwidth is more constrained than CPU (`-c` is
|
||||
`--checksum`, not a bandwidth option).
|
||||
- Tune `--chunk-size` for file sizes, memory limits, and network latency.
|
||||
- Use `-f` for large uncompressed TCP transfers where zero-copy I/O helps.
|
||||
- Use `--sendfile` for large uncompressed TCP transfers where zero-copy I/O
|
||||
helps (`-f` is `--filter`).
|
||||
- Use `--incremental` to avoid retransmitting unchanged files.
|
||||
- Use `--delta` for changed files when both endpoints are FastSync peers.
|
||||
- Use `--bwlimit` when sharing a link with other traffic.
|
||||
|
||||
+458
-268
@@ -6,13 +6,34 @@ This document maps rsync's full feature set to FastSync's current implementation
|
||||
|
||||
| Status | Count | Description |
|
||||
|--------|-------|-------------|
|
||||
| ✅ Implemented | 143 | Feature works end-to-end |
|
||||
| 🔀 Alt Arg | 0 | Functionality exists but under different flag/semantics |
|
||||
| ⛔ Impossible/Divergence | 4 | Flag is a documented divergence or cannot be implemented on any portable filesystem call |
|
||||
| ⚠️ Partial | 0 | Flag parsed/stored but behavior incomplete |
|
||||
| 🔄 Compatibility No-op | 0 | Flag is accepted for CLI compatibility but has no effect |
|
||||
| ❌ Not Implemented | 0 | Flag not recognized or no behavior |
|
||||
| **Total** | **147** | |
|
||||
| ✅ Parity | 83 | Reproduces rsync's semantics for this option's scope |
|
||||
| ⚠️ Caveat | 63 | Fully wired and tested, but carries a documented behavioral difference from rsync (named in the row and/or the wave notes) |
|
||||
| ❌ Divergent | 4 | Rejected, an accepted no-op, or impossible on any portable filesystem call |
|
||||
| **Total** | **150** | One row per rsync option/feature group; a row may name several spellings |
|
||||
|
||||
This matrix reports honest rsync parity, not "implemented" as a synonym for
|
||||
"parsed". A ✅ row matches rsync for the option's scope. A ⚠️ row is real and
|
||||
tested but diverges in at least one documented way — FastSync's push-only model,
|
||||
its own wire protocol, the delete timings that approximate rsync's engine modes,
|
||||
the safe-subset privilege model (`--super`/`--copy-as`), the stricter
|
||||
xattr/ACL and temp-dir policies, and the output counters that rsync computes on
|
||||
the generator side. An ❌ row is either rejected (`--stderr=client`, `--protocol`
|
||||
with any value but the current one), an accepted no-op (`-s`/`--secluded-args`),
|
||||
or impossible (`-N`/`--crtimes`). The counts are derived from the rows below;
|
||||
update them together with the table.
|
||||
|
||||
**Recently closed parity gaps (protocol 2.23.0).** The rsync-parity wave wired up
|
||||
the short options `-r`, `-b`, `-L`, `-B`; rsync short-option clustering
|
||||
(`-av`, `-aAX`, `-rlpt`) and attached/inline values (`--opt=value`, `-B1000`,
|
||||
`-essh`, `-MOPT`); `-c` now implies the checksum quick-check; `--checksum-choice`
|
||||
accepts `xxh64`/`xxhash`/`xxh3`/`xxh128`/`md5`/`auto` and rejects `md4`/`sha1`/
|
||||
`none` by name; `--compress-choice` accepts `zstd`/`none`/`auto`; `--checksum-seed=0`
|
||||
is randomized per transfer; `--skip-compress` uses rsync's default suffix list;
|
||||
`--timeout`/`--contimeout` match rsync's defaults; deletion gained
|
||||
`--max-delete` partial semantics with exit 25; symlinks are stored verbatim; and
|
||||
`--specials` recreates sockets. Every one of those still has an entry below with
|
||||
its remaining caveats. See the **Rsync-Parity Wave (protocol 2.23.0)** section
|
||||
near the end for the full list and the known limitations.
|
||||
|
||||
---
|
||||
|
||||
@@ -20,100 +41,100 @@ This document maps rsync's full feature set to FastSync's current implementation
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-a`, `--archive` | Archive mode is -rlptgoD | ✅ Implemented | Phase 7 Wave A: real rsync archive. `-a`/`--archive` now implies `--links` + metadata (perms/times/group/owner as FastSync's broad bundle) + `--devices` + `--specials`. FastSync is always recursive, so no `-r` is needed. It no longer implies compression or multithreading (those moved to `-z`/`-j`). The short-option namespace is now rsync-parity (see the Phase 7 note) |
|
||||
| `-v`, `--verbose` | Increase verbosity | ✅ Implemented | Sets `log_level=DEBUG` |
|
||||
| `-q`, `--quiet` | Suppress non-error messages | ✅ Implemented | Suppresses client output while preserving errors |
|
||||
| `--help` | Show help | ✅ Implemented | Prints usage and exits; `-h` is not accepted |
|
||||
| `-V`, `--version` | Print version | ✅ Implemented | |
|
||||
| `--info=FLAGS` | Fine-grained info verbosity | ✅ Implemented | Supports `copy`, `misc`, `skip`, `stats`, `all`, and `none`; explicit flags override `--verbose`, and `none` suppresses info output; unsupported names are rejected |
|
||||
| `--debug=FLAGS` | Fine-grained debug verbosity | ✅ Implemented | `io`, `proto`, `pack`, and `util` are supported; `--debug=help` lists flags; other rsync categories are rejected |
|
||||
| `--stderr=MODE` | Change stderr output mode | ⛔ Impossible/Divergence | `errors` (default) and `all` are supported; `client` is rejected with a clear error (`--stderr=client is not supported`) because FastSync has no rsync client-message channel — the rejection itself is the documented behavior (Phase 7 Wave B decision). The modes that exist work; the missing rsync channel cannot be emulated without a wire change |
|
||||
| `--no-motd` | Suppress daemon MOTD | ✅ Implemented | Client-only display switch (Wave C): the daemon still sends the configured `motd file` on a `host::module/path` connection; the client reads and discards the frame without showing it. Without the flag the MOTD is printed to stdout after the config/auth handshake and escaped so control bytes cannot inject terminal sequences |
|
||||
| `--exclude=PATTERN` | Exclude files matching pattern | ✅ Implemented | Glob matching in scanner |
|
||||
| `--include=PATTERN` | Include files matching pattern | ✅ Implemented | Glob matching in scanner |
|
||||
| `-C`, `--cvs-exclude` | Auto-ignore CVS files | ✅ Implemented | Applies the well-known rsync default exclude set as exclude rules during scanning (RCS SCCS CVS CVS.adm RCSLOG cvslog.* tags TAGS .make.state .nse_depinfo *~ #* .#* ,* _$* *$ *.old *.bak *.BAK *.orig *.rej .del-* *.a *.olb *.o *.obj *.so *.exe *.Z *.elc *.ln core .svn/ .git/ .hg/ .bzr/); `.git/`-style repo dirs are pruned without descending |
|
||||
| `-a`, `--archive` | Archive mode is -rlptgoD (rsync includes owner/group) | ✅ Parity | Phase 7 Wave A: real rsync archive. `-a`/`--archive` now implies `--links` + the four per-attribute preserve flags (perms/times/owner/group) + `--devices` + `--specials`, i.e. **`-rlptgoD`**. Owner/group **are** implied, but their application stays privilege-gated exactly like rsync: a receiver that cannot `chown` logs a warning and skips it (see the preserve-attribute split note below). FastSync is always recursive, so no `-r` is needed. It no longer implies compression or multithreading (those moved to `-z`/`-j`). The short-option namespace is now rsync-parity (see the Phase 7 note) |
|
||||
| `-v`, `--verbose` | Increase verbosity | ✅ Parity | Sets `log_level=DEBUG` |
|
||||
| `-q`, `--quiet` | Suppress non-error messages | ✅ Parity | Suppresses client output while preserving errors |
|
||||
| `--help` | Show help | ✅ Parity | Prints usage and exits; `-h` is not accepted |
|
||||
| `-V`, `--version` | Print version | ✅ Parity | |
|
||||
| `--info=FLAGS` | Fine-grained info verbosity | ⚠️ Caveat | Supports `copy`, `misc`, `skip`, `stats`, `all`, and `none`; explicit flags override `--verbose`, and `none` suppresses info output; unsupported names are rejected |
|
||||
| `--debug=FLAGS` | Fine-grained debug verbosity | ⚠️ Caveat | `io`, `proto`, `pack`, and `util` are supported; `--debug=help` lists flags; other rsync categories are rejected |
|
||||
| `--stderr=MODE` | Change stderr output mode | ❌ Divergent | `errors` (default) and `all` are supported; `client` is rejected with a clear error (`--stderr=client is not supported`) because FastSync has no rsync client-message channel — the rejection itself is the documented behavior (Phase 7 Wave B decision). The modes that exist work; the missing rsync channel cannot be emulated without a wire change |
|
||||
| `--no-motd` | Suppress daemon MOTD | ✅ Parity | Client-only display switch (Wave C): the daemon still sends the configured `motd file` on a `host::module/path` connection; the client reads and discards the frame without showing it. Without the flag the MOTD is printed to stdout after the config/auth handshake and escaped so control bytes cannot inject terminal sequences |
|
||||
| `--exclude=PATTERN` | Exclude files matching pattern | ✅ Parity | Glob matching in scanner |
|
||||
| `--include=PATTERN` | Include files matching pattern | ✅ Parity | Glob matching in scanner |
|
||||
| `-C`, `--cvs-exclude` | Auto-ignore CVS files | ✅ Parity | Applies the well-known rsync default exclude set as exclude rules during scanning (RCS SCCS CVS CVS.adm RCSLOG cvslog.* tags TAGS .make.state .nse_depinfo *~ #* .#* ,* _$* *$ *.old *.bak *.BAK *.orig *.rej .del-* *.a *.olb *.o *.obj *.so *.exe *.Z *.elc *.ln core .svn/ .git/ .hg/ .bzr/); `.git/`-style repo dirs are pruned without descending |
|
||||
|
||||
## 2. Modifying Output
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `--stats` | Give transfer stats | ✅ Implemented | Prints file/byte counts |
|
||||
| `-h`, `--human-readable` | Human-readable numbers | ✅ Implemented | Formats transfer byte sizes using binary units |
|
||||
| `-i`, `--itemize-changes` | Per-file change summary | ✅ Implemented | Prints rsync-style `>f+++++++++` lines to stdout only for files actually sent (also under `-j`/`--threads`); unchanged files print nothing, matching single-`-i` behavior |
|
||||
| `--progress` | Show progress | ✅ Implemented | Progress callback in sender |
|
||||
| `-P` | Same as --partial --progress | ✅ Implemented | Phase 7 Wave B: `-P` parses to `--partial` + `--progress`. On a failed/interrupted write the receiver now retains the already-written temp file at the destination path (best-effort rename instead of unlink when configured), so a later `--append`/`--append-verify` run can resume it; `--partial-dir` still stages completed files under the confined partial dir and installs them atomically. The retention never runs when `--partial` is off, when no data was actually written, or under `--ignore-existing`/`--existing` (the destination is not ours to overwrite), and it only ever renames the already-written temp (never a corrupt blend; a failed rename falls back to the normal unlink). See the `-S`/`--sparse` interplay note (a retained sparse temp has full logical size) |
|
||||
| `--out-format=FORMAT` | Custom output format | ✅ Implemented | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%M` `%%` (`%b` is the source length, always `== %l`; post-compression/delta wire bytes are not counted); unknown escapes preserved |
|
||||
| `--log-file=FILE` | Log to file | ✅ Implemented | `log_file` config field |
|
||||
| `--log-file-format=FMT` | Log format | ✅ Implemented | Requires `--log-file`; writes one template line per transferred file using the same token set as `--out-format` (including `%b` `==` source length) |
|
||||
| `--8-bit-output`, `-8` | Leave high-bit chars unescaped | ✅ Implemented | Applies to displayed paths and protocol debug output |
|
||||
| `--list-only` | List files instead of copying | ✅ Implemented | `ls -l`-style listing of files that would be transferred; scans the source only, contacts no server, writes nothing; also works with `-n` |
|
||||
| `--stats` | Give transfer stats | ⚠️ Caveat | Prints file/byte counts. **Divergence:** the receiver-only counters rsync derives during its generator pass (matched/unchanged data, file-list bytes, deleted-entry count) are reported as **0** by FastSync, and the byte total counts source bytes actually sent rather than the post-delta/post-compression wire volume. Counts that FastSync can observe locally (files, bytes, timing) are accurate |
|
||||
| `-h`, `--human-readable` | Human-readable numbers | ✅ Parity | Formats transfer byte and rate counts using rsync's **decimal** (base-1000) units, matching rsync `-h` (e.g. `1.23M`), not binary units |
|
||||
| `-i`, `--itemize-changes` | Per-file change summary | ✅ Parity | Prints rsync-style `>f+++++++++` lines to stdout only for files actually sent (also under `-j`/`--threads`); unchanged files print nothing, matching single-`-i` behavior |
|
||||
| `--progress` | Show progress | ⚠️ Caveat | Prints a periodic **aggregate** transfer line (bytes sent and current rate), not rsync's per-file progress block. With `-P` the partial-file retention behavior is fully implemented; only the progress presentation differs |
|
||||
| `-P` | Same as --partial --progress | ⚠️ Caveat | Phase 7 Wave B: `-P` parses to `--partial` + `--progress`. On a failed/interrupted write the receiver now retains the already-written temp file at the destination path (best-effort rename instead of unlink when configured), so a later `--append`/`--append-verify` run can resume it; `--partial-dir` still stages completed files under the confined partial dir and installs them atomically. The retention never runs when `--partial` is off, when no data was actually written, or under `--ignore-existing`/`--existing` (the destination is not ours to overwrite), and it only ever renames the already-written temp (never a corrupt blend; a failed rename falls back to the normal unlink). See the `-S`/`--sparse` interplay note (a retained sparse temp has full logical size) |
|
||||
| `--out-format=FORMAT` | Custom output format | ⚠️ Caveat | Per-transfer template on stdout; tokens `%f` `%n` `%l` `%b` `%M` `%%` (`%b` is the source length, always `== %l`; post-compression/delta wire bytes are not counted); unknown escapes preserved |
|
||||
| `--log-file=FILE` | Log to file | ✅ Parity | `log_file` config field |
|
||||
| `--log-file-format=FMT` | Log format | ✅ Parity | Requires `--log-file`; writes one template line per transferred file using the same token set as `--out-format` (including `%b` `==` source length) |
|
||||
| `--8-bit-output`, `-8` | Leave high-bit chars unescaped | ✅ Parity | Applies to displayed paths and protocol debug output |
|
||||
| `--list-only` | List files instead of copying | ✅ Parity | `ls -l`-style listing of files that would be transferred; scans the source only, contacts no server, writes nothing; also works with `-n` |
|
||||
|
||||
## 3. File Selection
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `--exclude-from=FILE` | Read exclude patterns from file | ✅ Implemented | Reads patterns from file |
|
||||
| `--include-from=FILE` | Read include patterns from file | ✅ Implemented | Reads patterns from file |
|
||||
| `--filter=RULE` | Add file-filtering rule | ✅ Implemented | Long option only: rsync's short `-f` conflicts with FastSync sendfile (see FastSync-specific list), so `-f` is not reassigned. Supported subset: `+`/`-` include/exclude, implicit-exclude patterns, `include`/`exclude` word forms, a leading `/` anchor (to the transfer root, or to a `.rsync-filter` file's directory), and a trailing `/` for dir-only rules; first match wins with a default of include inside the filter layer. Filters are an independent layer from `--exclude`/`--include` (an entry must pass both). Rejected with a clear error (no silent no-ops): `merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` words, rules that begin with `:`/`.`/`!` (merge/dir-merge/list-clear shorthands), and include/exclude modifiers other than `/` (`! C s r p x`) |
|
||||
| `--files-from=FILE` | Read source file list from file | ✅ Implemented | Entries are paths relative to the source root (leading `./` stripped, `..`/absolute entries rejected at parse time, blank lines ignored; NUL-delimited with `-0`). A listed regular file is transferred; a listed directory transfers its whole subtree (FastSync recursion is always on, unlike rsync's non-recursive default). Non-listed paths and their subtrees are pruned by the scanner. A listed entry that does not exist under the source (and an empty list) is a hard error reported before any transfer, unless `--ignore-missing-args` / `--delete-missing-args` is given (see the Safety & Security rows): those flags downgrade the listed-but-missing case to a skip and, for `--delete-missing-args`, a destination deletion; an empty list stays a hard error in every mode. Listing `.` (whole tree) and empty listed directories are fine. Scalability note: `file_list_affects` is O(list size) per scanned entry, so a very large `--files-from` list against a huge tree is quadratic; lists are typically small enough that this is acceptable, but it is the documented bound. The delete manifest still derives from what was actually sent, so `--delete` stays consistent with the subset |
|
||||
| `-0`, `--from0` | Delimit *-from files with NULs | ✅ Implemented | `--files-from` entries become NUL-delimited; the flag may appear before or after `--files-from` on the command line. NUL mode preserves entry bytes exactly (trailing CR/LF are part of the name; only newline mode trims them) |
|
||||
| `--max-size=SIZE` | Skip files larger than SIZE | ✅ Implemented | `max_size` in scanner |
|
||||
| `--min-size=SIZE` | Skip files smaller than SIZE | ✅ Implemented | `min_size` in scanner |
|
||||
| `-I`, `--ignore-times` | Don't skip files matching size+time | ✅ Implemented | `ignore_times` config field (crosses the wire). Disables the size+mtime quick-check in the `--incremental` per-file handshake and the basis-dir quick-match, forcing the file to be transferred rather than skipped as unchanged. Receiver-side policy: `match_by_metadata` (file_receive.c) is bypassed, so the receiver never replies `STATUS_OK` for a matching size+mtime. Requires `--incremental` to have the handshake to act on (rsync does its quick check by default; FastSync's `-I`/`--size-only`/`--modify-window` only take effect under `--incremental`, exactly like they take effect through the basis check) |
|
||||
| `--size-only` | Skip based on size only | ✅ Implemented | With `--incremental`, ignores mtime |
|
||||
| `-@`, `--modify-window=NUM` | Mod-time comparison accuracy | ✅ Implemented | Whole-second tolerance with nanosecond-aware comparisons |
|
||||
| `--existing` | Skip creating new files on receiver | ✅ Implemented | Existing destination files continue through normal update handling |
|
||||
| `--ignore-existing` | Skip updating existing files | ✅ Implemented | `ignore_existing` config field (crosses the wire; receiver-side policy). For a destination entry that already exists, the receiver skips the write: in the regular-file path, existing/delay-updates-staged, hardlink-sibling, and special/device handlers all return `FILE_SAVE_SKIPPED` without overwriting (passed as `no_replace` to the write engine), and `--backup` is disabled for skipped files. Note: it is applied at write time, so an existing dest whose size+mtime differ still has its data (or delta) transmitted before the write is discarded — functionally correct, bandwidth-suboptimal vs rsync, which short-circuits earlier. Like rsync, it does not apply to directories/symlinks (those return before the block). Combines with `-j`/`--threads` and `--delay-updates`. See Phase-4/— notes below |
|
||||
| `--remove-source-files` | Sender removes regular files after confirmed transfer | ✅ Implemented | |
|
||||
| `-x`, `--one-file-system` | Do not cross filesystem boundaries | ✅ Implemented | Sender scanner captures the root device and skips descending into mount-point crossings (`st_dev` differs); cross-filesystem mount-point subdirectories are dropped entirely, matching rsync |
|
||||
| `-F` | Add the default `.rsync-filter` rules | ✅ Implemented | Reads one filter rule per line from each directory's `.rsync-filter` file during traversal and applies it to that directory's subtree; the current directory's rules are evaluated before its ancestors', so deeper files override shallower ones and per-directory files override the command-line `--filter`/`-C` base by default (matching rsync's first-match-wins precedence); `.rsync-filter` files are never transferred. The rsync `-FF` behavior (also `.cvsignore`) is out of scope; unsupported rule types inside the file abort with a clear error |
|
||||
| `--exclude-from=FILE` | Read exclude patterns from file | ✅ Parity | Reads patterns from file |
|
||||
| `--include-from=FILE` | Read include patterns from file | ✅ Parity | Reads patterns from file |
|
||||
| `--filter=RULE` | Add file-filtering rule | ⚠️ Caveat | Long option only: rsync's short `-f` conflicts with FastSync sendfile (see FastSync-specific list), so `-f` is not reassigned. Supported subset: `+`/`-` include/exclude, implicit-exclude patterns, `include`/`exclude` word forms, a leading `/` anchor (to the transfer root, or to a `.rsync-filter` file's directory), and a trailing `/` for dir-only rules; first match wins with a default of include inside the filter layer. Filters are an independent layer from `--exclude`/`--include` (an entry must pass both). Rejected with a clear error (no silent no-ops): `merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` words, rules that begin with `:`/`.`/`!` (merge/dir-merge/list-clear shorthands), and include/exclude modifiers other than `/` (`! C s r p x`) |
|
||||
| `--files-from=FILE` | Read source file list from file | ⚠️ Caveat | Entries are paths relative to the source root (leading `./` stripped, `..`/absolute entries rejected at parse time, blank lines ignored; NUL-delimited with `-0`). A listed regular file is transferred; a listed directory transfers its whole subtree (FastSync recursion is always on, unlike rsync's non-recursive default). Non-listed paths and their subtrees are pruned by the scanner. A listed entry that does not exist under the source (and an empty list) is a hard error reported before any transfer, unless `--ignore-missing-args` / `--delete-missing-args` is given (see the Safety & Security rows): those flags downgrade the listed-but-missing case to a skip and, for `--delete-missing-args`, a destination deletion; an empty list stays a hard error in every mode. Listing `.` (whole tree) and empty listed directories are fine. Scalability note: `file_list_affects` is O(list size) per scanned entry, so a very large `--files-from` list against a huge tree is quadratic; lists are typically small enough that this is acceptable, but it is the documented bound. Delete scoping (protocol 2.23.0): the manifest carries the set of synchronized directories, and the extras walk only visits those subtrees, so `--delete` with a `--files-from` subset no longer removes destination paths outside the listed directory subtrees (a data-loss fix matching rsync) |
|
||||
| `-0`, `--from0` | Delimit *-from files with NULs | ✅ Parity | `--files-from` entries become NUL-delimited; the flag may appear before or after `--files-from` on the command line. NUL mode preserves entry bytes exactly (trailing CR/LF are part of the name; only newline mode trims them) |
|
||||
| `--max-size=SIZE` | Skip files larger than SIZE | ✅ Parity | `max_size` in scanner |
|
||||
| `--min-size=SIZE` | Skip files smaller than SIZE | ✅ Parity | `min_size` in scanner |
|
||||
| `-I`, `--ignore-times` | Don't skip files matching size+time | ✅ Parity | `ignore_times` config field (crosses the wire). Disables the size+mtime quick-check in the `--incremental` per-file handshake and the basis-dir quick-match, forcing the file to be transferred rather than skipped as unchanged. Receiver-side policy: `match_by_metadata` (file_receive.c) is bypassed, so the receiver never replies `STATUS_OK` for a matching size+mtime. Requires `--incremental` to have the handshake to act on (rsync does its quick check by default; FastSync's `-I`/`--size-only`/`--modify-window` only take effect under `--incremental`, exactly like they take effect through the basis check) |
|
||||
| `--size-only` | Skip based on size only | ✅ Parity | With `--incremental`, ignores mtime |
|
||||
| `-@`, `--modify-window=NUM` | Mod-time comparison accuracy | ✅ Parity | Whole-second tolerance with nanosecond-aware comparisons |
|
||||
| `--existing` | Skip creating new files on receiver | ✅ Parity | Existing destination files continue through normal update handling |
|
||||
| `--ignore-existing` | Skip updating existing files | ⚠️ Caveat | `ignore_existing` config field (crosses the wire; receiver-side policy). For a destination entry that already exists, the receiver skips the write: in the regular-file path, existing/delay-updates-staged, hardlink-sibling, and special/device handlers all return `FILE_SAVE_SKIPPED` without overwriting (passed as `no_replace` to the write engine), and `--backup` is disabled for skipped files. Note: it is applied at write time, so an existing dest whose size+mtime differ still has its data (or delta) transmitted before the write is discarded — functionally correct, bandwidth-suboptimal vs rsync, which short-circuits earlier. Like rsync, it does not apply to directories/symlinks (those return before the block). Combines with `-j`/`--threads` and `--delay-updates`. See Phase-4/— notes below |
|
||||
| `--remove-source-files` | Sender removes regular files after confirmed transfer | ✅ Parity | |
|
||||
| `-x`, `--one-file-system` | Do not cross filesystem boundaries | ✅ Parity | Sender scanner captures the root device and does not descend into mount-point crossings (`st_dev` differs). **Protocol 2.23.0 matches rsync's entry emission:** the mount-point directory itself is emitted as a payload-less directory entry (so the destination gets an empty directory) while its contents are skipped; previously the crossing subdirectory was dropped entirely |
|
||||
| `-F` | Add the default `.rsync-filter` rules | ⚠️ Caveat | Reads one filter rule per line from each directory's `.rsync-filter` file during traversal and applies it to that directory's subtree; the current directory's rules are evaluated before its ancestors', so deeper files override shallower ones and per-directory files override the command-line `--filter`/`-C` base by default (matching rsync's first-match-wins precedence); `.rsync-filter` files are never transferred. The rsync `-FF` behavior (also `.cvsignore`) is out of scope; unsupported rule types inside the file abort with a clear error |
|
||||
|
||||
## 4. Directory Options
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-r`, `--recursive` | Recurse into directories | ✅ Implemented | Default behavior |
|
||||
| `-R`, `--relative` | Use relative path names | ✅ Implemented | Meaningful together with `--files-from` (FastSync's default full-tree scan always mirrors the full source argument path below the destination root, so -R does not change it). With `-R` + `--files-from` each listed entry is transmitted under its bare relative destination path: an entry `sub/x.txt` lands at `<dest>/sub/x.txt` (its leading components preserved) instead of under the `<dest>/<full source path>` mirror. Only the path sent on the wire changes; the client still reads the absolute source path, and the delete manifest derives from the sent (relative) paths so `--delete` and `--remove-source-files` stay consistent in both layouts. Works single-threaded and under `-j`/`--threads` (including chunk serialization) |
|
||||
| `--no-implied-dirs` | Don't send implied dirs with -R | ✅ Implemented | Client-side, meaningful only with `-R` + `--files-from`. rsync would normally create the ancestor directories implied by a listed file so it can be written; with `--no-implied-dirs` a listed file whose parent directory is not itself (or via an ancestor) explicitly listed cannot be placed, and FastSync fails the whole run up front with a clear error (`--no-implied-dirs: cannot place file '...': parent directory '...' is not explicitly listed`). Listing the directory (or an ancestor of it, or the whole tree `.`) permits the file. In every other mode the option has no effect. FastSync has no per-entry skip channel, so the rsync "omit the file" case is surfaced as a hard pre-transfer error |
|
||||
| `-d`, `--dirs`, `--old-dirs`, `--old-d` | Transfer dirs without recursing | ✅ Implemented | `-d <dir>` transmits an explicit directory entry for the source-root directory, so the destination mirror is created empty and nothing is descended into. With `--files-from` exactly the listed items are transferred: a listed directory is created empty (no descent) and a listed file is transferred with its content; the dest layout follows the same -R rules as plain files. A new wire frame (`STATUS_MKDIR`) carries each directory entry — the path and, when `--preserve`/`-a` (metadata mode) is negotiated, the directory's metadata; the receiver creates it with the same confined mkdir-parent semantics as regular writes, in single-threaded and `-j`/`--threads` receivers (chunk serialization carries a per-entry type marker). Directory entries appear in the delete manifest so `--delete` prunes correctly. Directory TIMES are transmitted (the `STATUS_DIR_TIMES` frame carries every traversed source directory's captured times, including `--dirs` entries) and applied by the receiver at the END of the transfer, after all children and the delete/publication phases, so a later child write cannot clobber a directory's mtime (`-O`/`--omit-dir-times` skips this application). FastSync divergences: directory modes/ownership are still not applied (only times are), and empty directories are still never created (a `STATUS_DIR_TIMES` entry is record-only), filter/`--exclude` rules are not re-applied to the listed dirs mode (there is no descent during which they would apply), and `-d` never creates the intermediate directories between the destination root and a listed file beyond the usual on-demand parent creation. Under `--delay-updates` only regular files are staged: directory entries are created immediately, so a delayed run that fails part way can leave the already-created empty directories behind (matching rsync, which also creates directories as it processes the file list and only delays regular-file data) |
|
||||
| `--mkpath` | Create missing path components | ✅ Implemented | Wire option (client → server). At connection start the server creates the client's destination root directory (and any missing leading components below its own authorized root) when `--mkpath` is set, failing the connection cleanly if it cannot. Without `--mkpath` a destination root that does not exist yet is rejected up front (rsync semantics), so the flag is the only way to transfer into a not-yet-created destination directory. Creation is confined by the same secure mkdir walk as file writes (`O_NOFOLLOW`, no `..`) |
|
||||
| `-r`, `--recursive` | Recurse into directories | ✅ Parity | Default behavior |
|
||||
| `-R`, `--relative` | Use relative path names | ⚠️ Caveat | Meaningful together with `--files-from` (FastSync's default full-tree scan always mirrors the full source argument path below the destination root, so -R does not change it). With `-R` + `--files-from` each listed entry is transmitted under its bare relative destination path: an entry `sub/x.txt` lands at `<dest>/sub/x.txt` (its leading components preserved) instead of under the `<dest>/<full source path>` mirror. Only the path sent on the wire changes; the client still reads the absolute source path, and the delete manifest derives from the sent (relative) paths so `--delete` and `--remove-source-files` stay consistent in both layouts. Works single-threaded and under `-j`/`--threads` (including chunk serialization) |
|
||||
| `--no-implied-dirs` | Don't send implied dirs with -R | ⚠️ Caveat | Client-side, meaningful only with `-R` + `--files-from`. rsync would normally create the ancestor directories implied by a listed file so it can be written; with `--no-implied-dirs` a listed file whose parent directory is not itself (or via an ancestor) explicitly listed cannot be placed, and FastSync fails the whole run up front with a clear error (`--no-implied-dirs: cannot place file '...': parent directory '...' is not explicitly listed`). Listing the directory (or an ancestor of it, or the whole tree `.`) permits the file. In every other mode the option has no effect. FastSync has no per-entry skip channel, so the rsync "omit the file" case is surfaced as a hard pre-transfer error |
|
||||
| `-d`, `--dirs`, `--old-dirs`, `--old-d` | Transfer dirs without recursing | ⚠️ Caveat | `-d <dir>` transmits an explicit directory entry for the source-root directory, so the destination mirror is created empty and nothing is descended into. With `--files-from` exactly the listed items are transferred: a listed directory is created empty (no descent) and a listed file is transferred with its content; the dest layout follows the same -R rules as plain files. A new wire frame (`STATUS_MKDIR`) carries each directory entry — the path and, when `--preserve`/`-a` (metadata mode) is negotiated, the directory's metadata; the receiver creates it with the same confined mkdir-parent semantics as regular writes, in single-threaded and `-j`/`--threads` receivers (chunk serialization carries a per-entry type marker). Directory entries appear in the delete manifest so `--delete` prunes correctly. Directory TIMES are transmitted (the `STATUS_DIR_TIMES` frame carries every traversed source directory's captured times, including `--dirs` entries) and applied by the receiver at the END of the transfer, after all children and the delete/publication phases, so a later child write cannot clobber a directory's mtime (`-O`/`--omit-dir-times` skips this application). FastSync divergences: directory modes/ownership are still not applied (only times are), and empty directories are still never created (a `STATUS_DIR_TIMES` entry is record-only), filter/`--exclude` rules are not re-applied to the listed dirs mode (there is no descent during which they would apply), and `-d` never creates the intermediate directories between the destination root and a listed file beyond the usual on-demand parent creation. Under `--delay-updates` only regular files are staged: directory entries are created immediately, so a delayed run that fails part way can leave the already-created empty directories behind (matching rsync, which also creates directories as it processes the file list and only delays regular-file data) |
|
||||
| `--mkpath` | Create missing path components | ✅ Parity | Wire option (client → server). At connection start the server creates the client's destination root directory (and any missing leading components below its own authorized root) when `--mkpath` is set, failing the connection cleanly if it cannot. Without `--mkpath` a destination root that does not exist yet is rejected up front (rsync semantics), so the flag is the only way to transfer into a not-yet-created destination directory. Creation is confined by the same secure mkdir walk as file writes (`O_NOFOLLOW`, no `..`) |
|
||||
|
||||
## 5. Transfer Modifications
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-u`, `--update` | Skip files newer on receiver | ✅ Implemented | `update` config field (crosses the wire; receiver-side policy, implies `-M` metadata). Before writing a regular file, the receiver checks `file_destination_is_newer_secure()` (via `stat_is_newer`, second-then-nanosecond strict `>` on the existing destination) and skips the write when the destination is newer than the source (`FILE_SAVE_SKIPPED`); equal-or-older destination (or a newer source) is transferred normally. Applied at write time on the regular-file, delay-updates-staged, hardlink-sibling, and special/device paths. Only regular destinations can be guarded (the newer-check requires `S_ISREG`), and like the other write-time policies it does not short-circuit the data transfer for a differing-size dest. `--remove-source-files` correctly respects the receiver's skip outcome so a skipped source is not removed |
|
||||
| `--inplace` | Update files in-place | ✅ Implemented | Direct write mode |
|
||||
| `--append` | Append data to shorter files | ✅ Implemented | Tail-only resume. When an existing destination file is SHORTER than the source, the receiver negotiates a resume offset with the sender and only the tail is transferred; the receiver rebuilds the full file (retained prefix + tail) and installs it through the normal atomic store path, so the result is byte-identical to the source whenever the retained prefix matches. Plain `--append` does NOT content-verify that prefix (rsync parity): a destination whose prefix differs from the source is resumed anyway, so the result (wrong prefix + correct tail) is NOT byte-identical and the file is effectively left corrupt — the documented rsync-parity risk (use `--append-verify` when the prefix cannot be trusted). Non-content attributes (permissions/ownership/mtime, via `-M`) are still applied. Requires the per-file `STATUS_CHECK` handshake, so it implies `--incremental`; it takes precedence over block delta for a growing file and falls back to delta/full when the destination is not shorter. Incompatible with `-s` (chunk serialization) and `--whole-file` (both rejected up front so the mode never silently degrades to a full transfer). Combines with `--inplace`, `--partial`/`--partial-dir`, and `--delay-updates` (the reconstructed full file flows through those paths unchanged). Divergence: rsync appends in place; FastSync reconstructs and atomically installs, so an interrupted or failed resume never leaves a half-written file at the destination (no corruption window), and `--append` is thus safe to use with the normal atomic path — not only with in-place writes |
|
||||
| `--append-verify` | Append with old-data checksum | ✅ Implemented | Like `--append`, but the retained prefix IS verified before resuming: the sender transmits the source prefix checksum and the receiver compares it to the xxHash64 of the retained destination prefix; on a match only the tail is transferred, on a MISMATCH the run falls back to a clean full transfer so the result is always a byte-identical source copy (never a corrupt prefix+tail blend). Wire/protocol: the append handshake adds `STATUS_APPEND` / `STATUS_APPEND_SIG` / `STATUS_APPEND_OK` / `STATUS_APPEND_DATA` frames and `PROTOCOL_VERSION` was bumped **2.9.0 → 2.10.0** (peers must match, and both must be 2.10.0 or the run fails the version check). Same implications/incompatibilities as `--append`; when both spellings are given `--append-verify` wins (the safer semantics). See the Phase-3 append notes below |
|
||||
| `-W`, `--whole-file` | Copy whole file (no delta) | ✅ Implemented | `whole_file` config field. Forces a full (whole-file) copy, disabling the block-level delta machinery: the sender only sends `STATUS_NEXT` + full data (client_send.c) and the receiver never requests a delta signature/reconstruction — the receiver's `try_delta = use_delta && !whole_file && ...` short-circuits. `whole_file` crosses the wire folded into `use_delta` (the wire carries `use_delta && !whole_file`), so no separate field/bump is needed. Delta is opt-in (`--delta` needs `--incremental`); `-W` additionally makes `--fuzzy` inert (no similar-file delta basis). `--append`/`--append-verify` are incompatible with `-W` and rejected up front (both sides). See the delta/append notes below |
|
||||
| `--block-size=SIZE` | Force checksum block-size | ✅ Implemented | Phase 7 Wave B: `--block-size` is an alias for `--delta-block`; both set `config->delta_block_size` (default `DELTA_BLOCK_SIZE_DEFAULT`, bounds `DELTA_BLOCK_SIZE_MIN..MAX`, out-of-range values are rejected with the default kept). The value is genuinely honored by the delta engine end-to-end: `delta_signature_create_seeded(old, size, config->delta_block_size, seed)` on the sender and receiver, `delta_apply(old, ...)` with the same size, so a non-default block size changes the block count of every signature the harnesses exchange (verified by unit + integration tests) |
|
||||
| `-u`, `--update` | Skip files newer on receiver | ✅ Parity | `update` config field (crosses the wire; receiver-side policy, implies metadata transmission). Before writing a regular file, the receiver checks `file_destination_is_newer_secure()` (via `stat_is_newer`, second-then-nanosecond strict `>` on the existing destination) and skips the write when the destination is newer than the source (`FILE_SAVE_SKIPPED`); equal-or-older destination (or a newer source) is transferred normally. Applied at write time on the regular-file, delay-updates-staged, hardlink-sibling, and special/device paths. Only regular destinations can be guarded (the newer-check requires `S_ISREG`), and like the other write-time policies it does not short-circuit the data transfer for a differing-size dest. `--remove-source-files` correctly respects the receiver's skip outcome so a skipped source is not removed |
|
||||
| `--inplace` | Update files in-place | ✅ Parity | Direct write mode |
|
||||
| `--append` | Append data to shorter files | ⚠️ Caveat | Tail-only resume. When an existing destination file is SHORTER than the source, the receiver negotiates a resume offset with the sender and only the tail is transferred; the receiver rebuilds the full file (retained prefix + tail) and installs it through the normal atomic store path, so the result is byte-identical to the source whenever the retained prefix matches. Plain `--append` does NOT content-verify that prefix (rsync parity): a destination whose prefix differs from the source is resumed anyway, so the result (wrong prefix + correct tail) is NOT byte-identical and the file is effectively left corrupt — the documented rsync-parity risk (use `--append-verify` when the prefix cannot be trusted). Non-content attributes (permissions/ownership/mtime, via `-M`) are still applied. Requires the per-file `STATUS_CHECK` handshake, so it implies `--incremental`; it takes precedence over block delta for a growing file and falls back to delta/full when the destination is not shorter. Incompatible with `-s` (chunk serialization) and `--whole-file` (both rejected up front so the mode never silently degrades to a full transfer). Combines with `--inplace`, `--partial`/`--partial-dir`, and `--delay-updates` (the reconstructed full file flows through those paths unchanged). Divergence: rsync appends in place; FastSync reconstructs and atomically installs, so an interrupted or failed resume never leaves a half-written file at the destination (no corruption window), and `--append` is thus safe to use with the normal atomic path — not only with in-place writes |
|
||||
| `--append-verify` | Append with old-data checksum | ⚠️ Caveat | Like `--append`, but the retained prefix IS verified before resuming: the sender transmits the source prefix checksum and the receiver compares it to the xxHash64 of the retained destination prefix; on a match only the tail is transferred, on a MISMATCH the run falls back to a clean full transfer so the result is always a byte-identical source copy (never a corrupt prefix+tail blend). Wire/protocol: the append handshake adds `STATUS_APPEND` / `STATUS_APPEND_SIG` / `STATUS_APPEND_OK` / `STATUS_APPEND_DATA` frames and `PROTOCOL_VERSION` was bumped **2.9.0 → 2.10.0** (peers must match, and both must be 2.10.0 or the run fails the version check). Same implications/incompatibilities as `--append`; when both spellings are given `--append-verify` wins (the safer semantics). See the Phase-3 append notes below |
|
||||
| `-W`, `--whole-file` | Copy whole file (no delta) | ✅ Parity | `whole_file` config field. Forces a full (whole-file) copy, disabling the block-level delta machinery: the sender only sends `STATUS_NEXT` + full data (client_send.c) and the receiver never requests a delta signature/reconstruction — the receiver's `try_delta = use_delta && !whole_file && ...` short-circuits. `whole_file` crosses the wire folded into `use_delta` (the wire carries `use_delta && !whole_file`), so no separate field/bump is needed. Delta is opt-in (`--delta` needs `--incremental`); `-W` additionally makes `--fuzzy` inert (no similar-file delta basis). `--append`/`--append-verify` are incompatible with `-W` and rejected up front (both sides). See the delta/append notes below |
|
||||
| `--block-size=SIZE` | Force checksum block-size | ✅ Parity | Phase 7 Wave B: `--block-size` is an alias for `--delta-block`; both set `config->delta_block_size` (default `DELTA_BLOCK_SIZE_DEFAULT`, bounds `DELTA_BLOCK_SIZE_MIN..MAX`, out-of-range values are rejected with the default kept). The value is genuinely honored by the delta engine end-to-end: `delta_signature_create_seeded(old, size, config->delta_block_size, seed)` on the sender and receiver, `delta_apply(old, ...)` with the same size, so a non-default block size changes the block count of every signature the harnesses exchange (verified by unit + integration tests) |
|
||||
|
||||
## 6. Destination Handling
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-n`, `--dry-run` | Trial run with no changes | ✅ Implemented | `dry_run` config field |
|
||||
| `-b`, `--backup` | Make backups of overwritten files | ✅ Implemented | Backup before overwrite |
|
||||
| `--backup-dir=DIR` | Backup directory hierarchy | ✅ Implemented | `backup_dir` config field |
|
||||
| `--suffix=SUFFIX` | Backup suffix (default ~) | ✅ Implemented | `suffix` config field |
|
||||
| `--delay-updates` | Put updated files in place at end | ✅ Implemented | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). Works in single-threaded and `-j`/`--threads` modes |
|
||||
| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ✅ Implemented | `--temp-dir` with the rsync short `-T` (Phase 7 Wave A; the timeout alias moved to long-only `--timeout`). Scratch dir is resolved under the receive root; temp copies use a unique name there and are atomically renamed into place. If the scratch dir and destination are on different filesystems the atomic rename fails with EXDEV and the file save fails, which aborts the whole transfer (FastSync has no per-file skip/resume on a save error; rsync's non-atomic copy fallback is deliberately not used). `--inplace` and `--partial-dir` writes bypass the scratch dir |
|
||||
| `-n`, `--dry-run` | Trial run with no changes | ⚠️ Caveat | Server-contacting since protocol 2.21.0. The final routing predicate is `dry_run_targets_server()` in `src/client/client_send.c`: any target a real run would reach over the wire selects the server-contacting path — an SSH transport, a daemon `host::module` destination, an explicit `--server-host` or `--server-port`/`--port`, TLS, or a source-bind `--address` — and the client handshakes with the receiver, which runs the normal read-only per-file check and answers `STATUS_DRY_RUN_TRANSFER`/`STATUS_OK` without mutating anything. A plain local destination (none of those) keeps the original client-side manifest that never dials the default `127.0.0.1:8080`. Would-delete reporting for `--delete*` is deferred (dry-run never deletes). |
|
||||
| `-b`, `--backup` | Make backups of overwritten files | ✅ Parity | Backup before overwrite |
|
||||
| `--backup-dir=DIR` | Backup directory hierarchy | ✅ Parity | `backup_dir` config field |
|
||||
| `--suffix=SUFFIX` | Backup suffix (default ~) | ✅ Parity | `suffix` config field |
|
||||
| `--delay-updates` | Put updated files in place at end | ⚠️ Caveat | Successfully received files are staged under a private 0700 `.fastsync-stage` dir inside the receive root and atomically renamed into their final destinations only after the whole transfer (manifest/delete handling included) succeeds, just before the success/outcome frame is sent. The delete walker deliberately skips the staging dir at the receive root, so `--delete` removes genuine extras but never the staged files (deletion runs before publication; rsync's delete-after ordering is not implemented). `--existing`/`--ignore-existing`/`--update` decide against the final destination path at stage time; `--backup` moves the old file aside at publication, and **`--force` is honored at publication** (protocol 2.23.0): a staged regular file or symlink may replace a destination directory that blocks it. Incompatible with `--inplace` and with `--backup-dir=.fastsync-stage` (the internal staging name is reserved; both are rejected). The staging dir name is fixed, so two simultaneous delayed transfers to the same destination root are serialized with an exclusive advisory lock held for the whole transfer: the second session fails cleanly instead of corrupting the first. Aborting or failing before publication installs nothing and removes the staging tree; a crash between stage and publish leaves staged leftovers that the next delayed run wipes at start (process death releases the lock). A stage→publish failure aborts the transfer (best-effort cleanup of the not-yet-published staged files; already-published files are not rolled back). Works in single-threaded and `-j`/`--threads` modes |
|
||||
| `-T`, `--temp-dir=DIR` | Create temporary files in DIR | ⚠️ Caveat | `--temp-dir` with the rsync short `-T` (the timeout alias moved to long-only `--timeout`). **Protocol 2.23.0 receiver policy: the scratch dir is confined to the receive root — a relative dir is resolved below it, and an absolute path or one containing `..` is rejected by the receiver** (an absolute/foreign-filesystem scratch dir was the divergence; rsync's standalone mode would follow an absolute `--temp-dir`, while its daemon also confines). Temp copies use a unique name there and are atomically renamed into place. **On `EXDEV` (scratch dir and destination on different filesystems) the receiver falls back to a non-atomic copy instead of aborting the transfer**, matching rsync. `--inplace` and `--partial-dir` writes bypass the scratch dir |
|
||||
|
||||
## 7. Deletion
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `--delete` | Delete extraneous files from dest | ✅ Implemented | `use_delete` config field. Deletion is always derived from the transmitted keep-set manifest of the paths the sender sent/keeps (never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. FastSync's default timing when no timing flag is given is **delete-after** (extras are removed only once the whole transfer succeeded) — intentionally NOT rsync's `--del`/delete-during default, to preserve FastSync's commit-style safety. By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). The bounded deletion is **all-or-nothing**: if the destination holds more extras than the effective bound (a client `--max-delete=NUM` or the 100000-entry server bound) nothing is deleted and the run fails with a distinct error instead of silently truncating |
|
||||
| `--delete-before` | Delete before transfer | ✅ Implemented | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (all-or-nothing bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. Divergence: the keep-set is the pre-scan snapshot, so a file that appears on the source between the pre-scan and the data pass is still transferred but was not protected from deletion |
|
||||
| `--del`, `--delete-during` | Delete during transfer | ✅ Implemented | Both spellings accepted; imply `--delete`. FastSync streams the source in a single directory scan and has no per-directory generator pass, so deletions cannot be interleaved per-directory the way rsync's delete-during does. `--delete-during` therefore selects the same early engine mode as `--delete-before` (manifest transmitted before any data, extras removed and acknowledged before data is applied); observable success/failure behaviour equals `--delete-before`. That is the documented divergence from rsync, where `--del` is the default meaning of `--delete` |
|
||||
| `--delete-delay` | Find deletions during, delete after | ✅ Implemented | Implies `--delete`. Commit-mode timing: extras are removed only after the whole transfer succeeded. rsync's delete-delay records the deletion list during its scan and applies it at the end; FastSync never snapshots the destination while data flows (the keep-set is the transmitted manifest and the destination is listed only at deletion time), so `--delete-delay` is implemented as the same end-of-transfer commit as `--delete-after` with identical safety. That is the documented divergence |
|
||||
| `--delete-after` | Delete after transfer | ✅ Implemented | Implies `--delete`. The delete-after timing is also what plain `--delete` does: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing |
|
||||
| `--delete-excluded` | Also delete excluded files | ✅ Implemented | `delete_excluded` config field. Under `--delete` FastSync now protects (rsync's default) the destination mirror of paths the sender's source scan pruned by user-selection rules — the `--filter`/`-F`/`-C` layer, the legacy `--exclude`/`--include` layer, and `--max-size`/`--min-size`. The sender transmits those concrete pruned paths as **protected prefixes** in the delete-manifest frame (see the Phase-3 notes below); the walker never descends into or removes them. `--delete-excluded` opts back in: the sender sends an empty protected list, so the excluded destination mirrors become ordinary extras and are removed. Divergences (documented): protection is derived only from what the source scan actually pruned — a stray destination-only file that happens to match an exclude rule is not protected (FastSync never re-applies rules to the destination, keeping deletion sender-derived), and `--files-from` subset pruning stays keep-set-only (an unlisted source path is treated as absent and its mirror is deletable, matching the `--files-from` delete note below). The two are orthogonal: `--delete-excluded` removes filter-excluded mirrors; it does not make `--files-from` prune things |
|
||||
| `--max-delete=NUM` | Max files to delete | ✅ Implemented | `max_delete` config field (default -1 = no client limit; 0 = delete nothing). NUM bounds a `--delete` run with rsync's all-or-nothing semantics: the receiver rehearses the deletion first and, if the destination holds more than NUM extras, deletes NOTHING and fails the transfer with a distinct `--max-delete` error. A run at or below NUM deletes exactly the extras. NUM only applies together with `--delete` (it is inert otherwise, matching rsync). The hard server bound `MAX_SERVER_DELETE_COUNT` (100000) still caps the walk; a NUM above it never raises that cap, and exceeding the server bound is its own all-or-nothing error. Directories count toward the limit (each removed empty directory is one deletion), like rsync |
|
||||
| `--ignore-errors` | Delete even with I/O errors | ✅ Implemented | Sender-side, client-only config field. rsync suppresses `--delete` when the transfer had I/O errors; FastSync's equivalent is a source-scan I/O error (an unreadable directory, e.g. EACCES): by default the scan aborts the run so no deletion happens. With `--ignore-errors` the scan continues past the unreadable directory, the readable tree is transferred and the deletion still runs (the mirror of the unreadable directory is treated as an extra). The run still exits non-zero (the error is reported, matching rsync's error status). Divergence: without the flag FastSync aborts the whole run on the scan error, whereas rsync transfers the rest of the tree and merely skips the deletion; both leave the deletion undone |
|
||||
| `--force` | Force deletion of non-empty dirs | ✅ Implemented | `force_delete` receiver config field (crosses the wire). rsync's `--force` lets an incoming non-directory replace a destination directory; FastSync implements exactly that: when a regular file is written to a path that is currently a (possibly non-empty) destination directory, `--force` removes that directory tree first — confined to the receive root and symlink-safe (O_NOFOLLOW fd walk, symlinks removed by name, never followed) — so the atomic install can place the file. Without `--force` such a write fails and the run aborts. Divergence: `--force` acts on the immediate-install path only; under `--delay-updates` a blocking directory is not cleared (publication renames over regular files) |
|
||||
| `-m`, `--prune-empty-dirs` | Prune empty dir chains | ✅ Implemented | `-m`/`--prune-empty-dirs` (Phase 7 Wave A freed the rsync short `-m`; FastSync multithreading is now `-j`/`--threads`). FastSync's recursive transfer records directory times but never CREATES an empty directory (a `STATUS_DIR_TIMES` entry is record-only, and `--dirs` empty entries are pruned by this flag), so empty directories are inherently never transferred (which is rsync's `-m` behavior) and truly-empty destination directory chains are removed by `--delete` regardless of this flag. The flag's additional real effect is on the `--dirs` explicit directory-entry generator: a plain `-d <empty-dir>` run omits the empty source directory's entry, so nothing is created at the destination (no `STATUS_MKDIR`, no `-i`/`--out-format` change line, and an existing empty mirror becomes an extra that `--delete` prunes). Explicitly `--files-from`-listed directories always pass through (documented `--files-from` behavior). A directory that still holds an excluded-but-protected file survives, matching the `--delete-excluded` default |
|
||||
| `--delete` | Delete extraneous files from dest | ⚠️ Caveat | `use_delete` config field. Deletion is always derived from the transmitted keep-set manifest of the paths the sender sent/keeps (never from unchecked input), runs through the symlink-safe walker bounded by `MAX_SERVER_DELETE_COUNT`, and skips the `.fastsync-stage` staging dir under `--delay-updates`. FastSync's default timing when no timing flag is given is **delete-after** (extras are removed only once the whole transfer succeeded) — intentionally NOT rsync's `--del`/delete-during default, to preserve FastSync's commit-style safety. By default the destination mirror of a path the source scan pruned (filter/exclude/size rules) is **protected** from deletion — matching rsync, which does not delete excluded files under `--delete`; `--delete-excluded` opts back into deleting them (see below). Deletion is scoped to the **synchronized directories** sent in the manifest (protocol 2.23.0), so a `--files-from` subset no longer deletes untransmitted paths outside the listed directory subtrees. The walk is bounded: a client `--max-delete=NUM` (or the 100000-entry server bound) makes it **partial** — entries up to the bound are removed, the rest are skipped, and the client exits **25** (`RERR_PARTIAL`), matching rsync, rather than failing the transfer. Extraneous destination symlinks are unlinked by name (never followed); a directory still holding a kept/protected entry is left behind rather than failing |
|
||||
| `--delete-before` | Delete before transfer | ⚠️ Caveat | Implies `--delete`. The sender runs a full source pre-scan (paths only) and transmits the keep-set manifest BEFORE any file data; the receiver validates it, removes every destination entry not listed (bounded walk, staging-dir skip, protected prefixes honored), then acks `STATUS_OK`. The sender only starts streaming after the deletion committed, or aborts if the receiver reported a deletion error. By definition the deletions already happened when a later transfer phase fails — rsync's delete-before is destructive the same way; a subsequent failure does not restore the removed files. Divergence: the keep-set is the pre-scan snapshot, so a file that appears on the source between the pre-scan and the data pass is still transferred but was not protected from deletion |
|
||||
| `--del`, `--delete-during` | Delete during transfer | ⚠️ Caveat | Both spellings accepted; imply `--delete`. FastSync streams the source in a single directory scan and has no per-directory generator pass, so deletions cannot be interleaved per-directory the way rsync's delete-during does. `--delete-during` therefore selects the same early engine mode as `--delete-before` (manifest transmitted before any data, extras removed and acknowledged before data is applied); observable success/failure behaviour equals `--delete-before`. That is the documented divergence from rsync, where `--del` is the default meaning of `--delete` |
|
||||
| `--delete-delay` | Find deletions during, delete after | ⚠️ Caveat | Implies `--delete`. Commit-mode timing: extras are removed only after the whole transfer succeeded. rsync's delete-delay records the deletion list during its scan and applies it at the end; FastSync never snapshots the destination while data flows (the keep-set is the transmitted manifest and the destination is listed only at deletion time), so `--delete-delay` is implemented as the same end-of-transfer commit as `--delete-after` with identical safety. That is the documented divergence |
|
||||
| `--delete-after` | Delete after transfer | ✅ Parity | Implies `--delete`. The delete-after timing is also what plain `--delete` does: the keep-set manifest closes the data stream and the receiver commits the bounded deletion only after the terminal `STATUS_FINISHED` proves the whole transfer (every data frame received and stored) succeeded. A failed or aborted transfer removes nothing |
|
||||
| `--delete-excluded` | Also delete excluded files | ⚠️ Caveat | `delete_excluded` config field. Under `--delete` FastSync protects (rsync's default) the destination mirror of paths the sender's source scan pruned by the user-selection rules — the `--filter`/`-F`/`-C` layer and the legacy `--exclude`/`--include` layer. The sender transmits those concrete pruned paths as **protected prefixes** in the delete-manifest frame (see the Phase-3 notes below); the walker never descends into or removes them. `--delete-excluded` opts back in: the sender sends an empty protected list, so the excluded destination mirrors become ordinary extras and are removed. **`--max-size`/`--min-size` pruned mirrors are a separate, always-on protection** (protocol 2.23.0, rsync parity): size-pruned source mirrors survive `--delete` even with `--delete-excluded`. Divergences (documented): protection is derived only from what the source scan actually pruned — a stray destination-only file that happens to match an exclude rule is not protected (FastSync never re-applies rules to the destination, keeping deletion sender-derived) |
|
||||
| `--max-delete=NUM` | Max files to delete | ✅ Parity | `max_delete` config field (default -1 = no client limit; 0 = delete nothing). **Protocol 2.23.0 matches rsync's partial semantics:** the receiver deletes up to NUM entries (regular files, symlinks and empty directories; each directory removal counts as one) and then **stops deleting, skips the rest, and reports the run as partial**. The client prints a "deletions stopped due to `--max-delete` limit" message and exits **25** (rsync's `RERR_PARTIAL`), not a hard failure — the transfer itself succeeded. NUM only applies together with `--delete` (it is inert otherwise, matching rsync). A client NUM below the server hard bound `MAX_SERVER_DELETE_COUNT` (100000) replaces it; a NUM above it never raises that cap. Deleting an entire destination with no limit is still bounded by the server's 100000-entry ceiling. `--delete-missing-args` exact-path deletions and the ordinary extras walk draw from the same budget, matching rsync |
|
||||
| `--ignore-errors` | Delete even with I/O errors | ⚠️ Caveat | Sender-side, client-only config field. rsync suppresses `--delete` when the transfer had I/O errors; FastSync's equivalent is a source-scan I/O error (an unreadable directory, e.g. EACCES): by default the scan aborts the run so no deletion happens. With `--ignore-errors` the scan continues past the unreadable directory, the readable tree is transferred and the deletion still runs (the mirror of the unreadable directory is treated as an extra). The run still exits non-zero (the error is reported, matching rsync's error status). Divergence: without the flag FastSync aborts the whole run on the scan error, whereas rsync transfers the rest of the tree and merely skips the deletion; both leave the deletion undone |
|
||||
| `--force` | Force deletion of non-empty dirs | ⚠️ Caveat | `force_delete` receiver config field (crosses the wire). rsync's `--force` lets an incoming non-directory replace a destination directory; FastSync implements exactly that: when a regular file (or symlink) is written to a path that is currently a (possibly non-empty) destination directory, `--force` removes that directory tree first — confined to the receive root and symlink-safe (O_NOFOLLOW fd walk, symlinks removed by name, never followed) — so the install can place the file. **Protocol 2.23.0 honors `--force` on the `--delay-updates` publication path too**, not only the immediate-install path. Without `--force` such a write fails and the run aborts. Gated by the server `--allow-delete` policy (a client cannot use `--force` to remove a destination tree on a server that forbids deletion) |
|
||||
| `-m`, `--prune-empty-dirs` | Prune empty dir chains | ✅ Parity | `-m`/`--prune-empty-dirs` (Phase 7 Wave A freed the rsync short `-m`; FastSync multithreading is now `-j`/`--threads`). FastSync's recursive transfer records directory times but never CREATES an empty directory (a `STATUS_DIR_TIMES` entry is record-only, and `--dirs` empty entries are pruned by this flag), so empty directories are inherently never transferred (which is rsync's `-m` behavior) and truly-empty destination directory chains are removed by `--delete` regardless of this flag. The flag's additional real effect is on the `--dirs` explicit directory-entry generator: a plain `-d <empty-dir>` run omits the empty source directory's entry, so nothing is created at the destination (no `STATUS_MKDIR`, no `-i`/`--out-format` change line, and an existing empty mirror becomes an extra that `--delete` prunes). Explicitly `--files-from`-listed directories always pass through (documented `--files-from` behavior). A directory that still holds an excluded-but-protected file survives, matching the `--delete-excluded` default |
|
||||
|
||||
**Deletion-timing implementation notes (Phase 3):** the delete flags above are
|
||||
real. Two new config booleans (`delete_during`, `delete_delay`) join the already
|
||||
@@ -133,64 +154,66 @@ noted in the rows above.
|
||||
**Deletion-policy notes (Phase 3, delete-policy wave):** this wave made the
|
||||
deletion family real — `--delete-excluded`, `--max-delete`, `--ignore-errors`,
|
||||
`--force`, `--prune-empty-dirs` — and, to support them, the `STATUS_MANIFEST`
|
||||
frame now carries **two sections**: the keep-set paths followed by a list of
|
||||
**protected prefixes** (destination-relative paths the source scan pruned by
|
||||
user-selection rules, which the walker must never delete unless
|
||||
`--delete-excluded` opted out). Two config booleans were added for the wave:
|
||||
`force_delete` (crosses the wire; the receiver clears a directory that blocks an
|
||||
incoming file) and `ignore_errors` (client-only; the sender's scan continues
|
||||
past an unreadable directory). `max_delete`'s default became -1 ("no client
|
||||
limit"). These wire/layout changes bumped `PROTOCOL_VERSION` **2.8.0 → 2.9.0**
|
||||
(peers must match). All four wire additions — `force_delete`,
|
||||
`delete_excluded`, `prune_empty_dirs`, `max_delete` — round-trip unchanged and
|
||||
are validated on receive.
|
||||
frame carries the keep-set paths followed by a list of **protected prefixes**
|
||||
(destination-relative paths the source scan pruned by user-selection rules,
|
||||
which the walker must never delete unless `--delete-excluded` opted out).
|
||||
Two config booleans were added for the wave: `force_delete` (crosses the wire;
|
||||
the receiver clears a directory that blocks an incoming file) and
|
||||
`ignore_errors` (client-only; the sender's scan continues past an unreadable
|
||||
directory). `max_delete`'s default became -1 ("no client limit"). These
|
||||
wire/layout changes bumped `PROTOCOL_VERSION` **2.8.0 → 2.9.0** (peers must
|
||||
match). All four wire additions — `force_delete`, `delete_excluded`,
|
||||
`prune_empty_dirs`, `max_delete` — round-trip unchanged and are validated on
|
||||
receive.
|
||||
|
||||
**Missing-args note (Phase 3, missing-args wave):** `--ignore-missing-args` and
|
||||
`--delete-missing-args` are implemented as described in the Safety & Security
|
||||
rows. Wire impact: the `STATUS_MANIFEST` frame now carries a **third section** —
|
||||
**Missing-args note (Phase 3, missing-args wave; extended in 2.23.0):**
|
||||
`--ignore-missing-args` and `--delete-missing-args` are implemented as described
|
||||
in the Safety & Security rows. Wire impact: the `STATUS_MANIFEST` frame carries
|
||||
a list of destination-relative **exact-delete paths** (the missing entries'
|
||||
mirrors) — and the config frame gained a `delete_missing_args` boolean
|
||||
mirrors), and the config frame gained a `delete_missing_args` boolean
|
||||
(`ignore_missing_args` stays client-only, exactly like `ignore_errors`). These
|
||||
wire/layout changes bumped `PROTOCOL_VERSION` **2.9.0 → 2.10.0** (peers must
|
||||
match). The receiver validates the third section identically to the keep-set
|
||||
(non-empty, relative, traversal-free; `MAX_MANIFEST_ENTRIES` per section, a
|
||||
single `MAX_MANIFEST_BYTES` budget shared across all three). On commit the
|
||||
receiver runs the exact-path deletions FIRST (`manifest_delete_missing_args`:
|
||||
confined per-path unlink/rmdir, deep removal only under `--force`/`--delete`,
|
||||
staging/basis protected, never blocked by the protected-prefix list) and then
|
||||
the ordinary extras walk when `--delete` is active (`manifest_delete_all`). A
|
||||
client may request the exact-path deletions without `--delete`; the server's
|
||||
`--allow-delete` policy gates them exactly like `--delete`, so an unauthorized
|
||||
server ignores the request while the missing entries are still skipped.
|
||||
match). The receiver validates the section identically to the keep-set (non-empty,
|
||||
relative, traversal-free; the shared `MAX_MANIFEST_ENTRIES`/`MAX_MANIFEST_BYTES`
|
||||
budget spans every section). On commit the receiver runs the exact-path deletions
|
||||
FIRST (`manifest_delete_missing_args`: confined per-path unlink/rmdir, deep
|
||||
removal only under `--force`/`--delete`, staging/basis protected, never blocked
|
||||
by the protected-prefix list) and then the ordinary extras walk when `--delete`
|
||||
is active (`manifest_delete_all`). A client may request the exact-path deletions
|
||||
without `--delete`; the server's `--allow-delete` policy gates them exactly like
|
||||
`--delete`, so an unauthorized server ignores the request while the missing
|
||||
entries are still skipped.
|
||||
|
||||
The deletion walker is now **all-or-nothing**: before any unlink it rehearses
|
||||
the deletion (an fd-relative walk identical to the delete pass, counting every
|
||||
regular file it would unlink and every directory it would remove) and refuses to
|
||||
start when the extras exceed the effective bound — a client `--max-delete=NUM`
|
||||
below the hard bound, or the hard `MAX_SERVER_DELETE_COUNT` (100000) bound
|
||||
itself. Previously the walker removed up to `MAX_SERVER_DELETE_COUNT` extras and
|
||||
then reported an error (a truncated deletion); it now removes nothing and fails
|
||||
with an error naming the bound. Directories count toward the bound. A directory
|
||||
that still holds entries the walker leaves in place (a protected excluded file,
|
||||
a kept manifest entry, a symlink) is left behind rather than failing the run —
|
||||
matching rsync's "cannot delete non-empty directory" behaviour. The
|
||||
all-or-nothing guarantee holds only while the destination is not concurrently
|
||||
modified: rehearsal and delete are two separate walks, so a concurrent change
|
||||
between them (another process adding or removing destination entries) can make
|
||||
the actual deletion diverge from the counted set.
|
||||
**Delete scoping and partial limits (protocol 2.23.0).** The `STATUS_MANIFEST`
|
||||
frame now carries **four sections** — keep-set, protected prefixes, exact-delete
|
||||
(missing-args) paths, and the set of **synchronized directories**. The extras
|
||||
walk is scoped to the synchronized directories, so a `--files-from` subset no
|
||||
longer deletes untransmitted destination paths outside the listed directory
|
||||
subtrees (a data-loss fix, matching rsync). `--max-size`/`--min-size` pruned
|
||||
source mirrors are protected independently of `--delete-excluded`. Extraneous
|
||||
destination symlinks are unlinked by name (never followed). The `--max-delete`
|
||||
budget is **partial**: the walker deletes up to the effective bound (a client
|
||||
`--max-delete=NUM` below the hard bound, else the hard
|
||||
`MAX_SERVER_DELETE_COUNT` = 100000) and then stops, skips the rest, and reports
|
||||
the run as partial so the client exits **25** (`RERR_PARTIAL`) exactly like
|
||||
rsync — it is a successful transfer with an incomplete deletion, not a hard
|
||||
failure. The exact-path missing-args removals and the extras walk share that one
|
||||
budget. A directory that still holds entries the walker leaves in place (a
|
||||
protected excluded file, a kept manifest entry, a symlink) is left behind rather
|
||||
than failing the run — matching rsync's "cannot delete non-empty directory"
|
||||
behaviour.
|
||||
|
||||
Manifest size: the sender's keep-set and protected-prefix collections (streaming
|
||||
or early pre-scan) are unbounded, but the receiver rejects a manifest beyond
|
||||
`MAX_MANIFEST_ENTRIES` (1 048 576 entries, applied to EACH section — a frame can
|
||||
therefore total up to 2 097 152 entries) / `MAX_MANIFEST_BYTES` (16 MB of paths,
|
||||
counted across BOTH sections) as a hard protocol error. A heavily filtered
|
||||
source whose exclusion list grows large thus fails the run cleanly on the
|
||||
receiver (STATUS_ERROR) instead of being silently truncated. In the commit
|
||||
modes this only means the deletion is refused after the data already arrived; in
|
||||
the early modes (`--delete-before`/`--delete-during`) the manifest is the first
|
||||
frame, so an oversized keep-set or protected list aborts the whole transfer
|
||||
BEFORE any data is sent. Keep the source tree small enough for the receiver's
|
||||
manifest caps when using the early timing.
|
||||
Manifest size: the sender's collections (streaming or early pre-scan) are
|
||||
unbounded, but the receiver rejects a manifest whose **aggregate** count exceeds
|
||||
`MAX_MANIFEST_ENTRIES` (1 048 576 entries across ALL sections) or whose aggregate
|
||||
path bytes exceed `MAX_MANIFEST_BYTES` (16 MB across all sections) as a hard
|
||||
protocol error. A heavily filtered source whose exclusion list grows large thus
|
||||
fails the run cleanly on the receiver (STATUS_ERROR) instead of being silently
|
||||
truncated. In the commit modes this only means the deletion is refused after the
|
||||
data already arrived; in the early modes (`--delete-before`/`--delete-during`)
|
||||
the manifest is the first frame, so an oversized manifest aborts the whole
|
||||
transfer BEFORE any data is sent. Keep the source tree small enough for the
|
||||
receiver's manifest caps when using the early timing.
|
||||
|
||||
Early-delete ACK wait: after committing a large deletion (up to
|
||||
`MAX_SERVER_DELETE_COUNT` removals) the receiver's `STATUS_OK`/`STATUS_ERROR`
|
||||
@@ -238,33 +261,33 @@ why plain `--append` works on the normal atomic path, not only with `--inplace`.
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-M`, `--preserve` | Preserve file metadata | ✅ Implemented | Mode, uid, gid, mtime |
|
||||
| `-p`, `--perms` | Preserve permissions | ✅ Implemented | Phase 7 Wave A: `-p`/`--perms` now preserve permission bits, folded into FastSync's broad metadata bundle (`--preserve`); the SSH port moved to `--ssh-port`. rsync-parity short form |
|
||||
| `-o`, `--owner` | Preserve owner | ✅ Implemented | Part of -M |
|
||||
| `-g`, `--group` | Preserve group | ✅ Implemented | Part of -M |
|
||||
| `-t`, `--times` | Preserve modification times | ✅ Implemented | Part of -M |
|
||||
| `-E`, `--executability` | Preserve executability | ✅ Implemented | Preserves executable permission bits (implies metadata preservation) |
|
||||
| `--chmod=CHMOD` | Affect file permissions | ✅ Implemented | Supports numeric and symbolic `ugo` `rwx` changes; retains receiver safety masking |
|
||||
| `-A`, `--acls` | Preserve ACLs | ✅ Implemented | Implemented on Linux via the POSIX-ACL xattr representation: the sender captures the `system.posix_acl_access` / `system.posix_acl_default` xattrs into the same bounded whitelisted set as `-X`, transmits them per-file, and the receiver re-applies them fd-relative. Setting an ACL the receiver is not permitted to set (non-root on a file it does not own, unsupported filesystem) is logged and skipped, never fatal. libacl is **not** required. Only the `system.posix_acl_*` namespaces plus `user.*` are ever applied; privileged namespaces are never applied (see the Phase-4 xattr/ACL notes below). Implies metadata transmission |
|
||||
| `-X`, `--xattrs` | Preserve extended attributes | ✅ Implemented | Preserves unprivileged `user.*` extended attributes (Linux `listxattr`/`getxattr` on capture, `fsetxattr` on the written destination fd). Both capture (sender) and application (receiver) are restricted to the `user.*` namespace and the two POSIX ACL xattrs, so a client can **never** force a `security.*`/`trusted.*`/privileged attribute onto the destination; the receiver independently re-validates every incoming name against this whitelist and rejects anything else. Payloads are bounded (per-name ≤255B, per-value ≤1MiB, per-file count ≤256 total bytes ≤4MiB) on both ends, and an oversized/malformed frame is a clean protocol rejection (no OOM). Applied fd-relative to the exact written file. Implies metadata transmission. Incompatible with `-s` (chunk serialization), rejected up front (see the notes); a `--link-dest`/`-H` hard-link copy fallback re-applies the attributes so they are not dropped when a link is refused |
|
||||
| `-H`, `--hard-links` | Preserve hard links | ✅ Implemented | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below |
|
||||
| `-D` | Same as --devices --specials | ✅ Implemented | Implies `--devices --specials`. `-D` was unassigned in FastSync (verified: no collision), so it is free to imply both device-node and special-file preservation. See the `--devices`/`--specials` rows and the Phase-4 devices notes below |
|
||||
| `--devices` | Preserve device files | ✅ Implemented | Recreates char/block device nodes on the destination via `mknod` instead of transferring content. Type + rdev are validated strictly (S_IFMT from the transmitted mode; major/minor range-checked, non-negative), and creation is **privilege-gated**: `mknod` needs `CAP_MKNOD`, so a non-root receiver (CI runs via setpriv as non-root) logs a warning and **skips the device entry safely** — the whole transfer never aborts just because the node could not be made. The node is created fd-relative below the receive root (`mknodat` on the confined secure parent), so it can never be placed outside the authorized root, never follows a symlink, and never replaces an existing directory. Only a char/block mode is honored. Crosses the wire (a new `STATUS_SPECIAL` frame carries the path + metadata mode + rdev; `PROTOCOL_VERSION` bumped **2.12.0 → 2.13.0**). Divergence: per-entry skip (not a hard error) when the receiver lacks `CAP_MKNOD`, documented in the Phase-4 devices notes |
|
||||
| `--specials` | Preserve special files | ⛔ Impossible/Divergence | **FIFO recreation works**: FIFOs are recreated on the destination via `mkfifo` (unprivileged, so this is a real, assertable behavior under CI). **Only socket recreation is impossible**: a socket entry can be created only by `bind(2)` on a live socket, not by any filesystem call, so a source socket is skipped with an explicit note. That one unsupported node kind is why the flag is classified Impossible/Divergence even though FIFO recreation itself works; its normal path is otherwise complete. FIFO creation is privileged-gated only in the sense of graceful skip on any permission failure. Node creation is confined below the receive root (`mkfifoat` on the secure fd-relative parent; no `..`, no symlink follow). Crosses the wire like `--devices` (the `STATUS_SPECIAL` frame; `PROTOCOL_VERSION` bumped **2.12.0 → 2.13.0**). See the Phase-4 devices notes |
|
||||
| `--copy-devices` | Copy device contents as file | ✅ Implemented | Copy a device's CONTENT into an ordinary regular file on the destination instead of recreating the node — non-privileged and safe. FastSync scans a device/FIFO as a regular file: its reported size (`st_size`, typically 0 for char devices and FIFOs) is copied, so a FIFO or a non-readable device becomes an empty (or size-bounded) regular file. The default data path is size-bounded and never blocks (it sends exactly `st_size` bytes, never an unbounded pseudo-device stream); with `--sendfile`, a non-regular source (FIFO/device) is detected from its `stat` mode and falls back to that same buffered read, so `--copy-devices --sendfile` cannot hang either. The run always succeeds and never crashes on such input. **Deliberate, safe divergence from rsync's dd-like unbounded device read.** See the Phase-4 devices notes |
|
||||
| `--write-devices` | Write to devices as files | ✅ Implemented | Write the received data directly into an **existing** device node on the destination instead of creating a regular file. Restricted and best-effort: the destination must already exist and be a char/block device (opened only under the confined receive root, with `O_NOFOLLOW` + `O_NONBLOCK`); a missing, symlinked, FIFO-with-no-reader (`ENXIO`), non-device destination, or any write failure is **skipped with a warning** rather than allowed, so a run can never clobber the system, never blocks on a special-file target, and never aborts on an unusable target. See the Phase-4 devices notes |
|
||||
| `-U`, `--atimes` | Preserve access times | ✅ Implemented | Captures the source access time (from the scanner's pre-read stat, so it is not clobbered by reading the file for transfer) and transmits it over the wire; the receiver restores it together with the mtime via `futimens`/`utimensat`. Implies metadata transmission (the times travel inside the `-M` metadata payload), but does not enable ownership application (that stays opt-in via the identity flags). Wire: new `atime` fields on the metadata frame + a `preserve_atimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** |
|
||||
| `-N`, `--crtimes` | Preserve create times | ⛔ Impossible/Divergence | Birth-times cannot be set by any portable filesystem call (`utimensat`/`futimens` only set atime/mtime), so this row is an explicit **Impossible/Divergence** (Phase 7 Wave B). Capture + transmit stays: `statx(STATX_BTIME)` on Linux records the source birth time as a wire field; the receiver logs a debug note that it cannot be applied and continues — never failing the transfer and never pretending it worked. On platforms without `statx` it parses as a documented no-op (flag accepted; nothing is captured). Implies metadata transmission. Wire: new `crtime` fields + a `preserve_crtimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** (see the Phase-4 metadata-time notes) |
|
||||
| `-O`, `--omit-dir-times` | Omit dirs from --times | ✅ Implemented | Real modifier now that FastSync preserves directory times. With metadata on, the scanner captures every traversed source directory's mtime (and atime under `-U`) and the sender transmits them in trailing `STATUS_DIR_TIMES` frame(s) **after all file data and the optional delete manifest** (chunked at the receiver's `MAX_MANIFEST_ENTRIES` per-frame cap); a dir-time entry only RECORDS metadata and never creates the directory, so empty source directories stay untransferred. The receiver defers applying them until its delete / `--delay-updates` publication phases have committed, so writing or removing a child never clobbers a parent directory's mtime (rsync applies directory times at the end for exactly this reason). When `-O` is set (the boolean crosses the wire) the receiver does not apply any of them; without `-O` an `-a`/`--preserve` transfer now restores directory times (reversing the old "never preserves dir times" divergence). Wire change: the terminal `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** |
|
||||
| `-J`, `--omit-link-times` | Omit symlinks from --times | ✅ Implemented | Real modifier now that FastSync preserves symlink times. Symlink entries already carried their metadata on `STATUS_SYMLINK`; the receiver now applies it with **no-follow primitives only** (`utimensat(..., AT_SYMLINK_NOFOLLOW)`, plus best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)`), so the link itself is stamped without ever dereferencing it, confined fd-relative below the authorized receive root. A symlink has no children, so the times are applied immediately at creation. When `-J` is set (the boolean crosses the wire) the receiver skips the timestamps (mode/ownership are unaffected); without `-J` an `-a`/`-l` transfer restores symlink mtimes. Wire change alongside `-O`: the shared `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** |
|
||||
| `--super` | Receiver attempts super-user activities | ✅ Implemented | Phase 7 Wave E: receiver-side **safe-subset + clear-refusal** privilege model, tri-state `super_mode` (auto/on/off). `--super` **permits** the receiver to attempt super-user activities — ownership application and char/block device-node creation — that are already confined fd-relative below the authorized receive root; `--no-super` **forbids** them even when the receiver is root; the default (`auto`) preserves the pre-existing **best-effort** behavior of *attempting* them (not only when already root: an unprivileged attempt is refused by the kernel and skipped per entry, matching FastSync's history). The server additionally accepts an operator-level `--no-super` veto that forces `OFF` for every connection it accepts (so it also refuses any client `--copy-as`/`--super`); the `--fake-super` owner replay and the `--write-devices` write path are gated by the same policy. **FastSync never elevates**: no `setuid`/`seteuid`/`setgid` is ever called, and `--super` never bypasses the confinement floor (`file_open_secure_parent`, `O_NOFOLLOW`, root checks) — it only permits an attempt that is already confined. `--super` does **not** imply `--numeric-ids` and never enables client-chosen ownership on its own: ownership is applied only when an explicit identity policy (`--usermap`/`--groupmap`/`--chown`/`--numeric-ids`/`--copy-as`) is also given. A non-root receiver given `--super` logs exactly one warning at activation and each confined attempt is then refused by the kernel and skipped per entry (never aborts); `--no-super` suppresses ownership, char/block `mknod`, `--write-devices` and the fake-super owner replay, while unprivileged FIFO creation is unaffected. Wire: one trailing `super_mode` int on the config frame (validated 0..2), sent **before** the `--copy-as` block (fixed order: super int, then copy-as presence int + ids); `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Documented divergence from rsync:** rsync's `--super` runs the receiver with elevated privilege; FastSync only permits a confined attempt and never elevates |
|
||||
| `--fake-super` | Store/recover privileged attrs via xattrs | ✅ Implemented | Phase 7 Wave B: full record **and replay**. The receiver writes the source `uid:gid:mode:mtime_sec:mtime_nsec` into a reserved `user.fastsync.stat` xattr on each written file (best-effort, fd-relative, format unchanged), then immediately re-applies it via `fake_super_restore_fd`: `fchown` (only where privileged — a non-root EPERM/EACCES is skipped silently, matching FastSync's identity philosophy), `fchmod`, and `futimens`. The OWNER leg is additionally skipped unless an explicit ownership identity policy (`--numeric-ids`/`--usermap`/`--groupmap`/`--chown`/`--copy-as`) is active — `--fake-super` on its own only *records* the source owner and must not act as an un-gated chown primitive — when `--no-super` forbids super-user activities (even for root), or when an active `--copy-as` is authoritative, so the recorded source owner can never override a forced `--copy-as` owner; the xattr record is still stored/replayed for a later privileged restore and mode/mtime still apply, so unprivileged `--fake-super` keeps working. The restored mode goes through the same sanitization as the normal metadata path (group/other write bits are never granted, so a recorded 0666 restores as 0644), so fake-super replay can never grant group/other-write that plain `--preserve` would refuse. Absence or a malformed record is a silent no-op, never fatal. The recording format diverges from rsync's `user.rsync.%stat%`; no cross-tool conversion is attempted. Implies metadata transmission so the source uid/gid/mode/mtime are available. Both it and `-X`/`-A` are incompatible with `-s` (chunk serialization), rejected up front |
|
||||
| `--open-noatime` | Avoid changing access time when opening files | ✅ Implemented | Sender-side policy: the sender opens source files with `O_NOATIME` (Linux) when reading them for transfer, so the open/read does NOT bump the source's on-disk access time. Degrades safely when `O_NOATIME` is unavailable (not defined) or refused (`EPERM`, since it needs `CAP_FOWNER` or file ownership): the code falls back to a normal open, so the data always transfers — only the atime-bump is skipped. It does not itself capture/preserve atime; it only avoids modifying it. **Client-only, never crosses the wire.** Exposed as `file_open_for_read()` and applied to both the buffered data path and the sendfile path |
|
||||
| `--numeric-ids` | Do not map uid/gid by name | ✅ Implemented | Ownership is applied through FastSync's opt-in identity path (see the Phase-4 identity notes below). `--numeric-ids` is a mapping-policy modifier: when applying ownership it uses the transmitted numeric uid/gid directly, skipping the name lookup. Without an ownership-affecting option it is inert (FastSync only applies ownership when the user opts in). It does not need `-M` to be parsed, but ownership is only applied when metadata (hence the source uid/gid) is actually transmitted (see the notes) |
|
||||
| `--usermap=STRING` | Map usernames | ✅ Implemented | Opt-in ownership application. rsync subset implemented: comma-separated `FROM:TO` rules evaluated in order, first match wins; `FROM`/`TO` are group/user names (resolved on the SOURCE machine at parse time), `*` (FROM matches any id / TO = the receiving process's current euid), and an `@N` or bare `N` numeric id. Rules are carried over the wire as resolved numeric id pairs; the receiver applies a matching rule (else falls back to `--chown`, `--numeric-ids`, then a best-effort name lookup) via an fd-relative `fchown`. Malformed/unresolvable specs are rejected with a clear error, never a silent no-op. Implies metadata preservation so the source uid/gid travel. Only effective when the receiver can actually change ownership (root or membership); otherwise it warns and continues |
|
||||
| `--groupmap=STRING` | Map group names | ✅ Implemented | Same rsync subset and semantics as `--usermap` but for the group (gid) side and the group databases. See the Phase-4 identity notes |
|
||||
| `--chown=USER:GROUP` | Map owner and group | ✅ Implemented | Opt-in ownership override applied receiver-side. Forms: `USER:GROUP`, `USER` (owner only), `:GROUP` (group only); a `*` for USER/GROUP means the current/root user or group as appropriate; an `@N`/bare `N` numeric id is accepted. A `:` inside a name may be escaped as `\:`. Equivalent to a trailing `*:*` usermap+groupmap rule (so an explicit `--usermap`/`--groupmap` match wins). Malformed or unresolvable specs are clear parse errors. Implies metadata preservation. Only effective when the receiver has permission to chown; otherwise it warns and continues (rsync parity) |
|
||||
| `--copy-as=USER[:GROUP]` | Perform the copy as another user/group | ✅ Implemented | Safe-subset implementation, an explicit divergence from rsync's **real identity switching**. rsync makes the receiving process actually assume USER/GROUP (setuid/setgid); FastSync's receiver is multithreaded, so a real credential drop would be unsafe and is never attempted — FastSync never calls `setuid`/`seteuid`/`setgid`. Instead the receiver FORCES the ownership of every entry it writes to `copy_as_uid`/`copy_as_gid` through the existing confined, fd-relative identity path (the same `fchown`/`fchownat` mechanism as `--chown`/`--usermap`/`--groupmap`; symlinks use `fchownat(..., AT_SYMLINK_NOFOLLOW)`, and directories — including intermediate parents created implicitly while writing a nested file — and char/block/FIFO nodes are owned no-follow too, so a directory never keeps the receiver's owner while its children get the target owner), with `--copy-as` at the **highest priority** — it beats usermap/groupmap/`--chown`/`--numeric-ids` and the best-effort name lookup. This REQUIRES a privileged (root) receiver: an unprivileged receiver REFUSES the whole transfer up front at the config handshake (`server_module_gate`, running inside `config_receive_with_validate` before the `STATUS_OK` ack) with a clear error and no file data exchanged — never a silent wrong-ownership result. A server running with an operator `--no-super` veto also refuses it, and a **daemon** refuses `--copy-as`, like every other client-chosen-ownership request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/explicit `--super`), unless the selected module opts in with `client owner = yes`; without that per-module opt-in a daemon must not honor an arbitrary client-selected owner (the standalone listener and SSH `--stdio` server keep honoring these for their single operator-authorized root). `--fake-super` interaction: `--copy-as` is authoritative, so the recorded source owner is never replayed over the forced target owner. If the ownership apply still fails with EPERM/EACCES (capability-restricted root, root-squash, read-only mount) the failure is logged at ERROR and the **entry is reported as failed** rather than written with the wrong owner, which fails the transfer (fail-fast) so overall success is never reported with the wrong owner. USER is resolved on the client against the user database (a name, an `@N`/bare `N` numeric id, or `*` meaning the client's current euid); when `:GROUP` is present it is resolved against the group database (`*` meaning the client's egid). **Group-default rule:** when the group is omitted FastSync uses the user's primary gid (`getpwuid(uid)->pw_gid`); a numeric id with no local passwd entry has no primary gid to look up, so `gid` falls back to `uid` (documented divergence). Malformed/empty/unresolvable specs are clear parse errors, never a silent no-op. Never elevates privileges and never bypasses the confined receive root. Implies metadata preservation (the source uid/gid must be transmitted). Wire: a new trailing config-frame block **sent after** the `--super` int (presence int, then the two int32 ids, both validated `>= 0` on receive; the ids are also rejected if they do not fit int32 at CLI parse time); `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0** |
|
||||
| `--preserve` | (FastSync alias, not an rsync flag) | ✅ Parity | **FastSync-only alias** for `-p` + `-t` (mode + mtime), long-form only. It is not rsync's `--preserve` (rsync has no such option); the short `-M` that used to spell it is now rsync's `--remote-option`. The wire metadata also carries uid/gid for `-o`/`-g`/`-a`, and ownership is applied via `-o`/`-g`, `-a`, or an explicit identity flag (`--numeric-ids`/`--usermap`/`--groupmap`/`--chown`/`--copy-as`) |
|
||||
| `-p`, `--perms` | Preserve permissions | ✅ Parity | Real per-attribute flag (protocol 2.22.0): `preserve_perms` applies the source mode independently of times/owner/group. **Strict rsync parity (protocol 2.23.0): the source mode is copied exactly, including setuid/setgid/sticky and group/other-write bits — there is no masking.** Without `-p`, a new file gets `source_mode & ~umask` when metadata is present (else the historical fixed `0644`); new directories without `-p` still use FastSync's `0755` creation default, because directory metadata is only applied when a directory attribute is requested. `-A/--acls` implies `-p`; `--chmod` does **not** imply `-p` (rsync parity) and applies its own unsanitized changes to the new mode. `-X/--xattrs` does not imply `-p`. The SSH port moved to `--ssh-port`. rsync-parity short form |
|
||||
| `-o`, `--owner` | Preserve owner | ✅ Parity | Real per-attribute flag (`preserve_owner`): preserve the source uid, resolved on the receiver by name against its own user database with a raw-numeric fallback (only numeric ids cross the wire). `--usermap`/`--chown=USER` imply it. Application follows the `--super`/`--no-super` policy; a non-opted daemon module applies no ownership (see the Daemon Mode notes) |
|
||||
| `-g`, `--group` | Preserve group | ✅ Parity | Real per-attribute flag (`preserve_group`): preserve the source gid, resolved by name on the receiver with a raw-numeric fallback. `--groupmap`/`--chown=:GROUP` imply it. Same privilege/super-policy gating as `-o` |
|
||||
| `-t`, `--times` | Preserve modification times | ✅ Parity | Real per-attribute flag (`preserve_times`): apply the source mtime independently of the other attributes. `-O/--omit-dir-times` suppresses directories only and `-J/--omit-link-times` suppresses symlinks only; `-U`/`-N` do not imply it. `--preserve`/`-a` imply it, and `--incremental`/`--delta` auto-enable it unless `--no-times`/`--no-preserve` |
|
||||
| `-E`, `--executability` | Preserve executability | ✅ Parity | Preserves executable permission bits (implies metadata preservation) |
|
||||
| `--chmod=CHMOD` | Affect file permissions | ✅ Parity | Faithful port of rsync 3.4.1's `parse_chmod`/`tweak_mode`: numeric octal and symbolic `ugo`/`rwx` changes, `D`/`F` directory/file selectors, `X` (execute only on directories or already-executable files), `s`/`t` setuid/setgid/sticky, and append semantics — repeated clauses and repeated `--chmod` options accumulate in order (joined with commas). The changes are applied to the new mode **without sanitization** (matching rsync) and `--chmod` does **not** imply `-p` (rsync parity). Applied to files and directories on the receiver |
|
||||
| `-A`, `--acls` | Preserve ACLs | ⚠️ Caveat | Implemented on Linux via the POSIX-ACL xattr representation: the sender captures the `system.posix_acl_access` / `system.posix_acl_default` xattrs into the same bounded whitelisted set as `-X`, transmits them per-file, and the receiver re-applies them fd-relative. Setting an ACL the receiver is not permitted to set (non-root on a file it does not own, unsupported filesystem) is logged and skipped, never fatal. libacl is **not** required. Only the `system.posix_acl_*` namespaces plus `user.*` are ever applied; privileged namespaces are never applied (see the Phase-4 xattr/ACL notes below). Implies metadata transmission |
|
||||
| `-X`, `--xattrs` | Preserve extended attributes | ⚠️ Caveat | Preserves unprivileged `user.*` extended attributes (Linux `listxattr`/`getxattr` on capture, `fsetxattr` on the written destination fd). Both capture (sender) and application (receiver) are restricted to the `user.*` namespace and the two POSIX ACL xattrs, so a client can **never** force a `security.*`/`trusted.*`/privileged attribute onto the destination; the receiver independently re-validates every incoming name against this whitelist and rejects anything else. Payloads are bounded (per-name ≤255B, per-value ≤1MiB, per-file count ≤256 total bytes ≤4MiB) on both ends, and an oversized/malformed frame is a clean protocol rejection (no OOM). Applied fd-relative to the exact written file. Implies metadata transmission. Incompatible with `-s` (chunk serialization), rejected up front (see the notes); a `--link-dest`/`-H` hard-link copy fallback re-applies the attributes so they are not dropped when a link is refused |
|
||||
| `-H`, `--hard-links` | Preserve hard links | ✅ Parity | Files on the source that share an inode (`st_dev`+`st_ino`, e.g. a `cp -al` tree) are re-created as hard links to one another on the destination, so duplicate links stay deduplicated and only the first member's data is sent (later members are transmitted as payload-less `STATUS_HARDLINK` frames). The receiver links each sibling to the first member's installed file with an atomic link + rename; on `link()` failure it falls back to a byte-identical local copy of the first member, never a partial/corrupt file. Requires the sequential scan for ordering (the first member is always emitted and installed before any sibling is linked). Works single-threaded and under `-j`/`--threads`, `--inplace`, `--delay-updates` (links staged and published by rename) and `--partial`. Crosses the wire (`preserve_hard_links` bool; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0**, peers must match). Incompatible with `-s` (chunk serialization) and `--append`/`--append-verify`, rejected up front with a distinct error. See the Phase-4 hard-links notes below |
|
||||
| `-D` | Same as --devices --specials | ✅ Parity | Implies `--devices --specials`. `-D` was unassigned in FastSync (verified: no collision), so it is free to imply both device-node and special-file preservation. As of protocol 2.23.0 `--specials` genuinely covers **both FIFOs and unix sockets**, so `-D` covers the full rsync set. See the `--devices`/`--specials` rows and the Phase-4 devices notes below |
|
||||
| `--devices` | Preserve device files | ⚠️ Caveat | Recreates char/block device nodes on the destination via `mknod` instead of transferring content. Type + rdev are validated strictly (S_IFMT from the transmitted mode; major/minor range-checked, non-negative), and creation is **privilege-gated**: `mknod` needs `CAP_MKNOD`, so a non-root receiver (CI runs via setpriv as non-root) logs a warning and **skips the device entry safely** — the whole transfer never aborts just because the node could not be made. The node is created fd-relative below the receive root (`mknodat` on the confined secure parent), so it can never be placed outside the authorized root, never follows a symlink, and never replaces an existing directory. Only a char/block mode is honored. Crosses the wire (a `STATUS_SPECIAL` frame carries the path + metadata mode + rdev). Divergence: per-entry skip (not a hard error) when the receiver lacks `CAP_MKNOD`, documented in the Phase-4 devices notes |
|
||||
| `--specials` | Preserve special files | ✅ Parity | **FIFO and unix-socket recreation work** (protocol 2.23.0): FIFOs are recreated with `mkfifoat`, and sockets with `mknodat(..., S_IFSOCK)` — the latter is unprivileged on Linux because it materializes the socket *node*, not a live bound socket, so it is a real, assertable behavior under CI (it matches rsync, which also recreates a socket by `mknod`). Node creation is confined below the receive root (fd-relative parent; no `..`, no symlink follow) and type/rdev are validated strictly; a matching existing node is left in place and an unrelated entry is never replaced. Crosses the wire like `--devices` (the `STATUS_SPECIAL` frame). See the Phase-4 devices notes |
|
||||
| `--copy-devices` | Copy device contents as file | ⚠️ Caveat | Copy a device's CONTENT into an ordinary regular file on the destination instead of recreating the node — non-privileged and safe. FastSync scans a device/FIFO as a regular file: its reported size (`st_size`, typically 0 for char devices and FIFOs) is copied, so a FIFO or a non-readable device becomes an empty (or size-bounded) regular file. The default data path is size-bounded and never blocks (it sends exactly `st_size` bytes, never an unbounded pseudo-device stream); with `--sendfile`, a non-regular source (FIFO/device) is detected from its `stat` mode and falls back to that same buffered read, so `--copy-devices --sendfile` cannot hang either. The run always succeeds and never crashes on such input. **Deliberate, safe divergence from rsync's dd-like unbounded device read.** See the Phase-4 devices notes |
|
||||
| `--write-devices` | Write to devices as files | ⚠️ Caveat | Write the received data directly into an **existing** device node on the destination instead of creating a regular file. Restricted and best-effort: the destination must already exist and be a char/block device (opened only under the confined receive root, with `O_NOFOLLOW` + `O_NONBLOCK`); a missing, symlinked, FIFO-with-no-reader (`ENXIO`), non-device destination, or any write failure is **skipped with a warning** rather than allowed, so a run can never clobber the system, never blocks on a special-file target, and never aborts on an unusable target. See the Phase-4 devices notes |
|
||||
| `-U`, `--atimes` | Preserve access times | ✅ Parity | Captures the source access time (from the scanner's pre-read stat, so it is not clobbered by reading the file for transfer) and transmits it over the wire; the receiver restores it together with the mtime via `futimens`/`utimensat`. Implies metadata transmission (the times travel inside the shared metadata payload), but does not enable ownership application (that stays opt-in via the identity flags). Wire: `atime` fields on the metadata frame + a `preserve_atimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** |
|
||||
| `-N`, `--crtimes` | Preserve create times | ❌ Divergent | Birth-times cannot be set by any portable filesystem call (`utimensat`/`futimens` only set atime/mtime), so this row is an explicit **Divergent** entry (Phase 7 Wave B). Capture + transmit stays: `statx(STATX_BTIME)` on Linux records the source birth time as a wire field; the receiver logs a debug note that it cannot be applied and continues — never failing the transfer and never pretending it worked. On platforms without `statx` it parses as a documented no-op (flag accepted; nothing is captured). Implies metadata transmission. Wire: new `crtime` fields + a `preserve_crtimes` config boolean; `PROTOCOL_VERSION` bumped **2.11.0 → 2.12.0** (see the Phase-4 metadata-time notes) |
|
||||
| `-O`, `--omit-dir-times` | Omit dirs from --times | ✅ Parity | Real modifier now that FastSync preserves directory times. With metadata on, the scanner captures every traversed source directory's mtime (and atime under `-U`) and the sender transmits them in trailing `STATUS_DIR_TIMES` frame(s) **after all file data and the optional delete manifest** (chunked at the receiver's `MAX_MANIFEST_ENTRIES` per-frame cap); a dir-time entry only RECORDS metadata and never creates the directory, so empty source directories stay untransferred. The receiver defers applying them until its delete / `--delay-updates` publication phases have committed, so writing or removing a child never clobbers a parent directory's mtime (rsync applies directory times at the end for exactly this reason). When `-O` is set (the boolean crosses the wire) the receiver does not apply any of them; without `-O` an `-a`/`--preserve` transfer now restores directory times (reversing the old "never preserves dir times" divergence). Wire change: the terminal `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** |
|
||||
| `-J`, `--omit-link-times` | Omit symlinks from --times | ✅ Parity | Real modifier now that FastSync preserves symlink times. Symlink entries already carried their metadata on `STATUS_SYMLINK`; the receiver now applies it with **no-follow primitives only** (`utimensat(..., AT_SYMLINK_NOFOLLOW)`, plus best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)`), so the link itself is stamped without ever dereferencing it, confined fd-relative below the authorized receive root. A symlink has no children, so the times are applied immediately at creation. When `-J` is set (the boolean crosses the wire) the receiver skips the timestamps (mode/ownership are unaffected); without `-J` an `-a`/`-l` transfer restores symlink mtimes. Wire change alongside `-O`: the shared `STATUS_DIR_TIMES` frame; `PROTOCOL_VERSION` bumped **2.16.0 → 2.17.0** |
|
||||
| `--super` | Receiver attempts super-user activities | ⚠️ Caveat | Phase 7 Wave E: receiver-side **safe-subset + clear-refusal** privilege model, tri-state `super_mode` (auto/on/off). `--super` **permits** the receiver to attempt super-user activities — ownership application and char/block device-node creation — that are already confined fd-relative below the authorized receive root; `--no-super` **forbids** them even when the receiver is root; the default (`auto`) preserves the pre-existing **best-effort** behavior of *attempting* them (not only when already root: an unprivileged attempt is refused by the kernel and skipped per entry, matching FastSync's history). The server additionally accepts an operator-level `--no-super` veto that forces `OFF` for every connection it accepts (so it also refuses any client `--copy-as`/`--super`); a **privileged (root) standalone TCP listener now also defaults to `OFF`** unless the operator opts in with the new server-only `--allow-super` flag (the flag is **rejected with `--stdio`**, whose remote argv is composed by the client and must never defeat the secure default; operators exposing `fastsync-server --stdio` over SSH need a forced command if the default must hold. An unprivileged receiver is unchanged, since the kernel refuses the confined attempts anyway; the `--daemon` path keeps its per-module `client owner = yes` opt-in); the `--fake-super` owner replay and the `--write-devices` write path are gated by the same policy. **FastSync never elevates**: no `setuid`/`seteuid`/`setgid` is ever called, and `--super` never bypasses the confinement floor (`file_open_secure_parent`, `O_NOFOLLOW`, root checks) — it only permits an attempt that is already confined. `--super` does **not** imply `--numeric-ids` and never enables client-chosen ownership on its own: ownership is applied only when an explicit identity policy (`--usermap`/`--groupmap`/`--chown`/`--numeric-ids`/`--copy-as`) or a preserve-source request (`-o`/`-g`, or `-a`/`--archive`) is also given. A non-root receiver given `--super` logs exactly one warning at activation and each confined attempt is then refused by the kernel and skipped per entry (never aborts); `--no-super` suppresses ownership, char/block `mknod`, `--write-devices` and the fake-super owner replay, while unprivileged FIFO creation is unaffected. Wire: one trailing `super_mode` int on the config frame (validated 0..2), sent **before** the `--copy-as` block (fixed order: super int, then copy-as presence int + ids); `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Documented divergence from rsync:** rsync's `--super` runs the receiver with elevated privilege; FastSync only permits a confined attempt and never elevates |
|
||||
| `--fake-super` | Store/recover privileged attrs via xattrs | ⚠️ Caveat | Full record **and replay** (protocol 2.23.0 parity update). The receiver writes the resolved `uid:gid:mode:mtime_sec:mtime_nsec` into a reserved `user.fastsync.stat` xattr on each written file (best-effort, fd-relative), then immediately re-applies the mode and times via `fake_super_restore_fd` (`fchmod` + `futimens`; absent/malformed records are a silent no-op, never fatal). **`--fake-super` never performs a real `chown`**: when an explicit ownership mapping (`--chown`/`--usermap`/`--groupmap`/`--copy-as`) is active the receiver records the *resolved* id, otherwise the source's own id, but the owner leg is always suppressed so recording can never defeat the flag; the record is retained for a later privileged restore. The replayed mode goes through the shared `metadata_mode_for_policy` helper, so under `-p` it is copied exactly (including group/other-write and special bits — strict rsync parity, no masking) and under `-E` it follows the rsync executability rule. Directory ownership and directory xattrs/ACLs are preserved alongside file entries (mode/owner are applied to directories under the same per-attribute policy and `-A`/`-X` carry the directory ACL/xattr block). Implies metadata transmission so the source uid/gid/mode/mtime are available. The recording format diverges from rsync's `user.rsync.%stat%`; no cross-tool conversion is attempted. Both it and `-X`/`-A` are incompatible with `-s` (chunk serialization), rejected up front |
|
||||
| `--open-noatime` | Avoid changing access time when opening files | ✅ Parity | Sender-side policy: the sender opens source files with `O_NOATIME` (Linux) when reading them for transfer, so the open/read does NOT bump the source's on-disk access time. Degrades safely when `O_NOATIME` is unavailable (not defined) or refused (`EPERM`, since it needs `CAP_FOWNER` or file ownership): the code falls back to a normal open, so the data always transfers — only the atime-bump is skipped. It does not itself capture/preserve atime; it only avoids modifying it. **Client-only, never crosses the wire.** Exposed as `file_open_for_read()` and applied to both the buffered data path and the sendfile path |
|
||||
| `--numeric-ids` | Do not map uid/gid by name | ✅ Parity | **A mapping modifier only:** when ownership is being applied it uses the transmitted numeric uid/gid directly, skipping the name lookup. It does **not** request ownership application on its own — combine it with `-o`/`-g`, `-a`, or an explicit map (`--chown`/`--usermap`/`--groupmap`) — and it does not need any metadata flag merely to parse. Ownership is only applied when metadata (hence the source uid/gid) is actually transmitted (see the Phase-4 identity notes) |
|
||||
| `--usermap=STRING` | Map usernames | ⚠️ Caveat | Opt-in ownership application. rsync subset implemented (protocol 2.23.0): comma-separated `FROM:TO` rules evaluated in order, first match wins. `FROM` accepts a user name (resolved on the SOURCE machine at parse time), an `@N`/bare `N` numeric id, an inclusive `LOW-HIGH` **id range**, `*` (matches any id), or an **empty** field (matches ids with no name on the source). `TO` accepts a name (resolved on the **receiver**), an `@N`/bare `N` id, or `*` (the receiving process's current euid). Rules are carried over the wire as resolved numeric id pairs; the receiver applies a matching rule (else falls back to `--chown`, `--numeric-ids`, then a best-effort name lookup) via an fd-relative `fchown`, including directory entries. Malformed/unresolvable specs are rejected with a clear error, never a silent no-op. Implies metadata preservation so the source uid/gid travel. Only effective when the receiver can actually change ownership (root or membership); otherwise it warns and continues |
|
||||
| `--groupmap=STRING` | Map group names | ⚠️ Caveat | Same rsync subset and semantics as `--usermap` (names, `@N`/bare `N`, inclusive ranges, `*`, empty-FROM for unnamed ids, receiver-resolved `TO` names) but for the group (gid) side and the group databases. See the Phase-4 identity notes |
|
||||
| `--chown=USER:GROUP` | Map owner and group | ⚠️ Caveat | Opt-in ownership override applied receiver-side. Forms: `USER:GROUP`, `USER` (owner only), `:GROUP` (group only); a `*` for USER/GROUP means the current/root user or group as appropriate; an `@N`/bare `N` numeric id is accepted. A `:` inside a name may be escaped as `\:`. Equivalent to a trailing `*:*` usermap+groupmap rule (so an explicit `--usermap`/`--groupmap` match wins). **Protocol 2.23.0 makes `--chown` and `--usermap`/`--groupmap` mutually exclusive on the same side: combining them (in either order) is a clear configuration error** (`--usermap conflicts with prior --chown`), matching rsync and never an order-dependent silent winner. Malformed or unresolvable specs are clear parse errors. Implies metadata preservation. Only effective when the receiver has permission to chown; otherwise it warns and continues (rsync parity) |
|
||||
| `--copy-as=USER[:GROUP]` | Perform the copy as another user/group | ⚠️ Caveat | Safe-subset implementation, an explicit divergence from rsync's **real identity switching**. rsync makes the receiving process actually assume USER/GROUP (setuid/setgid); FastSync's receiver is multithreaded, so a real credential drop would be unsafe and is never attempted — FastSync never calls `setuid`/`seteuid`/`setgid`. Instead the receiver FORCES the ownership of every entry it writes to `copy_as_uid`/`copy_as_gid` through the existing confined, fd-relative identity path (the same `fchown`/`fchownat` mechanism as `--chown`/`--usermap`/`--groupmap`; symlinks use `fchownat(..., AT_SYMLINK_NOFOLLOW)`, and directories — including intermediate parents created implicitly while writing a nested file — and char/block/FIFO nodes are owned no-follow too, so a directory never keeps the receiver's owner while its children get the target owner), with `--copy-as` at the **highest priority** — it beats usermap/groupmap/`--chown`/`--numeric-ids` and the best-effort name lookup. This REQUIRES a privileged (root) receiver: an unprivileged receiver REFUSES the whole transfer up front at the config handshake (`server_module_gate`, running inside `config_receive_with_validate` before the `STATUS_OK` ack) with a clear error and no file data exchanged — never a silent wrong-ownership result. A server running with an operator `--no-super` veto also refuses it; a privileged (root) standalone TCP listener refuses it by default too and only honors it after the operator passes `--allow-super` (the flag is rejected with `--stdio`, where the client-composed remote argv could otherwise defeat the default; a forced command is required if the default must hold), and a **daemon** refuses `--copy-as`, like every other client-chosen-ownership request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/explicit `--super`), unless the selected module opts in with `client owner = yes`; without that per-module opt-in a daemon must not honor an arbitrary client-selected owner (a root standalone listener honors these for its single operator-authorized root only when started with `--allow-super`). `--fake-super` interaction: `--copy-as` is authoritative, so the recorded source owner is never replayed over the forced target owner. If the ownership apply still fails with EPERM/EACCES (capability-restricted root, root-squash, read-only mount) the failure is logged at ERROR and the **entry is reported as failed** rather than written with the wrong owner, which fails the transfer (fail-fast) so overall success is never reported with the wrong owner. USER is resolved on the client against the user database (a name, an `@N`/bare `N` numeric id, or `*` meaning the client's current euid); when `:GROUP` is present it is resolved against the group database (`*` meaning the client's egid). **Group-default rule:** when the group is omitted FastSync uses the user's primary gid (`getpwuid(uid)->pw_gid`); a numeric id with no local passwd entry has no primary gid to look up, so `gid` falls back to `uid` (documented divergence). Malformed/empty/unresolvable specs are clear parse errors, never a silent no-op. Never elevates privileges and never bypasses the confined receive root. Implies metadata preservation (the source uid/gid must be transmitted). Wire: a new trailing config-frame block **sent after** the `--super` int (presence int, then the two int32 ids, both validated `>= 0` on receive; the ids are also rejected if they do not fit int32 at CLI parse time); `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0** |
|
||||
|
||||
**Phase-4 metadata-time notes:** `-U/--atimes`, `-N/--crtimes`,
|
||||
`-O/--omit-dir-times`, `-J/--omit-link-times`, and `--open-noatime` are new.
|
||||
@@ -326,8 +349,13 @@ match, exactly as prior phases did).
|
||||
fatal.
|
||||
- **`--fake-super`**: see the row above; the reserved key is `user.fastsync.stat`
|
||||
with the documented `uid:gid:mode:mtime_sec:mtime_nsec` (mode octal) format.
|
||||
It is honest but partial — there is no replay, and it does not interoperate
|
||||
with rsync's `user.rsync.%stat%`.
|
||||
**Replay exists**: after each stored record the receiver immediately re-applies
|
||||
the recorded mode and times fd-relative (`fake_super_restore_fd`), but it
|
||||
deliberately never performs a real `chown` — `--fake-super` only *records*
|
||||
the resolved owner (the active `--chown`/`--usermap`/`--groupmap`/`--copy-as`
|
||||
mapping when one is in effect, otherwise the source's own id) for a later
|
||||
privileged restore. The recording format diverges from rsync's
|
||||
`user.rsync.%stat%`; no cross-tool conversion is attempted.
|
||||
- **Chunk serialization (`-s`) incompatibility:** the per-file xattr block rides
|
||||
the streaming per-file frame, which `-s` replaces with a fixed buffer format,
|
||||
so `-X` / `-A` combined with `-s` is rejected up front on both ends (mirroring
|
||||
@@ -359,7 +387,7 @@ symlink timestamps (ownership/mode application is unaffected and stays governed
|
||||
by the identity opt-in). Both config booleans already crossed the wire. See the
|
||||
`-O`/`-J` rows and the Wave D note below.
|
||||
|
||||
**-U/-N and -M interaction:** because FastSync carries all metadata (mode, uid,
|
||||
**-U/-N and metadata-bundle interaction:** because FastSync carries all metadata (mode, uid,
|
||||
gid, mtime, and now atime/crtime) in one bounded payload that is only sent when
|
||||
metadata transmission is on, `-U` and `-N` imply metadata transmission (the
|
||||
times travel inside that payload). They do **not** enable ownership application,
|
||||
@@ -458,19 +486,22 @@ CI runs the integration suite as a NON-ROOT user (via setpriv), so `mknod` fails
|
||||
with `EPERM`. The receiver treats this as a graceful, logged *skip of the entry*
|
||||
returned as a success/skip outcome — the whole transfer NEVER aborts just because
|
||||
the environment cannot create the node. `mkfifo` (FIFOs) is unprivileged, so
|
||||
`--specials` FIFO creation is a real, assertable behavior under CI; sockets cannot
|
||||
be recreated by any standard filesystem call and are skipped with an explicit
|
||||
note. The "device actually created" integration assertions are guarded to run
|
||||
only as root. User-facing expectation: point `--devices` at devices and a
|
||||
non-root receiver will faithfully skip them while transferring everything else.
|
||||
`--specials` FIFO creation is a real, assertable behavior under CI. **Sockets are
|
||||
recreated too** (protocol 2.23.0) with `mknodat(..., S_IFSOCK)`: Linux allows an
|
||||
unprivileged `mknod` of a socket node because no live bound socket is created,
|
||||
so a source socket materializes as a socket-type filesystem entry exactly as
|
||||
rsync does. The "device actually created" integration assertions are guarded to
|
||||
run only as root. User-facing expectation: point `--devices` at devices and a
|
||||
non-root receiver will faithfully skip them while transferring everything else;
|
||||
`--specials` recreates FIFOs and socket nodes for any receiver.
|
||||
|
||||
**Confinement & validation:** a special/device node is created with
|
||||
`mknodat`/`mkfifoat` on the parent directory opened fd-relative below the receive
|
||||
root (`file_open_secure_parent`: `O_NOFOLLOW`, no `..` components, root-checked),
|
||||
so a node can never be created outside the authorized destination root and never
|
||||
through a symlinked parent. The transmitted type is derived ONLY from the
|
||||
validated S_IFMT bits of the metadata mode (char/block/FIFO honored, socket
|
||||
skipped, regular/dir rejected as an invalid special), and the transmitted rdev is
|
||||
validated S_IFMT bits of the metadata mode (char/block/FIFO and socket honored;
|
||||
regular/dir rejected as an invalid special), and the transmitted rdev is
|
||||
validated both on the wire (`file_receive_special`, `chunk_deserialize`) and at
|
||||
the creation site (`file_special_rdev_valid`): a negative, oversize, or
|
||||
non-device-carrying rdev is rejected outright (receiver aborts the frame), and a
|
||||
@@ -496,13 +527,13 @@ warning + skip, never a system-clobbering write or an abort.
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-l`, `--links` | Copy symlinks as symlinks | ✅ Implemented | A symlink is transmitted as a real symlink: its target string crosses the wire (a new `STATUS_SYMLINK` frame / chunk entry type) and the receiver creates it with `symlinkat` beneath the receive root. This makes the previously-`-l`-included-but-targetless symlink handling complete. See the Phase-4 symlink-trust notes |
|
||||
| `-L`, `--copy-links` | Transform symlink to referent | ✅ Implemented | `copy_links` config field |
|
||||
| `--copy-unsafe-links` | Transform unsafe symlinks | ✅ Implemented | `copy_unsafe_links` config field |
|
||||
| `--safe-links` | Ignore symlinks outside tree | ✅ Implemented | `safe_links` config field |
|
||||
| `--munge-links` | Munge symlinks for safety | ✅ Implemented | Sender rewrites each transmitted symlink target with a `#SYMLINK/` marker; a target that could escape the receive root (absolute or containing `..`) is never transmitted (contained/skipped); the receiver strips the marker to restore the real target. See the Phase-4 symlink-trust notes |
|
||||
| `-k`, `--copy-dirlinks` | Transform symlink to dir | ✅ Implemented | A symlink whose referent is a directory is dereferenced and recursed as a real directory; a symlink to a regular file stays a symlink. Sender-side only. See the Phase-4 symlink-trust notes |
|
||||
| `-K`, `--keep-dirlinks` | Treat symlinked dir as dir | ✅ Implemented | On the receiver, an existing destination symlink-to-a-directory is used as that directory (followed) instead of being replaced; it is followed only when it resolves to a directory that stays beneath the receive root. See the Phase-4 symlink-trust notes |
|
||||
| `-l`, `--links` | Copy symlinks as symlinks | ⚠️ Caveat | A symlink is transmitted as a real symlink: its target string crosses the wire (`STATUS_SYMLINK` / chunk entry type) and the receiver creates it with `symlinkat` beneath the receive root, never following the target. **Targets are stored verbatim (protocol 2.23.0), matching rsync `-l`: an absolute target or one containing `..` is copied exactly, and the receiver no longer enforces a containment predicate by default.** `--safe-links` is the sender-side opt-in that drops unsafe targets before transmission; `--trust-sender` does **not** affect symlink targets (it only relaxes the receiver's path-list re-validation). The *placement* path is still hard-confined (`has_path_traversal`, O_NOFOLLOW fd walk), and the link's own mode/times are applied with no-follow primitives. See the Phase-4 symlink-trust notes and the residual-risk note below |
|
||||
| `-L`, `--copy-links` | Transform symlink to referent | ⚠️ Caveat | Sender-side: every symlink is replaced by its referent's content (`copy_links` config field). A referent that cannot be read, including a broken symlink, is treated as a non-error and the run exits 0 — where rsync exits 23 (`RERR_PARTIAL`). This is the documented status-code divergence |
|
||||
| `--copy-unsafe-links` | Transform unsafe symlinks | ⚠️ Caveat | Sender-side: only symlinks whose target is unsafe (absolute or escaping via `..`, matching rsync's `unsafe_symlink()` semantics) are dereferenced into their referent; safe links stay symlinks. Same broken-referent exit-0 caveat as `-L` (`copy_unsafe_links` config field) |
|
||||
| `--safe-links` | Ignore symlinks outside tree | ✅ Parity | Sender-side: a symlink whose target is unsafe is not transmitted at all (skipped), matching rsync's `--safe-links`. Because FastSync applies this while scanning the source, the receiver does not need to repeat it (`safe_links` config field) |
|
||||
| `--munge-links` | Munge symlinks for safety | ✅ Parity | Sender rewrites each transmitted symlink target with rsync's `/rsyncd-munged/` prefix; the receiver strips the marker (only when the negotiated `munge_links` policy is on, so a source link that genuinely begins with the marker round-trips verbatim) and restores the exact real target. Unlike rsync, FastSync prefixes on the *sender* and un-munges on the receiver, but the wire result and the stored marker match rsync. See the Phase-4 symlink-trust notes |
|
||||
| `-k`, `--copy-dirlinks` | Transform symlink to dir | ✅ Parity | A symlink whose referent is a directory is dereferenced and recursed as a real directory; a symlink to a regular file stays a symlink. Sender-side only. See the Phase-4 symlink-trust notes |
|
||||
| `-K`, `--keep-dirlinks` | Treat symlinked dir as dir | ✅ Parity | On the receiver, an existing destination symlink-to-a-directory is used as that directory (followed) instead of being replaced; it is followed only when it resolves to a directory that stays beneath the receive root. See the Phase-4 symlink-trust notes |
|
||||
|
||||
**Phase-4 symlink-trust notes:** `-l/--links`, `-k/--copy-dirlinks`,
|
||||
`-K/--keep-dirlinks`, and `--munge-links` form the "symlink trust boundaries"
|
||||
@@ -518,17 +549,20 @@ was bumped **2.12.0 → 2.13.0** (peers must match, exactly as prior phases did)
|
||||
|
||||
**Per-flag semantics and divergences.**
|
||||
- **`-l/--links`** copies a symlink as a symlink: the scanner `readlink`s the
|
||||
target, the sender transmits it, and the receiver `symlinkat`s it. FastSync
|
||||
`-l` never preserved symlink targets before (the flag was documented partial
|
||||
and, in fact, tried to read the referent as file data); it now does, matching
|
||||
rsync. Divergences: because the receiver enforces the symlink containment
|
||||
predicate unconditionally, a plain `-l` sync **refuses to round-trip a
|
||||
legitimate absolute symlink target** (it is dropped, never created pointing
|
||||
outside the root — see the `--munge-links` note for the symmetric trust
|
||||
boundary); a relative in-root target is copied as-is. As of P7 Wave D FastSync
|
||||
also applies the symlink's own metadata with no-follow primitives
|
||||
target, the sender transmits it, and the receiver `symlinkat`s it. **Targets
|
||||
are stored verbatim (protocol 2.23.0), matching rsync `-l`:** an absolute
|
||||
target or one containing `..` is copied exactly as-is. The receiver no longer
|
||||
enforces the strict containment predicate on the link *value*; target policy
|
||||
belongs to the sender (`--safe-links`/`--copy-unsafe-links`) exactly as in
|
||||
rsync. The link's *placement* path is still hard-confined
|
||||
(`has_path_traversal`, O_NOFOLLOW fd walk), and the link's own metadata is
|
||||
applied with no-follow primitives
|
||||
(`utimensat`/`fchownat`/`fchmodat` with `AT_SYMLINK_NOFOLLOW`), so `-J` is a
|
||||
real omit switch rather than a no-op.
|
||||
real omit switch rather than a no-op. **Residual risk:** because `-l` stores
|
||||
targets verbatim and does not enforce containment, a destination later
|
||||
consumed by a link-following tool can follow a link outside the receive root.
|
||||
Use `--safe-links` when the source is not trusted; a destination that only
|
||||
ever uses `openat`-style no-follow access is unaffected.
|
||||
- **`-k/--copy-dirlinks`** (sender): a symlink whose referent is a directory is
|
||||
dereferenced and recursed into as a real directory; a symlink to a regular
|
||||
file (or any non-directory) is kept as a symlink. This is rsync's `-k`. When
|
||||
@@ -547,40 +581,35 @@ was bumped **2.12.0 → 2.13.0** (peers must match, exactly as prior phases did)
|
||||
divergence for `--delete` over an existing symlinked dir). Without `-K` the
|
||||
destination symlink is not followed (the O_NOFOLLOW walk fails the write),
|
||||
which is the safe default.
|
||||
- **`--munge-links`** (sender security rewrite; crosses the wire so the receiver
|
||||
unmunges): every transmitted symlink target is prefixed with the marker
|
||||
`#SYMLINK/`; the receiver strips the marker (only when the negotiated
|
||||
`munge_links` policy is on — a plain `-l` run never strips the prefix, so a
|
||||
source symlink that genuinely begins with `#SYMLINK/` round-trips verbatim)
|
||||
and restores the exact real target. The trust boundary is **symmetric and
|
||||
enforced receiver-side**, independent of the sender: `file_symlink_at_secure`
|
||||
refuses any target that `file_symlink_target_contained` rejects (absolute
|
||||
`/...` or relative with a `..` component), and `file_save_to_disk_full`
|
||||
contains such an entry (skipped) rather than materializing it. A deliberate confinement trade-off: because the receiver
|
||||
enforces containment unconditionally, a plain `-l` (no `--munge-links`) sync
|
||||
*refuses to round-trip a legitimate absolute symlink target* — such target is
|
||||
dropped, never created pointing outside the root. This is a stricter subset of
|
||||
rsync: rsync stores munged targets on the RECEIVING side and depends on both
|
||||
ends running `--munge-links`; FastSync additionally enforces the containment
|
||||
predicate at the receiver regardless of what the sender transmitted. When no
|
||||
symlink is being transmitted (`-l`/`-k`/`-a` off) `--munge-links` has nothing
|
||||
to rewrite and is inert. -*K/`--keep-dirlinks` policy is installed per
|
||||
connection at config-accept (stable for the whole transfer, never racy under
|
||||
`-j`/`--threads`), and only ever follows an in-root symlink-to-directory.*
|
||||
- **`--munge-links`** (sender rewrite; crosses the wire so the receiver
|
||||
unmunges): every transmitted symlink target is prefixed with rsync's marker
|
||||
`SYMLINK_MUNGE_PREFIX` = `/rsyncd-munged/`; the receiver strips the marker
|
||||
(only when the negotiated `munge_links` policy is on — a plain `-l` run never
|
||||
strips the prefix, so a source symlink that genuinely begins with
|
||||
`/rsyncd-munged/` round-trips verbatim) and restores the exact real target.
|
||||
This matches rsync's stored marker and its both-ends-negotiated model, with the
|
||||
prefix applied on the sender rather than the receiver. The link *value* is
|
||||
otherwise stored verbatim; the *placement* path still goes through
|
||||
`file_symlink_at_secure`'s confined fd walk (`has_path_traversal` on the
|
||||
destination path, no symlink follow). When no symlink is being transmitted
|
||||
(`-l`/`-k`/`-a` off) `--munge-links` has nothing to rewrite and is inert.
|
||||
`-K`/`--keep-dirlinks` policy is installed per connection at config-accept
|
||||
(stable for the whole transfer, never racy under `-j`/`--threads`), and only
|
||||
ever follows an in-root symlink-to-directory.
|
||||
|
||||
**Compatibility (byte-identical when all three are absent):** `-k`, `-K` and
|
||||
`--munge-links` are opt-in. Without them the scanner's link handling, the wire
|
||||
frames, and the receiver's writes are unchanged for every other option set, so a
|
||||
run that previously worked continues to behave identically. `-l/--links` itself
|
||||
now transmits targets (the prior behavior was broken/partial); its status moved
|
||||
`⚠️ Partial → ✅ Implemented`.
|
||||
**Compatibility:** `-k`, `-K` and `--munge-links` are opt-in. Without them the
|
||||
scanner's link handling, the wire frames, and the receiver's writes are unchanged
|
||||
for every other option set. `--safe-links`/`--copy-unsafe-links` are applied
|
||||
sender-side; `--trust-sender` no longer changes how symlink targets are stored
|
||||
(it only skips the receiver's path-list re-validation). `-l/--links` stores
|
||||
targets verbatim, matching rsync.
|
||||
|
||||
## 10. Sparse & Device
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-S`, `--sparse` | Sparse block handling | ✅ Implemented | Phase 7 Wave B: real hole preservation with no wire change. The receiver's sparse-aware writer (`write_all_sparse`, next to `write_all` in `src/shared/file.c` and `src/shared/file_store.c`) walks the in-memory file image and emits any all-zero run ≥ 4096 bytes as a hole via `lseek(SEEK_CUR)` (the pre-size `ftruncate` guarantees the offset bookkeeping and logical size), `ftruncate(size)` after the last run pins the final size even with a hole tail. Wired into both the atomic temp+rename store and `--inplace` when `sparse` is set; the non-sparse path is byte-identical to before. **Sparse wins over `--preallocate`** (posix_fallocate is skipped when sparse is set, so the holes are not re-allocated). Interplay note: under `--partial` a retained sparse temp already has the full logical size (trailing content is holes), so `--append`'s "shorter destination" resume does not re-run; the retained file is still valid and a normal re-transfer (or `-W`/delta) repairs it — documented so the combination is never surprising |
|
||||
| `--preallocate` | Allocate dest files before writing | ✅ Implemented | The receiver preallocates the destination file's full expected space before any data is written, so a transfer that would overflow disk fails fast at allocation time (a clean error, not a half-written file) and the file is laid out contiguously, avoiding fragmentation. Crosses the wire (the config frame carries a `preallocate` boolean; `PROTOCOL_VERSION` bumped **2.10.0 → 2.11.0**, peers must match) so the sender knows the receiver will preallocate and the receiver performs it. **Allocation approach:** `posix_fallocate()` is preferred because it reserves *real* disk blocks (true fail-fast on ENOSPC), falling back to plain `ftruncate()` only when the filesystem reports the allocation is unsupported (`EOPNOTSUPP`/`ENOSYS`); `ftruncate` still extends the logical size so the intent degrades gracefully. **Fallback/error semantics:** `EOPNOTSUPP`/`ENOSYS` → clean fallback to `ftruncate` (best-effort, preallocates the logical size and never fails a transfer on filesystems that lack `posix_fallocate`); a genuine allocation failure (`ENOSPC`/`EDQUOT`/`EFBIG`/…) aborts the file/receive with a distinct `preallocate failed ... transfer aborted` error — it does **not** fall back to a normal non-preallocated write, preserving the fail-fast purpose. **Size-known requirement:** preallocation only runs when the final size is already known up front (the normal regular-file case); unknown-length data is skipped (never failed). **Orthogonality:** applies uniformly across the atomic temp+rename store path, `--inplace`, `--partial`/`--partial-dir`, `--delay-updates` (the staged temp file is preallocated before data flows) and the `--link-dest` copy fallback; it neither implies nor conflicts with `-s`, `--append`, or delta. rsync-divergence: rsync signals that `--preallocate` is ignored with `--sparse`; FastSync gives **sparse precedence** — when both are set, `posix_fallocate` is skipped so the holes the sparse writer creates are not re-allocated (the `ftruncate` presize sizing stays), matching the intent of "sparse wins". See the Phase-4 preallocate notes below |
|
||||
| `-S`, `--sparse` | Sparse block handling | ✅ Parity | Phase 7 Wave B: real hole preservation with no wire change. The receiver's sparse-aware writer (`write_all_sparse`, next to `write_all` in `src/shared/file.c` and `src/shared/file_store.c`) walks the in-memory file image and emits any all-zero run ≥ 4096 bytes as a hole via `lseek(SEEK_CUR)` (the pre-size `ftruncate` guarantees the offset bookkeeping and logical size), `ftruncate(size)` after the last run pins the final size even with a hole tail. Wired into both the atomic temp+rename store and `--inplace` when `sparse` is set; the non-sparse path is byte-identical to before. **Sparse wins over `--preallocate`** (posix_fallocate is skipped when sparse is set, so the holes are not re-allocated). Interplay note: under `--partial` a retained sparse temp already has the full logical size (trailing content is holes), so `--append`'s "shorter destination" resume does not re-run; the retained file is still valid and a normal re-transfer (or `-W`/delta) repairs it — documented so the combination is never surprising |
|
||||
| `--preallocate` | Allocate dest files before writing | ⚠️ Caveat | The receiver preallocates the destination file's full expected space before any data is written, so a transfer that would overflow disk fails fast at allocation time (a clean error, not a half-written file) and the file is laid out contiguously, avoiding fragmentation. Crosses the wire (the config frame carries a `preallocate` boolean; `PROTOCOL_VERSION` bumped **2.10.0 → 2.11.0**, peers must match) so the sender knows the receiver will preallocate and the receiver performs it. **Allocation approach:** `posix_fallocate()` is preferred because it reserves *real* disk blocks (true fail-fast on ENOSPC), falling back to plain `ftruncate()` only when the filesystem reports the allocation is unsupported (`EOPNOTSUPP`/`ENOSYS`); `ftruncate` still extends the logical size so the intent degrades gracefully. **Fallback/error semantics:** `EOPNOTSUPP`/`ENOSYS` → clean fallback to `ftruncate` (best-effort, preallocates the logical size and never fails a transfer on filesystems that lack `posix_fallocate`); a genuine allocation failure (`ENOSPC`/`EDQUOT`/`EFBIG`/…) aborts the file/receive with a distinct `preallocate failed ... transfer aborted` error — it does **not** fall back to a normal non-preallocated write, preserving the fail-fast purpose. **Size-known requirement:** preallocation only runs when the final size is already known up front (the normal regular-file case); unknown-length data is skipped (never failed). **Orthogonality:** applies uniformly across the atomic temp+rename store path, `--inplace`, `--partial`/`--partial-dir`, `--delay-updates` (the staged temp file is preallocated before data flows) and the `--link-dest` copy fallback; it neither implies nor conflicts with `-s`, `--append`, or delta. rsync-divergence: rsync signals that `--preallocate` is ignored with `--sparse`; FastSync gives **sparse precedence** — when both are set, `posix_fallocate` is skipped so the holes the sparse writer creates are not re-allocated (the `ftruncate` presize sizing stays), matching the intent of "sparse wins". See the Phase-4 preallocate notes below |
|
||||
|
||||
**Preallocate notes (Phase 4, preallocate wave):** `--preallocate` is implemented as a real receiver-side allocation of the destination file's space before data is written. It is a plain boolean config flag that crosses the wire (serialized in the config frame's selection-options block, mirroring `--inplace`/`--append`/`--force`), so the run requires matching ends: `PROTOCOL_VERSION` was bumped **2.10.0 → 2.11.0** (peers must match or the version check fails). The allocation is performed on the exact destination fd, immediately after it is opened, before any bytes are streamed; `posix_fallocate` (and the `ftruncate` fallback) leave the fd's file offset untouched, so the subsequent data write at offset 0 is unaffected and complete. Because FastSync writes each file's byte payload in one in-memory batch, the "full expected size" is exactly the known `data_size`, which is what gets preallocated. Unknown-length/streamed payloads are skipped rather than failed. A failed allocation logs a distinct `preallocate failed` error and aborts the file (the atomic temp is unlinked, the inplace target is left untrimmed) so the run fails cleanly and never silently degrades to a non-preallocated write — preserving rsync's fail-fast intent on a full disk.
|
||||
|
||||
@@ -589,56 +618,60 @@ now transmits targets (the prior behavior was broken/partial); its status moved
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `--checksum` | Skip based on checksum | ✅ Implemented | With `--incremental`, compares per-file whole-file content digests to skip unchanged files. The digest algorithm is `xxh64` with seed 0 by default and is selectable via `--checksum-choice`/`--cc` (xxh64/xxhash or md5) and `--checksum-seed=NUM` (see those rows); `-c` remains compression |
|
||||
| `--checksum-choice=STR`, `--cc=STR` | Choose checksum algorithm | ✅ Implemented | Real algorithm selection for the per-file whole-file digest used by the `--incremental`/`--checksum` handshake and by the basis-dir content verification. FastSync genuinely supports `xxh64` (the default, exact xxHash64, seeded by `--checksum-seed`) and `md5` (via OpenSSL EVP); `xxhash` is accepted as rsync's spelling of xxHash64. Any other name (md4/sha1/sha256/crc32/none/…) is rejected with a clear error at parse time — never a silent no-op. `--cc` is the alias (`--cc=ALG` and space forms both parse). The algorithm id and seed cross the wire with the config frame, so the receiver hashes its on-disk old file with the SAME algorithm+seed the sender used and both agree on a match; the sender's digest and the receiver's comparison live in the per-file `STATUS_CHECK` handshake, which now carries a length-prefixed, bounded (1..16 byte) digest instead of a fixed 64-bit value, and the receiver pins the received length to the negotiated algorithm's digest length (defense-in-depth: a mismatched/malicious length only forces a safe re-transfer). Note: `md5` is a FIPS-non-approved algorithm, so under an OpenSSL build with FIPS mode enabled `--checksum-choice=md5` fails loudly rather than silently falling back. Protocol/layout: `PROTOCOL_VERSION` bumped **2.9.0 → 2.10.0** (peers must match). Defaults preserve the pre-existing behavior byte-for-byte (xxh64, seed 0). Like rsync, the choice only takes effect where a whole-file digest is actually computed (`--checksum` on, or a basis-dir flag); it does not itself enable `--checksum`. Closely-related divergence: the delta BLOCK strong checksum (§11 delta) stays xxHash32 — `--checksum-choice` selects only the whole-file digest, matching rsync where the per-block checksum is independent of the whole-file checksum choice |
|
||||
| `--compare-dest=DIR` | Compare dest files relative to DIR | ✅ Implemented | DIR is a receiver-side basis relative to the destination root (confined below it; absolute/`..`/`.` rejected, `//` collapsed and trailing `/` dropped). On the receiver's per-file check (implies `--incremental`) an exact match = same size + mtime (unless `--size-only`; `-I` disables matching) **and** equal xxHash64 of the sender's file; a match suppresses the data transfer. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Divergences: when the destination already holds a *different* version rsync deletes it but FastSync instead transfers the data (keeps the mirror complete; never deletes without `--delete`); attribute-only differences on a match are not re-applied (data is skipped so the sender never sends metadata); content is verified by xxHash64, stricter than rsync's default quick check. Sizing: FastSync's whole-file payload limit is 256 MiB on **every** transfer path (not basis-specific); rsync applies basis dirs to arbitrary sizes, so FastSync refuses a basis run whose source contains a larger file up front with a clear error before any transfer. Wire: a basis-count field is always present on the config frame (protocol 2.9.0, so clients and servers must both be 2.9.0) |
|
||||
| `--copy-dest=DIR` | Include copies of unchanged files | ✅ Implemented | Same basis rules as `--compare-dest`, but an exact match materializes a **local copy** of the DIR file into the destination (via the normal atomic temp+rename store path, so `--existing`/`--ignore-existing`/`--update`/`--backup`/`--delay-updates` all still apply) instead of transferring data. Repeatable; command-line order = priority. Content is xxHash64-verified before the copy. Divergences: a basis-hit destination keeps the basis file's own mode/uid/gid and mtime (the sender sends no metadata on a skip), so with `--size-only` its mtime can differ from the source and attribute-only differences are copied with the basis attributes rather than rsync's "copy + fix attributes". Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
|
||||
| `--link-dest=DIR` | Hardlink to files when unchanged | ✅ Implemented | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy, never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Content is xxHash64-verified before linking. Divergences and caveats: an already up-to-date destination file is not re-linked to a basis file (only files that would otherwise be written are linked); a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); with `--size-only` the linked mtime can differ from the source; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
|
||||
| `-y`, `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ✅ Implemented | `-y/--fuzzy` is a pure bandwidth optimization on the existing receiver-driven delta path: when a file must be transferred and the destination holds no usable content at the exact path (file absent, or the destination file is outside the delta engine's size bounds), the receiver searches the SAME destination directory for an existing regular file whose basename is similar to the incoming name and uses it as the delta basis, so the sender transmits only the differences instead of the whole file. The output is always byte-exact regardless of which (or whether any) basis is chosen. Decision location: the receiver performs the candidate search inside `receive_incremental_check` and sends the normal `STATUS_DELTA_SIGNATURE`; the sender never learns the basis was a different file, so no new frame type or sender logic was needed — only the config frame grew a `fuzzy` boolean, so `PROTOCOL_VERSION` was bumped **2.8.0 → 2.9.0** (peers must match). Similarity heuristic (deterministic, simpler than rsync's deliberately-fuzzy matching, and documented precisely): candidates are the target's sibling entries in its destination directory, opened `O_NOFOLLOW`/`AT_SYMLINK_NOFOLLOW` under the confined root (symlinks never followed; nothing outside the destination root is ever read or hashed); dotfiles, directories, the target's own name, and the `.fastsync-stage`/temp scratch names are excluded; like the ordinary delta path, the block signature the receiver transmits is derived from on-disk content it may not otherwise send, so a negotiated `--fuzzy` run exposes the destination's sibling files (at block granularity) to the sender as a known-plaintext oracle — the same information class as the normal delta handshake over the file being replaced; the size gate is the delta engine's own bounds (both files ≥ 16 KiB, ≤ `--delta-max`, ratio ≤ 10×) rather than rsync's ~1.5× size window; the name gate is a Levenshtein edit distance between the basenames accepted only when ≤ half the length of the longer basename; the single best candidate (smallest distance, tie-break size closest to the incoming file then lexicographically smaller basename) is read; the directory scan is capped at 4096 entries so a pathological directory cannot stall a transfer. When fuzzy applies: only to files the receiver would otherwise send whole — the destination's own file is always preferred as the delta basis when it exists and fits the delta size bounds, so fuzzy does NOT replace an existing-but-different destination basis; FastSync's 10× delta size-ratio bound means an existing destination file that is too far away in size still lets the fuzzy search run. When no similar candidate exists the transfer falls back to the normal whole-file transfer. rsync-divergence note: rsync's own matching uses a fuzzy name/size rule set; FastSync implements the closest safe deterministic approximation above. Because FastSync's delta machinery is off by default (rsync's is on), `--fuzzy` implies `--incremental` + `--delta` (unless `--whole-file`/`-W` or an explicit `--no-delta` switched delta off, in which case fuzzy is inert — matching rsync where `--whole-file` makes fuzzy irrelevant). Unlike the basis-dir options, `--fuzzy` honors an explicit `--no-incremental` (it does not force the handshake back on); an explicit `--no-incremental` also suppresses the delta implication so no invalid `--delta requires --incremental` config results. `--no-fuzzy` negates it. All surrounding semantics are untouched: a fuzzy-reconstructed file is stored as a normal file, so `--remove-source-files`, itemize/`-i`, `--stats`, `--backup`, `--delay-updates`, `--existing`/`--ignore-existing`/`--update` behave exactly as for a whole-file transfer (the fuzzy delta does not skip the file) |
|
||||
| `--checksum` | Skip based on checksum | ✅ Parity | `-c`/`--checksum` compares per-file whole-file content digests to skip unchanged files. **As of protocol 2.23.0 the short `-c` implies the checksum quick-check**, so a plain `-c` run verifies content rather than only affecting the `--incremental` handshake. The digest algorithm is `xxh64` by default and is selectable via `--checksum-choice`/`--cc` (`xxh64`/`xxhash`/`xxh3`/`xxh128`/`md5`/`auto`) and `--checksum-seed=NUM` (see those rows) |
|
||||
| `--checksum-choice=STR`, `--cc=STR` | Choose checksum algorithm | ⚠️ Caveat | Real algorithm selection for the per-file whole-file digest used by the `--incremental`/`--checksum` handshake and by the basis-dir content verification. **Protocol 2.23.0 accepts `xxh64` (the default), `xxhash` (rsync's spelling of xxHash64), `xxh3`, `xxh128`, `md5`, and `auto` (which selects FastSync's default).** rsync choices FastSync does not implement — `md4`, `sha1`, `none`, and the two-name `transfer,pre-transfer` form — are **rejected by name** with a clear error at parse time, never a silent no-op. `--cc` is the alias (`--cc=ALG` and space forms both parse). The algorithm id and seed cross the wire with the config frame, so the receiver hashes its on-disk old file with the SAME algorithm+seed the sender used and both agree on a match; the sender's digest and the receiver's comparison live in the per-file `STATUS_CHECK` handshake, which carries a length-prefixed, bounded (1..16 byte) digest, and the receiver pins the received length to the negotiated algorithm's digest length (defense-in-depth: a mismatched/malicious length only forces a safe re-transfer). Digest lengths: `xxh64`/`xxh3` = 8 bytes, `xxh128`/`md5` = 16. Note: `md5` is a FIPS-non-approved algorithm, so under an OpenSSL build with FIPS mode enabled `--checksum-choice=md5` fails loudly rather than silently falling back. `PROTOCOL_VERSION` has moved well past the original 2.10.0 digest-frame bump. Like rsync, the choice only takes effect where a whole-file digest is actually computed (`--checksum` on, or a basis-dir flag). Closely-related divergence: the delta BLOCK strong checksum stays xxHash32 — `--checksum-choice` selects only the whole-file digest, matching rsync where the per-block checksum is independent of the whole-file choice |
|
||||
| `--compare-dest=DIR` | Compare dest files relative to DIR | ⚠️ Caveat | DIR is a receiver-side basis relative to the destination root (confined below it; absolute/`..`/`.` rejected, `//` collapsed and trailing `/` dropped). On the receiver's per-file check (implies `--incremental`) an exact match = same size + mtime (unless `--size-only`; `-I` disables matching) **and** equal xxHash64 of the sender's file; a match suppresses the data transfer. compare-dest never copies: it only skips a file the destination does **not** already hold (sparse destination, rsync parity), and is consulted before the normal delta/full paths. Repeatable; searched in command-line order, first match wins. Divergences: when the destination already holds a *different* version rsync deletes it but FastSync instead transfers the data (keeps the mirror complete; never deletes without `--delete`); attribute-only differences on a match are not re-applied (data is skipped so the sender never sends metadata); content is verified by xxHash64, stricter than rsync's default quick check. Sizing: FastSync's whole-file payload limit is 256 MiB on **every** transfer path (not basis-specific); rsync applies basis dirs to arbitrary sizes, so FastSync refuses a basis run whose source contains a larger file up front with a clear error before any transfer. Wire: a basis-count field is always present on the config frame (protocol 2.9.0, so clients and servers must both be 2.9.0) |
|
||||
| `--copy-dest=DIR` | Include copies of unchanged files | ⚠️ Caveat | Same basis rules as `--compare-dest`, but an exact match materializes a **local copy** of the DIR file into the destination (via the normal atomic temp+rename store path, so `--existing`/`--ignore-existing`/`--update`/`--backup`/`--delay-updates` all still apply) instead of transferring data. Repeatable; command-line order = priority. Content is xxHash64-verified before the copy. Divergences: a basis-hit destination keeps the basis file's own mode/uid/gid and mtime (the sender sends no metadata on a skip), so with `--size-only` its mtime can differ from the source and attribute-only differences are copied with the basis attributes rather than rsync's "copy + fix attributes". Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
|
||||
| `--link-dest=DIR` | Hardlink to files when unchanged | ⚠️ Caveat | Same basis rules as `--copy-dest`, but an exact match installs an atomic **hard link** to the DIR file (temp hard link + rename) so no data or disk space is used; where the link is impossible (basis on another filesystem, filesystem refuses links) it falls back cleanly to a byte-identical local copy, never a corrupt/partial file. `--delay-updates` stages the link and publishes by rename, so the final entry stays a real hard link. Repeatable (searched in command-line order, first match wins). Content is xxHash64-verified before linking. Divergences and caveats: an already up-to-date destination file is not re-linked to a basis file (only files that would otherwise be written are linked); a link keeps the basis inode's own mode/uid/gid and mtime — metadata is never written through the shared inode (that would mutate the basis file), so a later `--inplace` run that rewrites such a destination path **will mutate the basis snapshot** through the shared inode (use `--copy-dest` when the destination must stay independently writable); with `--size-only` the linked mtime can differ from the source; a `--remove-source-files` source satisfied by a basis dir is treated as skipped and therefore **retained** (never removed); basis dirs are excluded from `--delete`. Requires `--incremental` (implied); incompatible with `-s`. Wire: protocol 2.9.0 |
|
||||
| `-y`, `--fuzzy`, `--no-fuzzy` | Find similar file for basis | ⚠️ Caveat | `-y/--fuzzy` is a pure bandwidth optimization on the existing receiver-driven delta path: when a file must be transferred and the destination holds no usable content at the exact path (file absent, or the destination file is outside the delta engine's size bounds), the receiver searches the SAME destination directory for an existing regular file whose basename is similar to the incoming name and uses it as the delta basis, so the sender transmits only the differences instead of the whole file. The output is always byte-exact regardless of which (or whether any) basis is chosen. Decision location: the receiver performs the candidate search inside `receive_incremental_check` and sends the normal `STATUS_DELTA_SIGNATURE`; the sender never learns the basis was a different file, so no new frame type or sender logic was needed — only the config frame grew a `fuzzy` boolean, so `PROTOCOL_VERSION` was bumped **2.8.0 → 2.9.0** (peers must match). Similarity heuristic (deterministic, simpler than rsync's deliberately-fuzzy matching, and documented precisely): candidates are the target's sibling entries in its destination directory, opened `O_NOFOLLOW`/`AT_SYMLINK_NOFOLLOW` under the confined root (symlinks never followed; nothing outside the destination root is ever read or hashed); dotfiles, directories, the target's own name, and the `.fastsync-stage`/temp scratch names are excluded; like the ordinary delta path, the block signature the receiver transmits is derived from on-disk content it may not otherwise send, so a negotiated `--fuzzy` run exposes the destination's sibling files (at block granularity) to the sender as a known-plaintext oracle — the same information class as the normal delta handshake over the file being replaced; the size gate is the delta engine's own bounds (both files ≥ 16 KiB, ≤ `--delta-max`, ratio ≤ 10×) rather than rsync's ~1.5× size window; the name gate is a Levenshtein edit distance between the basenames accepted only when ≤ half the length of the longer basename; the single best candidate (smallest distance, tie-break size closest to the incoming file then lexicographically smaller basename) is read; the directory scan is capped at 4096 entries so a pathological directory cannot stall a transfer. When fuzzy applies: only to files the receiver would otherwise send whole — the destination's own file is always preferred as the delta basis when it exists and fits the delta size bounds, so fuzzy does NOT replace an existing-but-different destination basis; FastSync's 10× delta size-ratio bound means an existing destination file that is too far away in size still lets the fuzzy search run. When no similar candidate exists the transfer falls back to the normal whole-file transfer. rsync-divergence note: rsync's own matching uses a fuzzy name/size rule set; FastSync implements the closest safe deterministic approximation above. Because FastSync's delta machinery is off by default (rsync's is on), `--fuzzy` implies `--incremental` + `--delta` (unless `--whole-file`/`-W` or an explicit `--no-delta` switched delta off, in which case fuzzy is inert — matching rsync where `--whole-file` makes fuzzy irrelevant). Unlike the basis-dir options, `--fuzzy` honors an explicit `--no-incremental` (it does not force the handshake back on); an explicit `--no-incremental` also suppresses the delta implication so no invalid `--delta requires --incremental` config results. `--no-fuzzy` negates it. All surrounding semantics are untouched: a fuzzy-reconstructed file is stored as a normal file, so `--remove-source-files`, itemize/`-i`, `--stats`, `--backup`, `--delay-updates`, `--existing`/`--ignore-existing`/`--update` behave exactly as for a whole-file transfer (the fuzzy delta does not skip the file) |
|
||||
|
||||
## 12. Compression
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-z`, `--compress` | Compress file data | ✅ Implemented | Always uses zstd (rsync supports multiple algorithms — a documented divergence, selectable via `--compress-choice`). Phase 7 Wave A: `-z` is now the compression short form; `-c` is rsync's `--checksum` |
|
||||
| `--compress-choice=STR`, `--zc=STR` | Choose compression algorithm | ✅ Implemented | FastSync supports `zstd` and `none` |
|
||||
| `--compress-level=NUM`, `--zl=NUM` | Set compression level | ✅ Implemented | 1-22, default 5 |
|
||||
| `--compress-threads=NUM` | Set compression threads | ✅ Implemented | `compression_threads` config field (client-only; does not cross the wire). Sets the number of worker threads used by the zstd compression pool to NUM (1..64; 0/garbage/oversized rejected up front). Accepted in both `--compress-threads=NUM` and two-argument `--compress-threads NUM` forms. Composes with `-z`/compression; under the `-j`/`--threads` multithreaded pipeline it parallelizes compressed chunk encoding. See test_tcp.py `-z --compress-threads=2` and test_client_cli.c |
|
||||
| `--skip-compress=LIST` | Skip compress for suffixes | ✅ Implemented | Comma-separated, case-insensitive suffix list; empty list skips none; incompatible with FastSync chunk serialization (`-s`) |
|
||||
| `-z`, `--compress` | Compress file data | ⚠️ Caveat | Streaming zstd (rsync supports multiple algorithms — a documented divergence, selectable via `--compress-choice`). `-z` is the compression short form; `-c` is rsync's `--checksum`. `--skip-compress` applies rsync 3.4.1's default suffix list when no list is given |
|
||||
| `--compress-choice=STR`, `--zc=STR` | Choose compression algorithm | ⚠️ Caveat | FastSync supports `zstd` (default), `none`, and `auto`. rsync's other compiled-in choices (`lz4`, `zlib`, `zlibx`) are **rejected by name** at parse time with a clear error, never silently ignored. `--zc` is the alias |
|
||||
| `--compress-level=NUM`, `--zl=NUM` | Set compression level | ✅ Parity | 1-22, default 5 |
|
||||
| `--compress-threads=NUM` | Set compression threads | ✅ Parity | `compression_threads` config field (client-only; does not cross the wire). Sets the number of worker threads used by the zstd compression pool to NUM (1..64; 0/garbage/oversized rejected up front). Accepted in both `--compress-threads=NUM` and two-argument `--compress-threads NUM` forms. Composes with `-z`/compression; under the `-j`/`--threads` multithreaded pipeline it parallelizes compressed chunk encoding. See test_tcp.py `-z --compress-threads=2` and test_client_cli.c |
|
||||
| `--skip-compress=LIST` | Skip compress for suffixes | ⚠️ Caveat | Comma-separated (or `/`-separated, as in rsync) case-insensitive suffix list; a leading dot is optional; an empty list skips none. **When the option is omitted, rsync 3.4.1's built-in default suffix list applies** (`3g2 3gp 7z aac … zip zst`); an explicit list replaces that default entirely, matching rsync. A user-supplied list is a client-side compression choice; incompatible with FastSync chunk serialization (`-s`) |
|
||||
|
||||
## 13. Connectivity
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `-e`, `--rsh=COMMAND` | Remote shell to use | ✅ Implemented | `-e`/`--rsh` (and `--rsh=COMMAND`) select the remote-shell program used to build the SSH child argv, overriding the default `ssh`. The command is whitespace-split into the leading argv words so rsync's `-e "ssh -p 2222"` works; the standard `-o` family, an optional `-p` port, `user@host` and the quoted remote command (`fastsync-server --stdio`) follow. Stored in the `rsh_command` config field. **Client-only, never crosses the wire** (it is a launch concern, not a handshake property) |
|
||||
| `--rsync-path=PROGRAM` | rsync binary on remote | ✅ Implemented | Alias for `--fastsync-server-path`: both write the `fastsync_server_path` config field used as the remote-side server program (always quoted as one remote-shell word), which CROSSES the wire as before. Kept separate from `--rsh`, which names the local connecting program |
|
||||
| `--port=PORT` | Alternate daemon port | ✅ Implemented | rsync's daemon-port flag maps to the client-side `server_port` config field: a client connects to a TCP/TLS server (incl. `host::module/path` daemon destinations) with `--server-port`, and the `fastsync-server --daemon` listener's port is taken from its config's `port` key (default 873) or overridden by `--dparam port=` / `-p` |
|
||||
| `--sockopts=OPTIONS` | Custom TCP options | ✅ Implemented | Comma-separated allowlist of `OPT=VAL` applied via `setsockopt` after `socket()` before `connect()`/`bind()`. Only `TCP_NODELAY`, `SO_KEEPALIVE`, `SO_REUSEADDR` (0/1) and `SO_RCVBUF`/`SO_SNDBUF` (byte count) are accepted; an unknown option name or a bad value is rejected up front, never silently ignored. A value is required for every option (`OPT=VAL`; a bare name is an error). Applied to the outgoing TCP and TLS client socket; absent by default. `SockOptEntry`/`sockopts` config fields. Local socket concern: never crosses the wire |
|
||||
| `--blocking-io` | Use blocking I/O for remote shell | ✅ Implemented | With `--blocking-io` the SSH-transport socketpair socket is left without `SO_RCVTIMEO`/`SO_SNDTIMEO`, so the transfer blocks naturally; by default it gets the same read/write timeout as the TCP transport (see `--timeout`). `blocking_io` config bool. **Client-only, never crosses the wire** |
|
||||
| `--outbuf=N\|L\|B` | Set output buffering | ✅ Implemented | `N` (none/unbuffered) → `_IONBF`, `L` (line) → `_IOLBF`, `B` (block, the default) → `_IOFBF` via `setvbuf` on stdout and stderr. Garbage values are rejected. `outbuf` config field (`OutbufMode`). **Client-only, never crosses the wire** |
|
||||
| `--address=ADDRESS` | Bind address for outgoing socket | ✅ Implemented | Binds the outgoing client socket to a local source address before `connect()` (resolved with the same `-4`/`-6` family hints as the destination). Local socket concern: never crosses the wire |
|
||||
| `-4`, `--ipv4` | Prefer IPv4 | ✅ Implemented | Forces `AF_INET` in the `getaddrinfo` hints for client destination/source resolution and the server bind (see the Phase 5, Wave B note). Mutually exclusive with `-6` |
|
||||
| `-6`, `--ipv6` | Prefer IPv6 | ✅ Implemented | Forces `AF_INET6` in the `getaddrinfo` hints for client destination/source resolution and the server bind. Mutually exclusive with `-4` |
|
||||
| `--remote-option=OPT`, `-M` | Send an option only to the remote side | ✅ Implemented | Each value is appended to the remote server invocation over SSH as an individually single-quote-escaped shell word in `ssh_build_remote_command()`. Values are validated (non-empty, no control characters) and shell metacharacters cannot break out of the quoting (`;`, `&`, `|`, <code>`</code>, `$`, `(`, `)`, quotes are neutralized), so a value cannot inject an arbitrary remote command and a subsequent `--` on the client line cannot be turned into one. The options never cross the binary config frame. Phase 7 Wave A: the short `-M` form is now available (as `-M OPT` and `-M=OPT`), matching rsync; metadata mode moved to long-only `--preserve` |
|
||||
| `-e`, `--rsh=COMMAND` | Remote shell to use | ✅ Parity | `-e`/`--rsh` (and `--rsh=COMMAND`) select the remote-shell program used to build the SSH child argv, overriding the default `ssh`. The command is whitespace-split into the leading argv words so rsync's `-e "ssh -p 2222"` works; the standard `-o` family, an optional `-p` port, `user@host` and the quoted remote command (`fastsync-server --stdio`) follow. Stored in the `rsh_command` config field. **Client-only, never crosses the wire** (it is a launch concern, not a handshake property) |
|
||||
| `--rsync-path=PROGRAM` | rsync binary on remote | ✅ Parity | Alias for `--fastsync-server-path`: both write the `fastsync_server_path` config field used as the remote-side server program. The path is always quoted as one remote-shell word in the SSH argv. **Client-only: `fastsync_server_path` never crosses the wire** (it is a launch concern, not a handshake property), matching rsync, where `--rsync-path` likewise names the remote program locally. Kept separate from `--rsh`, which names the local connecting program |
|
||||
| `--port=PORT`, `--port PORT` | Alternate daemon port | ✅ Parity | rsync's daemon-port flag is an alias for `--server-port`: both spellings (and `--server-port=PORT`) map to the client-side `server_port` config field. The client connects to a TCP/TLS server (incl. `host::module/path` daemon destinations) on that port, and the `fastsync-server --daemon` listener's port is taken from its config's `port` key (default 873) or overridden by `--dparam port=` / `-p` |
|
||||
| `--sockopts=OPTIONS` | Custom TCP options | ✅ Parity | Comma-separated allowlist of `OPT=VAL` applied via `setsockopt` after `socket()` before `connect()`/`bind()`. Only `TCP_NODELAY`, `SO_KEEPALIVE`, `SO_REUSEADDR` (0/1) and `SO_RCVBUF`/`SO_SNDBUF` (byte count) are accepted; an unknown option name or a bad value is rejected up front, never silently ignored. A value is required for every option (`OPT=VAL`; a bare name is an error). Applied to the outgoing TCP and TLS client socket; absent by default. `SockOptEntry`/`sockopts` config fields. Local socket concern: never crosses the wire |
|
||||
| `--blocking-io` | Use blocking I/O for remote shell | ✅ Parity | With `--blocking-io` the SSH-transport socketpair socket is left without `SO_RCVTIMEO`/`SO_SNDTIMEO`, so the transfer blocks naturally; by default it gets the same read/write timeout as the TCP transport (see `--timeout`). `blocking_io` config bool. **Client-only, never crosses the wire** |
|
||||
| `--timeout=SEC`, `--contimeout=SEC` | Set I/O / connect timeouts | ✅ Parity | Protocol 2.23.0 matches rsync's defaults: **`--timeout` defaults to 0 (I/O deadlines disabled) and `--contimeout` to 60 s; `0` disables either.** A positive `--timeout` bounds both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline on the client; the server floors its session deadline so a client `0` can never hold a session open forever. `--no-timeout`/`--no-contimeout` are the negations. Both are client-side deadlines and are not sent on the wire |
|
||||
| `--outbuf=N\|L\|B` | Set output buffering | ✅ Parity | `N` (none/unbuffered) → `_IONBF`, `L` (line) → `_IOLBF`, `B` (block, the default) → `_IOFBF` via `setvbuf` on stdout and stderr. Garbage values are rejected. `outbuf` config field (`OutbufMode`). **Client-only, never crosses the wire** |
|
||||
| `--address=ADDRESS` | Bind address for outgoing socket | ✅ Parity | Binds the outgoing client socket to a local source address before `connect()` (resolved with the same `-4`/`-6` family hints as the destination). Local socket concern: never crosses the wire |
|
||||
| `-4`, `--ipv4` | Prefer IPv4 | ✅ Parity | Forces `AF_INET` in the `getaddrinfo` hints for client destination/source resolution and the server bind (see the Phase 5, Wave B note). Mutually exclusive with `-6` |
|
||||
| `-6`, `--ipv6` | Prefer IPv6 | ✅ Parity | Forces `AF_INET6` in the `getaddrinfo` hints for client destination/source resolution and the server bind. Mutually exclusive with `-4` |
|
||||
| `--remote-option=OPT`, `-M` | Send an option only to the remote side | ⚠️ Caveat | Each value is appended to the remote server invocation over SSH as an individually single-quote-escaped shell word in `ssh_build_remote_command()`. Values are validated (non-empty, no control characters) and shell metacharacters cannot break out of the quoting (`;`, `&`, `\|`, <code>`</code>, `$`, `(`, `)`, quotes are neutralized), so a value cannot inject an arbitrary remote command and a subsequent `--` on the client line cannot be turned into one. The short `-M` form (`-M OPT`, `-M=OPT`, and rsync-style attached `-MOPT`) is available, matching rsync; metadata mode moved to long-only `--preserve`. **Divergence:** `-M` is only meaningful for the SSH transport (`user@host:path`); a daemon (`host::module/path`) or local TCP destination **rejects** it (there is no remote command line to append to), whereas rsync applies it to its own remote process on every transport. The options never cross the binary config frame |
|
||||
|
||||
## 14. Daemon Mode
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `--daemon` | Run as rsync daemon | ✅ Implemented | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding |
|
||||
| `--config=FILE` | Alternate rsyncd.conf file | ✅ Implemented | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` |
|
||||
| `--dparam=OVERRIDE` | Override global daemon config | ✅ Implemented | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global scalar keys the grammar defines (`port`, `motd file`, `address`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` |
|
||||
| `--no-detach` | Don't detach from parent | ✅ Implemented | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` |
|
||||
| `--password-file=FILE` | Read daemon password from file | ✅ Implemented | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat |
|
||||
| `--early-input=FILE` | Use FILE for daemon early exec | ✅ Implemented | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) |
|
||||
| `--hash-credentials=FILE`, `--iterations N` | Hash a plaintext credential file | ✅ Implemented | Server-only offline tool (A7): reads the `user:password` lines of FILE (same owner-only 0600 check) and prints one new-format store line per entry to stdout, then exits. `--iterations` sets the PBKDF2 work factor (default 600000, range 100000–10000000). Dependency-free and does not run a listener. Use its output as `--password-file` for `--daemon`. There is no auto-upgrade: a legacy store line is hard-rejected by the loader and must be regenerated |
|
||||
| `--daemon` | Run as rsync daemon | ⚠️ Caveat | Wave A: a real persistent listener. `fastsync-server --daemon --config FILE` (plus `--no-detach` to stay foreground; without it the listener detaches to the background after binding) reads a FastSync-native module config file and serves each connection confined to the requested module's `path` root (never a client-chosen root; every client-chosen-ownership/super-user request (`--numeric-ids`/`--chown`/`--usermap`/`--groupmap`/`--fake-super`/`--copy-as`/explicit `--super`) is refused unless the module opts in with `client owner = yes`, and the operator `--no-super` veto is honored). TCP/TLS via the existing `--tls` stack; plaintext still requires `--allow-unauthenticated` (same secure default as the standalone server). Client destinations use rsync's `host::module/path` form. Wire/protocol: the config frame gained a trailing daemon-module string and `PROTOCOL_VERSION` was bumped **2.14.0 → 2.15.0** (see the Daemon Mode notes below). Daemon mode is built in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding |
|
||||
| `--config=FILE` | Alternate rsyncd.conf file | ⚠️ Caveat | Wave A: selects the daemon config file. Default when omitted (in `--daemon` mode): `~/.config/fastsync/fastsyncd.conf` if it exists, else `/etc/fastsyncd.conf`. The grammar is FastSync-native (documented in the Daemon Mode notes below) and strictly rejects unknown keys so a typo can never silently change what a module serves; requires `--daemon` |
|
||||
| `--dparam=OVERRIDE` | Override global daemon config | ⚠️ Caveat | Wave A: overrides one global scalar from the command line (`--dparam port=8734` and `--dparam=KEY=VALUE` both work). Limited to the global keys the grammar defines (`port`, `motd file`, `address`, `max connections`, `max connections per host`, `auth failure delay`, `auth lockout threshold`, `auth lockout duration`, `hosts allow`, `hosts deny`); keys are case-insensitive and unknown keys/invalid values are rejected. Requires `--daemon` |
|
||||
| `--no-detach` | Don't detach from parent | ✅ Parity | Wave A: with `--daemon`, keeps the listener in the foreground (what integration tests use). Without it the daemonizes (fork/setsid, stdio redirected to /dev/null) after the listening socket is bound. Requires `--daemon` |
|
||||
| `--password-file=FILE` | Read daemon password from file | ⚠️ Caveat | A7 daemon auth. Client: `--password-file` supplies `user:password` for a `host::module/path` destination (the username is taken from this file, so `user@host::module` stays rejected); the literal password is held client-side only for the SCRAM handshake and wiped at teardown. Server (`fastsync-server --daemon --password-file FILE`): the salted-PBKDF2 verifier store that modules with `auth users` are verified against. **Neither the password nor any replayable bearer value crosses the wire or is stored server-side** — the store holds a per-user salt plus derived keys, and the daemon proves the secret with a per-connection nonce challenge. The file must be private to its owner: both the client and server verify the exact inode they read (open-then-`fstat`, so the check cannot be raced) and refuse a `--password-file`/`--early-input` that is not owned by the current user or grants any group/other permission bit (mode 0600), mirroring the TLS private-key check. A process-substitution pipe (`--early-input <(vault ...)`) is still accepted when it satisfies those checks. See the Daemon Mode notes below for the file formats and the plaintext/TLS caveat |
|
||||
| `--early-input=FILE` | Use FILE for daemon early exec | ⚠️ Caveat | Server-only (requires `--daemon`): a second credential-store file, same new-format grammar as `--password-file`, read before the listener accepts connections (a secrets-manager / process-substitution source). Its entries layer over `--password-file`: byte-identical verifiers dedupe, a conflicting verifier for the same user is a startup error. A daemon whose modules declare `auth users` must be given at least one of the two, or it refuses to start (fail closed) |
|
||||
| `--hash-credentials=FILE`, `--iterations N` | Hash a plaintext credential file | ⚠️ Caveat | Server-only offline tool (A7): reads the `user:password` lines of FILE (same owner-only 0600 check) and prints one new-format store line per entry to stdout, then exits. `--iterations` sets the PBKDF2 work factor (default 600000, range 100000–10000000). Dependency-free and does not run a listener. Use its output as `--password-file` for `--daemon`. There is no auto-upgrade: a legacy store line is hard-rejected by the loader and must be regenerated |
|
||||
|
||||
**Daemon Mode notes (Wave A protocol 2.15.0; A7 auth protocol 2.19.0; MOTD no bump):** FastSync daemon mode is supported in FastSync's own protocol/config grammar, not rsync's SMB/daemon option encoding.
|
||||
|
||||
- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars). Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves.
|
||||
- **Config grammar** (`fastsyncd.conf`): line-based; an implicit global section first, then `[module]` sections. Keys are case-insensitive, values are trimmed and may be wrapped in one layer of double quotes (`path = "/srv/my dir"`). `#` and `;` at the start of a line (after leading whitespace) are full-line comments; inline comments and `\` continuations are not supported. Lines are bounded (4096 chars), and at most 256 `[module]` sections are accepted. Global keys: `port` (default 873), `motd file` (the daemon sends its bounded, escaped content to a client after the module gate/auth accepts, unless the client passes `--no-motd`), `address` (optional bind address), `max connections` (positive integer cap on concurrent connections, default 100; 0/negative/garbage is a parse error), `max connections per host` (concurrent-connection cap per source IP, default 0 = unlimited), `auth failure delay` (milliseconds to sleep after a failed authentication, default 500; 0 disables, capped at 5000), `auth lockout threshold` (failed authentications from one source before lockout, default 10; 0 disables), `auth lockout duration` (seconds a locked-out source is refused, default 300), `hosts allow` and `hosts deny` (comma- and/or whitespace-separated host access patterns — see the host access control note below). Module keys: `path` (required; the daemon-side authorized root for that module), `read only` (yes/no/true/false/1/0, default no), `client owner` (yes/no/true/false/1/0, default no; opts the module into client-chosen ownership — see below), `auth users` (comma list), `max connections` (optional per-module cap, 0 = unlimited; enforced across all connection children), `hosts allow`/`hosts deny` (per-module host access lists). **Unknown keys and malformed lines are parse-and-reject errors** (never silently ignored), so a typo cannot change what a module serves.
|
||||
- **Host access control (`hosts allow`/`hosts deny`):** both keys accept a comma- and/or whitespace-separated list of patterns and may appear globally and/or per module (multiple config-file lines append; a `--dparam` override replaces). Supported patterns are `*` (match all), an IPv4 or IPv6 literal (`10.0.0.1`, `2001:db8::1`), and an IPv4/IPv6 CIDR (`10.0.0.0/8`, `2001:db8::/32`). Hostname patterns are **not** supported: because the peer is always a numeric address and no reverse DNS is performed, a hostname/glob pattern would silently never match, so it is rejected at load time (fail-closed) instead of being accepted as a dead rule. An IPv4 peer on a dual-stack IPv6 listener is normalized from its `::ffff:a.b.c.d` form so IPv4 patterns match it. rsync-like semantics: a matching `hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of them is rejected; deny takes precedence over allow. The daemon enforces the global list first, then the selected module's list, **before authentication** in `server_module_gate`, with an audit log line naming the peer, the module and the outcome. The numeric peer address is obtained with `getpeername`+`inet_ntop` (`utils_fd_peer_ip`, handling both address families); when it cannot be obtained a module with any ACL fails closed (refused), while an ACL-free module continues and logs at debug. A malformed pattern (e.g. an out-of-range CIDR prefix) is a parse error at load time.
|
||||
- **Connection caps, shared registry and auth lockout:** the global `max connections` key (default 100) is plumbed into the listener (`transport_tcp.c`), which rejects a connection once the accept-loop parent's active-child count reaches it; the IPv4/IPv6 peer is logged for every accepted connection. Because the listener forks one child per connection, the per-module `max connections` cap, the global `max connections per host` cap, and the auth-failure counter live in a fixed-size registry carved from an anonymous shared mapping (`daemon_limits.c`, `mmap(MAP_SHARED|MAP_ANONYMOUS)`) created by the parent before the accept loop, so every forked child shares the same counters (C11 atomics only — never a pthread lock, which can deadlock in a forked child). The parent reserves a registry slot per accepted connection and the child records the selected module and source IP once known; the parent's `SIGCHLD` handler reclaims the slot when the child dies (including `SIGKILL`) and re-derives the per-module and per-source occupancy counts from the surviving REGISTERED slots, so a child killed mid-registration cannot leak a count. The per-source table has a bounded lifetime: an entry with no live connection is reclaimed after its lockout expires or it has been idle (300 s); if the table is genuinely full the per-source cap/lockout fails open for new sources (per-module cap and ACLs still apply) with a rate-limited warning. The per-module cap (0 = unlimited) is enforced after the module lookup and before auth; per-source identity reuses the normalized numeric peer address (`utils_fd_peer_ip`, IPv4-mapped IPv6 collapsed to IPv4), and a trusted loopback peer (127.0.0.0/8 / `::1`, `utils_fd_peer_is_local`) is exempt from the per-source cap and the auth lockout because all local clients share one address (the per-module/global caps still apply). Clients behind a shared NAT/proxy address likewise share one per-source budget and lockout counter. A failed authentication increments the shared per-source failure count and, once `auth lockout threshold` (default 10; 0 disables) is reached, the source is refused for `auth lockout duration` seconds (default 300) before any challenge is sent, even when the next attempt is handled by a different forked child; a successful authentication clears the counter. On a failed authentication the per-connection child still sleeps the global `auth failure delay` (default 500 ms, 0 disables, capped at 5000) via `nanosleep`, rate-limiting online guessing without delaying a success. A missing registry (allocation failure) degrades to the global cap and host ACLs rather than refusing to start.
|
||||
- **Module selection & confinement:** the client requests a module with an rsync-style `host::module[/path]` destination. The module name crosses the wire as a trailing string on the config frame (bumping `PROTOCOL_VERSION` 2.14.0 → 2.15.0; the bump is required because the config-frame layout changed and the strict same-version handshake is what prevents a peer from desynchronizing on the new trailing field). The daemon looks the module up in ITS OWN config and uses the module's `path` as the authorized root through the exact same `configure_authorization` confinement the standalone server applies to `--destination-root` (`file_open_secure_parent`, `has_path_traversal`, `path_is_within`); the client never supplies the root, every client-chosen-ownership/super-user request is refused unless the module declares `client owner = yes` (the daemon's per-module opt-in, see below), and the operator `--no-super` veto forces super-user activities off for every daemon connection. The client's `/path` part is relative inside the module and is rejected if absolute or if it contains `..`. Unknown modules are refused before any data moves (the run fails cleanly at the config handshake). An absolute destination and a module request against a non-daemon server are also refused.
|
||||
- **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. `client owner = yes` opts a single module in, allowing those requests within that module's root (the standalone listener and the SSH `--stdio` server always honor them for their single operator-authorized root). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible.
|
||||
- **`client owner` (client-chosen-ownership opt-in):** by default a daemon module refuses every request that would let the client pick an owner or ask for super-user activities — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and an explicit `--super` — at the config handshake (before `STATUS_OK`), because a daemon has no per-module opt-in for client-chosen ownership and any anonymous client could otherwise force arbitrary owner ids inside the module root. A plain preserve-source request (`-a`/`-o`/`-g`) is **not** refused: the module forces super-user activities off for that connection, so no ownership is applied, and it logs a warning that the requested ownership will not be applied (the transfer itself still succeeds). `client owner = yes` opts a single module in, allowing those requests within that module's root (a root standalone TCP listener honors them for its single operator-authorized root only when started with `--allow-super`; the flag is rejected with `--stdio`, whose client-composed remote argv must never opt back into super mode). Without the opt-in the daemon also forces super-user **device** activity off for that connection — char/block device-node creation (`--devices`) and `--write-devices` — even under the default `AUTO` mode, so a non-opted module can never be made to `mknod` or write a raw device; those entries are skipped (not refused) so an ordinary `-a` push still succeeds without device nodes. The opt-in does **not** lift the privilege requirement: `--copy-as` still needs a root receiver, and the operator `--no-super` veto still forces super-user activities off for every connection. The daemon logs a prominent startup warning for each `client owner = yes` module so the operator's deliberate choice is visible.
|
||||
- **`read only` safe default:** every network transfer FastSync currently supports is a push that writes under the module root, so a `read only` module refuses the connection (clear server log "module is read only"; the client exits non-zero, nothing is transferred). A future pull/list operation can be opened up when it exists; the knob is already stored.
|
||||
- **Direction — remote source / pull is intentionally unsupported:** FastSync is push-only. The first positional argument is always a **local** source directory and the second is the destination; only the destination is parsed for remote syntax (`user@host:path` SSH, `host::module[/path]` daemon). A remote source such as `fastsync user@host:src ./local` is deliberately **not** implemented: rsync has no pull flag (direction is positional), so supporting a remote source is an optional feature rather than a compatibility requirement, and it would require a protocol role reversal (server as sender, client as receiver) across both transports. FastSync documents this as an intentional limitation rather than a missing rsync option. <a id="direction"></a>
|
||||
- **`auth users` (A7 SCRAM-SHA-256 authentication):** a module that declares `auth users` requires the client to present credentials. The config frame carries ONLY the username; the daemon answers an auth-required module with `STATUS_AUTH_CHALLENGE` (PBKDF2 iteration count, 16-byte salt, 32-byte server nonce), the client answers with `STATUS_AUTH_RESPONSE` (fresh 32-byte client nonce + a 32-byte ClientProof), and the daemon accepts only when the proof verifies **and** the username is **on the module's `auth users` list** and has a store entry, replying `STATUS_AUTH_OK` with a 32-byte ServerSignature the client verifies before proceeding. Verification is constant-time over fixed 32-byte keys (the compare runs even for a miss), username membership uses a constant-time full-length scan, and an unknown/off-list user still receives a challenge and runs the same math against a dummy verifier: a deterministic per-username salt (`HMAC-SHA256(store dummy key, username)`), the store-wide uniform iteration count and dummy keys. Re-probing the same unknown username therefore yields an identical salt and iteration count while a different username yields a different salt, so there is no user-enumeration or timing oracle. The daemon logs the username but **never the password, proof or keys**. A module WITHOUT `auth users` stays open (legitimate rsync configuration); credentials sent to such a module are ignored. Read-only is orthogonal: even a correctly authenticated push to a `read only` module is still refused (all FastSync network transfers write). Fail-closed policy: a daemon whose config declares `auth users` on any module refuses to start unless a credential store was given (`--password-file` and/or `--early-input`); a missing or empty store is never silently treated as "open". A failed handshake (missing credentials, unknown/off-list user, wrong proof or malformed data) yields a single generic `STATUS_AUTH_FAILED` and the daemon closes before any data moves. The dummy key is persisted in an owner-only `<store_path>.dummykey` sidecar (auto-created on first load, mode 0600) so the dummy salt stays stable across daemon restarts, closing the restart-gated enumeration channel. The sidecar is secret material and must be protected like the credential store (owner-only 0600, included with the store in backups and rotation). It must be preserved across restarts for that guarantee; if it cannot be created (a process-substitution/FIFO store path such as `/dev/fd/N`, a read-only filesystem, a missing directory, or a create/write/fsync/link/fchmod failure), the daemon logs a warning and uses a transient per-run key, so unknown-user challenges change across restarts and the cross-restart guarantee does not hold for that deployment. One residual is accepted: the store iteration count is observable pre-auth by design, since the miss path must match a hit. **Transport policy (hardening A7-3/S1):** an auth-required module accepts credentials only when either (a) the connection is an encrypted, verified TLS connection whose client certificate matches `--client-cn`, or (b) the connection is plaintext from a loopback TCP peer **and** the operator explicitly passed `--allow-unauthenticated`. A remote plaintext peer, and a loopback plaintext peer without that flag, are refused at the config gate before any challenge is sent; `--allow-unauthenticated` never permits remote plaintext auth (remote peers still require verified TLS). Daemon modules are a `--daemon`-only feature — the SSH `--stdio` path never loads a daemon config and is not an auth transport for them. Because the loopback allowance trusts whichever peer the kernel reports as `127.0.0.1`, it assumes nothing relays remote connections to the daemon: a local TCP forwarder or TLS-terminating proxy in front of an auth-module listener makes remote clients appear as loopback and bypasses the mutual-TLS identity check, so do not front an auth-module listener with such a relay.
|
||||
- **Credential store format:** server `--password-file`/`--early-input` files are line-based `user:$fastsync$1$pbkdf2-sha256$<iters>$<salt_b64>$<stored_key_b64>$<server_key_b64>`, one per line (standard base64; 16-byte salt, 32-byte keys; `iters` in `[100000, 10000000]`, default 600000). Every entry in the resulting store must agree on `iters` (a store whose entries disagree, or where a layered `--early-input` disagrees with `--password-file`, is rejected). Generate lines with `fastsync-server --hash-credentials FILE [--iterations N]`; the emitted lines are secret material, so redirect them to an owner-only (mode 0600) file (the tool warns on stderr if stdout is a group/other-accessible regular file). Blank lines and lines starting with `#`/`;` are comments; the parser is strict (a malformed line fails the whole load, so a typo can never let a different set of users in). **The legacy `user:SHA256HEX` form is hard-rejected** with an actionable "legacy" error; there is no auto-upgrade, so a replayable bearer digest can never be loaded by a 2.19.0 daemon. The client `--password-file` holds `user:password` on its first meaningful line (the literal password, used only for the handshake then burned); keep both files readable only by their owner (mode 0600). Per-username wire length is bounded (256 chars) and every decoded salt/key length is validated. Loading the store also maintains an owner-only `<store_path>.dummykey` sidecar (auto-created, mode 0600, exactly 32 bytes) holding the store-wide dummy key that shapes unknown-user challenges; persist it across daemon restarts so those challenges stay stable, and treat a sidecar with the wrong owner, a mode other than exactly 0600, the wrong size or the wrong type as a fatal load error (fail closed). If the sidecar cannot be created (e.g. a process-substitution store path such as `/dev/fd/N`, a read-only filesystem, a missing directory, or a create/write/fsync/link/fchmod failure), the daemon logs a warning and uses a transient per-run key, so the cross-restart stability guarantee does not hold there.
|
||||
- **Plaintext caveat:** an auth-required module is refused, **before any challenge is sent**, unless the connection is encrypted and verified TLS whose client certificate matches the server's `--client-cn`, or it is plaintext from a loopback TCP peer **and** the operator passed `--allow-unauthenticated`. A remote plaintext peer, and a loopback plaintext peer without that flag, never receive a challenge, and `--allow-unauthenticated` never permits remote plaintext auth (remote peers still require verified TLS). On the loopback plaintext transport that remains permitted, a local sniffer could still read the challenge and response and mount an **offline dictionary attack** against a weak password, so use `--tls` for any real deployment. `--client-cn` matches the certificate CN only (not a subjectAltName), which is acceptable for a private CA. Clients sending daemon credentials with `--password-file` to a non-loopback daemon must use `--tls`; the client rejects such a destination before any network I/O. Unlike the old challenge-less exchange there is **no replay**: the proof is bound to the fresh per-connection server nonce, so a captured `STATUS_AUTH_RESPONSE` cannot be reused on another connection (an integration test proxies the daemon and proves this). TLS client-CN (`--client-cn`) is an independent transport identity check and composes with password auth; because `--tls` already mandates `--client-cn`, a TLS auth connection always verifies the client CN, so both checks necessarily apply together on such a connection.
|
||||
@@ -651,37 +684,37 @@ now transmits targets (the prior behavior was broken/partial); its status moved
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| Path escape detection | Ensure files stay within root | ✅ Implemented | `has_path_traversal()` + realpath |
|
||||
| Symlink-safe delete | Skip symlinks in delete walk | ✅ Implemented | `delete_extras_walk()` |
|
||||
| Protocol version check | Verify compatible versions | ✅ Implemented | `config_receive()` |
|
||||
| Max data/string/chunk sizes | Prevent OOM attacks | ✅ Implemented | Per-message limits |
|
||||
| Per-connection memory limit | 1GB per connection | ✅ Implemented | `MAX_CONNECTION_MEMORY` |
|
||||
| `--max-alloc=SIZE` | Limit a single memory allocation | ✅ Implemented | Caps the largest single allocation; binary units, default 1G |
|
||||
| `--trust-sender` | Trust remote sender's file list | ✅ Implemented | Long-form-only, receiver-local policy that never crosses the wire. The receiver skips its redundant up-front re-validation of the incoming file list (empty/`..` path rejection and the escaping-symlink-target containment), trusting the sender instead of double-checking (fewer checks, faster, potentially unsafe, matching rsync). Off by default. The low-level fd-relative confinement primitives (`file_open_secure_parent`, the O_NOFOLLOW parent walk, leaf/destination confinement) are deliberately KEPT even under `--trust-sender`, so a hostile sender still cannot write or link outside the authorized root (see Phase-5 notes below) |
|
||||
| `--old-args` | Disable modern arg protection | ✅ Implemented | SSH-only; accepted for CLI compatibility but is now a **documented no-op**: FastSync always single-quote-escapes the remote server path and each `--remote-option` value (`ssh_build_remote_command`), so a metacharacter-bearing `--rsync-path` can never be interpreted by the remote shell. The flag no longer disables that quoting (the old raw-construction behavior was an injection foot-gun and is removed); the safety-relevant behavior is identical either way |
|
||||
| `--ignore-missing-args` | Ignore missing source args | ✅ Implemented | FastSync has a single source-root argument (which always exists), so the "explicitly requested source arguments" are the `--files-from` entries and the flags only ever apply there (inert without `--files-from`, like `-R`). Without the flag a listed-but-missing entry stays a hard pre-transfer error (nothing is transferred). With it each missing entry is skipped: nothing is sent for it, it never enters the keep-set, and the run succeeds for the rest — an all-missing non-empty list succeeds transferring nothing, matching rsync. `--dirs` + `--files-from` missing entries are skipped the same way. Every skipped entry is logged and a per-run warning names the count, so the handling is never a silent no-op. Divergences: an EMPTY `--files-from` file stays a hard error in every mode (no argument was requested at all; rsync likewise reports "no source files specified"); missing-arg skipping only applies to the pre-transfer list validation, so an entry that is present at preflight and vanishes mid-transfer still fails (matching rsync, whose flag "does not affect subsequent vanished-file errors"); `--no-ignore-missing-args` is not a supported negation |
|
||||
| `--delete-missing-args` | Delete missing source args | ✅ Implemented | Implies `--ignore-missing-args` (order-independent) and additionally removes each missing entry's destination mirror receiver-side. The mirror is computed exactly like a present sibling's wire path: the bare relative entry under `-R`, otherwise the full source-mirror path below the destination root. rsync parity, verified against the man page: it does **not** imply `--delete` generally and is "independent of any other type of delete processing" — unrelated destination extras are untouched unless `--delete` is also present. Composition with `--delete` + timing: the exact-path deletions commit with the manifest, early for `--delete-before`/`--delete-during`, else only after a fully-successful transfer (delete-after/commit). A non-empty directory mirror is removed only when `--force` or `--delete` is in effect (otherwise it is left with a warning and the run continues, like rsync); an absent mirror is a no-op. An explicitly listed missing arg is a user request, not an excluded file: its deletion is never blocked by the filter-exclusion protection of excluded destination mirrors (a mirror sitting inside a filter-excluded directory is still removed). Safety/policy: gated by the server `--allow-delete` policy like `--delete`; the request paths cross the wire only in the delete-manifest frame and are confined by the same receiver validation as the keep-set (non-empty, relative, traversal-free, bounded by the per-section/per-frame manifest caps); the `--delay-updates` staging directory and basis snapshots are protected exactly as in the extras walker. Divergence: the missing-args deletions are not counted toward `--max-delete` (they are explicit per-path requests, not discovered extras). See the Phase-3 wire note below for the `PROTOCOL_VERSION` bump |
|
||||
| Path escape detection | Ensure files stay within root | ✅ Parity | `has_path_traversal()` + realpath |
|
||||
| Symlink-safe delete | Skip symlinks in delete walk | ✅ Parity | `delete_extras_walk()` |
|
||||
| Protocol version check | Verify compatible versions | ✅ Parity | `config_receive()` |
|
||||
| Max data/string/chunk sizes | Prevent OOM attacks | ✅ Parity | Per-message limits |
|
||||
| Per-connection memory limit | Cap memory per connection | ✅ Parity | `MAX_CONNECTION_MEMORY` is **256 MiB per connection** (256 * 1024 * 1024 bytes), charged across protocol reservations and decompression/chunk allocations. This is a FastSync-internal bound with no direct rsync analogue |
|
||||
| `--max-alloc=SIZE` | Limit a single memory allocation | ✅ Parity | Caps the largest single allocation; binary units, default 1G |
|
||||
| `--trust-sender` | Trust remote sender's file list | ⚠️ Caveat | Long-form-only, receiver-local policy that never crosses the wire. The receiver skips its redundant up-front re-validation of the incoming file list (empty/`..` path rejection), trusting the sender instead of double-checking (fewer checks, faster, potentially unsafe, matching rsync). Off by default. **It no longer affects symlink targets** (protocol 2.23.0): targets are stored verbatim under `-l` regardless of `--trust-sender`; the flag only relaxes the receiver's path-list checks. The low-level fd-relative confinement primitives (`file_open_secure_parent`, the O_NOFOLLOW parent walk, leaf/destination confinement) are deliberately KEPT even under `--trust-sender`, so a hostile sender still cannot write or link outside the authorized root (see Phase-5 notes below) |
|
||||
| `--old-args` | Disable modern arg protection | ⚠️ Caveat | SSH-only; accepted for CLI compatibility but is now a **documented no-op**: FastSync always single-quote-escapes the remote server path and each `--remote-option` value (`ssh_build_remote_command`), so a metacharacter-bearing `--rsync-path` can never be interpreted by the remote shell. The flag no longer disables that quoting (the old raw-construction behavior was an injection foot-gun and is removed); the safety-relevant behavior is identical either way |
|
||||
| `--ignore-missing-args` | Ignore missing source args | ⚠️ Caveat | FastSync has a single source-root argument (which always exists), so the "explicitly requested source arguments" are the `--files-from` entries and the flags only ever apply there (inert without `--files-from`, like `-R`). Without the flag a listed-but-missing entry stays a hard pre-transfer error (nothing is transferred). With it each missing entry is skipped: nothing is sent for it, it never enters the keep-set, and the run succeeds for the rest — an all-missing non-empty list succeeds transferring nothing, matching rsync. `--dirs` + `--files-from` missing entries are skipped the same way. Every skipped entry is logged and a per-run warning names the count, so the handling is never a silent no-op. Divergences: an EMPTY `--files-from` file stays a hard error in every mode (no argument was requested at all; rsync likewise reports "no source files specified"); missing-arg skipping only applies to the pre-transfer list validation, so an entry that is present at preflight and vanishes mid-transfer still fails (matching rsync, whose flag "does not affect subsequent vanished-file errors"); `--no-ignore-missing-args` is not a supported negation |
|
||||
| `--delete-missing-args` | Delete missing source args | ✅ Parity | Implies `--ignore-missing-args` (order-independent) and additionally removes each missing entry's destination mirror receiver-side. The mirror is computed exactly like a present sibling's wire path: the bare relative entry under `-R`, otherwise the full source-mirror path below the destination root. rsync parity, verified against the man page: it does **not** imply `--delete` generally and is "independent of any other type of delete processing" — unrelated destination extras are untouched unless `--delete` is also present. Composition with `--delete` + timing: the exact-path deletions commit with the manifest, early for `--delete-before`/`--delete-during`, else only after a fully-successful transfer (delete-after/commit). A non-empty directory mirror is removed only when `--force` or `--delete` is in effect (otherwise it is left with a warning and the run continues, like rsync); an absent mirror is a no-op. `--force` is deletion authority and is therefore gated by the server `--allow-delete` policy exactly like `--delete`/`--delete-missing-args`: without it the receiver clears the flag, so a client cannot use `--force` to recursively replace or remove a destination directory tree. An explicitly listed missing arg is a user request, not an excluded file: its deletion is never blocked by the filter-exclusion protection of excluded destination mirrors (a mirror sitting inside a filter-excluded directory is still removed). Safety/policy: gated by the server `--allow-delete` policy like `--delete`; the request paths cross the wire only in the delete-manifest frame and are confined by the same receiver validation as the keep-set (non-empty, relative, traversal-free, bounded by the per-section/per-frame manifest caps); the `--delay-updates` staging directory and basis snapshots are protected exactly as in the extras walker. Protocol 2.23.0 parity: the missing-args exact-path removals and the ordinary extras walk **draw from one shared `--max-delete` budget**, so a capped run stops part-way and exits 25 exactly like rsync. See the Phase-3 wire note below for the `PROTOCOL_VERSION` bump |
|
||||
|
||||
## 16. Batch Operations
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `--write-batch=FILE` | Write batched update to file | ✅ Implemented | Phase-6 residual-batch (client-only): runs the normal live transfer AND additionally emits a self-contained single-file batch of the whole source tree. The batch is a magic/format-version header followed by length-prefixed `chunk_serialize` blobs (full file images), replayable byte-identically by `--read-batch` on another machine with no source/server. `--write-batch` drives the single-threaded transfer path (the multithreaded path consumes the config before the separate batch scan pass). See the Phase-6 batch note below |
|
||||
| `--only-write-batch=FILE` | Write batch without updating dest | ✅ Implemented | Phase-6 residual-batch: emits the self-contained batch FILE only — NO destination update, NO server connection. Requires a source (scans it and serializes the full tree to FILE). Same single-file format as `--write-batch`, so the file is re-appliable via `--read-batch=FILE DEST`. See the Phase-6 batch note below |
|
||||
| `--read-batch=FILE` | Read batched update from file | ✅ Implemented | Phase-6 residual-batch: applies a previously written batch FILE locally to the destination. NO source and NO server — positional args are the destination only. Reads the magic/version header, then length-prefixed records, `chunk_deserialize`, and applies each via the confined `file_save_to_disk_full` path (same O_NOFOLLOW / `..`-rejection / root-confinement as the network receiver, so an attacker-controlled batch cannot escape the destination root). Malformed/truncated/oversized/traversal records are rejected cleanly. See the Phase-6 batch note below |
|
||||
| `--write-batch=FILE` | Write batched update to file | ⚠️ Caveat | Phase-6 residual-batch (client-only): runs the normal live transfer AND additionally emits a self-contained single-file batch of the whole source tree. The batch is a magic/format-version header followed by length-prefixed `chunk_serialize` blobs (full file images), replayable byte-identically by `--read-batch` on another machine with no source/server. `--write-batch` drives the single-threaded transfer path (the multithreaded path consumes the config before the separate batch scan pass). See the Phase-6 batch note below |
|
||||
| `--only-write-batch=FILE` | Write batch without updating dest | ⚠️ Caveat | Phase-6 residual-batch: emits the self-contained batch FILE only — NO destination update, NO server connection. Requires a source (scans it and serializes the full tree to FILE). Same single-file format as `--write-batch`, so the file is re-appliable via `--read-batch=FILE DEST`. See the Phase-6 batch note below |
|
||||
| `--read-batch=FILE` | Read batched update from file | ⚠️ Caveat | Phase-6 residual-batch: applies a previously written batch FILE locally to the destination. NO source and NO server — positional args are the destination only. Reads the magic/version header, then length-prefixed records, `chunk_deserialize`, and applies each via the confined `file_save_to_disk_full` path (same O_NOFOLLOW / `..`-rejection / root-confinement as the network receiver, so an attacker-controlled batch cannot escape the destination root). Malformed/truncated/oversized/traversal records are rejected cleanly. See the Phase-6 batch note below |
|
||||
|
||||
## 17. Advanced
|
||||
|
||||
| Flag | Rsync Description | FastSync Status | Notes |
|
||||
|------|-------------------|-----------------|-------|
|
||||
| `--stop-after=MINS` | Stop after N minutes | ✅ Implemented | Client-only sender stop deadline (Phase 6): computing `--stop-after=MINS` (a positive minute count; 0/negative/garbage rejected) and `--stop-at=TIME` (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`; a past time stops immediately). The transfer stops ELEGANTLY at the next chunk boundary: everything already fully sent is kept and applied, the run returns 0, and --delete (late/delete-after timing) does NOT wipe the destination — when the scan is cut short the partial keep-set manifest is suppressed with a warning (the delete walk is skipped rather than acting on an incomplete keep-set, so unscanned source mirrors survive). `--delete-before`/`--delete-during` still run their complete pre-scan (which ignores the deadline). Local client-only fields: never serialized into the wire config frame, so no PROTOCOL_VERSION bump. `--stop-after` uses CLOCK_MONOTONIC; `--stop-at` uses the wall clock. Works single-threaded and under `-j`/`--threads` (multithreaded). Divergence: rsync computes `--stop-after` from the run start; FastSync likewise. When both are given, the earlier of the two deadlines wins (checked per iteration). See the Phase-6 stop notes below |
|
||||
| `--stop-at=TIME` | Stop at specified time | ✅ Implemented | Same feature as `--stop-after` (deadline transfer stop), absolute wall-clock form (`HH:MM[:SS]` or `now+N[smhd]`). See the row above and the Phase-6 stop notes |
|
||||
| `--fsync` | Fsync every written file before publication | ✅ Implemented | |
|
||||
| `--protocol=NUM` | Force older protocol version | ✅ Implemented | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.19.0) with no downgrade/backward-compat code paths, so `--protocol=2.19.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.18.0`/`2.18`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below |
|
||||
| `--iconv=CONVERT_SPEC` | Charset conversion | ✅ Implemented | Charset conversion of FILE NAMES (not content) at the protocol boundary via iconv(3): `--iconv=LOCAL[,REMOTE]` — the sender converts each local filename LOCAL→REMOTE before transmitting, and the receiver converts each wire filename REMOTE→LOCAL before creating/writing. The full CONVERT_SPEC is serialized into the config frame as a new trailing string field so the peer knows the wire charset; **PROTOCOL_VERSION bumped 2.15.0 → 2.16.0**. `LOCAL[,REMOTE]` parse: single charset ⇒ LOCAL==REMOTE (identity both ways); garbage rejected up front. Validation probes BOTH directions (a spec that only opens one way is refused, as is a NUL-emitting target charset like utf-16/utf-32/ucs-2, since filenames cannot contain NUL). An unrepresentable name (EILSEQ/EINVAL) fails that path cleanly with a logged `--iconv: cannot convert file name ...` and is never written mangled/truncated. Conversion is applied at EVERY wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest, the incremental-check path, and the `-s`/`chunk_serialize` embedded blob path), on both client and server (`--iconv` is also a server/daemon option). Zero overhead when unset. See the Phase-6 iconv notes below |
|
||||
| `--checksum-seed=NUM` | Set checksum seed | ✅ Implemented | Sets the seed for FastSync's whole-file xxHash64 digest (full 64-bit seed) and for the delta path's per-block xxHash32 strong checksum (low 32 bits of the seed). An explicit seed deterministically changes every computed digest on BOTH endpoints (sender and receiver share the seed via the config frame, protocol 2.10.0), so identical runs with the same seed skip the same files and a changed seed changes the digests — the explicit-seed path that makes xxHash comparisons deterministic. `--checksum-choice=md5` has no seed and ignores it (documented). The value is a strict decimal 0..2⁶⁴-1 (blank, signed, or non-numeric values are rejected). Like rsync, a seed only matters where a digest is actually computed (`--checksum` or a basis-dir run, or a delta transfer); it does not by itself enable `--checksum`/`--delta`. Divergence from rsync: the default is seed 0, and FastSync never randomizes the seed (rsync uses a random per-transfer seed when `--checksum-seed` is unset); FastSync's unset default therefore reproduces its historical byte-for-byte behavior |
|
||||
| `--secluded-args`, `-s` | Use protocol to send args | ⛔ Impossible/Divergence | Accepted for CLI compatibility (including the rsync short `-s`, Phase 7 Wave A) but a documented **no-op / divergence**. rsync's `-s` protects arguments from shell expansion by shipping them over the protocol; FastSync never passes remote arguments through a shell expansion boundary in the first place — its SSH transport builds the remote argv as **single-quote-escaped shell words** (`ssh_build_remote_command`), so the injection/leak that `-s` guards against does not exist and there is nothing to "seclude". Implementing a true arg-send protocol would mean replacing the argv-based SSH launch with an in-band argument channel, a large redesign of the transport that buys no security here. Chunk serialization remains the long-only `--chunk-serialization`. |
|
||||
| `--no-OPTION` | Turn off implied option | ✅ Supported | Supported boolean FastSync options and archive-implied options; unsafe or value-taking options are rejected. |
|
||||
| `--stop-after=MINS` | Stop after N minutes | ✅ Parity | Client-only sender stop deadline (Phase 6): computing `--stop-after=MINS` (a positive minute count; 0/negative/garbage rejected) and `--stop-at=TIME` (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`; a past time stops immediately). The transfer stops ELEGANTLY at the next chunk boundary: everything already fully sent is kept and applied, the run returns 0, and --delete (late/delete-after timing) does NOT wipe the destination — when the scan is cut short the partial keep-set manifest is suppressed with a warning (the delete walk is skipped rather than acting on an incomplete keep-set, so unscanned source mirrors survive). `--delete-before`/`--delete-during` still run their complete pre-scan (which ignores the deadline). Local client-only fields: never serialized into the wire config frame, so no PROTOCOL_VERSION bump. `--stop-after` uses CLOCK_MONOTONIC; `--stop-at` uses the wall clock. Works single-threaded and under `-j`/`--threads` (multithreaded). Divergence: rsync computes `--stop-after` from the run start; FastSync likewise. When both are given, the earlier of the two deadlines wins (checked per iteration). See the Phase-6 stop notes below |
|
||||
| `--stop-at=TIME` | Stop at specified time | ⚠️ Caveat | Same feature as `--stop-after` (deadline transfer stop), absolute wall-clock form (`HH:MM[:SS]` or `now+N[smhd]`). See the row above and the Phase-6 stop notes |
|
||||
| `--fsync` | Fsync every written file before publication | ✅ Parity | |
|
||||
| `--protocol=NUM` | Force older protocol version | ❌ Divergent | Forces the wire protocol version for this transfer. FastSync has exactly ONE wire format (`PROTOCOL_VERSION`, currently 2.23.0) with no downgrade/backward-compat code paths, so `--protocol=2.23.0` is accepted (it sets the version claim the client sends, which the server already requires to match exactly) and **every other value is rejected up front** with a clear error before any connection — it does not and cannot speak an older or virtual wire format. Divergence from rsync (which negotiates a range and downgrades to an integer 0..31): FastSync's honest contract is force-to-the-one-supported-value; a genuine downgrade would require a per-version compatibility layer that does not exist. Client-only; the server-side exact-match check is unchanged. `--protocol=2.21.0`/`2.20.0`/`2.19.0`/`2.18.0`/`2.18`/`2.17.0`/`2.16.0`/`2.15.0`/`216`/`31`/garbage are all rejected. See the Phase-6 protocol note below |
|
||||
| `--iconv=CONVERT_SPEC` | Charset conversion | ⚠️ Caveat | Charset conversion of FILE NAMES (not content) at the protocol boundary via iconv(3): `--iconv=LOCAL[,REMOTE]` — the sender converts each local filename LOCAL→REMOTE before transmitting, and the receiver converts each wire filename REMOTE→LOCAL before creating/writing. The full CONVERT_SPEC is serialized into the config frame as a new trailing string field so the peer knows the wire charset; **PROTOCOL_VERSION bumped 2.15.0 → 2.16.0**. `LOCAL[,REMOTE]` parse: single charset ⇒ LOCAL==REMOTE (identity both ways); garbage rejected up front. Validation probes BOTH directions (a spec that only opens one way is refused, as is a NUL-emitting target charset like utf-16/utf-32/ucs-2, since filenames cannot contain NUL). An unrepresentable name (EILSEQ/EINVAL) fails that path cleanly with a logged `--iconv: cannot convert file name ...` and is never written mangled/truncated. Conversion is applied at EVERY wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest, the incremental-check path, and the `-s`/`chunk_serialize` embedded blob path), on both client and server (`--iconv` is also a server/daemon option). Zero overhead when unset. See the Phase-6 iconv notes below |
|
||||
| `--checksum-seed=NUM` | Set checksum seed | ✅ Parity | Sets the seed for FastSync's whole-file xxHash digest (full 64-bit seed) and for the delta path's per-block xxHash32 strong checksum (low 32 bits of the seed). **As of protocol 2.23.0 a seed of `0` — the default when the flag is unset — is randomized per transfer and the chosen seed is sent to the receiver**, exactly like rsync, so two runs against different content do not share a predictable seed; an explicit non-zero seed is used verbatim, so an explicit seed deterministically reproduces every computed digest on BOTH endpoints (the seed crosses in the config frame). `--checksum-choice=md5` has no seed and ignores it (documented). The value is a strict decimal 0..2⁶⁴-1 (blank, signed, or non-numeric values are rejected). Like rsync, a seed only matters where a digest is actually computed (`--checksum` or a basis-dir run, or a delta transfer); it does not by itself enable `--checksum`/`--delta` |
|
||||
| `--secluded-args`, `-s` | Use protocol to send args | ❌ Divergent | Accepted for CLI compatibility (including the rsync short `-s`, Phase 7 Wave A) but a documented **no-op / divergence**. rsync's `-s` protects arguments from shell expansion by shipping them over the protocol; FastSync never passes remote arguments through a shell expansion boundary in the first place — its SSH transport builds the remote argv as **single-quote-escaped shell words** (`ssh_build_remote_command`), so the injection/leak that `-s` guards against does not exist and there is nothing to "seclude". Implementing a true arg-send protocol would mean replacing the argv-based SSH launch with an in-band argument channel, a large redesign of the transport that buys no security here. Chunk serialization remains the long-only `--chunk-serialization`. |
|
||||
| `--no-OPTION` | Turn off implied option | ✅ Parity | Supported boolean FastSync options and archive-implied options; unsafe or value-taking options are rejected. |
|
||||
|
||||
---
|
||||
|
||||
@@ -689,7 +722,7 @@ now transmits targets (the prior behavior was broken/partial); its status moved
|
||||
|
||||
**Phase 5 notes (remote-option wave):** `--remote-option=OPT` (long form only) and `--trust-sender` landed here.
|
||||
- `--remote-option` is CLIENT-only and never serialized into the binary config frame. On the SSH transport the client forwards each value to the remote server by appending it to the remote command line in `ssh_build_remote_command()`, after ` --stdio`, as an individually single-quoted shell word (`'...'` with `'\''` for embedded quotes). Values are validated at CLI parse time (non-empty; no ASCII control characters) and rejected otherwise, and a non-conforming value is refused again in the command builder, so shell metacharacters (`;`, `&`, `|`, backticks, `$()`, quotes) can never break out of the quoting to inject an unrelated remote command — including after a client-side `--` separator, whose arguments are never forwarded anyway. Because the remote options affect the *remote server invocation*, not the transmitted config, the wire frame layout is unchanged, but `PROTOCOL_VERSION` was bumped **2.13.0 → 2.14.0** as the Phase-5 lockstep release marker (a 2.14 client against a 2.13 server fails the version check cleanly rather than the old server rejecting an unfamiliar forwarded argv later). Divergence: rsync's short `-M` form of `--remote-option` was intentionally NOT implemented at that time because `-M` was FastSync metadata mode; **Phase 7 Wave A later freed `-M` for `--remote-option` and moved metadata to long-only `--preserve`** (see the Sending Options table).
|
||||
- `--trust-sender` is a receiver-local policy: it never crosses the wire (the sender's value is never serialized, so a wire peer can never enable it). On the receiving process it skips the up-front re-validation of the incoming file list (empty/`..` path rejection and the escaping-symlink-target containment), trusting the sender's list instead of double-checking — fewer checks, faster, and potentially unsafe, matching rsync. It is OFF by default (`config.trust_sender`). As a deliberate safety floor, the low-level fd-relative confinement primitives are NOT disabled: `file_open_secure_parent()` (O_NOFOLLOW walk, `..` rejection, root containment) and leaf/destination confinement still hold, so even under `--trust-sender` a hostile sender cannot write or create a symlink outside the authorized root — the relaxation only removes the redundant list-layer double-checks, never the root-confinement guarantees.
|
||||
- `--trust-sender` is a receiver-local policy: it never crosses the wire (the sender's value is never serialized, so a wire peer can never enable it). On the receiving process it skips the up-front re-validation of the incoming file list (empty/`..` path rejection), trusting the sender's list instead of double-checking — fewer checks, faster, and potentially unsafe, matching rsync. It is OFF by default (`config.trust_sender`). Since protocol 2.23.0 it does **not** gate symlink-target handling: `-l` stores targets verbatim either way. As a deliberate safety floor, the low-level fd-relative confinement primitives are NOT disabled: `file_open_secure_parent()` (O_NOFOLLOW walk, `..` rejection, root containment) and leaf/destination confinement still hold, so even under `--trust-sender` a hostile sender cannot write or place a *path* outside the authorized root — the relaxation only removes the redundant list-layer double-checks, never the root-confinement guarantees for paths and placements.
|
||||
|
||||
The estimates below cover the currently unimplemented features in this document. They assume one engineer familiar with the codebase, include implementation and focused tests, and exclude production rollout time. A feature should not be marked implemented until its behavior is tested in both local and SSH/TCP paths where applicable.
|
||||
|
||||
@@ -793,7 +826,7 @@ These are the hardest compatibility items because they require durable formats o
|
||||
|
||||
**Phase 6, Wave B (iconv) shipping note (PROTOCOL 2.15.0 → 2.16.0):** `--iconv=LOCAL[,REMOTE]` converts file NAMES at the wire boundary (never content). The full CONVERT_SPEC is serialized into the config frame as a new trailing string field (empty→NULL canonicalized), so both ends share the same wire charset interpretation; this required the PROTOCOL bump because the frame is a strict ordered sequence and a peer that does not parse the new trailing field would desynchronize. Each end derives LOCAL (its own charset) and REMOTE (the wire charset): the sender opens LOCAL→REMOTE and converts every transmitted filename; the receiver opens REMOTE→LOCAL and converts every received filename before creating/writing. Conversion is applied at every wire-path site (regular/MKDIR/hardlink path+target/symlink path+target/SPECIAL, the delete manifest keep/protected/missing entries, the incremental-check path, and the embedded `-s`/chunk-blob path). A name it cannot convert (EILSEQ/EINVAL) is failed cleanly with a logged `--iconv: cannot convert file name ...` and is never written truncated/mangled. Validation probes both directions up front (both the sender local→remote and the receiver remote→local, and, for a server/daemon with its own `--iconv`, the client-REMOTE→server-LOCAL pair) so an unusable spec is rejected before the connection rather than mid-transfer, and NUL-emitting target charsets (utf-16/utf-32/ucs-2) are refused because filenames cannot contain NUL. Divergence documented upstream: the receiver does NOT half-swap; the wire charset always comes from the sender's REMOTE half, so a server whose local charset differs from the client's LOCAL must declare it with its own `--iconv`. Conversion is process-global and runs on a single thread per process (sender thread / receiver-loop thread), initialized before worker threads start and freed after they join.
|
||||
|
||||
**Phase 6, Wave C (protocol-version) shipping note (no PROTOCOL_VERSION change):** `--protocol=NUM` lets the client force the wire protocol version for a transfer. FastSync's protocol is a single lockstep format: the config frame is a strict ordered sequence and the server requires the client's version string to equal `PROTOCOL_VERSION` exactly (`config_receive_with_validate`, src/shared/config.c) — there are no older-format code paths and no downgrade/negotiation machinery, so a lower/higher/virtual version can never be spoken. The honest contract is therefore: `--protocol=2.19.0` (the current `PROTOCOL_VERSION`, as of the A7 auth redesign) is accepted and stored into the client's `version` claim (which `config_send` already transmits), and every other value — `2.18.0`, `2.18`, `2.17.0`, `2.16.0`, `2.15.0`, `3.0.0`, rsync-integer spellings like `216`/`31`, garbage, empty — is rejected up front in `validate_config()` before any connection, with a clear error that FastSync supports only its current wire protocol and cannot speak an older or virtual one. Implementation is client-only: a server-side `--protocol` is intentionally not added because the server has no negotiation (it only enforces exact match), and it could only ever be the current version. This preserves (and slightly tightens) existing validation: the client now also refuses to launch with a version it cannot actually speak, rather than only the server rejecting it later. A genuine downgrade would require a per-version compatibility layer for every frame/feature added since (append 2.10, preallocate 2.11, hardlinks 2.12, devices/specials/symlink-trust/xattr 2.13, remote-option 2.14, daemon module/auth 2.15, iconv 2.16, dir/symlink times 2.17, privilege flags --super/--copy-as 2.18, SCRAM daemon auth 2.19) and is intentionally out of scope — documented divergences from rsync's integer-negotiated downgrade remain.
|
||||
**Phase 6, Wave C (protocol-version) shipping note (no PROTOCOL_VERSION change):** `--protocol=NUM` lets the client force the wire protocol version for a transfer. FastSync's protocol is a single lockstep format: the config frame is a strict ordered sequence and the server requires the client's version string to equal `PROTOCOL_VERSION` exactly (`config_receive_with_validate`, src/shared/config.c) — there are no older-format code paths and no downgrade/negotiation machinery, so a lower/higher/virtual version can never be spoken. The honest contract is therefore: `--protocol=2.23.0` (the current `PROTOCOL_VERSION`, as of the rsync-parity wave) is accepted and stored into the client's `version` claim (which `config_send` already transmits), and every other value — `2.22.0`, `2.21.0`, `2.20.0`, `2.19.0`, `2.18.0`, `2.18`, `2.17.0`, `2.16.0`, `2.15.0`, `3.0.0`, rsync-integer spellings like `216`/`31`, garbage, empty — is rejected up front in `validate_config()` before any connection, with a clear error that FastSync supports only its current wire protocol and cannot speak an older or virtual one. Implementation is client-only: a server-side `--protocol` is intentionally not added because the server has no negotiation (it only enforces exact match), and it could only ever be the current version. This preserves (and slightly tightens) existing validation: the client now also refuses to launch with a version it cannot actually speak, rather than only the server rejecting it later. A genuine downgrade would require a per-version compatibility layer for every frame/feature added since (append 2.10, preallocate 2.11, hardlinks 2.12, devices/specials/symlink-trust/xattr 2.13, remote-option 2.14, daemon module/auth 2.15, iconv 2.16, dir/symlink times 2.17, privilege flags --super/--copy-as 2.18, SCRAM daemon auth 2.19, packed metadata 2.20, error-detail/dry-run 2.21, preserve-attribute split 2.22, rsync-parity wave 2.23) and is intentionally out of scope — documented divergences from rsync's integer-negotiated downgrade remain.
|
||||
|
||||
**Phase-1/2 selection-and-update status correction (docs):** `-I/--ignore-times`, `--size-only`, `-@/--modify-window`, `--existing`, `--ignore-existing`, `-u/--update`, `-W/--whole-file`, and `--compress-threads` were previously listed as not-implemented in this document but are in fact fully implemented and tested on `dev`. This pass corrects the matrix to match the code. The realistic model of these is that FastSync is a *sender-driven* whole-tree copy, so the size+mtime quick-check and all three receiver-policy skips (`--existing`, `--ignore-existing`, `-u`) are evaluated against the **destination** on the receiver side, and their booleans cross the wire in the config frame. `-I`/`--size-only`/`--modify-window` modify the `--incremental` per-file `STATUS_CHECK` handshake's match predicate (`-I` disables the mtime leg and forces transfer; `--size-only` drops only the mtime leg; `--modify-window` adds tolerance to `metadata_mtime_matches`); they require `--incremental` (or a basis dir) to have a handshake to affect, mirroring how they only matter where a quick-check exists in rsync. `--existing`/`--ignore-existing`/`-u` are receiver write-time policies (skipping the write / newer-destination guard) applied across the regular-file, `--delay-updates`-staged, hardlink-sibling, and special/device paths; `-u` implies `-M` metadata and uses a second-then-nanosecond strict `>` newer check; both correctly influence `--remove-source-files` (a skipped source is not removed). `-W/--whole-file` disables block-level delta (opt-in via `--delta`), folded into the wire `use_delta` so no protocol bump was needed, and makes `--fuzzy` inert; `--append`/`--append-verify` are rejected with `-W`. `--compress-threads=NUM` (1..64, client-only, never crosses the wire) sizes the zstd compression worker pool. No code was changed by this correction; the implementation had landed in earlier merge waves (feat/ignore-times, feat/ignore-existing via the newer `file_to_disk_secure_no_replace`/`linkat EEXIST` path, feat/size-only, feat/modify-window, feat/whole-file, feat/update, compression-threads).
|
||||
|
||||
@@ -803,7 +836,7 @@ These are the hardest compatibility items because they require durable formats o
|
||||
|
||||
These are the last compatibility items and the closing phase toward rsync flag parity. Per the project decision: every rsync flag (short **and** long) that is *possible* gets real rsync-parity behavior; anything physically impossible becomes an explicit **Impossible/Divergence** status (accepted for CLI compatibility, safely inert, with coverage tests proving that); and the two privilege flags (`--super`, `--copy-as`) adopt the deliberately-scoped **safe-subset + clear-refusal** model rather than blind elevation. The remaining `⚠️ Partial`, `🔄 Compatibility No-op`, `🔀 Alt Arg`, and `❌ Not Implemented` rows in the Summary are this phase's scope. All Wave A renames are **client-side only** (the wire config fields `use_compression`/`use_metadata`/`use_sendfile`/`use_chunk_serialization` are unchanged), so they require **no `PROTOCOL_VERSION` bump**.
|
||||
|
||||
**Wave A — CLI namespace parity (rename colliding FastSync short flags) — ✅ implemented.** This freed the short letters rsync needs and made the three `🔀 Alt Arg` rows real. `-c`→`--checksum`, `-m`→`--prune-empty-dirs`, `-M`→`--remote-option`, `-f`→`--filter`, `-s`→`--secluded-args`, `-p`→`--perms`, `-T`→`--temp-dir`, `-a`/`--archive`→real `-rlptgoD`. FastSync's own flags moved to long-form-only or new shorts: `-j`/`--threads` (multithreading), `--preserve` (metadata), `--sendfile`, `--chunk-serialization`, `--timeout`, `--ssh-port`. The server's independent little CLI keeps `-p` as its port. All client-side, no wire change, no `PROTOCOL_VERSION` bump. Unit tests 37/37, full integration 400 passed, cppcheck and clang-format clean. Known Wave-A limitation: `--no-perms`/`--no-compress`-style negation of the newly-aliased shorts is not wired into the negatable set (only the long-form `--preserve`/`--compress`/`--no-links` negations exist); `--archive --no-perms` is consequently not supported yet — a minor deviation from rsync, acceptable for Wave A.
|
||||
**Wave A — CLI namespace parity (rename colliding FastSync short flags) — ✅ implemented.** This freed the short letters rsync needs and made the three `🔀 Alt Arg` rows real. `-c`→`--checksum`, `-m`→`--prune-empty-dirs`, `-M`→`--remote-option`, `-f`→`--filter`, `-s`→`--secluded-args`, `-p`→`--perms`, `-T`→`--temp-dir`, `-a`/`--archive`→real `-rlptD`. FastSync's own flags moved to long-form-only or new shorts: `-j`/`--threads` (multithreading), `--preserve` (metadata), `--sendfile`, `--chunk-serialization`, `--timeout`, `--ssh-port`. The server's independent little CLI keeps `-p` as its port. All client-side, no wire change, no `PROTOCOL_VERSION` bump. Unit tests 37/37, full integration 400 passed, cppcheck and clang-format clean. Known Wave-A limitation: `--no-perms`/`--no-compress`-style negation of the newly-aliased shorts was not wired into the negatable set (only the long-form `--preserve`/`--compress`/`--no-links` negations existed), so `--archive --no-perms` was initially unsupported — a minor deviation from rsync. The preserve-attribute split wave below resolves the preservation side: `--no-perms`/`--no-times`/`--no-owner`/`--no-group` and `--no-preserve` now work, so `--archive --no-perms` is supported.
|
||||
|
||||
| FastSync flag today | rsync wants that name | Proposed rename |
|
||||
|---------------------|----------------------|-----------------|
|
||||
@@ -814,29 +847,186 @@ These are the last compatibility items and the closing phase toward rsync flag p
|
||||
| `-s` / `--chunk-serialization` | `-s` = `--secluded-args`/`--protect-args` | → `--chunk-serialization` (long-only) |
|
||||
| `-p` (SSH port) | `-p` = `--perms` | → `--port` (long-only; `--server-port` already exists) |
|
||||
| `-T` / `--timeout` | `-T` = `--temp-dir` | → `--timeout` (long-only) |
|
||||
| `-a` / `--archive` (= `-c -m -M`) | `-a` = `-rlptgoD` | → becomes **real rsync `-a`** after the renames |
|
||||
| `-a` / `--archive` (= `-c -m -M`) | `-a` = `-rlptD` | → becomes **real rsync `-a`** after the renames |
|
||||
|
||||
**Wave B — Output & filesystem completion (✅ implemented).** `-S`/`--sparse` (`⚠️→✅`): real hole preservation — a sparse-aware writer (`write_all_sparse`) skips all-zero runs ≥ 4096 bytes with `lseek(SEEK_CUR)` and `ftruncate`s the final size, wired into both the atomic temp+rename store and `--inplace` receiver-side with **no wire change** (the full file image is already in memory; the ftruncate presize is kept). `-P` (`⚠️→✅`): interrupted-write retention — on a save failure after data reached the temp fd, `--partial` now renames the already-written temp to the destination path (best-effort; falls through to the normal unlink on failure, never retains when `--partial` is off) so a later `--append`/`--append-verify` run can resume. `--block-size=SIZE` (`⚠️→✅`): promoted after verification — `--block-size` is now an alias for `--delta-block`, both set `config->delta_block_size`, which the delta engine already honored end-to-end (`delta_signature_create_seeded` + `delta_apply`); out-of-range values keep the default. `--fake-super` (`⚠️→✅`): added `fake_super_restore_fd` to parse and re-apply the recorded `user.fastsync.stat` record fd-relative (fchown best-effort/non-root skipped, fchmod, futimens); a save under `--fake-super` now re-applies the recorded attrs instead of only recording them, with the recording format unchanged. `--stderr=client` (`⚠️→⛔ Impossible/Divergence`): FastSync has no rsync client-message channel, and `client` is rejected at CLI parse — the rejection is the documented behavior (unit-tested). `-N`/`--crtimes` (`⚠️→⛔ Impossible/Divergence`): birth-times cannot be set by any portable fs call (`utimensat` sets only atime/mtime); capture/transmit stays, setting is impossible, the flag is accepted and safely inert. Review-hardening (post-eval): fake-super replay applies the mode through the same sanitization as the normal metadata path (group/other write bits are never granted); `--sparse` takes precedence over `--preallocate` (posix_fallocate skipped so holes survive); `--partial` retention is disabled for `--no_replace` (ignore/existing) and only marks a write-attempt after the actual write begins; `--block-size=SIZE`/`--delta-block=SIZE` inline forms are accepted.
|
||||
**Wave B — Output & filesystem completion (✅ implemented).** `-S`/`--sparse` (`⚠️→✅`): real hole preservation — a sparse-aware writer (`write_all_sparse`) skips all-zero runs ≥ 4096 bytes with `lseek(SEEK_CUR)` and `ftruncate`s the final size, wired into both the atomic temp+rename store and `--inplace` receiver-side with **no wire change** (the full file image is already in memory; the ftruncate presize is kept). `-P` (`⚠️→✅`): interrupted-write retention — on a save failure after data reached the temp fd, `--partial` now renames the already-written temp to the destination path (best-effort; falls through to the normal unlink on failure, never retains when `--partial` is off) so a later `--append`/`--append-verify` run can resume. `--block-size=SIZE` (`⚠️→✅`): promoted after verification — `--block-size` is now an alias for `--delta-block`, both set `config->delta_block_size`, which the delta engine already honored end-to-end (`delta_signature_create_seeded` + `delta_apply`); out-of-range values keep the default. `--fake-super` (`⚠️→✅`): added `fake_super_restore_fd` to parse and re-apply the recorded `user.fastsync.stat` record fd-relative (mode/time only — protocol 2.23.0: **never a real chown**; the resolved owner is recorded for a later privileged restore); a save under `--fake-super` now re-applies the recorded attrs instead of only recording them, with the recording format unchanged. `--stderr=client` (`⚠️→❌ Divergent`): FastSync has no rsync client-message channel, and `client` is rejected at CLI parse — the rejection is the documented behavior (unit-tested). `-N`/`--crtimes` (`⚠️→❌ Divergent`): birth-times cannot be set by any portable fs call (`utimensat` sets only atime/mtime); capture/transmit stays, setting is impossible, the flag is accepted and safely inert. Review-hardening (post-eval): fake-super replay applies the mode through the shared `metadata_mode_for_policy` helper (protocol 2.23.0: exactly the source mode under `-p`, with no masking); `--sparse` takes precedence over `--preallocate` (posix_fallocate skipped so holes survive); `--partial` retention is disabled for `--no_replace` (ignore/existing) and only marks a write-attempt after the actual write begins; `--block-size=SIZE`/`--delta-block=SIZE` inline forms are accepted.
|
||||
|
||||
**Wave C — Devices & special files (finalize statuses + tests) (✅ implemented).** The four special-file rows are finalized with coverage tests. `--devices`, `--copy-devices`, and `--write-devices` are **✅ Implemented**, each with a documented, safety-driven divergence: device-node creation is privilege-gated, so a receiver without `CAP_MKNOD` skips that entry with a warning (a per-entry skip, never a transfer failure); `--copy-devices` copies a device/FIFO's reported size into an ordinary regular file (a size-bounded safe divergence from rsync's unbounded dd-like read); `--write-devices` writes only into an existing char/block node under the confined receive root and skips every unusable target rather than clobbering or aborting. `--specials` is classified **⛔ Impossible/Divergence** for one reason only: **FIFO recreation works** (unprivileged `mkfifo`, asserted under CI), but **sockets cannot be recreated by any standard filesystem call**, so a source socket is skipped with an explicit note. Tests assert FIFO recreation, the safe socket skip, the regular-file result of `--copy-devices`, the skipped/missing and non-device `--write-devices` targets, and (root-gated) real device-node creation; a root runner additionally drops the receiver to an unprivileged user to assert the `CAP_MKNOD` skip is graceful.
|
||||
**Wave C — Devices & special files (finalize statuses + tests) (✅ implemented).** The four special-file rows are finalized with coverage tests. `--devices`, `--copy-devices`, and `--write-devices` are **✅ Implemented**, each with a documented, safety-driven divergence: device-node creation is privilege-gated, so a receiver without `CAP_MKNOD` skips that entry with a warning (a per-entry skip, never a transfer failure); `--copy-devices` copies a device/FIFO's reported size into an ordinary regular file (a size-bounded safe divergence from rsync's unbounded dd-like read); `--write-devices` writes only into an existing char/block node under the confined receive root and skips every unusable target rather than clobbering or aborting. `--specials` reclassified from **⛔ Impossible/Divergence** to **✅ Parity** in protocol 2.23.0: **FIFO recreation works** (unprivileged `mkfifo`) **and unix sockets are recreated** with `mknod(S_IFSOCK)`, which Linux permits unprivileged (the flag previously assumed sockets were impossible — see the `--specials` row). Tests assert FIFO recreation, socket recreation, the regular-file result of `--copy-devices`, the skipped/missing and non-device `--write-devices` targets, and (root-gated) real device-node creation; a root runner additionally drops the receiver to an unprivileged user to assert the `CAP_MKNOD` skip is graceful.
|
||||
|
||||
**Wave D — Times superstructure & arg-protection no-ops (✅ implemented, `--secluded-args` ⛔).** `-O`/`--omit-dir-times` and `-J`/`--omit-link-times` are now **real modifiers** (both `🔄 → ✅ Implemented`), reversing the old "never preserves directory/symlink times" divergence:
|
||||
**Wave D — Times superstructure & arg-protection no-ops (✅ implemented, `--secluded-args` ❌).** `-O`/`--omit-dir-times` and `-J`/`--omit-link-times` are now **real modifiers** (both `🔄 → ✅ Implemented`), reversing the old "never preserves directory/symlink times" divergence:
|
||||
|
||||
- **Directory times.** The recursive scanner captures every traversed source directory's metadata (mtime, plus atime under `-U`) into a per-transfer list — two paths are covered: the sequential `DirectoryScanner` captures each opened directory (including the transfer root), and the parallel scanner captures both the root in `parallel_scanner_create_with_options` and each worker's subdirectories in `open_next_directory` (appends are guarded by a mutex shared with the sender's pipeline context). The sender transmits them in trailing `STATUS_DIR_TIMES` frames (each: int count + count × (wire path, metadata) pairs) sent **after all file data and after the optional delete manifest**, just before `STATUS_FINISHED`. A tree larger than `MAX_MANIFEST_ENTRIES` (1 048 576) directories is chunked into repeated frames, each within the receiver's per-frame bound. A dir-time entry is RECORD-ONLY (`file->dir_time_only`): `file_save_to_disk_full` returns `FILE_SAVE_SKIPPED` without creating anything, so a source directory that was empty (or pruned by `-m/--prune-empty-dirs`) is never resurrected. The receiver accumulates received directory metadata in a `DirTimeList` and applies it only at the very end — after the entire stream, after the commit-style `--delete` deletion, and after `--delay-updates` publication — because creating or removing a child bumps the parent's mtime. Application is fd-relative/walk-confined (`file_open_secure_parent` + `utimensat(..., AT_SYMLINK_NOFOLLOW)`) and best-effort per entry: an absent path (an intentionally uncreated empty dir) is skipped QUIETLY and only a real existing directory is stamped. `-O` (config boolean, already on the wire) makes the receiver skip the whole set. The single-threaded sink applies in `receiver_send_success_frame`; the `-j`/`--threads` sink accumulates in `write_thread` and server.c applies after both threads join and the deletion commits.
|
||||
- **Symlink times/owner/mode.** `STATUS_SYMLINK` already carried metadata; the receiver now applies it with no-follow primitives only: `utimensat(..., AT_SYMLINK_NOFOLLOW)`, best-effort `fchmodat(..., AT_SYMLINK_NOFOLLOW)` (honest no-op where unsupported, e.g. Linux), and policy-gated `fchownat(..., AT_SYMLINK_NOFOLLOW)` via a new `identity_apply_ownership_link` that shares the identity resolver with the fd path. `-J` suppresses only the timestamps; ownership stays governed by the identity opt-in (`--numeric-ids`/`--usermap`/`--groupmap`/`--chown`) exactly like regular files. A symlink has no children, so this is applied immediately at creation.
|
||||
- **Wire:** the shared `STATUS_DIR_TIMES` frame (and metadata on `STATUS_MKDIR` for `--dirs` entries) is a frame-sequence change, so `PROTOCOL_VERSION` was bumped **2.16.0 → 2.17.0**; every version-sensitive test (`--protocol` accepted/rejected values) was updated. The config-frame layout itself is unchanged (the omit booleans already crossed). Non-metadata and `--no-preserve` transfers send no `STATUS_DIR_TIMES` frame and no directory metadata, keeping them byte-identical.
|
||||
|
||||
`--secluded-args` (`🔄 → ⛔ Impossible/Divergence`): a true arg-send protocol would replace the argv-based SSH launch with an in-band channel, and FastSync already builds the remote SSH argv injection-safe (single-quote-escaped shell words), so there is no argument-leak to close; the already-safe behavior is documented in the row and no transport change is made.
|
||||
`--secluded-args` (`🔄 → ❌ Divergent`): a true arg-send protocol would replace the argv-based SSH launch with an in-band channel, and FastSync already builds the remote SSH argv injection-safe (single-quote-escaped shell words), so there is no argument-leak to close; the already-safe behavior is documented in the row and no transport change is made.
|
||||
|
||||
**Wave E (LAST) — Privilege: `--super`/`--no-super` and `--copy-as=USER[:GROUP]` (✅ implemented).** FastSync adopts a **safe-subset + clear-refusal** privilege model: it never blind-elevates and never calls `setuid`/`seteuid`/`setgid`. All privileged operations remain fd-relative and confined below the authorized receive root.
|
||||
|
||||
`--super`/`--no-super` set a receiver-side tri-state `Config->super_mode` (`SUPER_MODE_AUTO`/`ON`/`OFF`). `privilege_super_permitted()` / `privilege_super_mode_permitted()` (src/shared/identity.c) return true for `ON` and `AUTO` (AUTO preserves FastSync's historical best-effort attempt, where the kernel refuses an unprivileged call and the caller skips it) and false only for `OFF`. The gate covers every super-user activity FastSync performs: ownership application (`identity_apply_ownership`/`_link`), char/block device-node creation (`file_save_special_to_disk`), writes into an existing device (`--write-devices`), and the `--fake-super` owner replay. Unprivileged FIFO creation is deliberately unaffected. `--super` does **not** imply `--numeric-ids`: ownership is applied only when an explicit identity policy (`--usermap`/`--groupmap`/`--chown`/`--numeric-ids`/`--copy-as`) is also given. `--no-super` suppresses those activities even for a root receiver. A non-root receiver given `--super` logs one warning at activation (`identity_set_active`); each confined attempt is then refused by the kernel and skipped, never aborting. The confinement floor is unchanged (`file_open_secure_parent`, `O_NOFOLLOW`, root/path checks). Operator control: the server CLI accepts `--no-super`, a veto that forces `OFF` for every connection, refuses any client `--copy-as`, and neutralizes an explicit `--super` (the connection is accepted but no super-user activity is attempted). On a daemon, a module that has not opted in with `client owner = yes` additionally has super-user device activity forced off (see the Daemon Mode notes).
|
||||
`--super`/`--no-super` set a receiver-side tri-state `Config->super_mode` (`SUPER_MODE_AUTO`/`ON`/`OFF`). `privilege_super_permitted()` / `privilege_super_mode_permitted()` (src/shared/identity.c) return true for `ON` and `AUTO` (AUTO preserves FastSync's historical best-effort attempt, where the kernel refuses an unprivileged call and the caller skips it) and false only for `OFF`. The gate covers every super-user activity FastSync performs: ownership application (`identity_apply_ownership`/`_link`), char/block device-node creation (`file_save_special_to_disk`), writes into an existing device (`--write-devices`), and the `--fake-super` owner replay. Unprivileged FIFO creation is deliberately unaffected. `--super` does **not** imply `--numeric-ids`: ownership is applied only when an explicit identity policy (`--usermap`/`--groupmap`/`--chown`/`--numeric-ids`/`--copy-as`) or a preserve-source request (`-o`/`-g`, or `-a`/`--archive`) is also given. `--no-super` suppresses those activities even for a root receiver. A non-root receiver given `--super` logs one warning at activation (`identity_set_active`); each confined attempt is then refused by the kernel and skipped, never aborting. The confinement floor is unchanged (`file_open_secure_parent`, `O_NOFOLLOW`, root/path checks). Operator control: the server CLI accepts `--no-super`, a veto that forces `OFF` for every connection, refuses any client `--copy-as`, and neutralizes an explicit `--super` (the connection is accepted but no super-user activity is attempted). A privileged (root) standalone TCP listener instead defaults to `OFF` and requires the server-only `--allow-super` opt-in to attempt any super-user activity (the flag is rejected with `--stdio`, whose client-composed remote argv must never defeat the default; use a forced command if the default must hold); a non-root server is unchanged. On a daemon, a module that has not opted in with `client owner = yes` additionally has super-user device activity forced off (see the Daemon Mode notes).
|
||||
|
||||
`--copy-as=USER[:GROUP]` is the safe subset. FastSync's receiver is multithreaded, so a real credential switch is unsafe; instead the receiver forces the ownership of **every entry it writes** — regular files, symlinks, directories (including implicitly-created parents), and special nodes — to the resolved target ids through the confined fd-relative identity path. USER is resolved on the client (name, `@N`/bare N, or `*` = client euid); when `:GROUP` is omitted the user's primary gid is used (falling back to `gid == uid` for a numeric id with no local passwd entry). It requires a privileged (root) receiver: an unprivileged receiver refuses the whole transfer at the config handshake, before `STATUS_OK`, so no data is ever written with the wrong ownership. A `--copy-as` chown failure on a capability-restricted root is logged at ERROR (never silently downgraded). `--copy-as` implies metadata (`--no-preserve` is rejected) and `--fake-super` cannot override it. Daemon policy: a `--daemon` receiver refuses **every** client-chosen-ownership / super-user request — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and explicit `--super` — unless the selected module opts in with `client owner = yes`; without that per-module opt-in any client could force arbitrary ownership inside the module root (the standalone listener and the SSH-launched `--stdio` server, which each serve one operator-authorized root, honor these requests). A `--copy-as` chown failure on a capability-restricted root marks the entry as failed rather than reporting success with the wrong owner.
|
||||
`--copy-as=USER[:GROUP]` is the safe subset. FastSync's receiver is multithreaded, so a real credential switch is unsafe; instead the receiver forces the ownership of **every entry it writes** — regular files, symlinks, directories (including implicitly-created parents), and special nodes — to the resolved target ids through the confined fd-relative identity path. USER is resolved on the client (name, `@N`/bare N, or `*` = client euid); when `:GROUP` is omitted the user's primary gid is used (falling back to `gid == uid` for a numeric id with no local passwd entry). It requires a privileged (root) receiver: an unprivileged receiver refuses the whole transfer at the config handshake, before `STATUS_OK`, so no data is ever written with the wrong ownership. A `--copy-as` chown failure on a capability-restricted root is logged at ERROR (never silently downgraded). `--copy-as` implies metadata (`--no-preserve` is rejected) and `--fake-super` cannot override it. Daemon policy: a `--daemon` receiver refuses **every** client-chosen-ownership / super-user request — `--numeric-ids`, `--chown`, `--usermap`/`--groupmap`, `--fake-super`, `--copy-as`, and explicit `--super` — unless the selected module opts in with `client owner = yes`; without that per-module opt-in any client could force arbitrary ownership inside the module root (a root standalone TCP listener, which serves one operator-authorized root, honors these requests only when started with `--allow-super`; the flag is rejected with `--stdio`). A `--copy-as` chown failure on a capability-restricted root marks the entry as failed rather than reporting success with the wrong owner.
|
||||
|
||||
**Wire:** two trailing config-frame blocks after the `--iconv` spec, in fixed order — `send_privilege_options`/`receive_privilege_options` (one `super_mode` int, validated `0..2`), then `send_copy_as_options`/`receive_copy_as_options` (presence int + two int32 ids, validated `>= 0`, with `copy_as_set ⇒ use_metadata`). `PROTOCOL_VERSION` bumped **2.17.0 → 2.18.0**. **Divergences from rsync:** rsync's `--super` elevates the receiver and `--copy-as` actually switches its credentials; FastSync never elevates and only permits/forwards confined attempts, and `--copy-as` forces ownership rather than switching identity.
|
||||
|
||||
**Post-Phase-7 Summary (after Waves A–E).** ✅143 / 🔀0 / ⛔4 / ⚠️0 / 🔄0 / ❌0 = 147. The 3 `🔀 Alt Arg` rows (`-a`, `-p`, `-z`) are ✅ (Wave A). All 10 prior `⚠️ Partial` rows are resolved to ✅ (`-S`, `-P`, `--block-size`, `--fake-super`, `--devices`, `--copy-devices`, `--write-devices`) or ⛔ (`--stderr=client`, `-N/--crtimes`, `--specials` for the impossible socket case). The 3 `🔄 Compatibility No-op` rows are resolved: `-O`/`-J` are now real ✅ (Wave D), `--secluded-args` is ⛔. The **Impossible/Divergence** bucket holds the 4 physically-impossible/divergent flags: `--stderr=client`, `-N/--crtimes`, `--specials` (sockets), `--secluded-args`. The last two `❌ Not Implemented` rows — `--super` and `--copy-as=USER[:GROUP]` — are now ✅ (Wave E). **No `❌ Not Implemented` rows remain.**
|
||||
**Current honest status (protocol 2.23.0).** ✅ Parity 83 / ⚠️ Caveat 63 / ❌ Divergent 4 = 150 rows. Earlier revisions of this document reported "143 ✅ / 0 divergence / 0 partial"; that conflated "parsed and tested" with "rsync parity", because many rows carried documented behavioral differences and some short options were not parsed at all. The reclassification makes the differences explicit and the rsync-parity wave closed the genuine gaps (short options, clustering, checksum/compression choices, seed randomization, timeout defaults, delete scoping and partial limits, verbatim symlink storage, socket recreation, `--chmod`, and more — see the next section). The four ❌ rows are `--stderr=client` (no rsync client-message channel), `-N/--crtimes` (no portable setter), `--protocol=NUM` (only the current wire version is accepted), and `-s/--secluded-args` (accepted no-op). `--specials` is now ✅ because sockets are recreated with `mknod(S_IFSOCK)`. **No `❌ Not Implemented` rows remain.**
|
||||
|
||||
**Preserve-attribute split (protocol 2.21.0 → 2.22.0) — ✅ implemented.** FastSync splits the former single metadata bundle into four independent, rsync-compatible per-attribute flags — `-p/--perms`, `-t/--times`, `-o/--owner`, `-g/--group` — each with a negation (`--no-perms`/`--no-times`/`--no-owner`/`--no-group`, short `--no-p`/`--no-t`/`--no-o`/`--no-g`), plus `--no-preserve` clearing all four. `-a/--archive` is now full rsync `-rlptgoD` (owner and group included, though their application stays privilege-gated), `-A/--acls` implies `-p`, `-X/--xattrs` does not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply `-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the user explicitly negated them. Wire: the binary config frame gains four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/`preserve_group`) after `omit_link_times`, so `PROTOCOL_VERSION` is bumped **2.21.0 → 2.22.0**; the fixed-width `FileMetadata` layout is unchanged and the receiver gates the metadata frame on a derived `use_metadata`. Receiver behavior: each attribute is applied independently, directory modes are applied under `-p` (at the end of the transfer, alongside dir times), symlink mode under `-p`, and `-O/--omit-dir-times` suppresses directory times only. Documented divergences as of 2.22.0, **all but (d)/(e) removed by the rsync-parity wave (protocol 2.23.0)**: (a) the mode-masking divergence is **gone** — under `-p` the source mode is now copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky; (b) a brand-new file without `-p` still gets `source_mode & ~umask` when metadata is present (else the historical fixed `0644`), and a new *directory* without `-p` still uses FastSync's `0755` default; (c) the `--chmod`-implies-`-p` divergence is **gone** — `--chmod` no longer implies `-p` (rsync parity); (d) `-o`/`-g` map by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); (e) a daemon module without `client owner = yes` does not refuse a plain `-a`/`-o`/`-g` — it forces super off, applies no ownership, and logs a warning, while explicit `--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused.
|
||||
|
||||
## Rsync-Parity Wave (protocol 2.23.0)
|
||||
|
||||
This wave closed the remaining CLI, filesystem, ownership, deletion, and output
|
||||
gaps against rsync 3.4.1. It is a wire change: `PROTOCOL_VERSION` moved
|
||||
**2.22.0 → 2.23.0** because the delete manifest gained a synchronized-directory
|
||||
section and the terminal status gained `STATUS_DELETE_LIMIT` (see the deletion
|
||||
notes above). Everything below is implemented and covered by unit and
|
||||
integration tests unless it is explicitly listed as a limitation.
|
||||
|
||||
### CLI parsing
|
||||
|
||||
- **Short options now parsed:** `-r` (`--recursive`), `-b` (`--backup`),
|
||||
`-L` (`--copy-links`), and `-B` (`--block-size`/`--delta-block`) are accepted
|
||||
as rsync spells them.
|
||||
- **rsync short-option clustering:** a token is expanded before parsing, so
|
||||
`-av` → `-a -v`, `-aAX` → `-a -A -X`, `-rlpt` → `-r -l -p -t`, and so on.
|
||||
A value-taking short option consumes the remainder of its token
|
||||
(`-B1000` → `-B 1000`, `-essh` → `-e ssh`, `-MOPT` → `-M OPT`), with an
|
||||
optional leading `=` dropped (`-B=1000`); a value-taking option written alone
|
||||
takes the next argv entry, which is copied verbatim so a value that happens to
|
||||
start with `-` (e.g. `--filter "- *.tmp"`) is not mistaken for a cluster.
|
||||
- **Inline/attached long values:** `--opt=value` is accepted uniformly, and each
|
||||
expanded token is mapped back to its original argv index so positional
|
||||
arguments stay correct.
|
||||
- **`-c` implies the checksum quick-check.** `-c`/`--checksum` sets the
|
||||
incremental checksum comparison rather than doing nothing on its own; like
|
||||
rsync, `-c` does not imply `-t`.
|
||||
|
||||
### Checksums and compression
|
||||
|
||||
- **`--checksum-choice`/`--cc`** accepts `xxh64` (default), `xxhash`, `xxh3`,
|
||||
`xxh128`, `md5`, and `auto`; `md4`, `sha1`, `none`, and the two-name
|
||||
`transfer,pre-transfer` form are **rejected by name**.
|
||||
- **`--checksum-seed=0` is randomized per transfer** (the chosen seed is sent to
|
||||
the receiver), matching rsync; an explicit non-zero seed is used verbatim.
|
||||
- **`--compress-choice`/`--zc`** accepts `zstd` (default), `none`, and `auto`;
|
||||
rsync's `lz4`/`zlib`/`zlibx` are **rejected by name**.
|
||||
- **`--skip-compress`** uses rsync 3.4.1's built-in default suffix list when no
|
||||
list is supplied; an explicit list replaces it.
|
||||
- **`--no-whole-file`** is accepted as the rsync spelling that clears
|
||||
`-W`/`--whole-file`.
|
||||
|
||||
### Timeouts and limits
|
||||
|
||||
- **`--timeout` defaults to 0 (disabled) and `--contimeout` to 60 s; `0`
|
||||
disables either**, matching rsync.
|
||||
- **`--max-alloc=0` means "no local allocation limit"** (rsync semantics). A
|
||||
standalone server still keeps its own ceiling for the peer it serves.
|
||||
|
||||
### Filesystem and deletion semantics
|
||||
|
||||
- **`--temp-dir` is confined to the receive root on the receiver:** a relative
|
||||
dir resolves below it; an absolute path or one containing `..` is rejected.
|
||||
An `EXDEV` install falls back to a non-atomic copy instead of aborting.
|
||||
- **Deletion scoping:** the manifest carries the synchronized directories, so
|
||||
the extras walk only visits their subtrees; `--files-from` subsets no longer
|
||||
delete untransmitted paths outside the listed directories.
|
||||
- **`--delete-excluded`** removes filter-excluded mirrors but never
|
||||
`--max-size`/`--min-size`-pruned mirrors (separate, always-on protection).
|
||||
- **Destination symlinks** are unlinked by name, never followed; a directory
|
||||
still holding one survives.
|
||||
- **`--max-delete=N` is partial:** delete up to N, skip the rest, exit **25**.
|
||||
`--delete-missing-args` removals draw from the same budget.
|
||||
- **`--force` is honored during `--delay-updates` publication.**
|
||||
- **`-x`/`--one-file-system` emits the mount-point directory entry** (an empty
|
||||
directory at the destination) without descending into it.
|
||||
- **`--include`/`--exclude` are an ordered first-match rule list**, evaluated
|
||||
like `--filter`/`-F`/`-C` (first match wins), so an earlier rule can override a
|
||||
later one.
|
||||
|
||||
### Ownership and metadata
|
||||
|
||||
- **`--numeric-ids` is a mapping modifier only** — it changes *how* ids map, not
|
||||
*whether* ownership is applied; combine it with `-o`/`-g`, `-a`, or an
|
||||
explicit map.
|
||||
- **`--usermap`/`--groupmap`** support names, `@N`/bare `N` ids, inclusive
|
||||
`LOW-HIGH` ranges, `*`, empty-`FROM` (unnamed ids), and receiver-resolved `TO`
|
||||
names.
|
||||
- **`--chown` conflicts with `--usermap`/`--groupmap` on the same side** and is a
|
||||
clear configuration error (matching rsync) instead of an order-dependent
|
||||
winner.
|
||||
- **`--fake-super` never real-chowns.** It records the *resolved* owner (the
|
||||
active mapping, else the source id) in `user.fastsync.stat` for a later
|
||||
privileged restore and replays only mode/times. Directory ownership and
|
||||
directory xattrs/ACLs are preserved alongside file entries.
|
||||
- **`--chmod`** implements rsync's `D`/`F`/`X` selectors, `s`/`t`, append
|
||||
semantics, does not imply `-p`, and applies its changes without sanitization.
|
||||
|
||||
### Symlinks and special files
|
||||
|
||||
- **`-l`/`--links` stores symlink targets verbatim** (absolute and `..`-bearing
|
||||
targets included), matching rsync. `--safe-links`, `--copy-unsafe-links`, and
|
||||
`--munge-links` (which now uses rsync's `/rsyncd-munged/` marker) match rsync
|
||||
and are applied sender-side.
|
||||
- **`--specials` recreates unix sockets** with `mknodat(..., S_IFSOCK)`, so
|
||||
`-D`/`--devices --specials` now covers the full rsync node set.
|
||||
- **`--copy-devices`** is implemented (see its caveat below).
|
||||
|
||||
### Output
|
||||
|
||||
- **`-i`/`--out-format`** print rsync-style change lines; **`--list-only`**
|
||||
scans the source only and contacts no server; **`-h`** uses rsync's decimal
|
||||
units; **`--progress`** is an aggregate line; **`--stats`** prints the counters
|
||||
FastSync can observe locally (receiver-only counters are 0).
|
||||
- **Server `--port`** is an alias of the `-p <port>` TCP listen port
|
||||
(`--dparam port=` overrides the daemon config).
|
||||
|
||||
### Known intentional divergences and limitations
|
||||
|
||||
These remain after the wave; they are the reasons a row above is ⚠️.
|
||||
|
||||
- **Symlink target containment is not enforced receiver-side by default.**
|
||||
Verbatim storage is rsync parity, but a destination later consumed by a
|
||||
link-following tool can follow a link outside the receive root. Use
|
||||
`--safe-links` when the source is untrusted. `--trust-sender` does **not**
|
||||
affect symlink targets.
|
||||
- **`--temp-dir` absolute/foreign-filesystem paths are rejected by the
|
||||
receiver** (rsync's daemon also confines; standalone rsync differs).
|
||||
- **`--copy-devices` reads a bounded `st_size`** rather than rsync's unbounded
|
||||
device read.
|
||||
- **A broken symlink referent under `--copy-links`/`--copy-unsafe-links` exits 0**
|
||||
where rsync exits 23.
|
||||
- **New directories without `-p` still use FastSync's `0755` creation default**
|
||||
rather than `source & ~umask`; directory metadata is only applied when a
|
||||
directory attribute is requested.
|
||||
- **`--stats` receiver-only counters** (matched data, file-list bytes, deleted
|
||||
count) are reported as 0; `--progress` is an aggregate line, not per-file.
|
||||
- **`--password-file`/`--early-input`/`--hash-credentials`/`--iterations` are
|
||||
FastSync-native** (SCRAM/PBKDF2), not rsync semantics; the batch format is not
|
||||
rsync-interoperable.
|
||||
- **xattr/ACL namespace policy** permits only `user.*` and
|
||||
`system.posix_acl_*` when `-A` is negotiated (stricter than rsync).
|
||||
- **`--stop-at` remains a FastSync-flexible parser** (client-only, not
|
||||
serialized); `--stop-after` matches rsync.
|
||||
- **Push-only model and a non-rsync wire protocol** remain by design;
|
||||
`--protocol` accepts only the current version and `-s`/`--secluded-args` is an
|
||||
accepted no-op.
|
||||
|
||||
## Packed Metadata Frame (protocol 2.20.0)
|
||||
|
||||
A file's metadata used to cross the wire as up to 12 separate per-field framed
|
||||
messages (a present flag followed by mode/uid/gid/mtime/atime/crtime writes),
|
||||
which cost ~11 extra protocol frames per file on many-small-file trees. FastSync
|
||||
now sends the metadata as ONE packed frame: a single `int32` present flag
|
||||
(`0` = absent) followed, when present, by the fixed
|
||||
`FILE_METADATA_WIRE_SIZE`-byte (68-byte) field record already emitted by the
|
||||
shared `metadata_to_buf()`/`metadata_from_buf()` chunk codec. Absent metadata is
|
||||
a lone `int32` zero. The encoded field layout is unchanged (only the framing
|
||||
collapses), so chunk-serialized blobs remain byte-identical. Protocol data is an
|
||||
unframed byte stream, so the packed encoding is byte-for-byte identical to the
|
||||
old field-by-field writes; `PROTOCOL_VERSION` was bumped `2.19.0 → 2.20.0` as a
|
||||
deliberate lockstep-release marker rather than because of a
|
||||
desynchronization. The strict same-version handshake rejects any mismatch before
|
||||
a byte of the frame is parsed.
|
||||
|
||||
### Recommended Delivery Order
|
||||
|
||||
@@ -874,7 +1064,7 @@ Ranked by user demand, implementation complexity, and interoperability impact (_
|
||||
|
||||
| Feature | Description |
|
||||
|---------|-------------|
|
||||
| `-j` / `--threads` | Multithreaded pipeline (scanner/loader/sender) (renamed from `-m` in Phase 7 Wave A; `-m` is now rsync `--prune-empty-dirs`) |
|
||||
| `-j` / `--threads[=N]` | Multithreaded pipeline (scanner/loader/sender); `N` (1–256) sizes the parallel scanner worker pool, bare `-j`/`--threads` uses the built-in default (renamed from `-m` in Phase 7 Wave A; `-m` is now rsync `--prune-empty-dirs`) |
|
||||
| `--chunk-serialization` | Chunk serialization mode (long form only; `-s` is now rsync `--secluded-args`) |
|
||||
| `--sendfile` | Zero-copy sendfile() syscall (TCP only) (long form only; `-f` is now rsync `--filter`) |
|
||||
| `-z [level]` / `--compress` | zstd compression level (1-22) (`-c` is now rsync `--checksum`) |
|
||||
|
||||
+412
-67
@@ -3,19 +3,24 @@
|
||||
|
||||
Compares FastSync configs against rsync (no compression) and rsync+zstd.
|
||||
Data is ~75% random/incompressible and ~25% structured/compressible by default,
|
||||
controllable via --random-ratio.
|
||||
controllable via --random-ratio. Transfers are verified by default (source and
|
||||
destination must match) so a fast-but-broken copy is never counted.
|
||||
|
||||
Usage:
|
||||
python3 benchmark/bench.py
|
||||
python3 benchmark/bench.py --runs 5 --profiles lan wan
|
||||
python3 benchmark/bench.py --random-ratio 0.5 --size-mb 50
|
||||
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
|
||||
python3 benchmark/bench.py --warm --runs 3
|
||||
python3 benchmark/bench.py --output json
|
||||
"""
|
||||
import argparse
|
||||
import filecmp
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import random
|
||||
import shlex
|
||||
import shutil
|
||||
import socket
|
||||
import statistics
|
||||
@@ -25,8 +30,11 @@ import tempfile
|
||||
import time
|
||||
|
||||
PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, "build")
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server")]
|
||||
DEFAULT_BUILD_DIR = "build-bench"
|
||||
# Populated by configure_build_dirs(); default to the dedicated bench dir so
|
||||
# importing this module never depends on the user's existing build/ tree.
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, DEFAULT_BUILD_DIR)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
BENCH_DIR = os.path.join(PROJECT_ROOT, "bench_data")
|
||||
|
||||
@@ -44,10 +52,11 @@ NETWORK_PROFILES = {
|
||||
|
||||
FASTSYNC_CONFIGS = [
|
||||
{"name": "fastsync", "flags": [], "tool": "fastsync"},
|
||||
{"name": "fastsync -c", "flags": ["-c"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m", "flags": ["-m"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m -c", "flags": ["-m", "-c"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m -c -s", "flags": ["-m", "-c", "-s"], "tool": "fastsync"},
|
||||
{"name": "fastsync -z", "flags": ["-z"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j", "flags": ["-j"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j -z", "flags": ["-j", "-z"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j -z --chunk-serialization", "flags": ["-j", "-z", "--chunk-serialization"], "tool": "fastsync"},
|
||||
{"name": "fastsync --sendfile", "flags": ["--sendfile"], "tool": "fastsync"},
|
||||
]
|
||||
|
||||
RSYNC_CONFIGS = [
|
||||
@@ -56,6 +65,7 @@ RSYNC_CONFIGS = [
|
||||
{"name": "rsync -z --zstd", "flags": ["-z", "--zc", "zstd"],"tool": "rsync"},
|
||||
]
|
||||
|
||||
|
||||
class RsyncDaemon:
|
||||
"""Manages an rsync daemon for network-fair benchmarking."""
|
||||
|
||||
@@ -117,6 +127,12 @@ STRUCTURED_FILES = {
|
||||
"nested/another.txt": b"another nested file\n" * 50,
|
||||
}
|
||||
|
||||
# Repeated text used to synthesize genuinely compressible filler of any size.
|
||||
COMPRESSIBLE_TEXT = (
|
||||
b"FastSync benchmark payload: the quick brown fox jumps over the lazy dog. "
|
||||
b"0123456789 ABCDEFGHIJKLMNOPQRSTUVWXYZ abcdefghijklmnopqrstuvwxyz\n"
|
||||
)
|
||||
|
||||
|
||||
class Progress:
|
||||
"""Simple progress bar with ETA."""
|
||||
@@ -151,35 +167,127 @@ class Progress:
|
||||
sys.stderr.flush()
|
||||
|
||||
|
||||
def write_compressible(path, nbytes):
|
||||
"""Write exactly nbytes of highly compressible, repeated text content."""
|
||||
if nbytes <= 0:
|
||||
return
|
||||
block = COMPRESSIBLE_TEXT * (max(1, 8192 // len(COMPRESSIBLE_TEXT)) + 1)
|
||||
remaining = nbytes
|
||||
with open(path, "wb") as f:
|
||||
while remaining > 0:
|
||||
piece = block if remaining >= len(block) else block[:remaining]
|
||||
f.write(piece)
|
||||
remaining -= len(piece)
|
||||
|
||||
|
||||
def generate_bench_data(source_dir, size_mb=25, random_ratio=0.75):
|
||||
"""Generate test data. ~random_ratio is incompressible, rest is structured."""
|
||||
"""Generate test data honouring the requested random/compressible split.
|
||||
|
||||
Exactly ``random_ratio * target`` bytes are incompressible random data and
|
||||
the remainder is genuinely compressible structured/repeated content. The
|
||||
measured byte counts are returned so callers can report the real mix.
|
||||
"""
|
||||
if os.path.exists(source_dir):
|
||||
shutil.rmtree(source_dir)
|
||||
os.makedirs(source_dir)
|
||||
|
||||
target = size_mb * 1024 * 1024
|
||||
structured_budget = int(target * (1 - random_ratio))
|
||||
written = 0
|
||||
random_budget = int(target * random_ratio)
|
||||
compressible_budget = target - random_budget
|
||||
compressible_written = 0
|
||||
random_written = 0
|
||||
files = 0
|
||||
|
||||
# A handful of fixed, human-meaningful files (directories, small files, a
|
||||
# binary blob) as long as they fit inside the compressible budget.
|
||||
for rel_path, content in STRUCTURED_FILES.items():
|
||||
if written >= structured_budget:
|
||||
if compressible_written + len(content) > compressible_budget:
|
||||
break
|
||||
full_path = os.path.join(source_dir, rel_path)
|
||||
os.makedirs(os.path.dirname(full_path), exist_ok=True)
|
||||
with open(full_path, "wb") as f:
|
||||
f.write(content)
|
||||
written += len(content)
|
||||
compressible_written += len(content)
|
||||
files += 1
|
||||
|
||||
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
|
||||
# Fill the rest of the compressible share with generated repeated content.
|
||||
if compressible_written < compressible_budget:
|
||||
os.makedirs(os.path.join(source_dir, "compressible"), exist_ok=True)
|
||||
i = 0
|
||||
while written < target:
|
||||
chunk_size = min(5 * 1024 * 1024, target - written)
|
||||
with open(os.path.join(source_dir, f"bulk/file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk_size))
|
||||
written += chunk_size
|
||||
while compressible_written < compressible_budget:
|
||||
chunk = min(1024 * 1024, compressible_budget - compressible_written)
|
||||
write_compressible(os.path.join(source_dir, "compressible", f"text_{i}.dat"), chunk)
|
||||
compressible_written += chunk
|
||||
files += 1
|
||||
i += 1
|
||||
|
||||
return written
|
||||
# Incompressible share.
|
||||
if random_written < random_budget:
|
||||
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
|
||||
i = 0
|
||||
while random_written < random_budget:
|
||||
chunk = min(5 * 1024 * 1024, random_budget - random_written)
|
||||
with open(os.path.join(source_dir, "bulk", f"file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk))
|
||||
random_written += chunk
|
||||
files += 1
|
||||
i += 1
|
||||
|
||||
return {
|
||||
"total_bytes": compressible_written + random_written,
|
||||
"compressible_bytes": compressible_written,
|
||||
"random_bytes": random_written,
|
||||
"files": files,
|
||||
}
|
||||
|
||||
|
||||
def list_relative_files(root):
|
||||
"""Return the set of file paths (relative to root) under a directory."""
|
||||
found = set()
|
||||
for dirpath, _dirnames, filenames in os.walk(root):
|
||||
for name in filenames:
|
||||
full = os.path.join(dirpath, name)
|
||||
found.add(os.path.relpath(full, root))
|
||||
return found
|
||||
|
||||
|
||||
def verify_transfer(source_dir, dest_dir):
|
||||
"""Recursively check dest matches source (paths, sizes, content).
|
||||
|
||||
Returns (ok, detail). Content is compared byte-for-byte, never hashed, so
|
||||
collisions are impossible. This is intentionally not part of the timing.
|
||||
"""
|
||||
if not os.path.isdir(dest_dir):
|
||||
return False, "destination directory missing"
|
||||
src_files = list_relative_files(source_dir)
|
||||
dst_files = list_relative_files(dest_dir)
|
||||
if src_files != dst_files:
|
||||
missing = src_files - dst_files
|
||||
extra = dst_files - src_files
|
||||
return False, f"path set mismatch (missing {len(missing)}, extra {len(extra)})"
|
||||
for rel in sorted(src_files):
|
||||
src = os.path.join(source_dir, rel)
|
||||
dst = os.path.join(dest_dir, rel)
|
||||
if os.path.getsize(src) != os.path.getsize(dst):
|
||||
return False, f"size mismatch: {rel}"
|
||||
if not filecmp.cmp(src, dst, shallow=False):
|
||||
return False, f"content mismatch: {rel}"
|
||||
return True, ""
|
||||
|
||||
|
||||
def percentile(values, pct):
|
||||
"""Linear-interpolation percentile (matches numpy's default method)."""
|
||||
if not values:
|
||||
return None
|
||||
ordered = sorted(values)
|
||||
if len(ordered) == 1:
|
||||
return ordered[0]
|
||||
rank = (len(ordered) - 1) * (pct / 100.0)
|
||||
low = math.floor(rank)
|
||||
high = math.ceil(rank)
|
||||
if low == high:
|
||||
return ordered[int(rank)]
|
||||
return ordered[low] + (ordered[high] - ordered[low]) * (rank - low)
|
||||
|
||||
|
||||
def find_free_port():
|
||||
@@ -207,18 +315,45 @@ def wait_proc(proc, timeout=5):
|
||||
proc.wait()
|
||||
|
||||
|
||||
def _tc_base_cmd():
|
||||
"""Return the command prefix for tc, honouring root vs sudo."""
|
||||
tc = shutil.which("tc")
|
||||
if not tc:
|
||||
raise RuntimeError(
|
||||
"tc (iproute2) not found in PATH; install iproute2 to use network profiles")
|
||||
if os.geteuid() == 0:
|
||||
return [tc]
|
||||
sudo = shutil.which("sudo")
|
||||
if sudo:
|
||||
return [sudo, tc]
|
||||
raise RuntimeError(
|
||||
"applying network limits requires root or sudo; "
|
||||
"re-run as root or install sudo")
|
||||
|
||||
|
||||
def _run_tc(args, check=True):
|
||||
return subprocess.run(_tc_base_cmd() + args, check=check, capture_output=True)
|
||||
|
||||
|
||||
def netem_apply(delay=None, jitter=None, throughput=None, loss=None):
|
||||
"""Apply tc/netem rules to loopback. Pass None to skip a parameter."""
|
||||
netem_reset()
|
||||
cmd = ["sudo", "tc", "qdisc", "add", "dev", "lo", "root", "netem"]
|
||||
params = []
|
||||
if throughput:
|
||||
cmd += ["rate", throughput]
|
||||
params += ["rate", throughput]
|
||||
if delay:
|
||||
cmd += ["delay", delay, jitter or "0ms"]
|
||||
params += ["delay", delay, jitter or "0ms"]
|
||||
if loss:
|
||||
cmd += ["loss", loss]
|
||||
if len(cmd) > 6:
|
||||
subprocess.run(cmd, check=True, capture_output=True)
|
||||
params += ["loss", loss]
|
||||
if not params:
|
||||
return
|
||||
try:
|
||||
_run_tc(["qdisc", "add", "dev", "lo", "root", "netem"] + params)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
detail = exc.stderr.decode(errors="replace").strip() if exc.stderr else str(exc)
|
||||
raise RuntimeError(f"failed to apply network profile via tc/netem: {detail}") from exc
|
||||
except RuntimeError:
|
||||
raise
|
||||
|
||||
|
||||
def netem_apply_profile(profile_name):
|
||||
@@ -235,7 +370,11 @@ def netem_apply_profile(profile_name):
|
||||
|
||||
|
||||
def netem_reset():
|
||||
subprocess.run("sudo tc qdisc del dev lo root".split(), capture_output=True)
|
||||
"""Best-effort removal of any loopback qdisc. Always safe to call."""
|
||||
try:
|
||||
_run_tc(["qdisc", "del", "dev", "lo", "root"], check=False)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def run_fastsync(source_dir, dest_dir, flags, port):
|
||||
@@ -252,8 +391,10 @@ def run_fastsync(source_dir, dest_dir, flags, port):
|
||||
duration = time.monotonic() - start
|
||||
if result.returncode == 0:
|
||||
return duration
|
||||
sys.stderr.write(f" fastsync failed (exit {result.returncode}): "
|
||||
f"{result.stderr.strip()[:500]}\n")
|
||||
except subprocess.TimeoutExpired:
|
||||
pass
|
||||
sys.stderr.write(" fastsync timed out after 120s\n")
|
||||
return None
|
||||
|
||||
|
||||
@@ -269,8 +410,10 @@ def run_rsync(source_dir, dest_dir, flags, rsync_daemon=None):
|
||||
duration = time.monotonic() - start
|
||||
if result.returncode == 0:
|
||||
return duration
|
||||
sys.stderr.write(f" rsync failed (exit {result.returncode}): "
|
||||
f"{result.stderr.strip()[:500]}\n")
|
||||
except subprocess.TimeoutExpired:
|
||||
pass
|
||||
sys.stderr.write(" rsync timed out after 120s\n")
|
||||
return None
|
||||
|
||||
|
||||
@@ -282,7 +425,81 @@ def run_transfer(config, source_dir, dest_dir, port=None, rsync_daemon=None):
|
||||
return run_fastsync(source_dir, dest_dir, config["flags"], port)
|
||||
|
||||
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=None):
|
||||
def apply_incremental_changes(source_dir, target_bytes):
|
||||
"""Add and modify a few files so a warm transfer has real work to do.
|
||||
|
||||
Returns a mutation record (changed byte count plus enough data to revert
|
||||
and re-apply it) so every warm run can start from a pristine source.
|
||||
"""
|
||||
modified_n = 3
|
||||
added_n = 2
|
||||
per_file = max(4096, target_bytes // (modified_n + added_n))
|
||||
modified = {}
|
||||
added = {}
|
||||
changed = 0
|
||||
|
||||
existing = sorted(list_relative_files(source_dir))
|
||||
if existing:
|
||||
step = max(1, len(existing) // modified_n)
|
||||
for rel in existing[::step][:modified_n]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
original_size = os.path.getsize(path)
|
||||
with open(path, "ab") as f:
|
||||
f.write(random.randbytes(per_file))
|
||||
modified[rel] = (original_size, per_file)
|
||||
changed += per_file
|
||||
|
||||
for i in range(added_n):
|
||||
os.makedirs(os.path.join(source_dir, "incremental"), exist_ok=True)
|
||||
rel = os.path.join("incremental", f"new_{i}.dat")
|
||||
write_compressible(os.path.join(source_dir, rel), per_file)
|
||||
added[rel] = per_file
|
||||
changed += per_file
|
||||
|
||||
return {"changed": changed, "modified": modified, "added": added}
|
||||
|
||||
|
||||
def revert_incremental_changes(source_dir, mutation):
|
||||
"""Undo apply_incremental_changes so the source is pristine again."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (original_size, _appended) in mutation["modified"].items():
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
with open(path, "r+b") as f:
|
||||
f.truncate(original_size)
|
||||
for rel in mutation["added"]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
os.remove(path)
|
||||
|
||||
|
||||
def reapply_incremental_changes(source_dir, mutation):
|
||||
"""Re-apply a mutation after an untimed pristine seed transfer."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (_original_size, appended) in mutation["modified"].items():
|
||||
with open(os.path.join(source_dir, rel), "ab") as f:
|
||||
f.write(random.randbytes(appended))
|
||||
for rel, size in mutation["added"].items():
|
||||
write_compressible(os.path.join(source_dir, rel), size)
|
||||
|
||||
|
||||
def expected_received_root(dest_dir, source_dir, tool):
|
||||
"""Where a tool places transferred files inside dest_dir.
|
||||
|
||||
FastSync mirrors the absolute source path under dest_dir (see the
|
||||
integration suite's get_dest_received_dir); rsync copies the source tree
|
||||
contents directly into dest_dir.
|
||||
"""
|
||||
if tool == "rsync":
|
||||
return dest_dir
|
||||
return os.path.join(dest_dir, os.path.abspath(source_dir).lstrip(os.sep))
|
||||
|
||||
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
measure_bytes, verify=True, warm=False, mutation=None,
|
||||
progress=None):
|
||||
"""Run benchmark for all configs, returns list of results."""
|
||||
is_limited = profile_name != "unlimited"
|
||||
has_rsync = any(c["tool"] == "rsync" for c in configs)
|
||||
@@ -298,7 +515,10 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
results = []
|
||||
for config in configs:
|
||||
times = []
|
||||
invalid = 0
|
||||
for run_idx in range(runs):
|
||||
if warm:
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
if os.path.exists(dest_dir):
|
||||
shutil.rmtree(dest_dir)
|
||||
os.makedirs(dest_dir, exist_ok=True)
|
||||
@@ -306,15 +526,33 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
port = find_free_port()
|
||||
server = None
|
||||
try:
|
||||
if config["tool"] == "fastsync":
|
||||
if config["tool"] == "fastsync" or warm:
|
||||
server = subprocess.Popen(
|
||||
SERVER_CMD + ["-p", str(port)],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
)
|
||||
wait_for_port(port)
|
||||
|
||||
if warm:
|
||||
seed = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if seed is None:
|
||||
invalid += 1
|
||||
sys.stderr.write(" warm-mode seeding failed; run not counted\n")
|
||||
continue
|
||||
reapply_incremental_changes(source_dir, mutation)
|
||||
|
||||
t = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if t is not None:
|
||||
if t is None:
|
||||
invalid += 1
|
||||
elif verify:
|
||||
root = expected_received_root(dest_dir, source_dir, config["tool"])
|
||||
ok, detail = verify_transfer(source_dir, root)
|
||||
if ok:
|
||||
times.append(t)
|
||||
else:
|
||||
invalid += 1
|
||||
sys.stderr.write(f" verification FAILED ({detail}); run not counted\n")
|
||||
else:
|
||||
times.append(t)
|
||||
finally:
|
||||
if server:
|
||||
@@ -327,15 +565,22 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
"config": config["name"],
|
||||
"tool": config["tool"],
|
||||
"profile": profile_name,
|
||||
"warm": warm,
|
||||
"runs": len(times),
|
||||
"invalid": invalid,
|
||||
"times": [round(t, 4) for t in times],
|
||||
}
|
||||
if times:
|
||||
entry["p50"] = round(statistics.median(times), 4)
|
||||
entry["p95"] = round(sorted(times)[int(len(times) * 0.95)], 4) if len(times) > 1 else entry["p50"]
|
||||
p50 = percentile(times, 50)
|
||||
p95 = percentile(times, 95)
|
||||
entry["p50"] = round(p50, 4)
|
||||
entry["p95"] = round(p95, 4)
|
||||
entry["min"] = round(min(times), 4)
|
||||
entry["max"] = round(max(times), 4)
|
||||
entry["stdev"] = round(statistics.stdev(times), 4) if len(times) > 1 else 0.0
|
||||
if measure_bytes:
|
||||
entry["throughput_mbps"] = round(
|
||||
(measure_bytes / (1024 * 1024)) / p50, 3)
|
||||
results.append(entry)
|
||||
return results
|
||||
finally:
|
||||
@@ -345,44 +590,59 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
netem_reset()
|
||||
|
||||
|
||||
def print_table(results, total_bytes, random_ratio):
|
||||
def print_table(results, measure_bytes, stats, warm):
|
||||
"""Print results as a human-readable table grouped by profile."""
|
||||
profiles = {}
|
||||
for r in results:
|
||||
profiles.setdefault(r["profile"], []).append(r)
|
||||
|
||||
total = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total * 100 if total else 0
|
||||
rand_pct = stats["random_bytes"] / total * 100 if total else 0
|
||||
|
||||
for profile, entries in profiles.items():
|
||||
params = NETWORK_PROFILES.get(profile, {})
|
||||
print(f"\n{'=' * 85}")
|
||||
print(f"\n{'=' * 95}")
|
||||
print(f" Profile: {profile.upper()}")
|
||||
if params.get("rate"):
|
||||
print(f" Network: {params['rate']}, {params['delay']} +/- {params['jitter']}, loss {params['loss']}")
|
||||
else:
|
||||
print(f" Network: unlimited")
|
||||
print(f" Data: {total_bytes / (1024*1024):.1f} MB ({random_ratio*100:.0f}% random, {(1-random_ratio)*100:.0f}% compressible)")
|
||||
print(f"{'=' * 85}")
|
||||
print(f" Data: {total / (1024*1024):.1f} MB "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)")
|
||||
if warm:
|
||||
print(f" Mode: warm (incremental) — measured {measure_bytes / (1024*1024):.2f} MB "
|
||||
f"changed after an untimed full seed")
|
||||
else:
|
||||
print(" Mode: cold (full copy)")
|
||||
print(f"{'=' * 95}")
|
||||
|
||||
fs_entries = [e for e in entries if e.get("tool") == "fastsync"]
|
||||
rsync_entries = [e for e in entries if e.get("tool") == "rsync"]
|
||||
|
||||
header = (f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} "
|
||||
f"{'stdev':>8} {'MB/s':>9} {'runs':>5} {'bad':>4}")
|
||||
rule = (f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} "
|
||||
f"{'-' * 8} {'-' * 9} {'-' * 5} {'-' * 4}")
|
||||
|
||||
if fs_entries:
|
||||
print(f"\n FastSync:")
|
||||
print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
print(header)
|
||||
print(rule)
|
||||
for e in sorted(fs_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if rsync_entries:
|
||||
print(f"\n rsync:")
|
||||
print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
print(header)
|
||||
print(rule)
|
||||
for e in sorted(rsync_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if params.get("rate_bps") and fs_entries and rsync_entries:
|
||||
fs_best = min((e["p50"] for e in fs_entries if "p50" in e), default=None)
|
||||
rsync_best = min((e["p50"] for e in rsync_entries if "p50" in e), default=None)
|
||||
theoretical = total_bytes / params["rate_bps"]
|
||||
theoretical = measure_bytes / params["rate_bps"]
|
||||
if fs_best and rsync_best:
|
||||
print(f"\n Theoretical max (line rate): {theoretical:.4f}s")
|
||||
print(f" FastSync best: {fs_best:.4f}s ({theoretical/fs_best:.2f}x vs line rate)")
|
||||
@@ -392,10 +652,43 @@ def print_table(results, total_bytes, random_ratio):
|
||||
|
||||
def _print_entry(e):
|
||||
if "p50" in e:
|
||||
print(f" {e['config']:<25} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} {e['runs']:>5}")
|
||||
tp = f"{e['throughput_mbps']:.2f}" if "throughput_mbps" in e else "N/A"
|
||||
print(f" {e['config']:<38} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} "
|
||||
f"{tp:>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
else:
|
||||
print(f" {e['config']:<25} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {e['runs']:>5}")
|
||||
print(f" {e['config']:<38} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} "
|
||||
f"{'N/A':>8} {'N/A':>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
|
||||
|
||||
def configure_build_dirs(build_dir):
|
||||
"""Install the selected build directory and derived binary paths."""
|
||||
global BUILD_DIR, SERVER_CMD, CLIENT_CMD
|
||||
if not os.path.isabs(build_dir):
|
||||
build_dir = os.path.join(PROJECT_ROOT, build_dir)
|
||||
BUILD_DIR = os.path.abspath(build_dir)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
|
||||
|
||||
def build_project():
|
||||
"""Configure (Release) and build into the dedicated bench build dir."""
|
||||
if shutil.which("cmake") is None:
|
||||
sys.stderr.write("cmake not found in PATH; cannot build\n")
|
||||
sys.exit(1)
|
||||
os.makedirs(BUILD_DIR, exist_ok=True)
|
||||
configure = ["cmake", "-B", BUILD_DIR, "-S", PROJECT_ROOT,
|
||||
"-DCMAKE_BUILD_TYPE=Release"]
|
||||
result = subprocess.run(configure, capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("CMake configure failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
jobs = str(os.cpu_count() or 1)
|
||||
result = subprocess.run(["cmake", "--build", BUILD_DIR, "-j", jobs],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("Build failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def main():
|
||||
@@ -412,12 +705,20 @@ Custom network limits (--delay/--jitter/--throughput) override profiles.
|
||||
|
||||
Data mix:
|
||||
Default is ~75%% random/incompressible + ~25%% structured/compressible,
|
||||
reflecting typical real-world file sets.
|
||||
reflecting typical real-world file sets. The actual mix is measured and
|
||||
reported. Transfers are verified (destination must match source) unless
|
||||
--no-verify is given.
|
||||
|
||||
Warm mode:
|
||||
--warm seeds the destination with an untimed full copy of a pristine base,
|
||||
then measures only the incremental transfer after modifying a few files.
|
||||
|
||||
Examples:
|
||||
%(prog)s --profiles wan --runs 5
|
||||
%(prog)s --throughput 50mbit --delay 30ms --jitter 5ms
|
||||
%(prog)s --random-ratio 0.5 --size-mb 100
|
||||
%(prog)s --warm --runs 3 --no-rsync
|
||||
%(prog)s --dry-run --size-mb 4 --random-ratio 0.25
|
||||
""")
|
||||
parser.add_argument("--runs", type=int, default=3,
|
||||
help="Number of runs per config (default: 3)")
|
||||
@@ -425,7 +726,8 @@ Examples:
|
||||
choices=list(NETWORK_PROFILES.keys()),
|
||||
help="Predefined network profiles (default: unlimited)")
|
||||
parser.add_argument("--configs", nargs="+", default=None,
|
||||
help="Custom FastSync config flags")
|
||||
help="Custom FastSync config flags (shell-quoted, e.g. "
|
||||
"\"-j -z --chunk-serialization\")")
|
||||
parser.add_argument("--size-mb", type=int, default=25,
|
||||
help="Test data size in MB (default: 25)")
|
||||
parser.add_argument("--random-ratio", type=float, default=0.75,
|
||||
@@ -440,6 +742,14 @@ Examples:
|
||||
help="Custom packet loss (e.g. 1%%)")
|
||||
parser.add_argument("--no-rsync", action="store_true",
|
||||
help="Skip rsync comparison")
|
||||
parser.add_argument("--no-verify", action="store_true",
|
||||
help="Skip source/destination verification after each run")
|
||||
parser.add_argument("--warm", action="store_true",
|
||||
help="Incremental mode: seed dest first, measure only changes")
|
||||
parser.add_argument("--build-dir", default=DEFAULT_BUILD_DIR,
|
||||
help=f"Build directory (default: {DEFAULT_BUILD_DIR})")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="Only generate data and report its composition, then exit")
|
||||
parser.add_argument("--progress", action="store_true",
|
||||
help="Show progress bar with ETA")
|
||||
parser.add_argument("--output", choices=["table", "json"], default="table",
|
||||
@@ -448,12 +758,48 @@ Examples:
|
||||
help="Don't clean up test data")
|
||||
args = parser.parse_args()
|
||||
|
||||
# Build
|
||||
print("Building...")
|
||||
if os.system(f"cmake -B {BUILD_DIR} -S {PROJECT_ROOT} > /dev/null 2>&1") != 0:
|
||||
print("CMake configure failed"); sys.exit(1)
|
||||
if os.system(f"cmake --build {BUILD_DIR} -j$(nproc) > /dev/null 2>&1") != 0:
|
||||
print("Build failed"); sys.exit(1)
|
||||
if not 0.0 <= args.random_ratio <= 1.0:
|
||||
parser.error("--random-ratio must be between 0.0 and 1.0")
|
||||
if args.size_mb <= 0:
|
||||
parser.error("--size-mb must be positive")
|
||||
|
||||
configure_build_dirs(args.build_dir)
|
||||
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
stats = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
total_bytes = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
rand_pct = stats["random_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB in {stats['files']} files "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)",
|
||||
file=sys.stderr)
|
||||
|
||||
if args.dry_run:
|
||||
print(f"size_mb={args.size_mb} random_ratio={args.random_ratio:.4f} "
|
||||
f"total_bytes={stats['total_bytes']} "
|
||||
f"compressible_bytes={stats['compressible_bytes']} "
|
||||
f"random_bytes={stats['random_bytes']} files={stats['files']}")
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
return
|
||||
|
||||
# Warm mode: keep a pristine base copy, then mutate the live source.
|
||||
base_dir = None
|
||||
measure_bytes = total_bytes
|
||||
mutation = None
|
||||
if args.warm:
|
||||
change_target = max(64 * 1024, min(int(total_bytes * 0.01), 4 * 1024 * 1024))
|
||||
mutation = apply_incremental_changes(source_dir, change_target)
|
||||
measure_bytes = mutation["changed"]
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
print(f"Warm mode: each run seeds a full copy, then measures "
|
||||
f"{measure_bytes / (1024*1024):.3f} MB of add/change deltas", file=sys.stderr)
|
||||
|
||||
# Build (Release: benchmarking a debug build is meaningless)
|
||||
print(f"Building (Release) into {BUILD_DIR}...", file=sys.stderr)
|
||||
build_project()
|
||||
|
||||
# Determine active profile for display
|
||||
has_custom_net = args.delay or args.jitter or args.throughput or args.loss
|
||||
@@ -473,18 +819,10 @@ Examples:
|
||||
else:
|
||||
profiles_to_run = args.profiles or ["unlimited"]
|
||||
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
total_bytes = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
compressible_pct = (1 - args.random_ratio) * 100
|
||||
random_pct = args.random_ratio * 100
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB "
|
||||
f"({random_pct:.0f}% random, {compressible_pct:.0f}% compressible)")
|
||||
|
||||
# Build config list
|
||||
# Build config list (shlex so quoted/space-separated flags survive)
|
||||
if args.configs:
|
||||
fastsync_configs = [{"name": c, "flags": c.split(), "tool": "fastsync"} for c in args.configs]
|
||||
fastsync_configs = [{"name": c, "flags": shlex.split(c), "tool": "fastsync"}
|
||||
for c in args.configs]
|
||||
else:
|
||||
fastsync_configs = list(FASTSYNC_CONFIGS)
|
||||
|
||||
@@ -496,14 +834,21 @@ Examples:
|
||||
total_runs = len(configs) * args.runs * len(profiles_to_run)
|
||||
progress = Progress(total_runs, "Benchmarking") if args.progress else None
|
||||
if progress:
|
||||
print(f"Running {total_runs} transfers...")
|
||||
print(f"Running {total_runs} transfers...", file=sys.stderr)
|
||||
|
||||
all_results = []
|
||||
try:
|
||||
for profile in profiles_to_run:
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile, progress)
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile,
|
||||
measure_bytes, verify=not args.no_verify,
|
||||
warm=args.warm, mutation=mutation,
|
||||
progress=progress)
|
||||
all_results.extend(results)
|
||||
except RuntimeError as exc:
|
||||
sys.stderr.write(f"error: {exc}\n")
|
||||
sys.exit(1)
|
||||
finally:
|
||||
netem_reset()
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
|
||||
@@ -511,7 +856,7 @@ Examples:
|
||||
if args.output == "json":
|
||||
print(json.dumps(all_results, indent=2))
|
||||
else:
|
||||
print_table(all_results, total_bytes, args.random_ratio)
|
||||
print_table(all_results, measure_bytes, stats, args.warm)
|
||||
print()
|
||||
|
||||
|
||||
|
||||
@@ -3,11 +3,35 @@
|
||||
}:
|
||||
|
||||
pkgs.mkShell {
|
||||
# Development shell for FastSync. Provides the host-side toolchain needed to
|
||||
# build, lint, unit-test, integration-test and benchmark the project.
|
||||
# It deliberately does NOT build on entry: run the CMake commands in README.md
|
||||
# (or use the CI Docker image for exact CI parity).
|
||||
nativeBuildInputs = with pkgs; [
|
||||
# build
|
||||
gcc
|
||||
cmake
|
||||
gnumake
|
||||
pkg-config
|
||||
# lint / static analysis (matches CI)
|
||||
clang-tools # clang-format
|
||||
cppcheck
|
||||
# tests
|
||||
(python3.withPackages (ps: with ps; [ pytest pytest-xdist psutil ]))
|
||||
openssh # SSH transport integration tests
|
||||
# debugging
|
||||
gdb
|
||||
valgrind
|
||||
# coverage
|
||||
lcov
|
||||
# benchmark tooling
|
||||
rsync
|
||||
iproute2 # tc/netem for network shaping
|
||||
# misc
|
||||
git
|
||||
curl
|
||||
nodejs
|
||||
nixpkgs-fmt
|
||||
docker
|
||||
tea
|
||||
];
|
||||
@@ -15,14 +39,21 @@ pkgs.mkShell {
|
||||
buildInputs = with pkgs; [
|
||||
zstd
|
||||
openssl
|
||||
(python3.withPackages (ps: with ps; [ pytest ]))
|
||||
];
|
||||
|
||||
# The CMake configure step fetches xxHash via FetchContent, which needs
|
||||
# network access; NIX_ENFORCE_PURITY must be off so the sandbox does not block.
|
||||
NIX_ENFORCE_PURITY = 0;
|
||||
|
||||
shellHook = ''
|
||||
export NIX_ENFORCE_PURITY=0
|
||||
cmake -B build
|
||||
# Make an existing build tree available on PATH, but never build here.
|
||||
if [ -d "$PWD/build" ]; then
|
||||
export PATH="$PWD/build:$PATH"
|
||||
fi
|
||||
echo "FastSync dev shell ready."
|
||||
echo " Build: cmake -B build -S . && cmake --build build -j\$(nproc)"
|
||||
echo " Unit: ./build/tests"
|
||||
echo " CI parity: docker run --rm --user \"\$(id -u):\$(id -g)\" -v \"\$PWD:/workspace\" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 ..."
|
||||
'';
|
||||
}
|
||||
+345
-152
@@ -7,21 +7,7 @@
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
/* Itemize code emitted for a transferred regular file.
|
||||
*
|
||||
* Layout (rsync-compatible 11-char item): `>f` marks a regular file that was
|
||||
* transferred to the remote host; the trailing nine markers are, in order,
|
||||
* c(hecksum) s(ize) t(ime) p(erms) o(wner) g(roup) u(ser/acl) a(ttrs) x(attrs).
|
||||
* Every marker is `+` (FastSync does not compare each attribute on the
|
||||
* receiving side, so a sent file is reported as fully updated). Files that
|
||||
* are already up to date print no line at all, matching rsync's single -i
|
||||
* which only itemizes changes.
|
||||
*
|
||||
* Because the scanner only yields regular-file transfer candidates, `>d`
|
||||
* (directory) lines are never produced; directories are not transferred as
|
||||
* items by FastSync. */
|
||||
#define ITEMIZE_SENT_FILE ">f+++++++++"
|
||||
#include <unistd.h>
|
||||
|
||||
typedef struct {
|
||||
char* data;
|
||||
@@ -80,103 +66,14 @@ static bool strbuf_append(StrBuf* buf, const char* text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool strbuf_append_ull(StrBuf* buf, unsigned long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
static bool strbuf_append_longlong(StrBuf* buf, long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%lld", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
bool change_list_enabled(const Config* config) {
|
||||
return config != NULL && (config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL));
|
||||
}
|
||||
|
||||
char* change_render_itemize(const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
const char* code = event->is_directory ? ">d+++++++++" : ITEMIZE_SENT_FILE;
|
||||
StrBuf line = {0};
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append(&line, " ") &&
|
||||
strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
/* ---- Itemize code ---- */
|
||||
|
||||
static const char* leaf_name(const char* path) {
|
||||
if (path == NULL)
|
||||
return "";
|
||||
const char* slash = strrchr(path, '/');
|
||||
return slash != NULL && slash[1] != '\0' ? slash + 1 : path;
|
||||
}
|
||||
|
||||
char* change_render_format(const char* format, const ChangeEvent* event) {
|
||||
if (format == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = strbuf_append(&line, leaf_name(event->path));
|
||||
break;
|
||||
case 'l':
|
||||
ok = strbuf_append_ull(&line, event->size);
|
||||
break;
|
||||
case 'b':
|
||||
ok = strbuf_append_ull(&line, event->bytes_sent);
|
||||
break;
|
||||
case 'M':
|
||||
ok = strbuf_append_longlong(&line, (long long)event->mtime_sec);
|
||||
break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* Format a mode as an `ls -l` permission string, e.g. `-rw-r--r--`. */
|
||||
/* Format the permission bits as an `ls -l` string, e.g. `-rw-r--r--`. */
|
||||
static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[0] = S_ISDIR(mode) ? 'd'
|
||||
: S_ISLNK(mode) ? 'l'
|
||||
@@ -198,29 +95,102 @@ static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[10] = '\0';
|
||||
}
|
||||
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime,
|
||||
const char* path) {
|
||||
char permission[11];
|
||||
mode_to_ls_string(mode, permission);
|
||||
char date[32];
|
||||
struct tm broken_down;
|
||||
if (localtime_r(&mtime, &broken_down) != NULL) {
|
||||
if (strftime(date, sizeof(date), "%Y/%m/%d %H:%M:%S", &broken_down) == 0)
|
||||
snprintf(date, sizeof(date), "?");
|
||||
} else {
|
||||
snprintf(date, sizeof(date), "?");
|
||||
static char itemize_type_char(const ChangeEvent* event) {
|
||||
if (event->is_directory)
|
||||
return 'd';
|
||||
if (event->is_symlink)
|
||||
return 'L';
|
||||
if (event->is_special) {
|
||||
if (S_ISCHR(event->mode) || S_ISBLK(event->mode))
|
||||
return 'D';
|
||||
return 'S';
|
||||
}
|
||||
return 'f';
|
||||
}
|
||||
|
||||
static bool times_match(const Config* config, const ChangeEvent* event) {
|
||||
if (!event->dest.known || !event->dest.existed)
|
||||
return false;
|
||||
if (event->mtime_sec == event->dest.mtime_sec)
|
||||
return event->mtime_nsec == event->dest.mtime_nsec;
|
||||
long long delta = (long long)event->mtime_sec - (long long)event->dest.mtime_sec;
|
||||
if (delta < 0)
|
||||
delta = -delta;
|
||||
return delta <= (long long)config->modify_window;
|
||||
}
|
||||
|
||||
/* Fill the 11-character itemize code (10 chars + NUL). `created` means the
|
||||
* destination entry did not exist, so every attribute marker is `+`. */
|
||||
static void itemize_code(const Config* config, const ChangeEvent* event, char code[12]) {
|
||||
bool known = event->dest.known;
|
||||
bool created = !known || !event->dest.existed;
|
||||
char update;
|
||||
if (event->is_hardlink)
|
||||
update = 'h';
|
||||
else if (created)
|
||||
update = (event->is_directory || event->is_symlink || event->is_special) ? 'c' : '>';
|
||||
else
|
||||
update = '>';
|
||||
code[0] = update;
|
||||
code[1] = itemize_type_char(event);
|
||||
if (created) {
|
||||
for (int i = 0; i < 9; i++)
|
||||
code[2 + i] = '+';
|
||||
code[11] = '\0';
|
||||
return;
|
||||
}
|
||||
bool size_diff = event->size != event->dest.size;
|
||||
bool time_diff = !times_match(config, event);
|
||||
bool perms_diff = (event->mode & 07777) != (event->dest.mode & 07777);
|
||||
bool owner_diff = event->uid != (uid_t)event->dest.uid;
|
||||
bool group_diff = event->gid != (gid_t)event->dest.gid;
|
||||
code[2] = '.'; /* checksum: no destination digest available */
|
||||
code[3] = size_diff ? 's' : '.';
|
||||
code[4] = time_diff ? 't' : '.';
|
||||
code[5] = (config->preserve_perms && perms_diff) ? 'p' : '.';
|
||||
code[6] = (config->preserve_owner && owner_diff) ? 'o' : '.';
|
||||
code[7] = (config->preserve_group && group_diff) ? 'g' : '.';
|
||||
code[8] = '.'; /* reserved */
|
||||
code[9] = '.'; /* acl: not compared */
|
||||
code[10] = '.';
|
||||
code[11] = '\0';
|
||||
}
|
||||
|
||||
char* change_render_itemize_code(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
return str_dup(code);
|
||||
}
|
||||
|
||||
/* rsync %n: the transfer-relative name, with a trailing slash for directories. */
|
||||
static bool append_name(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (!strbuf_append(buf, event->name != NULL ? event->name : ""))
|
||||
return false;
|
||||
if (event->is_directory && (event->name == NULL || event->name[0] == '\0' ||
|
||||
event->name[strlen(event->name) - 1] != '/'))
|
||||
return strbuf_append_char(buf, '/');
|
||||
return true;
|
||||
}
|
||||
|
||||
/* rsync %L: " -> target" for a symlink, " => target" for a hard link, else "". */
|
||||
static bool append_link_suffix(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (event->is_symlink && event->symlink_target != NULL)
|
||||
return strbuf_append(buf, " -> ") && strbuf_append(buf, event->symlink_target);
|
||||
if (event->is_hardlink && event->hardlink_target != NULL)
|
||||
return strbuf_append(buf, " => ") && strbuf_append(buf, event->hardlink_target);
|
||||
return true;
|
||||
}
|
||||
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
StrBuf line = {0};
|
||||
char size_field[32];
|
||||
int written = snprintf(size_field, sizeof(size_field), "%llu", size);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, size_field) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, date) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, path != NULL ? path : "");
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append_char(&line, ' ') &&
|
||||
append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
@@ -228,6 +198,141 @@ char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- --out-format / --log-file-format ---- */
|
||||
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event) {
|
||||
if (format == NULL || event == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'i': {
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
ok = strbuf_append(&line, code);
|
||||
break;
|
||||
}
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = append_name(&line, event);
|
||||
break;
|
||||
case 'L':
|
||||
ok = append_link_suffix(&line, event);
|
||||
break;
|
||||
case 'l': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->size);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'b': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_sent);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'M': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(event->mtime_sec, true, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 't': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(time(NULL), false, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 'o':
|
||||
ok = strbuf_append(&line, "send");
|
||||
break;
|
||||
case 'p': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%ld", (long)getpid());
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'B': {
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
ok = strbuf_append(&line, permission + 1);
|
||||
} break;
|
||||
case 'U': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->uid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'G': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->gid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- --list-only ---- */
|
||||
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event) {
|
||||
(void)config;
|
||||
if (event == NULL)
|
||||
return NULL;
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
char date[32];
|
||||
if (!format_rsync_datetime(event->mtime_sec, false, date, sizeof(date)))
|
||||
snprintf(date, sizeof(date), "?");
|
||||
StrBuf line = {0};
|
||||
char size_field[40];
|
||||
char grouped[32];
|
||||
if (!format_big_num(event->size, false, grouped, sizeof(grouped))) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
int written = snprintf(size_field, sizeof(size_field), "%15s", grouped);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
const char* name = event->name != NULL && event->name[0] != '\0' ? event->name : ".";
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append(&line, size_field) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, date) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, name);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- Event emission ---- */
|
||||
|
||||
static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_output) {
|
||||
char* escaped = output_escape(line, eight_bit_output);
|
||||
if (escaped != NULL) {
|
||||
@@ -247,15 +352,16 @@ void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
bool to_stdout = config->itemize_changes || config->out_format != NULL;
|
||||
bool to_log = config->log_file != NULL && config->log_file_format != NULL;
|
||||
if (to_stdout) {
|
||||
char* line = config->out_format != NULL ? change_render_format(config->out_format, event)
|
||||
: change_render_itemize(event);
|
||||
char* line = config->out_format != NULL
|
||||
? change_render_format(config->out_format, config, event)
|
||||
: change_render_itemize(config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
if (to_log) {
|
||||
char* line = change_render_format(config->log_file_format, event);
|
||||
char* line = change_render_format(config->log_file_format, config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(config->log_file, line, config->eight_bit_output);
|
||||
free(line);
|
||||
@@ -266,9 +372,6 @@ void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
static bool format_uses_mtime(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
/* Mirror change_render_format's tokenizer: "%%" is a literal percent (so
|
||||
* "%%M" does NOT expand %M) and unknown "%X" escapes consume both chars.
|
||||
* This keeps the optional stat() fallback below in step with the renderer. */
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
@@ -284,46 +387,136 @@ static bool format_uses_mtime(const char* format) {
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Relative path of an entry below the transfer root (no leading slash). Uses
|
||||
* the sender-side send_path override when present (bare-relative -R layout). */
|
||||
static char* relative_name(const Config* config, const File* file) {
|
||||
const char* full = file_wire_path(file);
|
||||
if (file->send_path != NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL || full == NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, full, root_len) == 0) {
|
||||
if (full[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (full[root_len] == '/')
|
||||
return str_dup(full + root_len + 1);
|
||||
}
|
||||
return str_dup(full);
|
||||
}
|
||||
|
||||
/* rsync %f long form: the source argument as typed (leading '/' removed,
|
||||
* trailing '/' removed, leading "./" removed) joined to the relative name. */
|
||||
static char* display_name(const Config* config, const char* name) {
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL)
|
||||
return str_dup(name != NULL ? name : "");
|
||||
const char* p = root;
|
||||
while (*p == '/')
|
||||
p++;
|
||||
if (p[0] == '.' && p[1] == '/')
|
||||
p += 2;
|
||||
size_t root_len = strlen(p);
|
||||
while (root_len > 0 && p[root_len - 1] == '/')
|
||||
root_len--;
|
||||
size_t name_len = name != NULL ? strlen(name) : 0;
|
||||
if (root_len == 0 && name_len == 0)
|
||||
return str_dup("");
|
||||
char* out = malloc(root_len + (root_len > 0 && name_len > 0 ? 1 : 0) + name_len + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t offset = 0;
|
||||
if (root_len > 0) {
|
||||
memcpy(out, p, root_len);
|
||||
offset = root_len;
|
||||
}
|
||||
if (root_len > 0 && name_len > 0)
|
||||
out[offset++] = '/';
|
||||
if (name_len > 0)
|
||||
memcpy(out + offset, name, name_len);
|
||||
out[offset + name_len] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
static void fill_event_from_file(const Config* config, const File* file, ChangeEvent* event,
|
||||
char** name_out, char** path_out) {
|
||||
char* name = relative_name(config, file);
|
||||
char* path = display_name(config, name);
|
||||
event->name = name;
|
||||
event->path = path;
|
||||
*name_out = name;
|
||||
*path_out = path;
|
||||
if (file->metadata != NULL) {
|
||||
event->mtime_sec = file->metadata->mtime_sec;
|
||||
event->mtime_nsec = file->metadata->mtime_nsec;
|
||||
event->mode = file->metadata->mode;
|
||||
event->uid = file->metadata->uid;
|
||||
event->gid = file->metadata->gid;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0) {
|
||||
event->mtime_sec = st.st_mtime;
|
||||
event->mtime_nsec = st.st_mtim.tv_nsec;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
/* The displayed path is the one transmitted (with -R + --files-from this is
|
||||
the bare relative destination path); the metadata fallback below still
|
||||
stats the local absolute path. */
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = false;
|
||||
event.is_special = false;
|
||||
event.is_hardlink = false;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
/* FastSync has no wire-byte counter yet, so %b reports the source length
|
||||
* that had to be delivered (always equal to %l); the actual bytes written
|
||||
* to the socket (compressed/delta) are not measured. */
|
||||
event.dest = file->dest_state;
|
||||
if (file->is_symlink) {
|
||||
event.is_symlink = true;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.size = file->symlink_target != NULL ? strlen(file->symlink_target) : 0;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->is_special) {
|
||||
event.is_special = true;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->link_group != 0 && !file->link_first) {
|
||||
event.is_hardlink = true;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.bytes_sent = 0;
|
||||
} else {
|
||||
/* Literal payload bytes delivered; compressed/delta wire bytes are not
|
||||
* separately counted. */
|
||||
event.bytes_sent = event.size;
|
||||
if (file->metadata != NULL) {
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
/* Best-effort fallback for %M when no metadata was captured (no -M): the
|
||||
* path is stat()ed just to fill the field, and any failure leaves 0. */
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0)
|
||||
event.mtime_sec = st.st_mtime;
|
||||
}
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
void change_emit_dir_sent(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = true;
|
||||
event.size = 0;
|
||||
event.bytes_sent = 0;
|
||||
if (file->metadata != NULL)
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
+34
-23
@@ -3,6 +3,7 @@
|
||||
|
||||
#include "config.h"
|
||||
#include "file_types.h"
|
||||
#include "format.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
@@ -26,42 +27,52 @@ typedef enum {
|
||||
} ChangeDecision;
|
||||
|
||||
typedef struct {
|
||||
const char* path; /* full source path */
|
||||
const char* path; /* long-form display path (rsync %f) */
|
||||
const char* name; /* transfer-relative path (rsync %n), no trailing slash */
|
||||
ChangeDecision decision;
|
||||
bool is_directory;
|
||||
bool is_symlink;
|
||||
bool is_special;
|
||||
bool is_hardlink; /* a hard-link sibling (linked, no data sent) */
|
||||
const char* symlink_target;
|
||||
const char* hardlink_target;
|
||||
unsigned long long size; /* source file length in bytes */
|
||||
/* The number of bytes reported for a sent file. FastSync has no wire-byte
|
||||
* counter, so this is always the source length (== size / %l); actual
|
||||
* post-compression/delta bytes on the wire are not counted. */
|
||||
unsigned long long bytes_sent;
|
||||
time_t mtime_sec; /* 0 when unknown */
|
||||
unsigned long long bytes_sent; /* literal data bytes actually transferred */
|
||||
time_t mtime_sec;
|
||||
long mtime_nsec;
|
||||
mode_t mode;
|
||||
uid_t uid;
|
||||
gid_t gid;
|
||||
/* Receiver-reported pre-transfer destination state (OutputDestState.known is
|
||||
* false when no report was requested/received). */
|
||||
OutputDestState dest;
|
||||
} ChangeEvent;
|
||||
|
||||
/* True when any output mode is active and per-file events matter. */
|
||||
bool change_list_enabled(const Config* config);
|
||||
|
||||
/* Render the rsync-style itemize line for a transferred file:
|
||||
* `>f+++++++++ <path>`
|
||||
* The 11-char code is `>f` (regular file transferred to the remote host)
|
||||
* followed by c/s/t/p/o/g/u/a/x markers that are all `+` (value will be set
|
||||
* / differs) because FastSync does not separately compare checksums, size,
|
||||
* mtime, perms, owner, group, uid, acl, or xattr on the receiving side, so a
|
||||
* sent file is reported as fully updated. Up-to-date files print no line
|
||||
* (rsync single `-i` only shows changes). Caller frees the result. */
|
||||
char* change_render_itemize(const ChangeEvent* event);
|
||||
/* Render the rsync-style itemize line for a transferred item
|
||||
* (`%i %n%L`): `>f+++++++++ sub/b.txt`. Caller frees the result. */
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Expand an --out-format/--log-file-format template. Tokens:
|
||||
* %f full source path %b "bytes sent" == the source length (%l);
|
||||
* %n leaf (base) name actual post-compression/delta wire bytes
|
||||
* %l file length in bytes are not counted
|
||||
* %M mtime in whole seconds %% a literal percent sign
|
||||
/* Render only the 11-character itemize code (rsync %i). Caller frees. */
|
||||
char* change_render_itemize_code(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Expand an --out-format/--log-file-format template. Supported tokens:
|
||||
* %i itemize code %n transfer-relative name (dir: trailing /)
|
||||
* %f long display path %l file length in bytes
|
||||
* %b bytes actually sent %M mtime (YYYY/MM/DD-HH:MM:SS)
|
||||
* %t current time %o operation ("send"/"del.")
|
||||
* %p pid %B permission bits without the type char
|
||||
* %U uid %G gid
|
||||
* %L " -> target" / " => target" %% a literal percent sign
|
||||
* Unknown %X sequences are preserved verbatim. Caller frees the result. */
|
||||
char* change_render_format(const char* format, const ChangeEvent* event);
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Render one --list-only long-listing entry:
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 <path>`
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 sub/b.txt`
|
||||
* (ls -l style columns; mtime in the local time zone). Caller frees it. */
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime, const char* path);
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Emit an event to every active destination:
|
||||
* stdout: --itemize-changes line, or the --out-format expansion when set;
|
||||
|
||||
+1465
-473
File diff suppressed because it is too large.
Load diff
+720
-124
File diff suppressed because it is too large.
Load diff
@@ -4,10 +4,24 @@
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "transport_tcp.h"
|
||||
#include <signal.h>
|
||||
#include <stdbool.h>
|
||||
|
||||
int send_chunk(Client* client, Chunk* chunk, Config* config);
|
||||
/* Set ONLY by the client's SIGINT/SIGTERM handler (async-signal-safe: the
|
||||
* handler stores 1 and does nothing else). The send loops poll it via
|
||||
* client_abort_pending() and, when set, best-effort send STATUS_ABORT so the
|
||||
* receiver can clean up before the client exits. */
|
||||
extern volatile sig_atomic_t client_abort_requested;
|
||||
bool client_abort_pending(void);
|
||||
/* Arm/disarm abort handling around the network phase. While disarmed, a
|
||||
* SIGINT/SIGTERM takes the default action (immediate termination) so local-only
|
||||
* modes are not left unresponsive. Defined in client_cli.c. */
|
||||
void client_set_abort_armed(bool armed);
|
||||
|
||||
/* Both sender entry points BORROW `config` for the duration of the call; they
|
||||
* never free it, and the caller retains ownership (freeing it with
|
||||
* config_delete() once the call returns). */
|
||||
int send_files(Config* config);
|
||||
/* Takes ownership only when *config is set to NULL on return. */
|
||||
int send_files_multithreaded(Config** config);
|
||||
/* Phase 6 residual-batch (client-only). See client_send.c. */
|
||||
int write_batch_from_source(const Config* config, const char* batch_path);
|
||||
|
||||
+27
-105
@@ -1,6 +1,4 @@
|
||||
#include "client_validation.h"
|
||||
#include "charset.h"
|
||||
#include "delay_updates.h"
|
||||
#include "log.h"
|
||||
#include "usage.h"
|
||||
#include "utils.h"
|
||||
@@ -24,6 +22,19 @@ bool validate_config(const Config* config) {
|
||||
"--write-batch, --only-write-batch, and --read-batch are mutually exclusive");
|
||||
return false;
|
||||
}
|
||||
/* A dry-run of a local batch apply is not meaningful: --read-batch bypasses
|
||||
the client-side scan/server decision entirely, so dry-run would have no
|
||||
wire state to report (and must not be used as a mutation escape hatch).
|
||||
--only-write-batch likewise never contacts a receiver. --write-batch DOES
|
||||
run a live transfer but additionally mutates the filesystem by emitting the
|
||||
batch file, so a dry-run must not write it either. Reject all three up
|
||||
front instead of silently ignoring --dry-run. */
|
||||
if (config->dry_run && (read_batch || only_write_batch || write_batch)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--dry-run cannot be combined with --read-batch, --only-write-batch, or "
|
||||
"--write-batch; a dry-run must not mutate anything, including batch files");
|
||||
return false;
|
||||
}
|
||||
if (read_batch) {
|
||||
if (!config->receive_root_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "--read-batch requires a destination directory");
|
||||
@@ -41,17 +52,6 @@ bool validate_config(const Config* config) {
|
||||
print_usage();
|
||||
return false;
|
||||
}
|
||||
if (config_has_basis(config) && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--compare-dest/--copy-dest/--link-dest require per-file incremental checks and "
|
||||
"cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s "
|
||||
"(chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->compression_threads > 0 && !config->use_compression) {
|
||||
log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-c or -z)");
|
||||
return false;
|
||||
@@ -60,8 +60,14 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport");
|
||||
return false;
|
||||
}
|
||||
if (config->use_incremental && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR, "--incremental is not supported with -s (chunk serialization)");
|
||||
/* -M/--remote-option appends an option to the REMOTE server's argv, which
|
||||
* only exists on the SSH (user@host:path) transport. A daemon
|
||||
* (host::module/path) or local TCP destination has no remote command line,
|
||||
* so the option would be silently ignored; reject it by name instead. */
|
||||
if (config->remote_option_count > 0 && config->transport != TRANSPORT_SSH) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"-M/--remote-option is only valid with the SSH transport (user@host:path); it "
|
||||
"cannot be used with a daemon (host::module/path) or local TCP destination");
|
||||
return false;
|
||||
}
|
||||
/* -4 and -6 are mutually exclusive: a socket address family cannot be both. */
|
||||
@@ -69,64 +75,6 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive");
|
||||
return false;
|
||||
}
|
||||
if (config->skip_compress_set && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--skip-compress cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->use_delta && !config->whole_file && !config->use_incremental) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta requires --incremental");
|
||||
return false;
|
||||
}
|
||||
if (config->use_delta && !config->whole_file && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->use_delta && !config->whole_file && config->use_sendfile) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -f (sendfile)");
|
||||
return false;
|
||||
}
|
||||
/* --append / --append-verify resume a shorter existing destination by
|
||||
transmitting only the tail. The resume needs the per-file STATUS_CHECK
|
||||
handshake (so the dest length is learned), which chunk serialization -s
|
||||
disables; and whole-file is the opposite intent (send everything), so the
|
||||
two would silently make the resume pointless. Both are rejected up front
|
||||
rather than silently degrading to a full transfer. */
|
||||
if ((config->append || config->append_verify) && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--append/--append-verify require the per-file incremental check and cannot be "
|
||||
"combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if ((config->append || config->append_verify) && config->whole_file) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--append/--append-verify are incompatible with --whole-file (which forces a "
|
||||
"full transfer)");
|
||||
return false;
|
||||
}
|
||||
/* --hard-links/-H transmits each later group member as a dedicated per-file
|
||||
STATUS_HARDLINK frame, which chunk serialization -s does not support; and a
|
||||
hard-links sibling carries no payload, so the tail-resume of --append is
|
||||
meaningless for it. Both combinations are rejected up front rather than
|
||||
silently degrading. */
|
||||
if (config->preserve_hard_links && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--hard-links/-H cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
/* -X/-A ride the per-file metadata frame; the buffer-based chunk-serialization
|
||||
wire format does not carry the xattr block, so the pair is rejected up front
|
||||
(mirroring -H + -s) rather than silently dropping attributes. */
|
||||
if ((config->preserve_xattrs || config->preserve_acls) && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--xattrs/-X and --acls/-A cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->preserve_hard_links && (config->append || config->append_verify)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--hard-links/-H cannot be combined with --append/--append-verify");
|
||||
return false;
|
||||
}
|
||||
if (config->log_file_format && !config->log_file) {
|
||||
log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file");
|
||||
return false;
|
||||
@@ -146,28 +94,12 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "sending daemon credentials to a non-local server requires --tls");
|
||||
return false;
|
||||
}
|
||||
if (config->delay_updates && config->inplace) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delay-updates does not work with --inplace");
|
||||
return false;
|
||||
}
|
||||
if (config->delay_updates && delay_updates_staging_name_conflict(config->backup_dir)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--backup-dir is reserved when --delay-updates is active (used for the internal "
|
||||
"staging directory)");
|
||||
return false;
|
||||
}
|
||||
if (!config_has_valid_delete_timing(config)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--delete-before/--delete-during/--delete-delay/--delete-after select the delete "
|
||||
"timing; at most one may be given and each implies --delete");
|
||||
return false;
|
||||
}
|
||||
/* --iconv: reject a malformed CONVERT_SPEC or an unsupported charset name at
|
||||
startup (a probe iconv_open is attempted), so a typo'd charset never fails
|
||||
the run mid-transfer with per-file errors. */
|
||||
if (!charset_spec_valid(config->iconv_spec)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv requires LOCAL[,REMOTE] charset names supported by iconv");
|
||||
/* Every cross-field invariant the receiver enforces lives in one shared
|
||||
predicate so the client and the server can never disagree. The client
|
||||
reports the specific reason here, before any network I/O. */
|
||||
const char* invariants_error = config_invariants_error(config);
|
||||
if (invariants_error) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s", invariants_error);
|
||||
return false;
|
||||
}
|
||||
/* --protocol: FastSync has exactly one wire format, so the forced version
|
||||
@@ -180,15 +112,5 @@ bool validate_config(const Config* config) {
|
||||
PROTOCOL_VERSION);
|
||||
return false;
|
||||
}
|
||||
/* --copy-as pushes the source ids through the metadata path (it implies
|
||||
--preserve). A later --no-preserve would clear use_metadata, leaving the
|
||||
transfer with nothing to chown while the receiver gate would still pass.
|
||||
Refuse the combination up front rather than silently chowning nothing. */
|
||||
if (config->copy_as_set && !config->use_metadata) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--copy-as requires metadata preservation and cannot be combined with "
|
||||
"--no-preserve");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
+405
-222
File diff suppressed because it is too large.
Load diff
+37
-58
@@ -14,6 +14,10 @@
|
||||
#include <sys/types.h>
|
||||
#include <threads.h>
|
||||
|
||||
/* Upper bound on the configurable parallel scanner worker count (--threads=N):
|
||||
* keeps one transfer from spawning an unbounded pool on a very large machine. */
|
||||
#define MAX_SCANNER_THREADS 256
|
||||
|
||||
typedef struct {
|
||||
bool use_metadata;
|
||||
/* Phase 4 metadata capture: -U/--atimes and -N/--crtimes tell the scanner to
|
||||
@@ -61,6 +65,10 @@ typedef struct {
|
||||
bool per_dir_filters; /* -F: read .rsync-filter per directory */
|
||||
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
|
||||
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
|
||||
/* --list-only: emit an is_dir File for every traversed directory (the listing
|
||||
* includes directory entries, matching rsync). Client-only; never set on a
|
||||
* real transfer, which relies on implicit parent creation. */
|
||||
bool list_dirs;
|
||||
/* --prune-empty-dirs (long only): in --dirs mode an empty source directory's
|
||||
explicit entry is omitted from the transfer file list (so nothing is
|
||||
created at the destination and it can be pruned by --delete); explicitly
|
||||
@@ -69,17 +77,32 @@ typedef struct {
|
||||
bool prune_empty_dirs;
|
||||
/* Delete-excluded protection sink (optional): when non-NULL the scanner
|
||||
* appends the destination-relative path of every entry it prunes because a
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules, the legacy
|
||||
* --exclude/--include layer, and --max-size/--min-size). The sender turns
|
||||
* this list into the manifest's protected prefixes so `--delete` leaves the
|
||||
* destination mirror of excluded source paths alone (rsync's default), and
|
||||
* empties it when --delete-excluded opts back into deleting them. NOT
|
||||
* recorded for --files-from subset pruning (whose delete semantics stay
|
||||
* keep-set-only) or for -R/--files-from relative wire paths. When
|
||||
* `excluded_mutex` is non-NULL it is taken around every append (the parallel
|
||||
* scanner shares one list across its worker threads). */
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules and the legacy
|
||||
* --exclude/--include layer). The sender turns this list into the manifest's
|
||||
* protected prefixes so `--delete` leaves the destination mirror of excluded
|
||||
* source paths alone (rsync's default), and drops it when --delete-excluded
|
||||
* opts back into deleting them. NOT recorded for --files-from subset pruning
|
||||
* (whose delete semantics derive from the synchronized-directory set) or for
|
||||
* -R/--files-from relative wire paths. When `excluded_mutex` is non-NULL it
|
||||
* is taken around every append (the parallel scanner shares one list across
|
||||
* its worker threads). */
|
||||
ArrayList* excluded_paths;
|
||||
mtx_t* excluded_mutex;
|
||||
/* Size-prune protection sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every entry it skipped because of
|
||||
* --max-size/--min-size. rsync never deletes a size-skipped source mirror,
|
||||
* even under --delete-excluded, so the sender always transmits this list as
|
||||
* protected prefixes (unlike excluded_paths, which --delete-excluded drops).
|
||||
* Guarded by `excluded_mutex` like excluded_paths. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Synchronized-directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it is about to traverse
|
||||
* that lies inside a --files-from listed directory (or of every traversed
|
||||
* directory when there is no list). The sender sends this set with the delete
|
||||
* manifest so the receiver confines its extras walk to synchronized
|
||||
* directories, exactly like rsync; the receive root is the "." sentinel.
|
||||
* Guarded by `excluded_mutex`. */
|
||||
ArrayList* synced_dirs;
|
||||
/* --ignore-errors: an unreadable directory during the scan is recorded as an
|
||||
* I/O error and skipped instead of aborting the scan. Client-only. */
|
||||
bool ignore_io_errors;
|
||||
@@ -117,37 +140,16 @@ typedef struct {
|
||||
typedef struct FilterNode FilterNode;
|
||||
|
||||
typedef struct {
|
||||
/* Scan inputs, copied once at create time. Everything that is also a
|
||||
ScannerOptions field lives here (with the normalized chunk_size); only
|
||||
scanner-owned bookkeeping stays as direct members below. */
|
||||
ScannerOptions options;
|
||||
Queue* directories;
|
||||
DIR* current_dir;
|
||||
char* current_path;
|
||||
bool use_metadata;
|
||||
bool preserve_atimes;
|
||||
bool preserve_crtimes;
|
||||
bool preserve_xattrs;
|
||||
bool preserve_acls;
|
||||
unsigned long long chunk_size;
|
||||
char** exclude_patterns;
|
||||
int exclude_count;
|
||||
char** include_patterns;
|
||||
int include_count;
|
||||
unsigned long long max_size;
|
||||
unsigned long long min_size;
|
||||
int max_depth;
|
||||
int current_depth;
|
||||
bool follow_symlinks;
|
||||
bool copy_links;
|
||||
bool safe_links;
|
||||
bool copy_unsafe_links;
|
||||
bool copy_dirlinks;
|
||||
bool munge_links;
|
||||
bool checksum;
|
||||
bool one_file_system;
|
||||
dev_t root_dev;
|
||||
bool failed;
|
||||
/* Phase 4 special/devices (see ScannerOptions). */
|
||||
bool preserve_devices;
|
||||
bool preserve_specials;
|
||||
bool copy_devices;
|
||||
/* Phase 2 (files-from / filter layer). */
|
||||
char* root_path; /* transfer root (fs path) for rel computation */
|
||||
char* current_rel; /* rel path of the open directory ("" == root) */
|
||||
@@ -155,40 +157,17 @@ typedef struct {
|
||||
FilterNode* seed_node; /* inherited context of the seed dir, or NULL */
|
||||
FilterNode* current_node; /* filter context of the open directory */
|
||||
ArrayList* filter_nodes; /* owned FilterNode arena (may be NULL) */
|
||||
const FileListSet* file_list;
|
||||
const FilterRuleList* base_filters;
|
||||
bool per_dir_filters;
|
||||
/* --dirs / -R state for the directory-entry generator (dirs_mode replaces
|
||||
/* --dirs / -R state for the directory-entry generator (options.dirs replaces
|
||||
the recursive scan). */
|
||||
bool dirs_mode;
|
||||
bool relative_mode; /* file_list && relative: send bare relative wire paths */
|
||||
bool prune_empty_dirs;
|
||||
bool dirs_root_emitted;
|
||||
int list_index;
|
||||
ArrayList* dirs_batch; /* owned when non-NULL */
|
||||
unsigned long long dirs_batch_size;
|
||||
/* Excluded-path sink (see ScannerOptions). `excluded_mutex` is shared across
|
||||
parallel worker threads. */
|
||||
ArrayList* excluded_paths;
|
||||
mtx_t* excluded_mutex;
|
||||
/* --ignore-errors: continue past unreadable directories (records io_error). */
|
||||
bool ignore_io_errors;
|
||||
/* --ignore-missing-args: --dirs listed-but-missing entries are skipped, not
|
||||
fatal (see ScannerOptions.ignore_missing_args). */
|
||||
bool ignore_missing_args;
|
||||
/* A directory could not be opened (I/O error, e.g. EACCES). With
|
||||
--ignore-errors the scan continues past it and the caller decides what to
|
||||
do; `failed` is reserved for fatal errors that always abort the scan. */
|
||||
bool io_error;
|
||||
/* --hard-links (-H): shared link-group detection table (see ScannerOptions).
|
||||
NULL when -H is off. */
|
||||
HardLinkTable* hardlinks;
|
||||
/* Phase 6: sender stop deadline (from ScannerOptions). */
|
||||
const StopCondition* stop_condition;
|
||||
/* P7 Wave D directory-time capture (see ScannerOptions). */
|
||||
bool capture_dir_times;
|
||||
ArrayList* dir_entries;
|
||||
mtx_t* dir_entries_mutex;
|
||||
} DirectoryScanner;
|
||||
|
||||
typedef struct {
|
||||
|
||||
+102
-56
@@ -2,6 +2,7 @@
|
||||
#include <stdio.h>
|
||||
#include <delta.h>
|
||||
#include <chunk.h>
|
||||
#include "scanner.h"
|
||||
|
||||
void print_usage(void) {
|
||||
printf("Usage:\n");
|
||||
@@ -19,11 +20,16 @@ void print_usage(void) {
|
||||
printf("Options:\n");
|
||||
printf(" -c, --checksum Verify content by checksum instead of size+mtime\n");
|
||||
printf(" -z, --compress [level] Enable compression (level 1-22, default 5)\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, metadata,\n");
|
||||
printf(" devices and specials (not compression/multithreading)\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, perms, times,\n");
|
||||
printf(" owner, group, devices and specials; not\n");
|
||||
printf(" compression/multithreading\n");
|
||||
printf(" -r, --recursive Recurse into directories (FastSync is always recursive)\n");
|
||||
printf(" -n, --dry-run Show what would be transferred\n");
|
||||
printf(" --remove-source-files Remove regular source files after successful transfer\n");
|
||||
printf(" -p, --perms Preserve permission bits (part of the metadata bundle)\n");
|
||||
printf(" -p, --perms Preserve permission bits\n");
|
||||
printf(" -t, --times Preserve modification times\n");
|
||||
printf(" -o, --owner Preserve owner (uid)\n");
|
||||
printf(" -g, --group Preserve group (gid)\n");
|
||||
printf(" --ssh-port <port> SSH port (default: 22)\n");
|
||||
printf(" -e, --rsh <command> Remote shell to launch on the client for the SSH\n");
|
||||
printf(" transport (default: ssh). The command may include\n");
|
||||
@@ -53,6 +59,8 @@ void print_usage(void) {
|
||||
printf(" Emit the batch file only (no destination, no server)\n");
|
||||
printf(" --read-batch=FILE Apply the batch file to the destination (no source, no\n");
|
||||
printf(" server); takes only the destination as an argument\n");
|
||||
printf(" NOTE: the FastSync batch format is NOT interoperable with rsync's batch\n");
|
||||
printf(" files (different container format); do not mix the two tools.\n");
|
||||
printf(" --delete Delete files on receiver not in source\n");
|
||||
printf(" (default timing: delete only after the whole\n");
|
||||
printf(" transfer has succeeded)\n");
|
||||
@@ -99,10 +107,10 @@ void print_usage(void) {
|
||||
printf(" parent directory is not itself listed\n");
|
||||
printf(" --mkpath Create the destination root directory on the server when it\n");
|
||||
printf(" does not exist yet\n");
|
||||
printf(" --exclude <pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file> Read include patterns from file\n");
|
||||
printf(" --exclude <pattern>, --exclude=<pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern>, --include=<pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file>, --exclude-from=<file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file>, --include-from=<file> Read include patterns from file\n");
|
||||
printf(" --files-from <file> Read the source file list from FILE (paths relative to the "
|
||||
"source root)\n");
|
||||
printf(" -0, --from0 Entries in --files-from are NUL-delimited\n");
|
||||
@@ -112,7 +120,8 @@ void print_usage(void) {
|
||||
printf(" -F Apply per-directory .rsync-filter files during the scan\n");
|
||||
printf(" --max-size <n> Skip files larger than n bytes\n");
|
||||
printf(" --min-size <n> Skip files smaller than n bytes\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G)\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G; 0 = no limit,\n");
|
||||
printf(" matching rsync)\n");
|
||||
printf(" --incremental Skip files unchanged since last transfer\n");
|
||||
printf(" --size-only Skip incremental files matching in size, ignoring mtime\n");
|
||||
printf(" -I, --ignore-times Transfer files even when size and mtime match\n");
|
||||
@@ -127,28 +136,36 @@ void print_usage(void) {
|
||||
printf(" --link-dest <dir> Like --copy-dest, but hard-links the unchanged file from DIR\n");
|
||||
printf(" into the destination (repeatable; earlier DIRs win)\n");
|
||||
printf(" --checksum-choice, --cc <alg> Whole-file checksum algorithm for --incremental/\n");
|
||||
printf(" --checksum compares (xxh64/xxhash or md5; default xxh64 with\n");
|
||||
printf(" seed 0). The seed comes from --checksum-seed\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash64 digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. The\n");
|
||||
printf(" digest algorithm and seed must match on sender and receiver\n");
|
||||
printf(" --checksum compares. Accepted: xxh64 (aka xxhash), xxh3,\n");
|
||||
printf(" xxh128, md5, or auto (default xxh64). rsync choices FastSync\n");
|
||||
printf(" does not implement (md4, sha1, none) and the two-name\n");
|
||||
printf(" transfer,pre-transfer form are rejected by name\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. A seed\n");
|
||||
printf(" of 0 (the default) is randomized per transfer, exactly like\n");
|
||||
printf(" rsync, and the chosen seed is sent to the receiver\n");
|
||||
printf(" --delta Delta transfer for changed files (requires --incremental)\n");
|
||||
printf(" -W, --whole-file Transfer changed files without delta processing\n");
|
||||
printf(" --no-whole-file rsync spelling that clears -W/--whole-file\n");
|
||||
printf(" -y, --fuzzy Use a similar-named file already in the destination\n");
|
||||
printf(" directory as the delta basis when the destination has no\n");
|
||||
printf(" usable file at the exact path (saves bandwidth; implies\n");
|
||||
printf(" --incremental and --delta; inert with --whole-file,\n");
|
||||
printf(" --no-delta, or --no-incremental)\n");
|
||||
printf(" --no-fuzzy Disable --fuzzy\n");
|
||||
printf(" --delta-block <n>, --block-size <n>\n");
|
||||
printf(" -B <n>, --block-size <n>, --delta-block <n>\n");
|
||||
printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT);
|
||||
printf(" --delta-max <n> Max file size for delta transfer (default: %llu)\n",
|
||||
DELTA_MAX_FILE_SIZE);
|
||||
printf(" -j, --threads Enable multithreading\n");
|
||||
printf(" -j, --threads[=N] Enable the multithreaded scanner/loader/sender\n");
|
||||
printf(" pipeline; N (1-%d) sets the parallel scanner worker\n",
|
||||
MAX_SCANNER_THREADS);
|
||||
printf(" count (bare -j/--threads uses the default)\n");
|
||||
printf(" --chunk-serialization Enable chunk serialization (long form only)\n");
|
||||
printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n");
|
||||
printf(" SSH argv is already built injection-safe)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only;\n");
|
||||
printf(" -f is bound to --filter, not --sendfile)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm (default: zstd)\n");
|
||||
printf(" --zc <alg> Alias for --compress-choice\n");
|
||||
printf(" -v, --verbose Enable debug logging\n");
|
||||
@@ -156,8 +173,16 @@ void print_usage(void) {
|
||||
printf(" --debug=FLAGS Fine-grained debug logging (use --debug=help for flags)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,misc,skip,stats,all,none\n");
|
||||
printf(" none suppresses info even with --verbose\n");
|
||||
printf(" --preserve Preserve file metadata (long form only)\n");
|
||||
printf(" --preserve Preserve permissions and times (= -pt; long form only)\n");
|
||||
printf(" --no-perms Negate -p/--perms\n");
|
||||
printf(" --no-times Negate -t/--times\n");
|
||||
printf(" --no-owner Negate -o/--owner\n");
|
||||
printf(" --no-group Negate -g/--group\n");
|
||||
printf(" --no-preserve Disable metadata preservation (negates --preserve)\n");
|
||||
printf(" -E, --executability Preserve executable permission bits\n");
|
||||
printf(" -U, --atimes Preserve access times\n");
|
||||
printf(" -N, --crtimes Capture birth time; cannot be applied (documented\n");
|
||||
printf(" divergence)\n");
|
||||
printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n");
|
||||
printf(" privileged security.*/trusted.* namespaces are\n");
|
||||
printf(" never captured or applied)\n");
|
||||
@@ -172,28 +197,32 @@ void print_usage(void) {
|
||||
printf(" (char/block device-node creation, --write-devices)\n");
|
||||
printf(" within the confined receive root. Never elevates\n");
|
||||
printf(" privileges and never bypasses confinement; ownership\n");
|
||||
printf(" is still applied only with an explicit identity flag\n");
|
||||
printf(" (--numeric-ids/--chown/--usermap/--groupmap/--copy-as)\n");
|
||||
printf(" is still applied only with -o/--owner, -g/--group, or an\n");
|
||||
printf(" explicit identity flag (--chown/--usermap/--groupmap/\n");
|
||||
printf(" --copy-as); --numeric-ids only changes how ids map\n");
|
||||
printf(" --no-super Forbid those super-user activities even when the\n");
|
||||
printf(" receiver is running as root\n");
|
||||
printf(" --chmod <changes> Modify transferred permissions (rsync syntax)\n");
|
||||
printf(" --numeric-ids Do not map uid/gid by name: use the source numeric\n");
|
||||
printf(" ids directly when applying ownership\n");
|
||||
printf(
|
||||
" --chmod <changes> Modify new/transferred permissions (rsync syntax; implies no -p)\n");
|
||||
printf(" --numeric-ids Map uid/gid by id instead of by name (a modifier, not\n");
|
||||
printf(" an ownership request: combine with -o/-g or a map)\n");
|
||||
printf(" --usermap=MAP Map usernames when applying ownership: comma-separated\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM/TO are names\n");
|
||||
printf(" (resolved on the source machine), * (match any /\n");
|
||||
printf(" current user), or @N numeric ids. e.g. *:nobody\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM is a name (from\n");
|
||||
printf(" the source), an id, an inclusive LOW-HIGH range, *\n");
|
||||
printf(" (any id), or empty (ids with no name). TO is an id, *\n");
|
||||
printf(" (current user), or a name resolved on the receiver.\n");
|
||||
printf(" e.g. 0-99:nobody,*:normal (cannot mix with --chown)\n");
|
||||
printf(" --groupmap=MAP Map group names when applying ownership (same syntax)\n");
|
||||
printf(" --chown=USER:GROUP Override the ownership of transferred files. Forms:\n");
|
||||
printf(" USER:GROUP, USER (owner only), :GROUP (group only); a\n");
|
||||
printf(" value of * means the current/root user as appropriate.\n");
|
||||
printf(" Names resolve on the source machine; @N for numerics.\n");
|
||||
printf(" (Metadata is enabled with --preserve; -M now means\n");
|
||||
printf(" rsync's --remote-option.)\n");
|
||||
printf(" (Implies owner/group metadata; -M now means rsync's\n");
|
||||
printf(" --remote-option.)\n");
|
||||
printf(" --copy-as=USER[:GROUP] Force every written entry (files, dirs, symlinks\n");
|
||||
printf(" and special nodes) to USER[:GROUP], resolved on the\n");
|
||||
printf(" source machine like --chown. Requires a privileged\n");
|
||||
printf(" (root) receiver and implies --preserve; an\n");
|
||||
printf(" (root) receiver and implies owner/group metadata; an\n");
|
||||
printf(" unprivileged receiver refuses the transfer. Never\n");
|
||||
printf(" switches process credentials (safe-subset; see\n");
|
||||
printf(" RSYNC_COMPAT.md). A daemon refuses it.\n");
|
||||
@@ -203,10 +232,12 @@ void print_usage(void) {
|
||||
printf(" --save-to-disk Write received files to disk\n");
|
||||
printf(" --server-host <ip> Server IP address (default: 127.0.0.1)\n");
|
||||
printf(" --server-port <n> Server port (default: 8080)\n");
|
||||
printf(" --port <n> Alias for --server-port\n");
|
||||
printf(" --password-file <f> Authenticate a host::module/path daemon destination.\n");
|
||||
printf(" The file's first user:password line supplies the\n");
|
||||
printf(" username and password (only a SHA-256 digest of the\n");
|
||||
printf(" password is sent; keep the file mode 0600)\n");
|
||||
printf(" FastSync-native SCRAM/PBKDF2 credential scheme (NOT\n");
|
||||
printf(" rsync's --password-file): the file's first user:password\n");
|
||||
printf(" line supplies the username and password; no password or\n");
|
||||
printf(" reusable digest is sent (keep the file mode 0600)\n");
|
||||
printf(" --no-motd Suppress display of the daemon's MOTD (the server\n");
|
||||
printf(" still sends it; the client just does not show it)\n");
|
||||
printf(" --bwlimit <KB/s> Bandwidth limit in kilobytes per second\n");
|
||||
@@ -214,21 +245,25 @@ void print_usage(void) {
|
||||
printf(" --cert <path> TLS certificate file (PEM)\n");
|
||||
printf(" --key <path> TLS private key file (PEM)\n");
|
||||
printf(" --ca <path> TLS CA certificate file (PEM)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 30; long form only)\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 10)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 0 = disabled, matching\n");
|
||||
printf(" rsync). 0 disables it; --no-timeout is the same\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 60, matching\n");
|
||||
printf(" rsync); 0 disables it (--no-contimeout)\n");
|
||||
printf(" --stop-after=MINS Stop the transfer after MINS minutes (a positive\n");
|
||||
printf(" integer); whatever was already transferred is kept\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time: HH:MM, HH:MM:SS, or\n");
|
||||
printf(" now+N[smhd] (a time already in the past stops the\n");
|
||||
printf(" transfer immediately; client-only). An early stop\n");
|
||||
printf(" skips the late --delete keep-set so it cannot delete\n");
|
||||
printf(" source mirrors that were not yet scanned\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time. Accepts rsync's date form\n");
|
||||
printf(" (Y-M-DTh:m, Y/M/DTh:m, abbreviable fields such as 12-31,\n");
|
||||
printf(" 14:00, :59, 1) plus FastSync's HH:MM[:SS] and now+N[smhd]\n");
|
||||
printf(" (a time already in the past stops the transfer\n");
|
||||
printf(" immediately; client-only). An early stop skips the late\n");
|
||||
printf(" --delete keep-set so it cannot delete source mirrors that\n");
|
||||
printf(" were not yet scanned\n");
|
||||
printf(" --address <ip> Bind the outgoing client socket to this source address\n");
|
||||
printf(" -4, --ipv4 Force IPv4 for destination resolution\n");
|
||||
printf(" -6, --ipv6 Force IPv6 for destination resolution\n");
|
||||
printf(" --sockopts=OPTS Comma-separated OPT=VAL socket options applied before connect:\n");
|
||||
printf(" TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR\n");
|
||||
printf(" --backup Backup existing files before overwriting\n");
|
||||
printf(" -b, --backup Backup existing files before overwriting\n");
|
||||
printf(" --backup-dir <dir> Directory for backups (requires --backup)\n");
|
||||
printf(" --suffix <str> Backup suffix (default: ~)\n");
|
||||
printf(" --stats Print transfer statistics at end\n");
|
||||
@@ -239,30 +274,40 @@ void print_usage(void) {
|
||||
printf(" -h, --human-readable Print byte sizes in human-readable form\n");
|
||||
printf(" --max-depth <n> Maximum directory depth (0=unlimited)\n");
|
||||
printf(" -x, --one-file-system Do not cross filesystem boundaries\n");
|
||||
printf(" --log-file <path> Write log messages to file\n");
|
||||
printf(" --log-file <path>, --log-file=<path> Write log messages to file\n");
|
||||
printf(" --stderr=MODE Route logging to stderr: errors or all\n");
|
||||
printf(" --partial Keep partial files on interrupted transfer\n");
|
||||
printf(" --partial-dir <dir> Directory for partial files\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install.\n");
|
||||
printf(" Relative dirs resolve below the destination root; absolute\n");
|
||||
printf(" dirs are used as-is (rsync semantics). The dir must\n");
|
||||
printf(" already exist; a different filesystem falls back to a\n");
|
||||
printf(" non-atomic copy instead of aborting\n");
|
||||
printf(" --fastsync-server-path <path>\n");
|
||||
printf(" Path to fastsync-server on remote (default: fastsync-server)\n");
|
||||
printf(" --old-args Accepted for rsync CLI compatibility; no effect (the\n");
|
||||
printf(" remote server path is always safely quoted now)\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation over SSH\n");
|
||||
printf(" (repeatable; each value is single-quote-escaped on the remote\n");
|
||||
printf(" command line; empty values and values with control characters\n");
|
||||
printf(" are rejected; -M OPT, -M=OPT and --remote-option=OPT work)\n");
|
||||
printf(" --trust-sender Trust the remote sender's file list: the receiver skips its\n");
|
||||
printf(" own up-front path-traversal/containment re-validation of the\n");
|
||||
printf(" incoming file list (fewer checks, faster, potentially unsafe).\n");
|
||||
printf(" Local receiver policy: never sent to the peer, off by default\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation. SSH\n");
|
||||
printf(" transport ONLY (user@host:path): a daemon (host::module) or\n");
|
||||
printf(" local TCP destination rejects it (no remote command line to\n");
|
||||
printf(" append to). Repeatable; each value is single-quote-escaped on\n");
|
||||
printf(" the remote command line; empty values and values with control\n");
|
||||
printf(" characters are rejected; -M OPT, -M=OPT and\n");
|
||||
printf(" --remote-option=OPT work\n");
|
||||
printf(" --trust-sender RECEIVER-LOCAL policy: trust the remote sender's file list\n");
|
||||
printf(" and skip the receiver's own up-front path-traversal/\n");
|
||||
printf(" containment re-validation of the incoming list (fewer checks,\n");
|
||||
printf(" faster, potentially unsafe). It is never sent to the peer, so\n");
|
||||
printf(" for a push it must be enabled on the receiving SERVER\n");
|
||||
printf(" (fastsync-server --trust-sender) or forwarded with\n");
|
||||
printf(" -M--trust-sender; the client flag alone has no effect\n");
|
||||
printf(" -l, --links Copy symlinks as symlinks\n");
|
||||
printf(" --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks that point outside transfer tree\n");
|
||||
printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n");
|
||||
printf(" -L, --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks whose target points outside the tree\n");
|
||||
printf(" --copy-unsafe-links Copy unsafe symlinks (outside tree) as referent files\n");
|
||||
printf(" -k, --copy-dirlinks Transform symlinks to directories into real dirs\n");
|
||||
printf(" -K, --keep-dirlinks Keep an existing symlink-to-dir as that dir\n");
|
||||
printf(" --munge-links Munge symlink targets on the wire (sender)\n");
|
||||
printf(" --munge-links Munge stored symlink targets (/rsyncd-munged/) on the receiver\n");
|
||||
printf(" -H, --hard-links Preserve hard-link relationships across the transfer\n");
|
||||
printf(" -S, --sparse Handle sparse files efficiently\n");
|
||||
printf(
|
||||
@@ -270,8 +315,7 @@ void print_usage(void) {
|
||||
printf(
|
||||
" --devices Recreate device nodes on the destination (privileged; skipped when\n");
|
||||
printf(" the receiver lacks CAP_MKNOD)\n");
|
||||
printf(" --specials Recreate special files (FIFOs) on the destination (sockets "
|
||||
"skipped)\n");
|
||||
printf(" --specials Recreate special files (FIFOs, sockets) on the destination\n");
|
||||
printf(" --copy-devices Copy a source device's content as a regular file instead\n");
|
||||
printf(" --write-devices Write received data into an existing destination device node\n");
|
||||
printf(" --inplace Update files in-place (no temp+rename)\n");
|
||||
@@ -284,7 +328,9 @@ void print_usage(void) {
|
||||
printf(" --fsync Fsync every written file before publication\n");
|
||||
printf(" --compress-level <n> Compression level (default: 5)\n");
|
||||
printf(" --zl <n> Alias for --compress-level\n");
|
||||
printf(" --skip-compress=LIST Skip compression for comma-separated suffixes\n");
|
||||
printf(" --skip-compress=LIST Skip compression for suffixes in LIST (separated by\n");
|
||||
printf(" '/' as in rsync, or ','); a leading dot is optional. The\n");
|
||||
printf(" default is rsync 3.4.1's built-in skip-compress list\n");
|
||||
printf(" --compress-threads <n> Compression worker threads (requires zstd threaded support)\n");
|
||||
printf(" --no-OPTION Disable a supported boolean option\n");
|
||||
printf(" --help Show this help\n");
|
||||
|
||||
+173
-24
@@ -12,6 +12,7 @@
|
||||
#include "utils.h"
|
||||
#include <stdlib.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code) {
|
||||
if (!outcomes)
|
||||
@@ -42,17 +43,19 @@ void receiver_outcomes_destroy(ReceiverOutcomes* outcomes) {
|
||||
/* End-of-transfer success frame. When --remove-source-files was negotiated
|
||||
each processed data file is acknowledged first (STATUS_NEXT = written,
|
||||
STATUS_OK = skipped) so the sender never removes a source the receiver did
|
||||
not actually store. The frame always ends with a plain STATUS_OK. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes) {
|
||||
not actually store. The frame ends with `final_status` (STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped). */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status) {
|
||||
if (!config->remove_source_files)
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
size_t count = outcomes ? outcomes->count : 0;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
Status per_file = outcomes->entries[i] == FILE_SAVE_WRITTEN ? STATUS_NEXT : STATUS_OK;
|
||||
if (!send_status(fd, per_file))
|
||||
return false;
|
||||
}
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
}
|
||||
|
||||
static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) {
|
||||
@@ -153,6 +156,93 @@ static bool receiver_process_batch(Config* config, int file_descriptor) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* ---- Anti-slowloris connection bounds ----
|
||||
* A legitimate transfer either streams data frames continuously or, when it
|
||||
* must pause, sends STATUS_KEEPALIVE so the peer sees the connection is alive.
|
||||
* An attacker can therefore squat on a connection slot indefinitely by sending
|
||||
* only keepalives under the per-message timeout. Two CLOCK_MONOTONIC bounds
|
||||
* defeat that without ever punishing a real transfer:
|
||||
*
|
||||
* MAX_SESSION_IDLE_SEC (1 h): the longest a stream may make no forward
|
||||
* progress. Data/status frames count as progress and refresh the timer;
|
||||
* keepalives do not. One hour is far longer than any real pause between
|
||||
* data frames, yet small enough to reap a slowloris well before the 24 h
|
||||
* session cap.
|
||||
*
|
||||
* MAX_SESSION_WALL_SEC (24 h): an absolute ceiling on one connection's
|
||||
* lifetime as defense-in-depth against a trickle of progress frames that
|
||||
* resets the idle timer just below its limit. Larger than any plausible
|
||||
* single transfer while still bounding resource occupancy.
|
||||
*
|
||||
* Both are wall-clock deltas, so the per-message poll timeout (60 s by default,
|
||||
* or --timeout) can never fool them, and both the single-threaded and the -m
|
||||
* receiver paths (receiver_process_pending) share the same logic. */
|
||||
#define MAX_SESSION_IDLE_SEC 3600u
|
||||
#define MAX_SESSION_WALL_SEC 86400u
|
||||
|
||||
static unsigned int g_max_session_idle_sec = MAX_SESSION_IDLE_SEC;
|
||||
static unsigned int g_max_session_wall_sec = MAX_SESSION_WALL_SEC;
|
||||
|
||||
void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec) {
|
||||
g_max_session_idle_sec = idle_sec;
|
||||
g_max_session_wall_sec = wall_sec;
|
||||
}
|
||||
|
||||
void receiver_reset_time_limits(void) {
|
||||
g_max_session_idle_sec = MAX_SESSION_IDLE_SEC;
|
||||
g_max_session_wall_sec = MAX_SESSION_WALL_SEC;
|
||||
}
|
||||
|
||||
bool receiver_time_limit_exceeded(const struct timespec* session_start,
|
||||
const struct timespec* last_progress,
|
||||
const struct timespec* now) {
|
||||
if (!session_start || !last_progress || !now)
|
||||
return false;
|
||||
if (now->tv_sec - session_start->tv_sec >= (time_t)g_max_session_wall_sec)
|
||||
return true;
|
||||
if (now->tv_sec - last_progress->tv_sec >= (time_t)g_max_session_idle_sec)
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/* A frame proves forward progress only when it cannot be fabricated for free.
|
||||
* KEEPALIVE/ABORT are pure liveness, and CHECK_BATCH/DIR_TIMES may carry zero
|
||||
* entries, so a peer must not be able to hold a connection slot forever by
|
||||
* merely emitting empty frames. */
|
||||
static bool status_counts_as_progress(Status status) {
|
||||
switch (status) {
|
||||
case STATUS_KEEPALIVE:
|
||||
case STATUS_ABORT:
|
||||
case STATUS_CHECK_BATCH:
|
||||
case STATUS_DIR_TIMES:
|
||||
return false;
|
||||
default:
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/* Refresh the progress timestamp for a forward-moving frame and enforce the
|
||||
* bounds above. Returns false when the connection must be dropped; the
|
||||
* terminal STATUS_ERROR is sent only when the sink owns error reporting (the
|
||||
* -m sink sets send_error=false so the main thread emits exactly one). */
|
||||
static bool receiver_note_status(const struct timespec* session_start,
|
||||
struct timespec* last_progress, Status status, int file_descriptor,
|
||||
const ReceiverSink* sink) {
|
||||
struct timespec now;
|
||||
if (clock_gettime(CLOCK_MONOTONIC, &now) != 0)
|
||||
now = *last_progress;
|
||||
if (status_counts_as_progress(status))
|
||||
*last_progress = now;
|
||||
if (!receiver_time_limit_exceeded(session_start, last_progress, &now))
|
||||
return true;
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"Receive session exceeded its time bound (idle %us / total %us); aborting connection",
|
||||
g_max_session_idle_sec, g_max_session_wall_sec);
|
||||
if (!sink || sink->send_error)
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) {
|
||||
return receiver_process_pending(config, file_descriptor, sink, NULL);
|
||||
}
|
||||
@@ -172,6 +262,15 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
Status status;
|
||||
if (!receive_status(file_descriptor, &status))
|
||||
return -1;
|
||||
/* Wall-clock (=CLOCK_MONOTONIC) anti-slowloris bookkeeping. session_start is
|
||||
* fixed for the whole connection; last_progress is refreshed by every frame
|
||||
* that is not a keepalive/abort. */
|
||||
struct timespec session_start;
|
||||
struct timespec last_progress;
|
||||
clock_gettime(CLOCK_MONOTONIC, &session_start);
|
||||
last_progress = session_start;
|
||||
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
|
||||
return -1;
|
||||
bool early_delete = config_delete_timing_early(config);
|
||||
/* Parked keep-set for the late/commit timing. Every exit path below frees it
|
||||
exactly once; the only exception is the successful FINISHED handoff, which
|
||||
@@ -191,10 +290,19 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
goto fail;
|
||||
}
|
||||
if (status == STATUS_CHECK) {
|
||||
bool skipped;
|
||||
File* file = receive_incremental_check(file_descriptor, config, &skipped);
|
||||
if (!skipped && (!file || !sink->store_file(file, sink->context)))
|
||||
bool skipped = false;
|
||||
bool would_transfer = false;
|
||||
File* file = receive_incremental_check_ex(file_descriptor, config, &skipped, &would_transfer);
|
||||
if (config->dry_run) {
|
||||
/* Server-contacting --dry-run: the reply has already been sent
|
||||
(STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and
|
||||
nothing may be stored. Both flags false means a genuine protocol
|
||||
error (STATUS_ERROR already sent or sent by receive_error below). */
|
||||
if (!skipped && !would_transfer)
|
||||
goto receive_error;
|
||||
} else if (!skipped && (!file || !sink->store_file(file, sink->context))) {
|
||||
goto receive_error;
|
||||
}
|
||||
} else if (status == STATUS_CHUNK) {
|
||||
Chunk* chunk = receive_chunk_data(file_descriptor, config);
|
||||
if (!chunk || !receiver_process_chunk(chunk, sink))
|
||||
@@ -226,20 +334,33 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
DeleteManifest* manifest = receive_manifest_entries(file_descriptor);
|
||||
if (!manifest)
|
||||
goto fail; /* receive_manifest_entries already sent STATUS_ERROR */
|
||||
if (config->dry_run) {
|
||||
/* Server-contacting --dry-run mutates nothing, so a keep-set manifest
|
||||
is consumed and discarded. The early-delete mode still needs its ACK
|
||||
so a sender blocked on the delete handshake is not left hanging. */
|
||||
delete_manifest_free(manifest);
|
||||
if (early_delete && !send_status(file_descriptor, STATUS_OK))
|
||||
goto fail;
|
||||
goto next_status;
|
||||
}
|
||||
if (early_delete) {
|
||||
/* --delete-before / --delete-during: the manifest is authoritative the
|
||||
moment it arrives, before any file data. Delete now and acknowledge
|
||||
so the sender only starts streaming once the deletion committed (or
|
||||
failed). This is the rsync delete-before/delete-during window: a
|
||||
later transfer failure does not restore these deletions. */
|
||||
bool deletion_ok = (config->use_delete || config->delete_missing_args)
|
||||
later transfer failure does not restore these deletions. A
|
||||
--max-delete-capped commit still succeeds and the transfer proceeds;
|
||||
the terminal success frame reports the cap. */
|
||||
DeleteCommitResult deletion = (config->use_delete || config->delete_missing_args)
|
||||
? manifest_delete_all(config, manifest)
|
||||
: true;
|
||||
: DELETE_COMMIT_OK;
|
||||
delete_manifest_free(manifest);
|
||||
if (!deletion_ok) {
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
if (!send_status(file_descriptor, STATUS_OK))
|
||||
goto fail;
|
||||
} else if (config->use_delete || config->delete_missing_args) {
|
||||
@@ -271,6 +392,8 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
next_status:
|
||||
if (!receive_status(file_descriptor, &status))
|
||||
goto receive_error;
|
||||
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
|
||||
goto fail;
|
||||
}
|
||||
if (status != STATUS_FINISHED) {
|
||||
log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status");
|
||||
@@ -290,13 +413,15 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
*pending_manifest = deferred_manifest;
|
||||
deferred_manifest = NULL;
|
||||
} else {
|
||||
bool deletion_ok = manifest_delete_all(config, deferred_manifest);
|
||||
DeleteCommitResult deletion = manifest_delete_all(config, deferred_manifest);
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
if (!deletion_ok) {
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
}
|
||||
}
|
||||
if (sink->send_success) {
|
||||
@@ -337,29 +462,41 @@ typedef struct {
|
||||
after the whole transfer (and its delete/publication phases) has run so a
|
||||
child write never clobbers a directory mtime. */
|
||||
DirTimeList dir_times;
|
||||
/* Set when a --max-delete commit was capped; the terminal frame then carries
|
||||
STATUS_DELETE_LIMIT so the sender exits 25 like rsync. */
|
||||
bool delete_limit_reached;
|
||||
} ReceiverSaveContext;
|
||||
|
||||
static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
FileSaveResult result = FILE_SAVE_ERROR;
|
||||
if (!context->config->save_to_disk) {
|
||||
if (context->config->dry_run) {
|
||||
/* Defense in depth: a dry-run receiver mutates nothing even if a data
|
||||
frame reaches the sink (the sender is not supposed to send one). */
|
||||
result = FILE_SAVE_SKIPPED;
|
||||
} else if (!context->config->save_to_disk) {
|
||||
/* Nothing is stored; report the file as not-written so a
|
||||
--remove-source-files sender keeps its source. */
|
||||
result = FILE_SAVE_SKIPPED;
|
||||
} else {
|
||||
result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
|
||||
}
|
||||
/* A directory's times are deferred, never applied inline: collect the
|
||||
metadata now and apply it at the end. -O/--omit-dir-times is honored by
|
||||
dir_time_list_apply's caller (see receiver_send_success_frame). */
|
||||
/* A directory's metadata is deferred, never applied inline: collect it now
|
||||
and apply it at the end. -O/--omit-dir-times and --preserve_perms/-times
|
||||
are honored by dir_metadata_list_apply's caller (see
|
||||
receiver_send_success_frame). */
|
||||
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
context->config->use_metadata && !context->config->omit_dir_times &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir &&
|
||||
!file->is_special && !file->skip &&
|
||||
/* A dry-run receiver mutates nothing AND records no per-file outcomes: a
|
||||
hostile dry-run client that streamed data frames anyway must not be able to
|
||||
grow `outcomes` without bound (receiver_outcomes_append reallocs uncharged)
|
||||
or force a per-frame ack. */
|
||||
if (!context->config->dry_run && result != FILE_SAVE_ERROR &&
|
||||
context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip &&
|
||||
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
return false;
|
||||
@@ -368,8 +505,18 @@ static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
return result != FILE_SAVE_ERROR;
|
||||
}
|
||||
|
||||
static void receiver_note_delete_limit(void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static bool receiver_send_success_frame(int fd, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK;
|
||||
/* Server-contacting --dry-run: nothing was staged or written, so there is
|
||||
nothing to publish and no directory times to stamp. */
|
||||
if (context->config->dry_run)
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
/* --delay-updates: the whole protocol stream (including manifest/delete
|
||||
handling, which ran inside receiver_process) has succeeded and every
|
||||
staged file was fully written. Publish them atomically now, before the
|
||||
@@ -385,14 +532,16 @@ static bool receiver_send_success_frame(int fd, void* context_pointer) {
|
||||
phases have committed, so it is finally safe to stamp directory times.
|
||||
This runs after the deferred deletion because receiver_process commits it
|
||||
before calling this success frame. */
|
||||
dir_time_list_apply(&context->dir_times, context->config->receive_root_directory);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes);
|
||||
dir_metadata_list_apply(&context->dir_times, context->config->receive_root_directory,
|
||||
context->config);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
}
|
||||
|
||||
int receiver_receive_files(Config* config, int file_descriptor) {
|
||||
ReceiverSaveContext context = {.config = config, .outcomes = {0}};
|
||||
dir_time_list_init(&context.dir_times);
|
||||
ReceiverSink sink = {receiver_save_file, &context, true, true, receiver_send_success_frame};
|
||||
ReceiverSink sink = {receiver_save_file, &context, true, true, receiver_send_success_frame,
|
||||
receiver_note_delete_limit};
|
||||
int ret = receiver_process(config, file_descriptor, &sink);
|
||||
if (ret != 0 && config->delay_updates && config->delay_context)
|
||||
delay_updates_cleanup(config->delay_context);
|
||||
|
||||
+34
-2
@@ -4,6 +4,9 @@
|
||||
#include "config.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include <stdbool.h>
|
||||
#include <time.h>
|
||||
|
||||
typedef bool (*ReceiverFileSink)(File* file, void* context);
|
||||
|
||||
@@ -19,6 +22,12 @@ typedef struct {
|
||||
|
||||
typedef bool (*ReceiverSuccessFrame)(int fd, void* context);
|
||||
|
||||
/* Records that a --max-delete commit stopped with extras left over, so the
|
||||
caller's terminal success frame can carry STATUS_DELETE_LIMIT instead of
|
||||
STATUS_OK. The commit runs on the receiver thread, so the flag is stored in
|
||||
the sink's own context rather than in a shared global. */
|
||||
typedef void (*ReceiverNoteDeleteLimit)(void* context);
|
||||
|
||||
typedef struct {
|
||||
ReceiverFileSink store_file;
|
||||
void* context;
|
||||
@@ -26,13 +35,18 @@ typedef struct {
|
||||
bool send_success;
|
||||
/* Emits the end-of-transfer success frame. When the sender requested
|
||||
--remove-source-files this includes one per-file status per processed
|
||||
data file followed by the final STATUS_OK; otherwise just STATUS_OK. */
|
||||
data file followed by the final status; otherwise just the final status. */
|
||||
ReceiverSuccessFrame send_success_frame;
|
||||
/* Optional; may be NULL when the sink has no --max-delete handling. */
|
||||
ReceiverNoteDeleteLimit note_delete_limit;
|
||||
} ReceiverSink;
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code);
|
||||
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes);
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes);
|
||||
/* Send the terminal success frame. `final_status` is usually STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status);
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink);
|
||||
/* receiver_process with an escape hatch for the commit-style (late) deletion:
|
||||
@@ -45,4 +59,22 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
DeleteManifest** pending_manifest);
|
||||
int receiver_receive_files(Config* config, int file_descriptor);
|
||||
|
||||
/* ---- Connection time bounds (anti-slowloris) ----
|
||||
* receiver_process_pending() aborts a connection that makes no forward progress
|
||||
* (only STATUS_KEEPALIVE/STATUS_ABORT frames) beyond a wall-clock idle limit,
|
||||
* and enforces a hard cap on the whole session. Both are CLOCK_MONOTONIC
|
||||
* deltas, independent of the per-message poll deadline, so a 60 s (or
|
||||
* --timeout) receive window can never reset them. Defaults are deliberately
|
||||
* generous (see MAX_SESSION_IDLE_SEC / MAX_SESSION_WALL_SEC in receiver.c). */
|
||||
|
||||
/* Test seam: override the idle/session wall-clock limits (0 = abort on the
|
||||
* next status). Always restore with receiver_reset_time_limits(). */
|
||||
void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec);
|
||||
void receiver_reset_time_limits(void);
|
||||
/* Pure predicate over explicit monotonic timestamps, exposed so the bound is
|
||||
* unit-testable without sleeping. True when either the idle or the overall
|
||||
* session limit has elapsed. */
|
||||
bool receiver_time_limit_exceeded(const struct timespec* session_start,
|
||||
const struct timespec* last_progress, const struct timespec* now);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,269 @@
|
||||
#include "receiver_pipeline.h"
|
||||
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <threads.h>
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue,
|
||||
int file_descriptor, SSL* ssl) {
|
||||
PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver));
|
||||
if (context == NULL)
|
||||
return NULL;
|
||||
context->config = config;
|
||||
context->queue = queue;
|
||||
context->file_descriptor = file_descriptor;
|
||||
context->ssl = ssl;
|
||||
context->outcomes.entries = NULL;
|
||||
context->outcomes.count = 0;
|
||||
context->outcomes.capacity = 0;
|
||||
dir_time_list_init(&context->dir_times);
|
||||
protocol_session_init(&context->session, file_descriptor, file_descriptor);
|
||||
protocol_session_set_ssl(&context->session, ssl);
|
||||
context->receiver_done = false;
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->deferred_manifest = NULL;
|
||||
context->delete_limit_reached = false;
|
||||
atomic_init(&context->cancelled, false);
|
||||
int init = 0;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_full) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_empty) != thrd_success)
|
||||
goto fail;
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
return context;
|
||||
|
||||
fail:
|
||||
log_perror("Error initializing synchronization objects");
|
||||
if (init >= 3)
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
if (init >= 2)
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&context->mutex);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
|
||||
config_delete(context->config);
|
||||
if (context->deferred_manifest)
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
queue_destroy(context->queue);
|
||||
receiver_outcomes_destroy(&context->outcomes);
|
||||
dir_time_list_free(&context->dir_times);
|
||||
mtx_destroy(&context->mutex);
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
free(context);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes) {
|
||||
if (context == NULL)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
context->max_queue_bytes = max_bytes;
|
||||
context->queued_bytes = 0;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes) {
|
||||
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
if (released_bytes >= context->queued_bytes)
|
||||
context->queued_bytes = 0;
|
||||
else
|
||||
context->queued_bytes -= released_bytes;
|
||||
cnd_signal(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) {
|
||||
if (context == NULL || file == NULL)
|
||||
return false;
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
mtx_lock(&context->mutex);
|
||||
while (!atomic_load(&context->cancelled)) {
|
||||
bool blocked_by_count = queue_is_full(context->queue);
|
||||
bool blocked_by_budget = false;
|
||||
if (context->max_queue_bytes > 0) {
|
||||
size_t budget = context->max_queue_bytes;
|
||||
size_t used = context->queued_bytes;
|
||||
if (used >= budget) {
|
||||
blocked_by_budget = true;
|
||||
} else if (file_bytes > budget - used) {
|
||||
/* A single payload larger than the whole budget (not possible with
|
||||
the per-file receive cap) is only admitted to an empty pipeline so
|
||||
the wait can never deadlock. */
|
||||
blocked_by_budget = used != 0;
|
||||
}
|
||||
}
|
||||
if (!blocked_by_count && !blocked_by_budget)
|
||||
break;
|
||||
cnd_wait(&context->condition_not_full, &context->mutex);
|
||||
}
|
||||
if (atomic_load(&context->cancelled)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (!queue_enqueue(context->queue, file)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
context->queued_bytes += file_bytes;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool receiver_enqueue_file(File* file, void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
return pipeline_context_receiver_enqueue_file(context, file);
|
||||
}
|
||||
|
||||
/* Early delete modes (--delete-before/--delete-during) commit the manifest
|
||||
inside receiver_process_pending on this thread; record a capped commit so
|
||||
server.c's terminal frame can report STATUS_DELETE_LIMIT. The plain bool is
|
||||
safe: receive_thread writes it before the main thread joins the thread. */
|
||||
static void receiver_pipeline_note_delete_limit(void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static void receiver_thread_fail(PipelineContextReceiver* context) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
int receive_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
int file_descriptor = context->file_descriptor;
|
||||
const Config* config = context->config;
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
ReceiverSink sink = {
|
||||
receiver_enqueue_file, context, false, false, NULL, receiver_pipeline_note_delete_limit};
|
||||
if (receiver_process_pending((Config*)config, file_descriptor, &sink,
|
||||
&context->deferred_manifest) != 0) {
|
||||
receiver_thread_fail(context);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
mtx_lock(&context->mutex);
|
||||
context->receiver_done = true;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
|
||||
int write_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
bool save_to_disk = context->config->save_to_disk;
|
||||
char* root_directory = str_dup(context->config->receive_root_directory);
|
||||
mtx_unlock(&context->mutex);
|
||||
if (save_to_disk && !root_directory) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
File* file =
|
||||
queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty,
|
||||
&context->condition_not_full, &context->receiver_done);
|
||||
if (file == NULL) {
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
FileSaveResult result = FILE_SAVE_SKIPPED;
|
||||
/* Server-contacting --dry-run: never write. The receiver thread does not
|
||||
enqueue anything on the dry-run path, but this keeps the writer thread
|
||||
provably mutation-free if a data frame ever reached it. */
|
||||
bool dry_run = context->config->dry_run;
|
||||
if (save_to_disk && !dry_run) {
|
||||
result = file_save_to_disk_full(root_directory, file, context->config);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
}
|
||||
/* P7 Wave D: a directory's times are never applied inline (a later child
|
||||
write would clobber them); accumulate the metadata here and let the
|
||||
caller apply it once every writer has drained. */
|
||||
if (!dry_run && result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
/* Record the per-file outcome so a --remove-source-files sender learns
|
||||
which sources were actually written versus skipped on the receiver.
|
||||
Explicit directory entries and recreated device/special nodes have no
|
||||
source and are never acknowledged (mirrors receiver.c). */
|
||||
if (!dry_run && context->config->remove_source_files && !file->is_dir && !file->is_special &&
|
||||
!file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,72 @@
|
||||
#ifndef RECEIVER_PIPELINE_H
|
||||
#define RECEIVER_PIPELINE_H
|
||||
|
||||
#include <stdatomic.h>
|
||||
#include <stdbool.h>
|
||||
#include <threads.h>
|
||||
|
||||
#include "config.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "receiver.h"
|
||||
#include <openssl/ssl.h>
|
||||
|
||||
typedef struct PipelineContextReceiver {
|
||||
Queue* queue;
|
||||
Config* config;
|
||||
int file_descriptor;
|
||||
SSL* ssl;
|
||||
ProtocolSession session;
|
||||
ReceiverOutcomes outcomes;
|
||||
mtx_t mutex;
|
||||
cnd_t condition_not_full;
|
||||
cnd_t condition_not_empty;
|
||||
bool receiver_done;
|
||||
atomic_bool cancelled;
|
||||
/* Aggregate payload bytes that have been received but not yet released by
|
||||
the disk writer (queued or in the writer's hand). Guarded by `mutex`.
|
||||
When `max_queue_bytes` is non-zero the receiver blocks before enqueuing
|
||||
once this total would exceed it, so decompressed/copied file payloads
|
||||
buffered ahead of a slow disk writer respect the per-connection memory
|
||||
budget instead of growing without bound. */
|
||||
size_t queued_bytes;
|
||||
size_t max_queue_bytes;
|
||||
/* Keep-set manifest for the commit-style (late) deletion
|
||||
(--delete/--delete-after/--delete-delay). receive_thread parses the whole
|
||||
protocol stream but hands the manifest here instead of deleting while the
|
||||
disk writer may still be draining; the caller (server.c) commits the
|
||||
deletion after both threads have joined, so no extra is removed unless the
|
||||
transfer truly succeeded. NULL in the early delete modes (which delete at
|
||||
the manifest). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* Set by server.c when the deferred delete commit hit the --max-delete
|
||||
budget; the terminal success frame then carries STATUS_DELETE_LIMIT
|
||||
(rsync exit 25) while the transfer itself still succeeds. */
|
||||
bool delete_limit_reached;
|
||||
/* P7 Wave D: directory metadata collected by write_thread from received
|
||||
directory entries. Only write_thread mutates it (before it joins); the
|
||||
caller (server.c) applies it after the delete/delay-updates phase. */
|
||||
DirTimeList dir_times;
|
||||
} PipelineContextReceiver;
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
|
||||
int file_descriptor, SSL* ssl);
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context);
|
||||
/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes);
|
||||
/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue
|
||||
is full by element count or when adding `file` would push queued_bytes over
|
||||
the configured byte limit; waits until the disk writer releases bytes.
|
||||
Takes ownership of `file` on success and destroys it on failure/cancel. */
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file);
|
||||
/* Account for `released_bytes` of payload memory that has been freed by the
|
||||
disk writer, unblocking a receiver that is waiting on the byte limit. */
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes);
|
||||
int receive_thread(void* pipeline_context);
|
||||
int write_thread(void* pipeline_context);
|
||||
|
||||
#endif
|
||||
+599
-243
File diff suppressed because it is too large.
Load diff
+32
-2
@@ -179,6 +179,8 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
opts->trust_sender = true;
|
||||
} else if (arg_is(argv[i], "--no-super")) {
|
||||
opts->no_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-super")) {
|
||||
opts->allow_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-unauthenticated")) {
|
||||
opts->allow_unauthenticated = true;
|
||||
} else if (arg_has_value(argv[i], "--iconv", &inline_value)) {
|
||||
@@ -190,14 +192,20 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->iconv_spec = inline_value;
|
||||
} else if (arg_is(argv[i], "-p")) {
|
||||
} else if (arg_is(argv[i], "-p") || arg_has_value(argv[i], "--port", &inline_value)) {
|
||||
if (inline_value) {
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(inline_value, &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
} else {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for -p");
|
||||
set_error(err, err_size, "missing argument for %s", argv[i]);
|
||||
return -1;
|
||||
}
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
}
|
||||
} else {
|
||||
if (arg_has_value(argv[i], "--config", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
@@ -259,6 +267,28 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
set_error(err, err_size, "--hash-credentials cannot be combined with --daemon or --stdio");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->no_super) {
|
||||
set_error(err, err_size, "--allow-super and --no-super are mutually exclusive");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->daemon_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is for a locally-launched standalone TCP server; daemon modules opt "
|
||||
"in per module with 'client owner = yes'");
|
||||
return -1;
|
||||
}
|
||||
/* --stdio is the SSH transport: the remote server argv is composed by the
|
||||
* CLIENT (directly and via --remote-option), so a client could otherwise pass
|
||||
* --allow-super to a root --stdio receiver and defeat the C3 secure default.
|
||||
* Never honor it there; the super mode stays forced OFF. An operator who
|
||||
* must keep the historical permissive behavior over SSH has to launch the
|
||||
* receiver through a forced command, not via client-composed argv. */
|
||||
if (opts->allow_super && opts->stdio_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is not accepted with --stdio (the remote argv is client-composed; "
|
||||
"use a forced command if the default must hold)");
|
||||
return -1;
|
||||
}
|
||||
if (opts->hash_iterations_set && opts->hash_credentials_file == NULL) {
|
||||
set_error(err, err_size, "--iterations requires --hash-credentials");
|
||||
return -1;
|
||||
|
||||
@@ -45,6 +45,17 @@ typedef struct ServerCliOptions {
|
||||
* device-node creation) even when running as root. Applies to --stdio and
|
||||
* --daemon alike; also makes the server refuse any client --copy-as. */
|
||||
bool no_super; /* --no-super */
|
||||
/* --allow-super: locally-launched standalone TCP listener opt-in that keeps
|
||||
* the historical permissive behavior for a PRIVILEGED (root) receiver.
|
||||
* Without it a root standalone server forces SUPER_MODE_OFF, so a client
|
||||
* --devices / --write-devices / --super / ownership request cannot make it
|
||||
* create device nodes, write raw devices, or apply client-chosen ownership.
|
||||
* It is rejected for --stdio: that path's remote argv is composed by the
|
||||
* client (directly and via --remote-option), so it must never opt a root
|
||||
* receiver back into super mode. Non-root receivers are unaffected (the
|
||||
* kernel refuses the confined attempts). The daemon path instead uses the
|
||||
* per-module `client owner = yes` opt-in. */
|
||||
bool allow_super; /* --allow-super */
|
||||
/* --iconv=CONVERT_SPEC: the server's own LOCAL charset declaration. The
|
||||
* client's full spec rides the wire config frame anyway; when the server is
|
||||
* started with its own --iconv, its LOCAL half overrides the local charset
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "log.h"
|
||||
#include "array_list.h"
|
||||
#include "protocol.h"
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -39,6 +40,8 @@ void array_list_delete(ArrayList* array_list) {
|
||||
static bool array_list_extend(ArrayList* array_list) {
|
||||
if (array_list == NULL)
|
||||
return false;
|
||||
if (array_list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_capacity = array_list->capacity * 2;
|
||||
if (new_capacity == 0)
|
||||
new_capacity = INITIAL_ARRAY_SIZE;
|
||||
|
||||
+53
-18
@@ -2,6 +2,7 @@
|
||||
#include "data.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include <errno.h>
|
||||
#include <stdlib.h>
|
||||
@@ -10,11 +11,15 @@
|
||||
|
||||
/* Serialization metadata mode for the batch stream, captured from the config at
|
||||
* batch_write_header time. The header persists it into the file so a batch is
|
||||
* self-describing: batch_read_apply re-reads it from the file (not from the
|
||||
* reading config), so a batch written with -M is applied identically by an
|
||||
* invoking process regardless of its own -M setting. The batch driver is a
|
||||
* single sequential scan pass within one thread, so this module-level flag is
|
||||
* safe. */
|
||||
* self-describing about whether per-entry metadata was CAPTURED in the stream:
|
||||
* batch_read_apply re-reads it from the file (not from the reading config) to
|
||||
* decode the chunk records correctly. Which attributes are actually APPLIED,
|
||||
* however, comes from the INVOKING process's per-attribute config (the
|
||||
* FileAttrPolicy and the dir-metadata gate), so a batch written with -M is NOT
|
||||
* automatically applied identically by an invoking process with a different
|
||||
* -p/-t/-o/-g: --read-batch must be invoked with the same -p/-t/-o/-g as the
|
||||
* write side (rsync requires the same options). The batch driver is a single
|
||||
* sequential scan pass within one thread, so this module-level flag is safe. */
|
||||
static bool batch_metadata_mode = false;
|
||||
|
||||
static bool write_all_bytes(int fd, const void* data, size_t size) {
|
||||
@@ -91,22 +96,37 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
if (fd < 0 || dest_root == NULL || dest_root[0] == '\0')
|
||||
return -1;
|
||||
|
||||
/* Directory metadata is deferred to the end of the apply (a child write would
|
||||
* otherwise clobber its parent's mtime/mode). The batch header's single
|
||||
* metadata bit only says whether metadata is present in the stream; which
|
||||
* attributes are APPLIED comes from the invoking process's config, so
|
||||
* --read-batch must be invoked with the same -p/-t/-o/-g as the write side
|
||||
* (rsync requires the same options). The identity snapshot is activated so
|
||||
* -o/-g and the explicit ownership flags can apply. */
|
||||
DirTimeList dir_times;
|
||||
dir_time_list_init(&dir_times);
|
||||
int result = -1;
|
||||
if (!identity_set_active(config)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not activate the identity policy");
|
||||
goto done;
|
||||
}
|
||||
|
||||
char magic[BATCH_MAGIC_LEN];
|
||||
bool eof = false;
|
||||
if (!read_exact(fd, magic, BATCH_MAGIC_LEN, &eof) || eof ||
|
||||
memcmp(magic, BATCH_MAGIC, BATCH_MAGIC_LEN) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad magic)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char version;
|
||||
if (!read_exact(fd, &version, 1, &eof) || eof || version != BATCH_FORMAT_VERSION) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad or missing format version)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char mode;
|
||||
if (!read_exact(fd, &mode, 1, &eof) || eof || (mode != 0 && mode != 1)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad metadata flag)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
bool use_metadata = mode == 1;
|
||||
|
||||
@@ -114,47 +134,62 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
unsigned long long length;
|
||||
if (!read_exact(fd, &length, sizeof(length), &eof)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated length prefix");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (eof)
|
||||
break; /* clean end of stream */
|
||||
if (length == 0 || length > BATCH_MAX_RECORD) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected record length %llu (valid range 1..%llu)",
|
||||
length, (unsigned long long)BATCH_MAX_RECORD);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
char* record = (char*)malloc((size_t)length);
|
||||
if (record == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not allocate a %llu-byte record", length);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (!read_exact(fd, record, (size_t)length, &eof) || eof) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated chunk record");
|
||||
free(record);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
Data* data = data_create(record, (size_t)length);
|
||||
if (data == NULL)
|
||||
return -1; /* data_create frees `record` on failure */
|
||||
goto done; /* data_create frees `record` on failure */
|
||||
Chunk* chunk = chunk_deserialize(data, use_metadata);
|
||||
data_destroy(data);
|
||||
if (chunk == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected malformed chunk record");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* file = chunk->items[i];
|
||||
chunk->items[i] = NULL;
|
||||
if (file == NULL)
|
||||
continue;
|
||||
FileSaveResult result = file_save_to_disk_full(dest_root, file, config);
|
||||
FileSaveResult save = file_save_to_disk_full(dest_root, file, config);
|
||||
/* Accumulate directory metadata (when it applies) before the File is
|
||||
* destroyed; applied once the whole stream has been consumed. */
|
||||
if (save != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(config) &&
|
||||
!dir_time_list_add(&dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
chunk_destroy(chunk);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
file_destroy(file);
|
||||
if (save == FILE_SAVE_ERROR) {
|
||||
chunk_destroy(chunk);
|
||||
goto done;
|
||||
}
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
return 0;
|
||||
dir_metadata_list_apply(&dir_times, dest_root, config);
|
||||
result = 0;
|
||||
|
||||
done:
|
||||
identity_clear_active();
|
||||
dir_time_list_free(&dir_times);
|
||||
return result;
|
||||
}
|
||||
+26
-1
@@ -21,6 +21,20 @@ bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t
|
||||
return true;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_XXH3) {
|
||||
uint64_t digest = XXH3_64bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_XXH128) {
|
||||
XXH128_hash_t digest = XXH3_128bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5) {
|
||||
/* md5 takes no seed; the caller's seed is deliberately ignored (documented
|
||||
* in RSYNC_COMPAT.md). OpenSSL's one-shot EVP_Digest needs a non-NULL
|
||||
@@ -45,6 +59,10 @@ int checksum_algo_from_name(const char* name) {
|
||||
return -1;
|
||||
if (strcasecmp(name, "xxh64") == 0 || strcasecmp(name, "xxhash") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH64;
|
||||
if (strcasecmp(name, "xxh3") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH3;
|
||||
if (strcasecmp(name, "xxh128") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH128;
|
||||
if (strcasecmp(name, "md5") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD5;
|
||||
return -1;
|
||||
@@ -54,6 +72,10 @@ const char* checksum_algo_name(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
return "xxh64";
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return "xxh3";
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
return "xxh128";
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
return "md5";
|
||||
}
|
||||
@@ -61,13 +83,16 @@ const char* checksum_algo_name(ChecksumAlgo algo) {
|
||||
}
|
||||
|
||||
bool checksum_algo_valid(int algo) {
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5;
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5 ||
|
||||
algo == (int)CHECKSUM_ALGO_XXH3 || algo == (int)CHECKSUM_ALGO_XXH128;
|
||||
}
|
||||
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return 8;
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
return 16;
|
||||
}
|
||||
|
||||
+15
-6
@@ -9,10 +9,17 @@
|
||||
* seeded with --checksum-seed. The ids are the values actually placed on the
|
||||
* wire (config frame), so they must be kept stable and validated on receive.
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the default and is byte-for-byte what FastSync
|
||||
* computed before these options existed (xxHash64 with seed 0). */
|
||||
typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
|
||||
* computed before these options existed (xxHash64 with seed 0). The set mirrors
|
||||
* the algorithms rsync 3.4.1 can be built with; the ones FastSync does not
|
||||
* implement (md4, sha1, none) are rejected by name at parse time. */
|
||||
typedef enum {
|
||||
CHECKSUM_ALGO_XXH64 = 0,
|
||||
CHECKSUM_ALGO_MD5 = 1,
|
||||
CHECKSUM_ALGO_XXH3 = 2,
|
||||
CHECKSUM_ALGO_XXH128 = 3
|
||||
} ChecksumAlgo;
|
||||
|
||||
/* md5 digest is 16 bytes, the longest supported. */
|
||||
/* xxh128 digest is 16 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 16
|
||||
|
||||
/* Compute the whole-file digest of the first `size` bytes of `data`.
|
||||
@@ -29,8 +36,10 @@ bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "xxh64" and "xxhash" (both map to CHECKSUM_ALGO_XXH64, rsync's
|
||||
* xxhash spelling) and "md5". Returns -1 for any unsupported name. */
|
||||
* Accepts "xxh64"/"xxhash", "xxh3", "xxh128" and "md5". "auto", rsync's
|
||||
* default automatic choice, is resolved to the default by the caller (it is not
|
||||
* a distinct algorithm here). Returns -1 for any name FastSync does not
|
||||
* implement (md4/sha1/none included). */
|
||||
int checksum_algo_from_name(const char* name);
|
||||
|
||||
/* Canonical name of an algorithm (used in CLI error messages). */
|
||||
@@ -39,7 +48,7 @@ const char* checksum_algo_name(ChecksumAlgo algo);
|
||||
/* True when `algo` is a supported id (used by config receive validation). */
|
||||
bool checksum_algo_valid(int algo);
|
||||
|
||||
/* Digest length in bytes for an algorithm (xxx64 = 8, md5 = 16). */
|
||||
/* Digest length in bytes for an algorithm (xxh64/xxh3 = 8, md5/xxh128 = 16). */
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo);
|
||||
|
||||
#endif /* CHECKSUM_H */
|
||||
+170
-77
@@ -1,90 +1,183 @@
|
||||
#include "chmod.h"
|
||||
#include "file.h"
|
||||
#include <stddef.h>
|
||||
#include <string.h>
|
||||
|
||||
static bool parse_clause(mode_t* mode, const char* begin, const char* end) {
|
||||
const char* p = begin;
|
||||
unsigned who = 0;
|
||||
while (p < end && strchr("ugoa", *p)) {
|
||||
if (*p == 'a')
|
||||
who = 7;
|
||||
else
|
||||
who |= *p == 'u' ? 1U : (*p == 'g' ? 2U : 4U);
|
||||
p++;
|
||||
}
|
||||
if (who == 0)
|
||||
who = 7;
|
||||
if (p == end || (*p != '+' && *p != '-' && *p != '='))
|
||||
return false;
|
||||
char operation = *p++;
|
||||
mode_t bits = 0;
|
||||
while (p < end) {
|
||||
mode_t bit;
|
||||
switch (*p++) {
|
||||
case 'r':
|
||||
bit = 4;
|
||||
break;
|
||||
case 'w':
|
||||
bit = 2;
|
||||
break;
|
||||
case 'x':
|
||||
bit = 1;
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
bits |= bit;
|
||||
}
|
||||
for (unsigned class_index = 0; class_index < 3; class_index++) {
|
||||
unsigned class_bit = 1U << class_index;
|
||||
if (!(who & class_bit))
|
||||
continue;
|
||||
mode_t shift = (mode_t)((2U - class_index) * 3U);
|
||||
mode_t mask = (mode_t)(7U << shift);
|
||||
mode_t class_bits = (mode_t)(bits << shift);
|
||||
if (operation == '+')
|
||||
*mode |= class_bits;
|
||||
else if (operation == '-')
|
||||
*mode &= ~class_bits;
|
||||
else
|
||||
*mode = (*mode & ~mask) | class_bits;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
/* rsync's --chmod parser (parse_chmod + tweak_mode). A single clause is
|
||||
* applied as it is completed, so repeated clauses and repeated --chmod options
|
||||
* (joined with commas by the CLI) accumulate exactly like rsync. The D/F
|
||||
* selectors restrict a clause to directories/files; X adds execute only to
|
||||
* directories or files that were already executable. */
|
||||
|
||||
#define CHMOD_BITS 07777
|
||||
#define CHMOD_FLAG_X_KEEP (1U << 0)
|
||||
#define CHMOD_FLAG_DIRS_ONLY (1U << 1)
|
||||
#define CHMOD_FLAG_FILES_ONLY (1U << 2)
|
||||
|
||||
enum chmod_op { CHMOD_OP_ADD = 1, CHMOD_OP_SUB, CHMOD_OP_EQ, CHMOD_OP_SET };
|
||||
enum chmod_state {
|
||||
CHMOD_STATE_ERROR,
|
||||
CHMOD_STATE_1ST_HALF,
|
||||
CHMOD_STATE_2ND_HALF,
|
||||
CHMOD_STATE_OCTAL
|
||||
};
|
||||
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result) {
|
||||
if (!spec || !*spec || !result)
|
||||
return false;
|
||||
bool numeric = true;
|
||||
size_t length = strlen(spec);
|
||||
if (length > 4)
|
||||
numeric = false;
|
||||
for (size_t i = 0; i < length && numeric; i++)
|
||||
numeric = spec[i] >= '0' && spec[i] <= '7';
|
||||
if (numeric) {
|
||||
if (length == 0 || length > 4)
|
||||
return false;
|
||||
mode_t parsed = 0;
|
||||
for (size_t i = 0; i < length; i++)
|
||||
parsed = (mode_t)((parsed << 3) | (spec[i] - '0'));
|
||||
*result = parsed;
|
||||
return true;
|
||||
}
|
||||
|
||||
const mode_t nonperm = mode & ~(mode_t)CHMOD_BITS;
|
||||
const bool initially_executable = (mode & 0111) != 0;
|
||||
mode_t changed = mode;
|
||||
const char* begin = spec;
|
||||
while (*begin) {
|
||||
const char* end = strchr(begin, ',');
|
||||
if (!end)
|
||||
end = begin + strlen(begin);
|
||||
if (!parse_clause(&changed, begin, end))
|
||||
return false;
|
||||
if (*end == '\0')
|
||||
int state = CHMOD_STATE_1ST_HALF;
|
||||
unsigned where = 0;
|
||||
int what = 0, op = 0, topbits = 0, topoct = 0, flags = 0;
|
||||
const char* p = spec;
|
||||
while (state != CHMOD_STATE_ERROR) {
|
||||
if (*p == '\0' || *p == ',') {
|
||||
int bits;
|
||||
if (!op) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
begin = end + 1;
|
||||
if (!*begin)
|
||||
return false;
|
||||
}
|
||||
*result = changed;
|
||||
if (where)
|
||||
bits = (int)(where * (unsigned)what);
|
||||
else {
|
||||
where = 0111;
|
||||
bits = (int)((where * (unsigned)what) & ~(unsigned)file_process_umask());
|
||||
}
|
||||
int mode_and, mode_or;
|
||||
switch (op) {
|
||||
case CHMOD_OP_ADD:
|
||||
mode_and = CHMOD_BITS;
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
case CHMOD_OP_SUB:
|
||||
mode_and = CHMOD_BITS - bits - topoct;
|
||||
mode_or = 0;
|
||||
break;
|
||||
case CHMOD_OP_EQ:
|
||||
mode_and = CHMOD_BITS - (int)(where * 7U) - (topoct ? topbits : 0);
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
default:
|
||||
mode_and = 0;
|
||||
mode_or = bits;
|
||||
break;
|
||||
}
|
||||
bool is_dir = S_ISDIR(nonperm);
|
||||
if (!((flags & CHMOD_FLAG_DIRS_ONLY) && !is_dir) &&
|
||||
!((flags & CHMOD_FLAG_FILES_ONLY) && is_dir)) {
|
||||
changed &= (mode_t)mode_and;
|
||||
if ((flags & CHMOD_FLAG_X_KEEP) && !initially_executable && !is_dir)
|
||||
changed |= (mode_t)(mode_or & ~0111);
|
||||
else
|
||||
changed |= (mode_t)mode_or;
|
||||
}
|
||||
if (*p == '\0')
|
||||
break;
|
||||
p++;
|
||||
state = CHMOD_STATE_1ST_HALF;
|
||||
where = 0;
|
||||
what = op = topoct = topbits = flags = 0;
|
||||
continue;
|
||||
}
|
||||
switch (state) {
|
||||
case CHMOD_STATE_1ST_HALF:
|
||||
switch (*p) {
|
||||
case 'D':
|
||||
if (flags & CHMOD_FLAG_FILES_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_DIRS_ONLY;
|
||||
break;
|
||||
case 'F':
|
||||
if (flags & CHMOD_FLAG_DIRS_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_FILES_ONLY;
|
||||
break;
|
||||
case 'u':
|
||||
where |= 0100;
|
||||
topbits |= 04000;
|
||||
break;
|
||||
case 'g':
|
||||
where |= 0010;
|
||||
topbits |= 02000;
|
||||
break;
|
||||
case 'o':
|
||||
where |= 0001;
|
||||
break;
|
||||
case 'a':
|
||||
where |= 0111;
|
||||
break;
|
||||
case '+':
|
||||
op = CHMOD_OP_ADD;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '-':
|
||||
op = CHMOD_OP_SUB;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '=':
|
||||
op = CHMOD_OP_EQ;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7' && !where) {
|
||||
op = CHMOD_OP_SET;
|
||||
state = CHMOD_STATE_OCTAL;
|
||||
where = 1;
|
||||
what = *p - '0';
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case CHMOD_STATE_2ND_HALF:
|
||||
switch (*p) {
|
||||
case 'r':
|
||||
what |= 4;
|
||||
break;
|
||||
case 'w':
|
||||
what |= 2;
|
||||
break;
|
||||
case 'X':
|
||||
flags |= CHMOD_FLAG_X_KEEP;
|
||||
/* fall through */
|
||||
case 'x':
|
||||
what |= 1;
|
||||
break;
|
||||
case 's':
|
||||
if (topbits)
|
||||
topoct |= topbits;
|
||||
else
|
||||
topoct = 04000;
|
||||
break;
|
||||
case 't':
|
||||
topoct |= 01000;
|
||||
break;
|
||||
default:
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7') {
|
||||
what = what * 8 + (*p - '0');
|
||||
if (what > CHMOD_BITS)
|
||||
state = CHMOD_STATE_ERROR;
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
p++;
|
||||
}
|
||||
if (state == CHMOD_STATE_ERROR)
|
||||
return false;
|
||||
*result = (changed & (mode_t)CHMOD_BITS) | nonperm;
|
||||
return true;
|
||||
}
|
||||
+4
-1
@@ -4,7 +4,10 @@
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Apply the supported rsync --chmod syntax to a permission mode. */
|
||||
/* Apply rsync's --chmod syntax to a permission mode, including the D/F/X
|
||||
* selectors and the s/t special bits. `mode` should carry the file type bits
|
||||
* (S_IFDIR/S_IFREG) so D/F/X can be evaluated; the type bits are preserved in
|
||||
* `result`. A spec may contain comma-separated clauses, which accumulate. */
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result);
|
||||
|
||||
#endif
|
||||
+100
-96
@@ -1,6 +1,7 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -20,6 +21,33 @@
|
||||
#define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024)
|
||||
#define MAX_FILES_PER_CHUNK 65536U
|
||||
|
||||
/* Reserve `charge` against `session`'s connection budget. This mirrors the
|
||||
static protocol_reserve_memory() in protocol.c: the receive-side call sites
|
||||
only have the Data.owner pointer (a ProtocolSession*), and protocol.c is out
|
||||
of scope for this fix, so the same atomic CAS accounting is reproduced here.
|
||||
The matching release always goes through data_destroy()'s Data.owner path. */
|
||||
static bool chunk_session_reserve(ProtocolSession* session, size_t charge) {
|
||||
unsigned long long allocated = atomic_load(&session->total_allocated_bytes);
|
||||
while (true) {
|
||||
if (allocated > MAX_CONNECTION_MEMORY ||
|
||||
(unsigned long long)charge > MAX_CONNECTION_MEMORY - allocated)
|
||||
return false;
|
||||
if (atomic_compare_exchange_weak(&session->total_allocated_bytes, &allocated,
|
||||
allocated + (unsigned long long)charge))
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge) {
|
||||
if (!data || charge == 0 || session == NULL)
|
||||
return true;
|
||||
if (!chunk_session_reserve(session, charge))
|
||||
return false;
|
||||
data->owner = session;
|
||||
data->protocol_charge = charge;
|
||||
return true;
|
||||
}
|
||||
|
||||
Chunk* chunk_create(File** items, int element_count) {
|
||||
if (element_count < 0 || (element_count > 0 && items == NULL))
|
||||
return NULL;
|
||||
@@ -207,17 +235,20 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
return NULL;
|
||||
char* data_pointer = data->data;
|
||||
size_t remaining_size = data->size;
|
||||
/* The element currently being parsed is owned by `files` only after the
|
||||
* array_list_add() at the end of the iteration; until then the error
|
||||
* epilogue destroys it directly. Keeping this one pointer nulled after the
|
||||
* hand-off makes the single cleanup path correct for every failure. */
|
||||
File* file = NULL;
|
||||
|
||||
while (remaining_size > 0) {
|
||||
if ((unsigned int)files->size >= MAX_FILES_PER_CHUNK) {
|
||||
log_message(LOG_LEVEL_ERROR, "Chunk contains too many files");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path length");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
size_t path_len;
|
||||
@@ -227,26 +258,19 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (path_len > SIZE_MAX - 1 || remaining_size < path_len) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
if (path_len == SIZE_MAX) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
char* path = protocol_alloc(path_len + 1);
|
||||
if (path == NULL) {
|
||||
log_perror("Could not allocate memory for file path");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
memcpy(path, data_pointer, path_len);
|
||||
path[path_len] = '\0';
|
||||
if (memchr(path, '\0', path_len) != NULL) {
|
||||
free(path);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
data_pointer += path_len;
|
||||
remaining_size -= path_len;
|
||||
@@ -260,8 +284,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (local_path == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv: received chunk file name cannot be converted to the local charset");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
path = local_path;
|
||||
path_len = strlen(path);
|
||||
@@ -269,30 +292,23 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (path_len == 0 || has_path_traversal(path)) {
|
||||
free(path);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
File* file = file_create(path);
|
||||
file = file_create(path);
|
||||
free(path);
|
||||
if (file == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (file == NULL)
|
||||
goto error;
|
||||
|
||||
if (remaining_size < sizeof(int)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for entry type");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
int entry_type;
|
||||
memcpy(&entry_type, data_pointer, sizeof(int));
|
||||
if (entry_type != 0 && entry_type != 1 && entry_type != 2 && entry_type != 3) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad entry type");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
file->is_dir = entry_type == 1;
|
||||
file->is_symlink = entry_type == 2;
|
||||
@@ -303,9 +319,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (file->is_special) {
|
||||
if (remaining_size < 2 * (int32_t)sizeof(int32_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for special rdev");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
int32_t special_major, special_minor;
|
||||
memcpy(&special_major, data_pointer, sizeof(special_major));
|
||||
@@ -320,9 +334,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (special_major < 0 || special_minor < 0 || special_major > 0xffff ||
|
||||
special_minor > 0x00ffffff) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: out-of-range special rdev");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
file->rdev_major = special_major;
|
||||
file->rdev_minor = special_minor;
|
||||
@@ -331,37 +343,32 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (use_metadata) {
|
||||
if (remaining_size < sizeof(int)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
// Peek at present flag to determine total size needed before reading
|
||||
/* Peek at the present flag to determine the total record size before
|
||||
decoding. metadata_from_buf() independently bounds-checks every read
|
||||
against remaining_size, so a short body can never over-read. */
|
||||
int present_flag;
|
||||
memcpy(&present_flag, data_pointer, sizeof(int));
|
||||
if ((present_flag != 0 && present_flag != 1) ||
|
||||
(present_flag == 1 && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata body");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
file->metadata = metadata_from_buf(&data_pointer);
|
||||
remaining_size -= sizeof(int);
|
||||
file->metadata = metadata_from_buf((const uint8_t*)data_pointer, remaining_size);
|
||||
size_t metadata_consumed = sizeof(int);
|
||||
if (present_flag == 1) {
|
||||
if (file->metadata == NULL) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
remaining_size -= FILE_METADATA_WIRE_SIZE;
|
||||
if (file->metadata == NULL)
|
||||
goto error;
|
||||
metadata_consumed += FILE_METADATA_WIRE_SIZE;
|
||||
}
|
||||
data_pointer += metadata_consumed;
|
||||
remaining_size -= metadata_consumed;
|
||||
}
|
||||
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for data size");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
size_t file_data_size;
|
||||
@@ -371,34 +378,34 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (remaining_size < file_data_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for file content");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
// Reject individual file data larger than the maximum allowed size.
|
||||
if (file_data_size > MAX_FILE_DATA_SIZE) {
|
||||
log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size,
|
||||
(unsigned long long)MAX_FILE_DATA_SIZE);
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
size_t allocation_size = file_data_size > 0 ? file_data_size : 1;
|
||||
void* file_data = protocol_alloc(allocation_size);
|
||||
if (file_data == NULL) {
|
||||
log_perror("Could not allocate memory for file data");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
memcpy(file_data, data_pointer, file_data_size);
|
||||
Data* replacement = data_create(file_data, file_data_size);
|
||||
if (replacement == NULL) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
if (replacement == NULL)
|
||||
goto error;
|
||||
/* Charge the retained per-file copy to the connection budget (when the
|
||||
inbound chunk carries an owning session) so the queued copies are not
|
||||
held outside MAX_CONNECTION_MEMORY (B6). A NULL owner (e.g. a local
|
||||
batch apply) leaves the copy uncharged. */
|
||||
if (!data_charge_session(replacement, data->owner, allocation_size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for chunk file data");
|
||||
data_destroy(replacement);
|
||||
goto error;
|
||||
}
|
||||
data_destroy(file->data);
|
||||
file->data = replacement;
|
||||
@@ -408,9 +415,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (file->is_symlink) {
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for symlink target");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
size_t target_len;
|
||||
memcpy(&target_len, data_pointer, sizeof(size_t));
|
||||
@@ -418,24 +423,18 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
remaining_size -= sizeof(size_t);
|
||||
if (target_len == 0 || remaining_size < target_len) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad symlink target");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
char* target = protocol_alloc(target_len + 1);
|
||||
if (!target) {
|
||||
log_perror("Could not allocate memory for symlink target");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
memcpy(target, data_pointer, target_len);
|
||||
target[target_len] = '\0';
|
||||
if (memchr(target, '\0', target_len) != NULL) {
|
||||
free(target);
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
/* The symlink target also rides the wire charset; decode it to the local
|
||||
charset like the path (a target is a path). */
|
||||
@@ -446,9 +445,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv: received chunk symlink target cannot be converted to the local "
|
||||
"charset");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
target = local_target;
|
||||
}
|
||||
@@ -457,29 +454,27 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
remaining_size -= target_len;
|
||||
}
|
||||
|
||||
if (!array_list_add(files, file)) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (!array_list_add(files, file))
|
||||
goto error;
|
||||
file = NULL;
|
||||
}
|
||||
|
||||
File** file_array = (File**)array_list_to_array(files);
|
||||
if (files->size > 0 && file_array == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (files->size > 0 && file_array == NULL)
|
||||
goto error;
|
||||
Chunk* chunk = chunk_create(file_array, files->size);
|
||||
|
||||
free(file_array);
|
||||
if (chunk == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (chunk == NULL)
|
||||
goto error;
|
||||
files->item_destroyer = NULL;
|
||||
array_list_delete(files);
|
||||
|
||||
return chunk;
|
||||
|
||||
error:
|
||||
if (file)
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) {
|
||||
@@ -508,12 +503,21 @@ Chunk* receive_chunk_data(int fd, const Config* config) {
|
||||
}
|
||||
Data* data_to_process = chunk_data;
|
||||
if (config->use_compression) {
|
||||
/* Preserve the inbound session across decompression so the (larger)
|
||||
decompressed chunk is charged to the same connection budget; the
|
||||
compressed buffer's own charge is released by data_destroy below. */
|
||||
ProtocolSession* owner = chunk_data->owner;
|
||||
data_to_process = data_decompress_limited(chunk_data, MAX_CHUNK_SIZE);
|
||||
data_destroy(chunk_data);
|
||||
if (data_to_process == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to decompress chunk");
|
||||
return NULL;
|
||||
}
|
||||
if (!data_charge_session(data_to_process, owner, data_to_process->size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for decompressed chunk");
|
||||
data_destroy(data_to_process);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
// Reject chunks larger than the maximum allowed size to prevent OOM.
|
||||
|
||||
@@ -23,4 +23,15 @@ Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_
|
||||
int compression_threads);
|
||||
Chunk* receive_chunk_data(int fd, const Config* config);
|
||||
|
||||
/* Charge `charge` retained bytes of `data` against `session`'s per-connection
|
||||
* budget (MAX_CONNECTION_MEMORY), mirroring the protocol layer's accounting, and
|
||||
* record them on `data` so data_destroy() returns the charge through the
|
||||
* Data.owner path. Returns false (leaving `data` uncharged) when the ceiling
|
||||
* would be exceeded. A NULL/zero-size charge or a NULL session is a no-op
|
||||
* success. The receive-side decompression and chunk-copy paths know the owning
|
||||
* session only through the Data.owner of the buffer they are processing, so
|
||||
* this is the entry point that lets them participate in the connection budget
|
||||
* without a session handle (B6). */
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
+239
-57
@@ -2,43 +2,161 @@
|
||||
#include "data.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include <stdlib.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <zstd.h>
|
||||
|
||||
#define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024)
|
||||
#define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */
|
||||
|
||||
static char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv",
|
||||
".zip", ".gz", ".xz", ".zst", NULL};
|
||||
/* rsync 3.4.1's built-in skip-compress suffix list (the `--skip-compress`
|
||||
* defaults, in the man page's order). rsync stores it as space-separated
|
||||
* "*.suffix" globs; FastSync matches the plain suffix after the final dot, so
|
||||
* the leading "*." is omitted here. A user --skip-compress list replaces this
|
||||
* default entirely (matching rsync). */
|
||||
#define DEFAULT_SKIP_COMPRESS_SUFFIXES \
|
||||
"3g2 3gp 7z aac ace apk avi bz2 deb dmg ear f4v flac flv gpg gz iso jar jpeg jpg lrz lz lz4 " \
|
||||
"lzma " \
|
||||
"lzo m1a m1v m2a m2ts m2v m4a m4b m4p m4r m4v mka mkv mov mp1 mp2 mp3 mp4 mpa mpeg mpg mpv mts " \
|
||||
"odb odf odg odi odm odp ods odt oga ogg ogm ogv ogx opus otg oth otp ots ott oxt png qt rar " \
|
||||
"rpm " \
|
||||
"rz rzip spx squashfs sxc sxd sxg sxm sxw sz tbz tbz2 tgz tlz ts txz tzo vob war webm webp xz " \
|
||||
"z " \
|
||||
"zip zst"
|
||||
|
||||
bool compression_should_skip(const char* path) {
|
||||
return compression_should_skip_with_suffixes(path, NULL, -1);
|
||||
/* Case-insensitive match of a bare suffix (no leading dot) against a
|
||||
* space-separated suffix list. */
|
||||
static bool suffix_in_list(const char* name, const char* list) {
|
||||
size_t name_len = strlen(name);
|
||||
while (*list) {
|
||||
while (*list == ' ')
|
||||
list++;
|
||||
const char* start = list;
|
||||
while (*list && *list != ' ')
|
||||
list++;
|
||||
size_t len = (size_t)(list - start);
|
||||
if (len == name_len && strncasecmp(name, start, len) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count) {
|
||||
if (!path)
|
||||
return false;
|
||||
const char* dot = strrchr(path, '.');
|
||||
if (!dot)
|
||||
if (!dot || dot[1] == '\0')
|
||||
return false;
|
||||
if (count < 0) {
|
||||
suffixes = SKIP_COMPRESSION_EXTENSIONS;
|
||||
count = 0;
|
||||
while (SKIP_COMPRESSION_EXTENSIONS[count])
|
||||
count++;
|
||||
}
|
||||
const char* name = dot + 1;
|
||||
/* count < 0 (the user gave no --skip-compress) selects rsync's built-in
|
||||
* default list; a non-negative count is the user's explicit list. */
|
||||
if (count < 0)
|
||||
return suffix_in_list(name, DEFAULT_SKIP_COMPRESS_SUFFIXES);
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (strcasecmp(dot, suffixes[i]) == 0)
|
||||
const char* suffix = suffixes[i];
|
||||
if (suffix[0] == '.')
|
||||
suffix++;
|
||||
if (strcasecmp(name, suffix) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Per-thread cache of zstd contexts plus the grow-only compression scratch
|
||||
* buffer. zstd contexts are stateful and not safe to share between threads,
|
||||
* so each thread keeps its own (see compression_get_thread_ctx). The cache is
|
||||
* stored in a C11 thread-specific storage slot whose destructor releases the
|
||||
* contexts when the thread exits; this keeps LeakSanitizer clean for the
|
||||
* short-lived sender/receiver/scanner worker threads without every worker
|
||||
* entry point having to remember to call compression_free_thread_contexts().
|
||||
* The main thread's slot is not torn down by tss at process exit, so an atexit
|
||||
* hook releases it (and compression_free_thread_contexts allows eager
|
||||
* release). */
|
||||
typedef struct {
|
||||
ZSTD_CCtx* cctx;
|
||||
ZSTD_DCtx* dctx;
|
||||
void* out_buf; /* reusable ZSTD_compressBound-sized output scratch */
|
||||
size_t out_cap; /* bytes currently allocated for out_buf */
|
||||
int level; /* compression level currently applied to cctx */
|
||||
int workers; /* nbWorkers currently applied to cctx */
|
||||
bool params_set;
|
||||
bool cached; /* false when the TSS slot could not be used: caller owns */
|
||||
} CompressionThreadCtx;
|
||||
|
||||
static once_flag compression_tls_once = ONCE_FLAG_INIT;
|
||||
static tss_t compression_tls_key;
|
||||
static bool compression_tls_ready;
|
||||
|
||||
static void compression_tls_make_key(void);
|
||||
|
||||
static void compression_ctx_free(CompressionThreadCtx* ctx) {
|
||||
if (!ctx)
|
||||
return;
|
||||
if (ctx->cctx)
|
||||
ZSTD_freeCCtx(ctx->cctx);
|
||||
if (ctx->dctx)
|
||||
ZSTD_freeDCtx(ctx->dctx);
|
||||
free(ctx->out_buf);
|
||||
free(ctx);
|
||||
}
|
||||
|
||||
static void compression_tls_destructor(void* value) {
|
||||
compression_ctx_free((CompressionThreadCtx*)value);
|
||||
}
|
||||
|
||||
void compression_free_thread_contexts(void) {
|
||||
call_once(&compression_tls_once, compression_tls_make_key);
|
||||
if (!compression_tls_ready)
|
||||
return;
|
||||
CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key);
|
||||
if (!ctx)
|
||||
return;
|
||||
/* Clear the slot first so the thread-exit destructor cannot free it twice. */
|
||||
tss_set(compression_tls_key, NULL);
|
||||
compression_ctx_free(ctx);
|
||||
}
|
||||
|
||||
static void compression_atexit_cleanup(void) {
|
||||
compression_free_thread_contexts();
|
||||
}
|
||||
|
||||
static void compression_tls_make_key(void) {
|
||||
if (tss_create(&compression_tls_key, compression_tls_destructor) == thrd_success) {
|
||||
compression_tls_ready = true;
|
||||
atexit(compression_atexit_cleanup);
|
||||
}
|
||||
}
|
||||
|
||||
static CompressionThreadCtx* compression_get_thread_ctx(void) {
|
||||
call_once(&compression_tls_once, compression_tls_make_key);
|
||||
if (!compression_tls_ready) {
|
||||
/* Extremely unlikely: fall back to an uncached context the caller frees. */
|
||||
return (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx));
|
||||
}
|
||||
CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key);
|
||||
if (ctx)
|
||||
return ctx;
|
||||
ctx = (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx));
|
||||
if (!ctx)
|
||||
return NULL;
|
||||
ctx->cached = true;
|
||||
if (tss_set(compression_tls_key, ctx) != thrd_success)
|
||||
ctx->cached = false;
|
||||
return ctx;
|
||||
}
|
||||
|
||||
/* Release an uncached context immediately; cached contexts are owned by the
|
||||
* thread's TSS slot and freed on thread exit / compression_free_thread_contexts. */
|
||||
static void compression_ctx_put(CompressionThreadCtx* ctx) {
|
||||
if (ctx && !ctx->cached)
|
||||
compression_ctx_free(ctx);
|
||||
}
|
||||
|
||||
Data* data_compress(Data* data_to_compress, int compression_level) {
|
||||
return data_compress_with_threads(data_to_compress, compression_level, 0);
|
||||
}
|
||||
@@ -50,68 +168,102 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
return NULL;
|
||||
log_message(LOG_LEVEL_DEBUG, "Starting to compress data");
|
||||
size_t dst_size = ZSTD_compressBound(data_to_compress->size);
|
||||
Data* compressed_data = data_create_empty(dst_size);
|
||||
if (compressed_data == NULL)
|
||||
return NULL;
|
||||
|
||||
ZSTD_CCtx* cctx = ZSTD_createCCtx();
|
||||
if (!cctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context");
|
||||
data_destroy(compressed_data);
|
||||
CompressionThreadCtx* ctx = compression_get_thread_ctx();
|
||||
if (ctx == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD compression context");
|
||||
return NULL;
|
||||
}
|
||||
Data* compressed_data = NULL;
|
||||
|
||||
size_t zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_compressionLevel, compression_level);
|
||||
if (!ctx->cctx) {
|
||||
ctx->cctx = ZSTD_createCCtx();
|
||||
if (!ctx->cctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context");
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->params_set = false;
|
||||
}
|
||||
|
||||
/* Reset only the session: parameters (and any already-allocated zstd worker
|
||||
* pool) stay attached to the context, so compressing the next file does not
|
||||
* rebuild the pool. */
|
||||
ZSTD_CCtx_reset(ctx->cctx, ZSTD_reset_session_only);
|
||||
|
||||
if (!ctx->params_set || ctx->level != compression_level) {
|
||||
size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_compressionLevel, compression_level);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression level: %s", ZSTD_getErrorName(zret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->level = compression_level;
|
||||
}
|
||||
|
||||
int available_threads = 0;
|
||||
if (compression_threads > 0) {
|
||||
long online_cpus = sysconf(_SC_NPROCESSORS_ONLN);
|
||||
int available_threads = online_cpus > 0 && online_cpus < compression_threads
|
||||
? (int)online_cpus
|
||||
available_threads = online_cpus > 0 && online_cpus < compression_threads ? (int)online_cpus
|
||||
: compression_threads;
|
||||
zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_nbWorkers, available_threads);
|
||||
}
|
||||
if (!ctx->params_set || ctx->workers != available_threads) {
|
||||
size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_nbWorkers, available_threads);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression threads: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->workers = available_threads;
|
||||
}
|
||||
ctx->params_set = true;
|
||||
|
||||
if (available_threads > 0) {
|
||||
/* Streaming compression needs the source size before threaded mode can end a frame. */
|
||||
zret = ZSTD_CCtx_setPledgedSrcSize(cctx, data_to_compress->size);
|
||||
size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, data_to_compress->size);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
}
|
||||
|
||||
if (ctx->out_cap < dst_size) {
|
||||
void* grown = protocol_realloc(ctx->out_buf, dst_size);
|
||||
if (grown == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate compression buffer");
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->out_buf = grown;
|
||||
ctx->out_cap = dst_size;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0};
|
||||
ZSTD_outBuffer output = {compressed_data->data, dst_size, 0};
|
||||
ZSTD_outBuffer output = {ctx->out_buf, dst_size, 0};
|
||||
|
||||
size_t ret;
|
||||
do {
|
||||
ret = ZSTD_compressStream2(cctx, &output, &input, ZSTD_e_end);
|
||||
ret = ZSTD_compressStream2(ctx->cctx, &output, &input, ZSTD_e_end);
|
||||
if (ZSTD_isError(ret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Compression failed: %s", ZSTD_getErrorName(ret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
} while (ret > 0);
|
||||
|
||||
/* Hand off an exactly-sized copy; the scratch buffer stays cached so the next
|
||||
* call does not reallocate a ZSTD_compressBound-sized block. */
|
||||
compressed_data = data_create_empty(output.pos);
|
||||
if (compressed_data == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate compressed data");
|
||||
goto cleanup;
|
||||
}
|
||||
if (output.pos > 0)
|
||||
memcpy(compressed_data->data, ctx->out_buf, output.pos);
|
||||
compressed_data->size = output.pos;
|
||||
ZSTD_freeCCtx(cctx);
|
||||
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu",
|
||||
data_to_compress->size, compressed_data->size);
|
||||
|
||||
cleanup:
|
||||
compression_ctx_put(ctx);
|
||||
return compressed_data;
|
||||
}
|
||||
|
||||
@@ -122,9 +274,13 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Start to decompress data");
|
||||
unsigned long long dst_size =
|
||||
ZSTD_getFrameContentSize(compressed_data->data, compressed_data->size);
|
||||
if (ZSTD_isError(dst_size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: %s",
|
||||
ZSTD_getErrorName(dst_size));
|
||||
/* ZSTD_isError() is also true for ZSTD_CONTENTSIZE_ERROR and
|
||||
* ZSTD_CONTENTSIZE_UNKNOWN (both are encoded near (size_t)-1), so test the
|
||||
* sentinels explicitly instead of blanket-rejecting every error-ish value:
|
||||
* only CONTENTSIZE_ERROR means an unreadable header, while CONTENTSIZE_UNKNOWN
|
||||
* must reach the estimate fallback below. */
|
||||
if (dst_size == ZSTD_CONTENTSIZE_ERROR) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: invalid zstd frame");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
@@ -144,20 +300,30 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
ZSTD_DCtx* dctx = ZSTD_createDCtx();
|
||||
if (!dctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context");
|
||||
CompressionThreadCtx* ctx = compression_get_thread_ctx();
|
||||
if (ctx == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD decompression context");
|
||||
return NULL;
|
||||
}
|
||||
Data* uncompressed_data = NULL;
|
||||
|
||||
if (!ctx->dctx) {
|
||||
ctx->dctx = ZSTD_createDCtx();
|
||||
if (!ctx->dctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context");
|
||||
goto cleanup;
|
||||
}
|
||||
}
|
||||
/* Reset only the session; decompression parameters are sticky. */
|
||||
ZSTD_DCtx_reset(ctx->dctx, ZSTD_reset_session_only);
|
||||
|
||||
size_t buf_size = (dst_size > 0) ? (size_t)dst_size : INITIAL_DECOMPRESS_BUF_SIZE;
|
||||
if (buf_size > maximum_size)
|
||||
buf_size = maximum_size;
|
||||
Data* uncompressed_data = data_create_empty(buf_size);
|
||||
uncompressed_data = data_create_empty(buf_size);
|
||||
if (!uncompressed_data) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate decompression buffer");
|
||||
ZSTD_freeDCtx(dctx);
|
||||
return NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {compressed_data->data, compressed_data->size, 0};
|
||||
@@ -165,20 +331,20 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
|
||||
size_t ret;
|
||||
do {
|
||||
ret = ZSTD_decompressStream(dctx, &output, &input);
|
||||
ret = ZSTD_decompressStream(ctx->dctx, &output, &input);
|
||||
if (ZSTD_isError(ret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Decompression failed: %s", ZSTD_getErrorName(ret));
|
||||
ZSTD_freeDCtx(dctx);
|
||||
data_destroy(uncompressed_data);
|
||||
return NULL;
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
if (ret > 0 && output.pos == output.size) {
|
||||
if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) {
|
||||
log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes",
|
||||
(unsigned long long)MAX_DECOMPRESSED_SIZE);
|
||||
ZSTD_freeDCtx(dctx);
|
||||
data_destroy(uncompressed_data);
|
||||
return NULL;
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
buf_size *= 2;
|
||||
if (buf_size > hard_limit)
|
||||
@@ -186,20 +352,36 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
void* new_data = protocol_realloc(uncompressed_data->data, buf_size);
|
||||
if (!new_data) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to grow decompression buffer");
|
||||
ZSTD_freeDCtx(dctx);
|
||||
data_destroy(uncompressed_data);
|
||||
return NULL;
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
uncompressed_data->data = new_data;
|
||||
output.dst = new_data;
|
||||
output.size = buf_size;
|
||||
/* Re-attempt with the larger output buffer; the truncated-frame check
|
||||
* below must not reject a complete frame that merely filled the previous
|
||||
* buffer exactly. */
|
||||
continue;
|
||||
}
|
||||
/* A positive hint with all input consumed means the frame is incomplete: a
|
||||
* truncated stream would otherwise spin here forever (ZSTD_decompressStream
|
||||
* keeps returning the same hint). Fail instead of burning CPU. */
|
||||
if (ret != 0 && input.pos == input.size) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"Truncated zstd frame: input exhausted with %zu bytes still expected", ret);
|
||||
data_destroy(uncompressed_data);
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
} while (ret > 0);
|
||||
|
||||
uncompressed_data->size = output.pos;
|
||||
ZSTD_freeDCtx(dctx);
|
||||
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Decompressed data successfully");
|
||||
|
||||
cleanup:
|
||||
compression_ctx_put(ctx);
|
||||
return uncompressed_data;
|
||||
}
|
||||
|
||||
|
||||
@@ -11,7 +11,14 @@ Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress(Data* compressed_data);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
bool compression_should_skip(const char* path);
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count);
|
||||
|
||||
/* Release the calling thread's cached zstd contexts (compressor, decompressor
|
||||
* and scratch buffer). The cache is thread-local and is also released
|
||||
* automatically when a worker thread exits (via a C11 tss destructor) and for
|
||||
* the main thread at process exit; this explicit entry point exists so tests
|
||||
* and long-lived callers can drop the cache deterministically. Safe to call
|
||||
* when no context has been created, and idempotent. */
|
||||
void compression_free_thread_contexts(void);
|
||||
|
||||
#endif
|
||||
+634
-662
File diff suppressed because it is too large.
Load diff
+614
-296
File diff suppressed because it is too large.
Load diff
@@ -158,13 +158,13 @@ static int hex_value(char c) {
|
||||
* digits. Such a line is refused loudly (and never accepted) so an operator
|
||||
* cannot keep a replayable bearer digest in place after the protocol bump. */
|
||||
static bool secret_is_legacy_hex(const char* s) {
|
||||
if (!s)
|
||||
if (!s || strlen(s) != 64)
|
||||
return false;
|
||||
for (int i = 0; i < 64; i++) {
|
||||
if (hex_value(s[i]) < 0)
|
||||
return false;
|
||||
}
|
||||
return s[64] == '\0';
|
||||
return true;
|
||||
}
|
||||
|
||||
bool credentials_b64_encode(const uint8_t* in, size_t n, char* out, size_t out_sz) {
|
||||
|
||||
+344
-4
@@ -1,8 +1,13 @@
|
||||
#include "daemon_conf.h"
|
||||
#include "credentials.h"
|
||||
#include "utils.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <ctype.h>
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <netinet/in.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -48,6 +53,191 @@ static bool parse_bool_value(const char* value, bool* out) {
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Parse an IPv4/IPv6 CIDR "addr/prefix" into `bytes`/`*family`. Returns false
|
||||
* for a malformed address, a missing/oversized prefix, or a prefix that does
|
||||
* not fit the address family. */
|
||||
static bool parse_cidr(const char* cidr, int* prefix_out, uint8_t* bytes, int* family_out) {
|
||||
const char* slash = strchr(cidr, '/');
|
||||
if (!slash)
|
||||
return false;
|
||||
size_t addr_len = (size_t)(slash - cidr);
|
||||
if (addr_len == 0 || addr_len >= INET6_ADDRSTRLEN)
|
||||
return false;
|
||||
char addr[INET6_ADDRSTRLEN];
|
||||
memcpy(addr, cidr, addr_len);
|
||||
addr[addr_len] = '\0';
|
||||
char* end = NULL;
|
||||
long prefix = strtol(slash + 1, &end, 10);
|
||||
if (end == slash + 1 || *end != '\0')
|
||||
return false;
|
||||
struct in_addr v4;
|
||||
struct in6_addr v6;
|
||||
if (inet_pton(AF_INET, addr, &v4) == 1) {
|
||||
if (prefix < 0 || prefix > 32)
|
||||
return false;
|
||||
memcpy(bytes, &v4, sizeof(v4));
|
||||
*prefix_out = (int)prefix;
|
||||
*family_out = AF_INET;
|
||||
return true;
|
||||
}
|
||||
if (inet_pton(AF_INET6, addr, &v6) == 1) {
|
||||
if (prefix < 0 || prefix > 128)
|
||||
return false;
|
||||
memcpy(bytes, &v6, sizeof(v6));
|
||||
*prefix_out = (int)prefix;
|
||||
*family_out = AF_INET6;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* A host pattern is valid when it is `*`, a valid IPv4/IPv6 literal, or a valid
|
||||
* CIDR. Peer addresses reaching the matcher are always numeric, so hostname
|
||||
* globs are rejected at parse time: accepting one would create a deny rule that
|
||||
* silently never matches (fail-open). */
|
||||
static bool host_pattern_valid(const char* pattern) {
|
||||
if (!pattern || *pattern == '\0')
|
||||
return false;
|
||||
if (strcmp(pattern, "*") == 0)
|
||||
return true;
|
||||
if (strchr(pattern, '/')) {
|
||||
uint8_t bytes[16];
|
||||
int prefix;
|
||||
int family;
|
||||
return parse_cidr(pattern, &prefix, bytes, &family);
|
||||
}
|
||||
struct in_addr v4;
|
||||
struct in6_addr v6;
|
||||
return inet_pton(AF_INET, pattern, &v4) == 1 || inet_pton(AF_INET6, pattern, &v6) == 1;
|
||||
}
|
||||
|
||||
/* Append every comma- and/or whitespace-separated host pattern in `value` to
|
||||
* the heap-owned list (or replace the list when `replace` is set, which --dparam
|
||||
* uses so an override can narrow access rather than only widen it). Returns
|
||||
* false (err filled) on an invalid pattern or an allocation failure. */
|
||||
static bool store_host_list(char*** list, int* count, const char* value, const char* key,
|
||||
const char* module_name, bool replace, char* err, size_t err_size) {
|
||||
if (replace) {
|
||||
for (int i = 0; i < *count; i++)
|
||||
free((*list)[i]);
|
||||
free(*list);
|
||||
*list = NULL;
|
||||
*count = 0;
|
||||
}
|
||||
char* copy = str_dup(value);
|
||||
if (!copy) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
|
||||
else
|
||||
set_error(err, err_size, "out of memory parsing '%s'", key);
|
||||
return false;
|
||||
}
|
||||
char* save = NULL;
|
||||
int added = 0;
|
||||
for (char* token = strtok_r(copy, ", \t", &save); token; token = strtok_r(NULL, ", \t", &save)) {
|
||||
if (!host_pattern_valid(token)) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': invalid host pattern '%s' in '%s'", module_name,
|
||||
token, key);
|
||||
else
|
||||
set_error(err, err_size, "invalid host pattern '%s' in '%s'", token, key);
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
char** grown = realloc(*list, (size_t)(*count + 1) * sizeof(char*));
|
||||
if (!grown) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
|
||||
else
|
||||
set_error(err, err_size, "out of memory parsing '%s'", key);
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
*list = grown;
|
||||
char* dup = str_dup(token);
|
||||
if (!dup) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
|
||||
else
|
||||
set_error(err, err_size, "out of memory parsing '%s'", key);
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
(*list)[(*count)++] = dup;
|
||||
added++;
|
||||
}
|
||||
free(copy);
|
||||
/* A present key with an empty (or separator-only) value would otherwise
|
||||
* install a zero-length list, i.e. no ACL at all: a strict-parse config must
|
||||
* never silently turn a restrictive directive into "allow everyone". */
|
||||
if (added == 0) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': '%s' must list at least one host pattern", module_name,
|
||||
key);
|
||||
else
|
||||
set_error(err, err_size, "'%s' must list at least one host pattern", key);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse a `max connections` value: a positive integer (0/negative/garbage are
|
||||
* rejected because they would silently disable the cap or admit nothing). */
|
||||
static bool store_max_connections(int* slot, const char* value, const char* module_name, char* err,
|
||||
size_t err_size) {
|
||||
char* end = NULL;
|
||||
errno = 0;
|
||||
long n = strtol(value, &end, 10);
|
||||
if (*value == '\0' || errno != 0 || *end != '\0' || n <= 0 || n > INT_MAX) {
|
||||
if (module_name)
|
||||
set_error(err, err_size,
|
||||
"module '%s': invalid 'max connections' '%s' (must be a positive "
|
||||
"integer)",
|
||||
module_name, value);
|
||||
else
|
||||
set_error(err, err_size, "invalid 'max connections' '%s' (must be a positive integer)",
|
||||
value);
|
||||
return false;
|
||||
}
|
||||
*slot = (int)n;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse a non-negative concurrency cap where 0 means unlimited/disabled
|
||||
* (per-module `max connections`, `max connections per host`,
|
||||
* `auth lockout threshold`). Negative/garbage/oversized values are rejected. */
|
||||
static bool store_optional_cap(int* slot, const char* value, int max_value, const char* key,
|
||||
const char* module_name, char* err, size_t err_size) {
|
||||
char* end = NULL;
|
||||
errno = 0;
|
||||
long n = strtol(value, &end, 10);
|
||||
if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 || n > max_value) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': invalid '%s' '%s' (must be 0-%d)", module_name, key,
|
||||
value, max_value);
|
||||
else
|
||||
set_error(err, err_size, "invalid '%s' '%s' (must be 0-%d)", key, value, max_value);
|
||||
return false;
|
||||
}
|
||||
*slot = (int)n;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse an `auth failure delay` value: 0 (disabled) through the configured cap. */
|
||||
static bool store_auth_failure_delay(int* slot, const char* value, char* err, size_t err_size) {
|
||||
char* end = NULL;
|
||||
errno = 0;
|
||||
long n = strtol(value, &end, 10);
|
||||
if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 ||
|
||||
n > DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS) {
|
||||
set_error(err, err_size, "invalid 'auth failure delay' '%s' (must be 0-%d milliseconds)", value,
|
||||
DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS);
|
||||
return false;
|
||||
}
|
||||
*slot = (int)n;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool daemon_module_name_valid(const char* name) {
|
||||
if (!name || *name == '\0')
|
||||
return false;
|
||||
@@ -68,14 +258,28 @@ DaemonConf* daemon_conf_create(void) {
|
||||
if (!conf)
|
||||
return NULL;
|
||||
conf->global.port = DAEMON_CONF_DEFAULT_PORT;
|
||||
conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS;
|
||||
conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS;
|
||||
conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST;
|
||||
conf->global.auth_lockout_threshold = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD;
|
||||
conf->global.auth_lockout_duration_sec = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC;
|
||||
return conf;
|
||||
}
|
||||
|
||||
/* Free a heap-owned pattern list of `count` entries. */
|
||||
static void free_string_list(char** list, int count) {
|
||||
for (int i = 0; i < count; i++)
|
||||
free(list[i]);
|
||||
free(list);
|
||||
}
|
||||
|
||||
void daemon_conf_free(DaemonConf* conf) {
|
||||
if (!conf)
|
||||
return;
|
||||
free(conf->global.motd_file);
|
||||
free(conf->global.address);
|
||||
free_string_list(conf->global.hosts_allow, conf->global.hosts_allow_count);
|
||||
free_string_list(conf->global.hosts_deny, conf->global.hosts_deny_count);
|
||||
for (int i = 0; i < conf->module_count; i++) {
|
||||
DaemonModule* m = &conf->modules[i];
|
||||
free(m->name);
|
||||
@@ -83,6 +287,8 @@ void daemon_conf_free(DaemonConf* conf) {
|
||||
for (int j = 0; j < m->auth_user_count; j++)
|
||||
free(m->auth_users[j]);
|
||||
free(m->auth_users);
|
||||
free_string_list(m->hosts_allow, m->hosts_allow_count);
|
||||
free_string_list(m->hosts_deny, m->hosts_deny_count);
|
||||
}
|
||||
free(conf->modules);
|
||||
free(conf);
|
||||
@@ -122,8 +328,8 @@ static bool store_port(int* slot, const char* value, char* err, size_t err_size)
|
||||
|
||||
/* Apply a global scalar key/value. Keys are case-insensitive. Returns false
|
||||
* (err filled) on an unknown key or an invalid value. */
|
||||
static bool apply_global_key(DaemonConf* conf, char* key, const char* value, char* err,
|
||||
size_t err_size) {
|
||||
static bool apply_global_key(DaemonConf* conf, char* key, const char* value, bool replace_hosts,
|
||||
char* err, size_t err_size) {
|
||||
if (key_equals(key, "port"))
|
||||
return store_port(&conf->global.port, value, err, err_size);
|
||||
if (key_equals(key, "motd file")) {
|
||||
@@ -140,6 +346,28 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, cha
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (key_equals(key, "max connections"))
|
||||
return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size);
|
||||
if (key_equals(key, "max connections per host"))
|
||||
return store_optional_cap(&conf->global.max_connections_per_host, value,
|
||||
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "max connections per host", NULL,
|
||||
err, err_size);
|
||||
if (key_equals(key, "auth failure delay"))
|
||||
return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size);
|
||||
if (key_equals(key, "auth lockout threshold"))
|
||||
return store_optional_cap(&conf->global.auth_lockout_threshold, value,
|
||||
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "auth lockout threshold", NULL,
|
||||
err, err_size);
|
||||
if (key_equals(key, "auth lockout duration"))
|
||||
return store_optional_cap(&conf->global.auth_lockout_duration_sec, value,
|
||||
DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC, "auth lockout duration",
|
||||
NULL, err, err_size);
|
||||
if (key_equals(key, "hosts allow"))
|
||||
return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value,
|
||||
"hosts allow", NULL, replace_hosts, err, err_size);
|
||||
if (key_equals(key, "hosts deny"))
|
||||
return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value,
|
||||
"hosts deny", NULL, replace_hosts, err, err_size);
|
||||
set_error(err, err_size, "unknown global key '%s'", key);
|
||||
return false;
|
||||
}
|
||||
@@ -188,10 +416,17 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char*
|
||||
return false;
|
||||
}
|
||||
char* save = NULL;
|
||||
int added = 0;
|
||||
for (char* token = strtok_r(list, ",", &save); token; token = strtok_r(NULL, ",", &save)) {
|
||||
const char* user = trim_ws(token);
|
||||
if (*user == '\0')
|
||||
continue;
|
||||
if (!credentials_username_valid(user)) {
|
||||
set_error(err, err_size, "module '%s': invalid 'auth users' entry '%s'", module->name,
|
||||
user);
|
||||
free(list);
|
||||
return false;
|
||||
}
|
||||
char** grown =
|
||||
realloc(module->auth_users, (size_t)(module->auth_user_count + 1) * sizeof(char*));
|
||||
if (!grown) {
|
||||
@@ -209,10 +444,27 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char*
|
||||
return false;
|
||||
}
|
||||
module->auth_users[module->auth_user_count++] = dup;
|
||||
added++;
|
||||
}
|
||||
free(list);
|
||||
/* An empty/separator-only value must not silently disable authentication:
|
||||
* the key's presence is an explicit request for an allow-list. */
|
||||
if (added == 0) {
|
||||
set_error(err, err_size, "module '%s': 'auth users' must list at least one user",
|
||||
module->name);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (key_equals(key, "max connections"))
|
||||
return store_optional_cap(&module->max_connections, value, DAEMON_CONF_MAX_CONCURRENCY_LIMIT,
|
||||
"max connections", module->name, err, err_size);
|
||||
if (key_equals(key, "hosts allow"))
|
||||
return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow",
|
||||
module->name, false, err, err_size);
|
||||
if (key_equals(key, "hosts deny"))
|
||||
return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny",
|
||||
module->name, false, err, err_size);
|
||||
set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name);
|
||||
return false;
|
||||
}
|
||||
@@ -250,6 +502,11 @@ static int open_module(DaemonConf* conf, int* current_module, const char* name,
|
||||
set_error(err, err_size, "duplicate module '%s'", name);
|
||||
return -1;
|
||||
}
|
||||
if (conf->module_count >= DAEMON_CONF_MAX_MODULES) {
|
||||
set_error(err, err_size, "too many modules (limit %d); module '%s' rejected",
|
||||
DAEMON_CONF_MAX_MODULES, name);
|
||||
return -1;
|
||||
}
|
||||
DaemonModule* grown =
|
||||
realloc(conf->modules, (size_t)(conf->module_count + 1) * sizeof(DaemonModule));
|
||||
if (!grown) {
|
||||
@@ -393,7 +650,7 @@ DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size) {
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
if (!apply_global_key(conf, key, value, err, err_size)) {
|
||||
if (!apply_global_key(conf, key, value, false, err, err_size)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
@@ -448,7 +705,90 @@ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err
|
||||
set_error(err, err_size, "--dparam '%s' has an empty value", assignment);
|
||||
return -1;
|
||||
}
|
||||
bool ok = apply_global_key(conf, key, value, err, err_size);
|
||||
bool ok = apply_global_key(conf, key, value, true, err, err_size);
|
||||
free(copy);
|
||||
return ok ? 0 : -1;
|
||||
}
|
||||
|
||||
/* Compare the first `prefix` bits of two 16-byte address buffers. */
|
||||
static bool bit_prefix_match(const uint8_t* a, const uint8_t* b, int prefix) {
|
||||
int whole = prefix / 8;
|
||||
if (whole > 0 && memcmp(a, b, (size_t)whole) != 0)
|
||||
return false;
|
||||
int remainder = prefix % 8;
|
||||
if (remainder == 0)
|
||||
return true;
|
||||
uint8_t mask = (uint8_t)(0xffu << (8 - remainder));
|
||||
return (a[whole] & mask) == (b[whole] & mask);
|
||||
}
|
||||
|
||||
/* Case-insensitive glob match used for hostname patterns. Falls back to the
|
||||
* shared case-sensitive matcher when an operand is too long for the stack
|
||||
* buffers. */
|
||||
static bool host_glob_match(const char* pattern, const char* str) {
|
||||
char pbuf[256];
|
||||
char sbuf[256];
|
||||
size_t plen = strlen(pattern);
|
||||
size_t slen = strlen(str);
|
||||
if (plen >= sizeof(pbuf) || slen >= sizeof(sbuf))
|
||||
return glob_match(pattern, str);
|
||||
for (size_t i = 0; i <= plen; i++)
|
||||
pbuf[i] = (char)tolower((unsigned char)pattern[i]);
|
||||
for (size_t i = 0; i <= slen; i++)
|
||||
sbuf[i] = (char)tolower((unsigned char)str[i]);
|
||||
return glob_match(pbuf, sbuf);
|
||||
}
|
||||
|
||||
bool daemon_host_pattern_match(const char* pattern, const char* peer_ip) {
|
||||
if (!pattern || *pattern == '\0' || !peer_ip || *peer_ip == '\0')
|
||||
return false;
|
||||
if (strcmp(pattern, "*") == 0)
|
||||
return true;
|
||||
if (strchr(pattern, '/')) {
|
||||
uint8_t pattern_bytes[16];
|
||||
uint8_t peer_bytes[16];
|
||||
int prefix = 0;
|
||||
int family = AF_UNSPEC;
|
||||
if (!parse_cidr(pattern, &prefix, pattern_bytes, &family))
|
||||
return false;
|
||||
if (inet_pton(family, peer_ip, peer_bytes) != 1)
|
||||
return false;
|
||||
return bit_prefix_match(pattern_bytes, peer_bytes, prefix);
|
||||
}
|
||||
struct in_addr pattern_v4;
|
||||
struct in_addr peer_v4;
|
||||
if (inet_pton(AF_INET, pattern, &pattern_v4) == 1)
|
||||
return inet_pton(AF_INET, peer_ip, &peer_v4) == 1 && pattern_v4.s_addr == peer_v4.s_addr;
|
||||
struct in6_addr pattern_v6;
|
||||
struct in6_addr peer_v6;
|
||||
if (inet_pton(AF_INET6, pattern, &pattern_v6) == 1)
|
||||
return inet_pton(AF_INET6, peer_ip, &peer_v6) == 1 &&
|
||||
memcmp(&pattern_v6, &peer_v6, sizeof(pattern_v6)) == 0;
|
||||
/* Not a literal: a hostname/glob pattern. */
|
||||
return host_glob_match(pattern, peer_ip);
|
||||
}
|
||||
|
||||
bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count,
|
||||
char* const* deny, int deny_count) {
|
||||
if (!peer_ip)
|
||||
return false;
|
||||
for (int i = 0; i < deny_count; i++) {
|
||||
if (daemon_host_pattern_match(deny[i], peer_ip))
|
||||
return false;
|
||||
}
|
||||
if (allow_count > 0) {
|
||||
for (int i = 0; i < allow_count; i++) {
|
||||
if (daemon_host_pattern_match(allow[i], peer_ip))
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny,
|
||||
int deny_count) {
|
||||
(void)allow;
|
||||
(void)deny;
|
||||
return allow_count > 0 || deny_count > 0;
|
||||
}
|
||||
@@ -52,6 +52,15 @@ typedef struct DaemonModule {
|
||||
activities. Without it the daemon refuses all of them. */
|
||||
char** auth_users; /* `auth users = a,b`; Wave B credential list */
|
||||
int auth_user_count;
|
||||
/* `max connections = N` (optional per-module cap). 0 means unlimited. The
|
||||
* per-connection child records the selected module in the shared registry
|
||||
* (daemon_limits.c) once the config frame names it, so the cap is enforced
|
||||
* across all forked children; the parent reclaims the slot on SIGCHLD. */
|
||||
int max_connections;
|
||||
char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
char** hosts_deny; /* `hosts deny = a,b`; host access deny patterns */
|
||||
int hosts_deny_count;
|
||||
} DaemonModule;
|
||||
|
||||
/* Global (pre-module) scalar keys. `motd file` is parsed and stored but has
|
||||
@@ -60,6 +69,25 @@ typedef struct DaemonConfGlobals {
|
||||
int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */
|
||||
char* motd_file; /* `motd file`, may be NULL */
|
||||
char* address; /* `address` (optional bind address), may be NULL */
|
||||
int max_connections; /* `max connections`, default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */
|
||||
int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */
|
||||
int max_connections_per_host; /* `max connections per host`, concurrent cap per
|
||||
source IP; default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST (0 =
|
||||
unlimited) */
|
||||
int auth_lockout_threshold; /* `auth lockout threshold`, failed attempts from
|
||||
one source before lockout; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD (0
|
||||
disables) */
|
||||
int auth_lockout_duration_sec; /* `auth lockout duration`, seconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC
|
||||
(0 disables) */
|
||||
char** hosts_allow; /* `hosts allow`; global host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
char** hosts_deny; /* `hosts deny`; global host access deny patterns */
|
||||
int hosts_deny_count;
|
||||
} DaemonConfGlobals;
|
||||
|
||||
typedef struct DaemonConf {
|
||||
@@ -69,6 +97,31 @@ typedef struct DaemonConf {
|
||||
} DaemonConf;
|
||||
|
||||
#define DAEMON_CONF_DEFAULT_PORT 873
|
||||
/* Default global connection cap when `max connections` is absent. Matches the
|
||||
* historical hardcoded listener value. */
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100
|
||||
/* Default `auth failure delay` in milliseconds (0 disables the throttle). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500
|
||||
/* Default `max connections per host` (0 = unlimited). */
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST 0
|
||||
/* Default cross-process auth lockout: 10 failed attempts from one source lock
|
||||
* it out for 300 s (0 disables either knob). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD 10
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC 300
|
||||
/* Upper bound on a `max connections per host` or `auth lockout threshold`
|
||||
* value, so a typo cannot size the shared registry absurdly. */
|
||||
#define DAEMON_CONF_MAX_CONCURRENCY_LIMIT 1000000
|
||||
/* Upper bound on `auth lockout duration` (7 days). */
|
||||
#define DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC 604800
|
||||
/* Largest accepted `auth failure delay`, so a typo cannot pin a connection
|
||||
* child in nanosleep for an absurd time. */
|
||||
/* Bounded well below the socket I/O timeout so a failed-auth child cannot hold
|
||||
* a connection slot for long enough to amplify connection-cap exhaustion. */
|
||||
#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000
|
||||
/* Upper bound on the number of [module] sections, so the shared registry's
|
||||
* per-module counter array stays fixed-size. The parser rejects the next
|
||||
* section past this bound. */
|
||||
#define DAEMON_CONF_MAX_MODULES 256
|
||||
/* Longest accepted config line (excluding the trailing newline). Longer lines
|
||||
* are rejected rather than buffered unboundedly. */
|
||||
#define DAEMON_CONF_MAX_LINE 4096
|
||||
@@ -99,9 +152,31 @@ const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char*
|
||||
bool daemon_module_name_valid(const char* name);
|
||||
|
||||
/* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and
|
||||
* apply it to the global scalars only. Keys are case-insensitive and limited
|
||||
* to the global scalar keys defined by the grammar (port, motd file, address).
|
||||
* apply it to the global keys only. Keys are case-insensitive and limited to
|
||||
* the global keys defined by the grammar (port, motd file, address,
|
||||
* max connections, max connections per host, auth failure delay,
|
||||
* auth lockout threshold, auth lockout duration, hosts allow, hosts deny).
|
||||
* Returns 0 on success, -1 on error (err filled). */
|
||||
int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size);
|
||||
|
||||
/* Host access-control matching (pure; no I/O). `daemon_host_pattern_match`
|
||||
* matches one configured pattern against a numeric peer IP string. Supported
|
||||
* patterns: `*` (match anything), an IPv4/IPv6 literal, an IPv4/IPv6 CIDR
|
||||
* (`10.0.0.0/8`, `2001:db8::/32`), or a glob (`*.example.com`) evaluated with
|
||||
* the same matcher as file globs; a glob only matches a peer string of the
|
||||
* same shape, so a numeric peer never matches a hostname glob. */
|
||||
bool daemon_host_pattern_match(const char* pattern, const char* peer_ip);
|
||||
|
||||
/* rsync-like combined decision over a deny list and an allow list: a matching
|
||||
* deny rejects (deny takes precedence); otherwise, when any allow entries
|
||||
* exist, a peer that matches none is rejected; with no allow entries every
|
||||
* peer not denied is accepted. An empty/unset pair returns true. */
|
||||
bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count,
|
||||
char* const* deny, int deny_count);
|
||||
|
||||
/* True when at least one allow or deny pattern is configured (i.e. an
|
||||
* unprovable peer must fail closed rather than being treated as unrestricted). */
|
||||
bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny,
|
||||
int deny_count);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,494 @@
|
||||
#include "daemon_limits.h"
|
||||
#include "daemon_conf.h"
|
||||
#include "log.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <netinet/in.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <time.h>
|
||||
|
||||
/* The two module-count bounds must agree: the daemon config parser never
|
||||
* produces more than DAEMON_CONF_MAX_MODULES modules, so the shared registry's
|
||||
* per-module counter array is sized from the same bound. */
|
||||
_Static_assert(DAEMON_LIMITS_MAX_MODULES == DAEMON_CONF_MAX_MODULES,
|
||||
"daemon_limits module bound must match daemon_conf");
|
||||
|
||||
/* Slot lifecycle states (stored in slot_state). */
|
||||
enum {
|
||||
SLOT_FREE = 0,
|
||||
SLOT_CLAIMED = 1,
|
||||
SLOT_REGISTERED = 2,
|
||||
};
|
||||
|
||||
/* The registry header lives at the base of the shared mapping; the pointer
|
||||
* fields point at the arrays carved out of the same mapping. Absolute pointers
|
||||
* remain valid in a forked child because fork() clones the address space and
|
||||
* mapping, so parent and child observe the same virtual addresses. */
|
||||
struct DaemonLimitRegistry {
|
||||
int max_slots;
|
||||
int module_count;
|
||||
int host_slots; /* power of two; 1 when no per-source tracking is needed */
|
||||
int per_host_cap;
|
||||
int lockout_threshold;
|
||||
int lockout_duration_sec;
|
||||
size_t map_size;
|
||||
_Atomic long long host_full_warn; /* last "table full" warning epoch */
|
||||
_Atomic int* slot_state;
|
||||
_Atomic int* slot_pid;
|
||||
_Atomic int* slot_module;
|
||||
_Atomic int* slot_host; /* per-source table bucket, or -1 */
|
||||
_Atomic int* module_active;
|
||||
_Atomic uint64_t* host_key; /* 0 == empty bucket */
|
||||
_Atomic int* host_active;
|
||||
_Atomic int* host_fail;
|
||||
_Atomic long long* host_until; /* epoch seconds the lockout expires */
|
||||
_Atomic long long* host_last_use; /* epoch seconds the bucket was last touched */
|
||||
};
|
||||
|
||||
static size_t round_up(size_t n, size_t align) {
|
||||
return (n + align - 1) & ~(align - 1);
|
||||
}
|
||||
|
||||
static size_t next_pow2(size_t n) {
|
||||
size_t p = 1;
|
||||
while (p < n)
|
||||
p <<= 1;
|
||||
return p;
|
||||
}
|
||||
|
||||
/* Parse a numeric IPv4/IPv6 peer string into family + raw bytes. */
|
||||
static bool parse_peer_ip(const char* peer_ip, int* family, unsigned char* bytes) {
|
||||
if (!peer_ip || *peer_ip == '\0')
|
||||
return false;
|
||||
struct in_addr v4;
|
||||
if (inet_pton(AF_INET, peer_ip, &v4) == 1) {
|
||||
memcpy(bytes, &v4, sizeof(v4));
|
||||
*family = AF_INET;
|
||||
return true;
|
||||
}
|
||||
struct in6_addr v6;
|
||||
if (inet_pton(AF_INET6, peer_ip, &v6) == 1) {
|
||||
memcpy(bytes, &v6, sizeof(v6));
|
||||
*family = AF_INET6;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) {
|
||||
if (ok)
|
||||
*ok = false;
|
||||
unsigned char bytes[16];
|
||||
int family = AF_UNSPEC;
|
||||
if (!parse_peer_ip(peer_ip, &family, bytes))
|
||||
return 0;
|
||||
uint64_t hash = 14695981039346656037ULL ^ (uint64_t)(uint32_t)family;
|
||||
size_t length = family == AF_INET ? 4 : 16;
|
||||
for (size_t i = 0; i < length; i++) {
|
||||
hash ^= bytes[i];
|
||||
hash *= 1099511628211ULL;
|
||||
}
|
||||
if (hash == 0)
|
||||
hash = 0x9e3779b97f4a7c15ULL;
|
||||
if (ok)
|
||||
*ok = true;
|
||||
return hash;
|
||||
}
|
||||
|
||||
/* True when the registry must maintain per-source buckets: either the per-host
|
||||
* cap is configured, or the auth lockout is (threshold AND duration > 0). A
|
||||
* lockout threshold without a duration is a no-op, so it must not size or intern
|
||||
* the table. create(), register() and the lockout paths all agree on this. */
|
||||
static bool registry_tracks_hosts(const DaemonLimitRegistry* registry) {
|
||||
return registry->per_host_cap > 0 ||
|
||||
(registry->lockout_threshold > 0 && registry->lockout_duration_sec > 0);
|
||||
}
|
||||
|
||||
/* Find the bucket holding `peer_ip`, or -1 when it has no entry. Finding a
|
||||
* bucket refreshes its last-use time so the eviction policy sees it as live. */
|
||||
static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], (long long)time(NULL),
|
||||
memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0)
|
||||
return -1; /* no tombstones: an empty bucket ends the probe chain */
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* A bucket with no live connection may be repurposed: immediately when its
|
||||
* lockout deadline has already passed (the review's "expired" case), or after an
|
||||
* idle window when it holds no pending lockout. A bucket with a future lockout
|
||||
* deadline is retained so the lockout actually lasts its configured duration. */
|
||||
static bool host_bucket_reclaimable(DaemonLimitRegistry* registry, size_t idx, long long now) {
|
||||
if (atomic_load_explicit(®istry->host_active[idx], memory_order_relaxed) != 0)
|
||||
return false;
|
||||
long long until = atomic_load_explicit(®istry->host_until[idx], memory_order_relaxed);
|
||||
if (until != 0)
|
||||
return until <= now;
|
||||
long long last_use = atomic_load_explicit(®istry->host_last_use[idx], memory_order_relaxed);
|
||||
/* A bucket whose key is published but whose last_use has not yet been stamped
|
||||
* (last_use == 0) must be treated as live: reclaiming it here would steal a
|
||||
* bucket a racing child just claimed. The claim path also stamps last_use
|
||||
* before publishing the key, so this window cannot persist. */
|
||||
return last_use != 0 && now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC;
|
||||
}
|
||||
|
||||
/* Emit at most one "per-source table full" warning per
|
||||
* DAEMON_LIMITS_HOST_FULL_WARN_SEC across all forked children. Called from a
|
||||
* normal (non-signal) child path, so logging is safe here. */
|
||||
static void host_warn_table_full(DaemonLimitRegistry* registry, long long now) {
|
||||
long long last = atomic_load_explicit(®istry->host_full_warn, memory_order_relaxed);
|
||||
if (last != 0 && now - last < DAEMON_LIMITS_HOST_FULL_WARN_SEC)
|
||||
return;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_full_warn, &last, now,
|
||||
memory_order_relaxed, memory_order_relaxed)) {
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"daemon: per-source registry is full (%d slots) and no bucket can be reclaimed; "
|
||||
"'max connections per host' and the auth lockout are temporarily not enforced for "
|
||||
"new sources (the per-module cap and host ACLs still apply)",
|
||||
registry->host_slots);
|
||||
}
|
||||
}
|
||||
|
||||
/* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two
|
||||
* forked children racing on the same source converge on one bucket.
|
||||
*
|
||||
* When the probe finds no empty bucket it reclaims, via a key CAS, the first
|
||||
* bucket that is reclaimable (expired lockout or idle, and no active
|
||||
* connection) and resets its counters. This bounds the table's lifetime so it
|
||||
* cannot fill permanently and stay fail-open. Returns -1 only when the address
|
||||
* is unparseable or the table is genuinely full of live/locked buckets
|
||||
* (callers fail open: the global/module caps and ACLs still apply). */
|
||||
static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
long long now = (long long)time(NULL);
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
/* A couple of passes bound the work: the first normally claims/seeds a bucket;
|
||||
* a lost eviction CAS retries once against the freshly observed table. */
|
||||
for (int pass = 0; pass < 2; pass++) {
|
||||
int evict = -1;
|
||||
uint64_t evict_key = 0;
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0) {
|
||||
/* Stamp last_use *before* publishing the key so a reclaimer racing the
|
||||
* claim can never observe a claimed bucket with last_use == 0 and
|
||||
* evict it. A pre-stamp is harmless if the CAS loses: the bucket is
|
||||
* either still empty (never inspected for reclaim) or has just been
|
||||
* taken by another source that wants a fresh timestamp anyway. */
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
uint64_t expected = 0;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
return (int)idx;
|
||||
}
|
||||
if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) {
|
||||
return (int)idx;
|
||||
}
|
||||
continue; /* another child won this empty bucket; keep probing */
|
||||
}
|
||||
if (evict < 0 && host_bucket_reclaimable(registry, idx, now)) {
|
||||
evict = (int)idx;
|
||||
evict_key = current;
|
||||
}
|
||||
}
|
||||
if (evict >= 0) {
|
||||
/* Refresh the timestamp before the key changes hands so the reused bucket
|
||||
* is not seen as immediately idle by a racing reclaimer. */
|
||||
atomic_store_explicit(®istry->host_last_use[evict], now, memory_order_relaxed);
|
||||
uint64_t expected = evict_key;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[evict], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
/* The bucket now belongs to the new source; clear the evicted source's
|
||||
* stale lockout/failure state. */
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_fail[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_until[evict], 0, memory_order_relaxed);
|
||||
/* Two children can race to intern the same brand-new key into different
|
||||
* eviction targets, leaving the table with duplicate buckets for `key`.
|
||||
* Re-scan for the first (canonical) bucket holding `key`; when it
|
||||
* precedes `evict`, drop our duplicate's occupancy and hand back the
|
||||
* canonical bucket so per-source counts are not orphaned on the
|
||||
* duplicate. The duplicate keeps its key, so no tombstone hole is
|
||||
* created and probe chains stay intact; it ages out normally. */
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t candidate = (start + i) & mask;
|
||||
uint64_t found =
|
||||
atomic_load_explicit(®istry->host_key[candidate], memory_order_acquire);
|
||||
if (found == key) {
|
||||
if (candidate != (size_t)evict) {
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
return (int)candidate;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (found == 0)
|
||||
break; /* the key is present at `evict`, so this cannot happen first */
|
||||
}
|
||||
return evict;
|
||||
}
|
||||
continue; /* lost the race; re-probe with fresh observations */
|
||||
}
|
||||
break; /* no free and no reclaimable bucket: genuinely full */
|
||||
}
|
||||
host_warn_table_full(registry, now);
|
||||
return -1;
|
||||
}
|
||||
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec) {
|
||||
if (max_slots < DAEMON_LIMITS_MIN_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MIN_SLOTS;
|
||||
if (max_slots > DAEMON_LIMITS_MAX_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MAX_SLOTS;
|
||||
if (module_count < 1)
|
||||
module_count = 1;
|
||||
if (module_count > DAEMON_LIMITS_MAX_MODULES)
|
||||
module_count = DAEMON_LIMITS_MAX_MODULES;
|
||||
if (per_host_cap < 0)
|
||||
per_host_cap = 0;
|
||||
if (lockout_threshold < 0)
|
||||
lockout_threshold = 0;
|
||||
if (lockout_duration_sec < 0)
|
||||
lockout_duration_sec = 0;
|
||||
|
||||
bool need_hosts = per_host_cap > 0 || (lockout_threshold > 0 && lockout_duration_sec > 0);
|
||||
int host_slots = 1;
|
||||
if (need_hosts) {
|
||||
size_t want = (size_t)max_slots * 4;
|
||||
if (want < 64)
|
||||
want = 64;
|
||||
if (want > DAEMON_LIMITS_MAX_HOST_SLOTS)
|
||||
want = DAEMON_LIMITS_MAX_HOST_SLOTS;
|
||||
host_slots = (int)next_pow2(want);
|
||||
}
|
||||
|
||||
size_t header = round_up(sizeof(DaemonLimitRegistry), 16);
|
||||
size_t slot_bytes =
|
||||
round_up((size_t)max_slots * sizeof(_Atomic int), 16) * 4; /* state,pid,module,host */
|
||||
size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16);
|
||||
size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16);
|
||||
size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2;
|
||||
size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16) * 2;
|
||||
size_t total =
|
||||
header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16;
|
||||
|
||||
void* map = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0);
|
||||
if (map == MAP_FAILED)
|
||||
return NULL;
|
||||
memset(map, 0, total);
|
||||
|
||||
DaemonLimitRegistry* registry = (DaemonLimitRegistry*)map;
|
||||
registry->max_slots = max_slots;
|
||||
registry->module_count = module_count;
|
||||
registry->host_slots = host_slots;
|
||||
registry->per_host_cap = per_host_cap;
|
||||
registry->lockout_threshold = lockout_threshold;
|
||||
registry->lockout_duration_sec = lockout_duration_sec;
|
||||
registry->map_size = total;
|
||||
|
||||
unsigned char* cursor = (unsigned char*)map + header;
|
||||
registry->slot_state = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_pid = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_module = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_host = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->module_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)module_count * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_key = (_Atomic uint64_t*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic uint64_t);
|
||||
registry->host_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
registry->host_fail = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_until = (atomic_llong*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic long long);
|
||||
registry->host_last_use = (atomic_llong*)cursor;
|
||||
|
||||
for (int i = 0; i < max_slots; i++) {
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
}
|
||||
return registry;
|
||||
}
|
||||
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
munmap(registry, registry->map_size);
|
||||
}
|
||||
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
int expected = SLOT_FREE;
|
||||
if (atomic_compare_exchange_strong(®istry->slot_state[i], &expected, SLOT_CLAIMED)) {
|
||||
atomic_store(®istry->slot_pid[i], 0);
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
}
|
||||
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_store(®istry->slot_pid[slot], (int)pid);
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel);
|
||||
atomic_store_explicit(®istry->slot_pid[slot], 0, memory_order_relaxed);
|
||||
/* The module/host occupancy arrays are derived from the slot table; do not
|
||||
* decrement here or a SIGKILL between a child's increment and its REGISTERED
|
||||
* publish would leak a count. Callers that need the derived counts call
|
||||
* daemon_limits_recompute. */
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) {
|
||||
if (!registry || pid <= 0)
|
||||
return;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load(®istry->slot_state[i]) == SLOT_FREE)
|
||||
continue;
|
||||
if (atomic_load(®istry->slot_pid[i]) == (int)pid) {
|
||||
daemon_limits_reclaim_slot(registry, i);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
/* Zero the derived arrays, then re-derive solely from the REGISTERED slots.
|
||||
* A child that was SIGKILLed after incrementing a counter but before
|
||||
* publishing REGISTERED is not counted, and its leaked increment is erased by
|
||||
* the zeroing, so the leak cannot persist. */
|
||||
for (int m = 0; m < registry->module_count; m++)
|
||||
atomic_store_explicit(®istry->module_active[m], 0, memory_order_relaxed);
|
||||
for (int h = 0; h < registry->host_slots; h++)
|
||||
atomic_store_explicit(®istry->host_active[h], 0, memory_order_relaxed);
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load_explicit(®istry->slot_state[i], memory_order_acquire) != SLOT_REGISTERED)
|
||||
continue;
|
||||
int module = atomic_load_explicit(®istry->slot_module[i], memory_order_relaxed);
|
||||
if (module >= 0 && module < registry->module_count)
|
||||
atomic_fetch_add_explicit(®istry->module_active[module], 1, memory_order_relaxed);
|
||||
int host = atomic_load_explicit(®istry->slot_host[i], memory_order_relaxed);
|
||||
if (host >= 0 && host < registry->host_slots)
|
||||
atomic_fetch_add_explicit(®istry->host_active[host], 1, memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (module_index < 0 || module_index >= registry->module_count)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (atomic_load_explicit(®istry->slot_state[slot], memory_order_acquire) != SLOT_CLAIMED)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
|
||||
int host = -1;
|
||||
if (registry_tracks_hosts(registry))
|
||||
host = host_intern(registry, peer_ip);
|
||||
|
||||
int module_count = atomic_fetch_add(®istry->module_active[module_index], 1) + 1;
|
||||
if (module_cap > 0 && module_count > module_cap) {
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_MODULE_FULL;
|
||||
}
|
||||
if (host >= 0) {
|
||||
int host_count = atomic_fetch_add(®istry->host_active[host], 1) + 1;
|
||||
if (registry->per_host_cap > 0 && host_count > registry->per_host_cap) {
|
||||
atomic_fetch_sub(®istry->host_active[host], 1);
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_HOST_FULL;
|
||||
}
|
||||
}
|
||||
atomic_store(®istry->slot_module[slot], module_index);
|
||||
atomic_store(®istry->slot_host[slot], host);
|
||||
atomic_store_explicit(®istry->slot_state[slot], SLOT_REGISTERED, memory_order_release);
|
||||
return DAEMON_LIMIT_OK;
|
||||
}
|
||||
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return false;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return false;
|
||||
long long until = atomic_load(®istry->host_until[bucket]);
|
||||
long long now = (long long)time(NULL);
|
||||
if (until > now) {
|
||||
if (seconds_remaining)
|
||||
*seconds_remaining = (int)(until - now);
|
||||
return true;
|
||||
}
|
||||
if (until != 0) {
|
||||
/* The previous lockout has expired: clear the stale counter so the source
|
||||
* gets a fresh allowance. */
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return;
|
||||
int bucket = host_intern(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
int failures = atomic_fetch_add(®istry->host_fail[bucket], 1) + 1;
|
||||
if (failures >= registry->lockout_threshold) {
|
||||
long long now = (long long)time(NULL);
|
||||
atomic_store(®istry->host_until[bucket], now + (long long)registry->lockout_duration_sec);
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry)
|
||||
return;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
#ifndef DAEMON_LIMITS_H
|
||||
#define DAEMON_LIMITS_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* Cross-process daemon connection registry.
|
||||
*
|
||||
* The daemon listener forks ONE child per accepted connection, so any
|
||||
* per-module / per-source accounting must live in state shared across the
|
||||
* forked children. This module owns a fixed-size registry carved out of an
|
||||
* anonymous shared mapping (mmap(MAP_SHARED | MAP_ANONYMOUS)) created by the
|
||||
* accept-loop PARENT before it forks; every child inherits the mapping (and the
|
||||
* pointer to it) across fork().
|
||||
*
|
||||
* Rules:
|
||||
* - ONLY C11 atomics (atomic_*); never mtx_t/pthread locks, which can deadlock
|
||||
* in a forked child if another thread held them at fork time.
|
||||
* - No heap allocation after fork: the mapping is fixed-size and all access is
|
||||
* atomic load/store/CAS over preallocated arrays.
|
||||
*
|
||||
* Slot lifecycle (the parent reclaims even when a child is SIGKILLed):
|
||||
* FREE --(parent claim_slot)--> CLAIMED
|
||||
* CLAIMED --(child register)--> REGISTERED
|
||||
* any --(parent reclaim)--> FREE
|
||||
* The child records its module index and per-source bucket into the slot before
|
||||
* publishing REGISTERED; the parent's SIGCHLD handler matches the reaped pid to
|
||||
* the slot and, when REGISTERED, decrements the module/per-source counters.
|
||||
* A child killed before registering holds no counts, so reclaiming a CLAIMED
|
||||
* slot only frees the slot.
|
||||
*
|
||||
* Per-source identity is the normalized numeric peer IP (IPv4-mapped IPv6 is
|
||||
* already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an
|
||||
* open-addressed, linear-probing table keyed by a 64-bit hash. The same table
|
||||
* also carries the cross-process auth-failure counter and lockout deadline.
|
||||
*
|
||||
* Per-source table lifetime: a bucket's key is never cleared back to empty (that
|
||||
* would break every later probe chain that passed through it). Instead the
|
||||
* table has a bounded-lifetime eviction policy: when no empty bucket exists, the
|
||||
* first bucket that is reclaimable -- no active connection AND (its lockout
|
||||
* deadline has passed OR it has been idle for
|
||||
* DAEMON_LIMITS_HOST_EVICT_IDLE_SEC) -- is atomically repurposed for the new
|
||||
* source via a CAS of its key, and its counters are reset. The table therefore
|
||||
* cannot fill permanently, and a full table degrades to fail-open for the
|
||||
* per-source cap/lockout of new sources (the per-module cap and host ACLs still
|
||||
* apply) instead of staying fail-open forever. A rate-limited warning is logged
|
||||
* on the fail-open path. The eviction race with a concurrent
|
||||
* registration/reclaim on the same bucket is benign: it can at worst lose one
|
||||
* source's counter (fail-open), never corrupt memory or the module caps.
|
||||
*/
|
||||
|
||||
typedef struct DaemonLimitRegistry DaemonLimitRegistry;
|
||||
|
||||
/* Result of a per-connection admission check. */
|
||||
typedef enum {
|
||||
DAEMON_LIMIT_OK = 0, /* admitted; slot is now REGISTERED */
|
||||
DAEMON_LIMIT_MODULE_FULL, /* module's `max connections` cap reached */
|
||||
DAEMON_LIMIT_HOST_FULL, /* global `max connections per host` cap reached */
|
||||
DAEMON_LIMIT_UNAVAILABLE, /* registry/slot unusable (caller fails open) */
|
||||
} DaemonLimitResult;
|
||||
|
||||
/* Bounds for registry sizing. A slot is one concurrently live child. */
|
||||
#define DAEMON_LIMITS_MIN_SLOTS 16
|
||||
#define DAEMON_LIMITS_MAX_SLOTS 65536
|
||||
#define DAEMON_LIMITS_MAX_HOST_SLOTS 65536
|
||||
#define DAEMON_LIMITS_NO_SLOT (-1)
|
||||
/* Upper bound on `module_count`, matching daemon_conf.h's DAEMON_CONF_MAX_MODULES
|
||||
* (asserted in daemon_limits.c) so a caller can never size the per-module counter
|
||||
* array larger than the config parser can produce. */
|
||||
#define DAEMON_LIMITS_MAX_MODULES 256
|
||||
|
||||
/* Per-source table lifetime: a bucket with no active connection and no pending
|
||||
* lockout is reclaimable once it has been idle this long, so a flood of distinct
|
||||
* sources cannot pin the table full forever. A bucket whose lockout deadline
|
||||
* has passed is reclaimable immediately (independent of this idle window). */
|
||||
#define DAEMON_LIMITS_HOST_EVICT_IDLE_SEC 300
|
||||
/* Minimum spacing between "per-source table is full" warnings, so a table-full
|
||||
* attack cannot flood the log. */
|
||||
#define DAEMON_LIMITS_HOST_FULL_WARN_SEC 60
|
||||
|
||||
/* Create the shared registry in the calling (parent) process. `max_slots` is
|
||||
* the number of concurrently live children to track (clamped to
|
||||
* [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the
|
||||
* number of daemon modules (clamped to
|
||||
* [1, DAEMON_LIMITS_MAX_MODULES]); `per_host_cap` and the lockout pair come
|
||||
* from the daemon config (0 disables). Returns NULL on failure (e.g. mmap
|
||||
* allocation); callers must degrade gracefully (global cap + ACLs still
|
||||
* apply). */
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec);
|
||||
|
||||
/* Unmap the registry. Only the creating process may call this. */
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Parent side: reserve a slot for the next fork. Returns the slot index or
|
||||
* DAEMON_LIMITS_NO_SLOT when every slot is in use. */
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry);
|
||||
/* Parent side: record the forked child's pid in a claimed slot. */
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid);
|
||||
/* Parent side: release a slot. The slot becomes FREE; the module/per-source
|
||||
* occupancy arrays are DERIVED state and are only refreshed by
|
||||
* daemon_limits_recompute, which callers must invoke afterwards when they rely
|
||||
* on the derived counts (the SIGCHLD handler batches one recompute for the whole
|
||||
* reap). Idempotent. */
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot);
|
||||
/* Parent SIGCHLD side: release the slot owned by `pid` (no-op when not found).
|
||||
* Like reclaim_slot this does not touch the derived occupancy arrays; call
|
||||
* daemon_limits_recompute after a batch of releases. */
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid);
|
||||
|
||||
/* Parent side (async-signal-safe; atomics only, no malloc/log): rebuild
|
||||
* module_active[] / host_active[] from scratch by scanning the REGISTERED slots.
|
||||
* The slot table is the single source of truth, so this self-heals any
|
||||
* count leaked by a child that was SIGKILLed mid-registration (it zeroes the
|
||||
* arrays and re-derives them). Bounded by max_slots + host_slots. A
|
||||
* registration racing this call can be transiently undercounted until the next
|
||||
* recompute, which can only relax a cap briefly -- never corrupt memory. */
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Child side: admit the connection for `module_index` from `peer_ip`. Always
|
||||
* tracks the module/per-source occupancy (so the parent's reclaim is
|
||||
* symmetric); when `module_cap` > 0 it additionally enforces the per-module
|
||||
* cap. A NULL/empty or non-numeric `peer_ip` skips the per-source track (the
|
||||
* callers use that to exempt a trusted loopback peer from the per-host cap; the
|
||||
* per-module cap still applies). Returns DAEMON_LIMIT_OK and publishes the
|
||||
* slot, or a refusal reason. */
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap);
|
||||
|
||||
/* Child side: true when `peer_ip` is currently locked out after too many failed
|
||||
* authentications. `seconds_remaining` may be NULL. */
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining);
|
||||
/* Child side: count one failed authentication for `peer_ip`; once the threshold
|
||||
* is reached the source is locked out for the configured duration. */
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
/* Child side: clear the failure counter/lockout for a source that authenticated
|
||||
* successfully (no-op when the source has no table entry). */
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
|
||||
/* Pure helper: 64-bit FNV-1a hash of a numeric peer IP plus its family, used to
|
||||
* index the per-source table. *ok is set false (and 0 returned) for a NULL or
|
||||
* non-numeric address. Exposed for unit testing. */
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok);
|
||||
|
||||
#endif
|
||||
+7
-1
@@ -23,6 +23,7 @@ Data* data_create_reserve(size_t size) {
|
||||
d->data = NULL;
|
||||
d->size = size;
|
||||
d->protocol_charge = 0;
|
||||
d->owner = NULL;
|
||||
return d;
|
||||
}
|
||||
|
||||
@@ -36,14 +37,19 @@ Data* data_create(void* data, size_t data_size) {
|
||||
new_data->data = data;
|
||||
new_data->size = data_size;
|
||||
new_data->protocol_charge = 0;
|
||||
new_data->owner = NULL;
|
||||
return new_data;
|
||||
}
|
||||
|
||||
void data_destroy(Data* data) {
|
||||
if (data == NULL)
|
||||
return;
|
||||
if (data->protocol_charge != 0)
|
||||
if (data->protocol_charge != 0) {
|
||||
if (data->owner != NULL)
|
||||
protocol_release_memory_for_session(data->owner, data->protocol_charge);
|
||||
else
|
||||
protocol_release_memory(data->protocol_charge);
|
||||
}
|
||||
free(data->data);
|
||||
free(data);
|
||||
}
|
||||
@@ -3,11 +3,25 @@
|
||||
|
||||
#include <stdlib.h>
|
||||
|
||||
/* Forward declaration for the connection budget a received Data is charged
|
||||
* against; defined in protocol.h (which includes this header). */
|
||||
typedef struct ProtocolSession ProtocolSession;
|
||||
|
||||
typedef struct {
|
||||
void* data;
|
||||
size_t size;
|
||||
/* Non-zero only for a buffer charged to the protocol connection budget. */
|
||||
size_t protocol_charge;
|
||||
/* Session whose budget `protocol_charge` was reserved from. When non-NULL,
|
||||
* the charge is returned to this session directly, regardless of which
|
||||
* session (if any) is bound to the destroying thread. owner is not
|
||||
* guaranteed to be set whenever protocol_charge is non-zero: it is NULL for
|
||||
* uncharged Data and for Data that has no recorded owner, in which case any
|
||||
* charge falls back to the session bound at destroy time.
|
||||
*
|
||||
* Lifetime contract: a Data with a non-NULL owner must not outlive that
|
||||
* ProtocolSession -- data_destroy dereferences owner to return the charge. */
|
||||
ProtocolSession* owner;
|
||||
} Data;
|
||||
|
||||
Data* data_create_empty(size_t data_size);
|
||||
@@ -15,5 +29,9 @@ Data* data_create_reserve(size_t size);
|
||||
Data* data_create(void* data, size_t data_size);
|
||||
void data_destroy(Data* data);
|
||||
void protocol_release_memory(size_t charge);
|
||||
/* Release `charge` against `session` directly instead of the thread-local bound
|
||||
* session. Used by data_destroy to honor Data.owner; `session` must outlive
|
||||
* the Data whose charge is being returned. A NULL session is a no-op. */
|
||||
void protocol_release_memory_for_session(ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
@@ -264,6 +264,20 @@ static bool delay_publish_entry(DelayUpdatesContext* context, const Config* conf
|
||||
const StagedFileEntry* entry) {
|
||||
if (!delay_publish_backup(context, config, entry))
|
||||
return false;
|
||||
/* --force: an incoming regular file/symlink may replace a destination
|
||||
DIRECTORY (possibly non-empty). The immediate-install path handles this in
|
||||
file_receive; a --delay-updates run stages elsewhere and only discovers the
|
||||
blocking directory here, so clear it before the rename (rsync's
|
||||
"could not make way for new regular file" without --force). */
|
||||
if (config && config->force_delete && file_directory_exists_secure(entry->final_path)) {
|
||||
if (!file_remove_tree_secure(entry->final_path)) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not remove destination directory blocking '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!file_rename_secure(entry->staged_path, entry->final_path)) {
|
||||
if (errno == EXDEV) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
|
||||
@@ -18,8 +18,9 @@ typedef struct {
|
||||
/* Receiver-side --delay-updates staging registry. All successfully written
|
||||
files land under a private staging directory inside the receive root and are
|
||||
atomically renamed into their final destination only at the very end of the
|
||||
transfer. A single PipelineContextReceiver has exactly one writer thread,
|
||||
but the registry is still mutex-protected so the same object can be safely
|
||||
transfer. A single receiver pipeline (see src/server/receiver_pipeline.h)
|
||||
has exactly one writer thread, but the registry is still mutex-protected so
|
||||
the same object can be safely
|
||||
shared with the publish/cleanup phase that runs after the threads join. */
|
||||
typedef struct DelayUpdatesContext {
|
||||
char* root_directory; /* receive root the staging dir lives under */
|
||||
|
||||
+308
-128
@@ -6,6 +6,7 @@
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <limits.h>
|
||||
#include <pthread.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
@@ -72,6 +73,58 @@ static unsigned long long next_temp_sequence(void) {
|
||||
return atomic_fetch_add_explicit(&sequence, 1, memory_order_relaxed);
|
||||
}
|
||||
|
||||
/* Process-wide umask, captured exactly once. Reading the umask requires a
|
||||
* get+set round trip (umask(0); umask(old)); doing that per write would be racy
|
||||
* in the multithreaded receiver, so the value is captured at process startup by
|
||||
* file_umask_capture() (called at the top of main(), before any threads exist).
|
||||
* The pthread_once fallback keeps a caller that never called the capture (e.g. a
|
||||
* unit test) correct. */
|
||||
static unsigned g_process_umask;
|
||||
static atomic_bool g_process_umask_captured;
|
||||
static pthread_once_t g_process_umask_once = PTHREAD_ONCE_INIT;
|
||||
|
||||
static void file_capture_umask_now(void) {
|
||||
mode_t mask = umask(0);
|
||||
umask(mask);
|
||||
g_process_umask = (unsigned)mask;
|
||||
atomic_store_explicit(&g_process_umask_captured, true, memory_order_release);
|
||||
}
|
||||
|
||||
static void file_capture_umask_once(void) {
|
||||
if (atomic_load_explicit(&g_process_umask_captured, memory_order_acquire))
|
||||
return;
|
||||
file_capture_umask_now();
|
||||
}
|
||||
|
||||
/* Re-captures the umask. Must only be called while the process is still
|
||||
* single-threaded (startup, or the daemon's post-fork setup after umask(0)),
|
||||
* so a later re-capture can refresh the cached value before any receiver
|
||||
* thread exists. */
|
||||
void file_umask_capture(void) {
|
||||
file_capture_umask_now();
|
||||
}
|
||||
|
||||
unsigned file_process_umask(void) {
|
||||
if (!atomic_load_explicit(&g_process_umask_captured, memory_order_acquire))
|
||||
pthread_once(&g_process_umask_once, file_capture_umask_once);
|
||||
return g_process_umask;
|
||||
}
|
||||
|
||||
/* Base mode applied when the policy does not take the source mode wholesale
|
||||
* (i.e. --perms is off). A pre-existing destination keeps its own mode; a
|
||||
* brand-new file is created like rsync: source_mode & 0777 & ~umask (special
|
||||
* bits are not part of a mode-preserving transfer without -p). Only when no
|
||||
* metadata is available at all does the historical fixed 0644 default apply.
|
||||
* The -E rule (and no-op for a plain -t) is layered on top of this base. */
|
||||
static mode_t file_mode_base(const FileMetadata* metadata, bool existing_known,
|
||||
mode_t existing_mode) {
|
||||
if (existing_known)
|
||||
return existing_mode;
|
||||
if (metadata)
|
||||
return metadata->mode & 0777 & ~(mode_t)file_process_umask();
|
||||
return S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH;
|
||||
}
|
||||
|
||||
bool file_checksum(File* file, ChecksumAlgo algo, uint64_t seed, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len) {
|
||||
if (!file || !out || !out_len || !file->data)
|
||||
@@ -124,6 +177,7 @@ File* file_create(const char* path) {
|
||||
file->rdev_major = 0;
|
||||
file->rdev_minor = 0;
|
||||
file->xattrs = NULL;
|
||||
file->dest_state = (OutputDestState){0};
|
||||
return file;
|
||||
}
|
||||
|
||||
@@ -284,28 +338,6 @@ size_t file_content_to_buffer(File* file) {
|
||||
|
||||
/* ---- Secure filesystem primitives ---- */
|
||||
|
||||
static int authorized_root_fd = -1;
|
||||
static char* authorized_root_path;
|
||||
|
||||
static bool path_is_within_root(const char* root, const char* path) {
|
||||
size_t root_len = strlen(root);
|
||||
return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/');
|
||||
}
|
||||
|
||||
bool file_set_authorized_root(int fd, const char* canonical_path) {
|
||||
char* path_copy = canonical_path ? str_dup(canonical_path) : NULL;
|
||||
if (canonical_path && !path_copy) {
|
||||
authorized_root_fd = -1;
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = NULL;
|
||||
return false;
|
||||
}
|
||||
authorized_root_fd = fd;
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = path_copy;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool file_path_exists_secure(const char* path) {
|
||||
if (!path)
|
||||
return false;
|
||||
@@ -362,10 +394,6 @@ void file_set_keep_dirlinks(bool enable) {
|
||||
file_keep_dirlinks = enable;
|
||||
}
|
||||
|
||||
bool file_get_keep_dirlinks(void) {
|
||||
return file_keep_dirlinks;
|
||||
}
|
||||
|
||||
/* --trust-sender (Phase 5) receiver process-wide policy: when set, the receiver
|
||||
* trusts the sender's file list and skips its own redundant up-front re-
|
||||
* validation (empty/".." path rejection, escaping-symlink-target containment).
|
||||
@@ -383,10 +411,69 @@ bool file_get_trust_sender(void) {
|
||||
return file_trust_sender;
|
||||
}
|
||||
|
||||
/* True when `target` is a lexical symlink target that can never escape the
|
||||
* receive root once created beneath it: relative (not absolute) and containing
|
||||
* no ".." path component. Used by --munge-links' sender-side containment: an
|
||||
* escaping target is never transmitted (the entry is skipped/contained). */
|
||||
/* rsync 3.4.1 unsafe_symlink(): true when `target` (the link's destination
|
||||
* string) points outside the transfer tree rooted at the symlink's own
|
||||
* location. `link_path` is the symlink's path relative to the top of the
|
||||
* transfer (including its name). This is a purely lexical test matching
|
||||
* rsync's util1.c: absolute/empty targets are always unsafe; leading "../"
|
||||
* components are counted against the symlink's own directory depth; a ".."
|
||||
* that would climb above the transfer root is unsafe. rsync 3.4.1 additionally
|
||||
* rejects any INTERNAL "/../" component and a trailing "/..". */
|
||||
bool file_symlink_unsafe(const char* target, const char* link_path) {
|
||||
if (!target || target[0] == '\0' || target[0] == '/')
|
||||
return true;
|
||||
const char* rest = target;
|
||||
while (strncmp(rest, "../", 3) == 0) {
|
||||
rest += 3;
|
||||
while (*rest == '/')
|
||||
rest++;
|
||||
}
|
||||
if (strstr(rest, "/../") != NULL)
|
||||
return true;
|
||||
size_t target_len = strlen(target);
|
||||
if (target_len > 3 && strcmp(&target[target_len - 3], "/..") == 0)
|
||||
return true;
|
||||
|
||||
int depth = 0;
|
||||
const char* name;
|
||||
const char* slash;
|
||||
const char* src = link_path ? link_path : "";
|
||||
for (name = src; (slash = strchr(name, '/')) != NULL; name = slash + 1) {
|
||||
if (*name == '.' && (name[1] == '/' || (name[1] == '.' && name[2] == '/'))) {
|
||||
if (name[1] == '.')
|
||||
depth = 0;
|
||||
} else {
|
||||
depth++;
|
||||
}
|
||||
while (slash[1] == '/')
|
||||
slash++;
|
||||
}
|
||||
if (*name == '.' && name[1] == '.' && name[2] == '\0')
|
||||
depth = 0;
|
||||
|
||||
for (name = target; (slash = strchr(name, '/')) != NULL; name = slash + 1) {
|
||||
if (*name == '.' && (name[1] == '/' || (name[1] == '.' && name[2] == '/'))) {
|
||||
if (name[1] == '.') {
|
||||
if (--depth < 0)
|
||||
return true;
|
||||
}
|
||||
} else {
|
||||
depth++;
|
||||
}
|
||||
while (slash[1] == '/')
|
||||
slash++;
|
||||
}
|
||||
if (*name == '.' && name[1] == '.' && name[2] == '\0')
|
||||
depth--;
|
||||
return depth < 0;
|
||||
}
|
||||
|
||||
/* Strict lexical helper: true when `target` is relative (not absolute) and
|
||||
* contains no ".." component at all, so it can never escape the directory it
|
||||
* is created in. This is stricter than rsync's unsafe_symlink() (which allows
|
||||
* an in-tree ".."); the scanner/receiver use file_symlink_unsafe()/--safe-links
|
||||
* for rsync parity, and this helper is retained for callers that want the
|
||||
* ".."-free guarantee. */
|
||||
bool file_symlink_target_contained(const char* target) {
|
||||
if (!target || target[0] == '\0' || target[0] == '/')
|
||||
return false;
|
||||
@@ -417,8 +504,9 @@ bool file_symlink_unmunge(char* target) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Owned copy of `target` prefixed with SYMLINK_MUNGE_PREFIX (the sender-side
|
||||
* --munge-links rewriting). Returns NULL on allocation failure. */
|
||||
/* Owned copy of `target` prefixed with SYMLINK_MUNGE_PREFIX (the receiver-side
|
||||
* --munge-links rewriting, matching rsync's receiver). Returns NULL on
|
||||
* allocation failure. */
|
||||
char* file_symlink_munge(const char* target) {
|
||||
if (!target)
|
||||
return NULL;
|
||||
@@ -439,23 +527,14 @@ char* file_symlink_munge(const char* target) {
|
||||
* the target is ever followed. The final component is never dereferenced: an
|
||||
* existing non-directory entry at `path` is unlinked by name before the link is
|
||||
* placed; an existing directory there is left untouched (returns false, so a
|
||||
* caller can treat it as a collision). As a receiver-side trust-boundary
|
||||
* invariant, `target` must be file_symlink_target_contained() (relative and
|
||||
* ".."-free): an absolute or escaping target is rejected outright (returns
|
||||
* false) so a malicious sender can never materialize a symlink that points
|
||||
* outside the receive root. */
|
||||
* caller can treat it as a collision). The link VALUE `target` is copied
|
||||
* verbatim, matching rsync -l (which stores absolute and ".."-bearing targets
|
||||
* as-is); target policy is the caller's job -- the scanner applies
|
||||
* --safe-links/--copy-unsafe-links, and the receiver applies --munge-links.
|
||||
* The PLACEMENT path is always confined below the authorized root. */
|
||||
bool file_symlink_at_secure(const char* path, const char* target) {
|
||||
/* The link itself (`path`) is always kept below the authorized root. The
|
||||
TARGET may point anywhere: normally only a contained (relative, ".."-free)
|
||||
target is permitted so a malicious sender can never plant a symlink that
|
||||
later dereferences outside the root. Under --trust-sender that target
|
||||
containment check is relaxed (the receiver trusts the sender and copies the
|
||||
link verbatim, matching rsync -l), but path/leaf confinement is never
|
||||
disabled, so the link still cannot be placed outside the tree. */
|
||||
if (!path || !target || has_path_traversal(path))
|
||||
return false;
|
||||
if (!file_trust_sender && !file_symlink_target_contained(target))
|
||||
return false;
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(path, &leaf, true);
|
||||
if (parent_fd < 0)
|
||||
@@ -492,7 +571,10 @@ static int open_dir_beneath_root(const char* resolved, const char* root) {
|
||||
rel++;
|
||||
if (*rel == '\0')
|
||||
return -1;
|
||||
int fd = dup(authorized_root_fd);
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
if (root_fd < 0)
|
||||
return -1;
|
||||
int fd = dup(root_fd);
|
||||
if (fd < 0)
|
||||
return -1;
|
||||
char* copy = str_dup(rel);
|
||||
@@ -534,20 +616,21 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs)
|
||||
return -1;
|
||||
}
|
||||
int fd;
|
||||
if (authorized_root_fd >= 0) {
|
||||
if (!authorized_root_path || path[0] != '/' ||
|
||||
!path_is_within_root(authorized_root_path, path)) {
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
if (root_fd >= 0) {
|
||||
if (!root_path || path[0] != '/' || !path_is_within_root(root_path, path)) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
fd = dup(authorized_root_fd);
|
||||
fd = dup(root_fd);
|
||||
if (fd < 0) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
size_t root_len = strlen(authorized_root_path);
|
||||
size_t root_len = strlen(root_path);
|
||||
char* relative = str_dup(path + root_len);
|
||||
if (!relative) {
|
||||
free(copy);
|
||||
@@ -611,15 +694,14 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs)
|
||||
O_NOFOLLOW walk. Only honoured when the symlink resolves to a
|
||||
directory that stays beneath the authorized root, so a malicious link
|
||||
can never redirect the write outside it. */
|
||||
if (next < 0 && file_keep_dirlinks && authorized_root_path != NULL &&
|
||||
if (next < 0 && file_keep_dirlinks && root_path != NULL &&
|
||||
(errno == ELOOP || errno == ENOTDIR || errno == EACCES)) {
|
||||
struct stat lst;
|
||||
if (fstatat(fd, component, &lst, AT_SYMLINK_NOFOLLOW) == 0 && S_ISLNK(lst.st_mode)) {
|
||||
char candidate[PATH_MAX];
|
||||
char root[PATH_MAX];
|
||||
if (realpath(authorized_root_path, root) &&
|
||||
snprintf(candidate, sizeof(candidate), "%s%s/%s", root, rel_buf, component) <
|
||||
(int)sizeof(candidate)) {
|
||||
if (realpath(root_path, root) && snprintf(candidate, sizeof(candidate), "%s%s/%s", root,
|
||||
rel_buf, component) < (int)sizeof(candidate)) {
|
||||
char resolved[PATH_MAX];
|
||||
if (realpath(candidate, resolved) && strcmp(resolved, root) != 0 &&
|
||||
strncmp(root, resolved, strlen(root)) == 0 &&
|
||||
@@ -694,8 +776,9 @@ bool file_ensure_directory_secure(const char* path) {
|
||||
return false;
|
||||
/* The authorized root is already an open directory, and the filesystem root
|
||||
is always present: there is no final component left to create for them. */
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
bool root_is_open =
|
||||
authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0;
|
||||
utils_get_authorized_root_fd() >= 0 && root_path && strcmp(norm, root_path) == 0;
|
||||
if (root_is_open || strcmp(norm, "/") == 0) {
|
||||
free(norm);
|
||||
return true;
|
||||
@@ -742,8 +825,9 @@ bool file_directory_exists_secure(const char* path) {
|
||||
char* norm = normalize_directory_path(path);
|
||||
if (!norm)
|
||||
return false;
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
bool root_is_open =
|
||||
authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0;
|
||||
utils_get_authorized_root_fd() >= 0 && root_path && strcmp(norm, root_path) == 0;
|
||||
if (root_is_open || strcmp(norm, "/") == 0) {
|
||||
free(norm);
|
||||
return true;
|
||||
@@ -876,29 +960,48 @@ int file_open_private_dir(const char* dir_path) {
|
||||
return fd;
|
||||
}
|
||||
|
||||
/* Open a --temp-dir scratch directory exactly as rsync does: the directory must
|
||||
* already exist and is used as given (an absolute path is used verbatim, a
|
||||
* relative one was already resolved against the destination root by the
|
||||
* caller). Unlike file_open_private_dir this neither creates it nor confines
|
||||
* it below the receive root, because rsync accepts any temp dir -- including
|
||||
* one outside the destination tree or on another filesystem. Returns an
|
||||
* O_DIRECTORY|O_CLOEXEC fd, or -1 on error. */
|
||||
int file_open_temp_dir(const char* dir_path) {
|
||||
if (!dir_path)
|
||||
return -1;
|
||||
return open(dir_path, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
|
||||
}
|
||||
|
||||
/* After the content and mode/times are restored on the just-written file, apply
|
||||
* the per-file xattrs (-X/-A) and, for --fake-super, park the source's
|
||||
* uid/gid/mode/mtime in the reserved xattr. All fd-relative (confined to the
|
||||
* destination file) and best-effort: a per-attribute or privilege failure is
|
||||
* logged and skipped, never fatal. */
|
||||
static void restore_extra_fd(int fd, const FileMetadata* metadata, const FileXattrList* xattrs,
|
||||
bool fake_super) {
|
||||
bool fake_super, FileAttrPolicy policy) {
|
||||
xattr_apply_fd(fd, xattrs);
|
||||
if (fake_super && metadata) {
|
||||
fake_super_store_fd(fd, (uint32_t)metadata->uid, (uint32_t)metadata->gid,
|
||||
(uint32_t)metadata->mode, metadata->mtime_sec, metadata->mtime_nsec);
|
||||
/* Replay: re-apply the recorded uid/gid/mode/mtime fd-relative so a save
|
||||
under --fake-super restores the attrs (when privileged) instead of only
|
||||
recording them. Best-effort; fake_super_restore_fd silently skips a
|
||||
non-root fchown EPERM/EACCES and never fatal. */
|
||||
fake_super_restore_fd(fd);
|
||||
/* Record the ownership that WOULD have been applied: when an explicit
|
||||
ownership request (--chown/--usermap/--groupmap/--copy-as or -o/-g) is
|
||||
active, the resolved mapping; otherwise the source's own id. The real
|
||||
chown is suppressed (identity_apply_ownership early-returns under
|
||||
--fake-super) so recording never defeats the flag. Mode/mtime are still
|
||||
replayed (policy-gated) so unprivileged --fake-super keeps working. */
|
||||
uint32_t store_uid;
|
||||
uint32_t store_gid;
|
||||
identity_resolve_storage_ids((int32_t)metadata->uid, (int32_t)metadata->gid, &store_uid,
|
||||
&store_gid);
|
||||
fake_super_store_fd(fd, store_uid, store_gid, (uint32_t)metadata->mode, metadata->mtime_sec,
|
||||
metadata->mtime_nsec);
|
||||
fake_super_restore_fd(fd, policy);
|
||||
}
|
||||
}
|
||||
|
||||
static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool update, bool no_replace,
|
||||
FileAttrPolicy policy, bool update, bool no_replace,
|
||||
bool use_fsync, const char* temp_dir,
|
||||
const FileXattrList* xattrs, bool fake_super,
|
||||
bool keep_partial) {
|
||||
@@ -908,16 +1011,55 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
return false;
|
||||
int fd = -1;
|
||||
bool ok = false;
|
||||
/* Set when a --temp-dir install fails with EXDEV: rsync then falls back to a
|
||||
* non-atomic write directly in the destination directory (see the tail of
|
||||
* this function). */
|
||||
bool cross_device_fallback = false;
|
||||
/* The base mode applied when --perms is off (neither the source mode nor an
|
||||
* exec-only change is taken wholesale): a pre-existing destination keeps its
|
||||
* own mode (special bits dropped), while a brand-new file uses
|
||||
* source&~umask when metadata is available (see file_mode_base) or 0644 when
|
||||
* there is none. Captured from the destination probe before the write. */
|
||||
mode_t existing_mode = S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH;
|
||||
bool existing_mode_known = false;
|
||||
if (inplace) {
|
||||
/* --inplace writes directly into the destination; a scratch --temp-dir
|
||||
does not apply and must never redirect these writes. */
|
||||
fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_CLOEXEC | O_NOFOLLOW, 0644);
|
||||
/* Type gate BEFORE opening: an existing destination entry that is not a
|
||||
regular file (FIFO, socket, char/block device, directory) must never be
|
||||
opened for writing. Opening a FIFO would block the receive thread
|
||||
forever and writing into a device would bypass the --write-devices /
|
||||
super-mode gate (a client-controlled device write). fstatat with
|
||||
AT_SYMLINK_NOFOLLOW does not follow a symlink and does not block. */
|
||||
struct stat pre_stat;
|
||||
if (fstatat(dirfd, leaf, &pre_stat, AT_SYMLINK_NOFOLLOW) == 0) {
|
||||
if (!S_ISREG(pre_stat.st_mode)) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
/* Capture the old destination mode before the overwrite so a no--p/-E
|
||||
* write can restore it (the write itself may clear setuid/setgid). */
|
||||
existing_mode = pre_stat.st_mode & 0777;
|
||||
existing_mode_known = true;
|
||||
}
|
||||
/* O_NONBLOCK: a no-op for a regular file, but a raced-in FIFO cannot block
|
||||
the open before the post-open S_ISREG re-check rejects it. */
|
||||
fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK, 0644);
|
||||
if (fd >= 0) {
|
||||
struct stat destination_stat;
|
||||
/* Re-check the opened descriptor: a concurrent replacement between the
|
||||
fstatat probe and the open (or a device/FIFO raced in) must never be
|
||||
written through. */
|
||||
if (fstat(fd, &destination_stat) != 0 || !S_ISREG(destination_stat.st_mode)) {
|
||||
close(fd);
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
bool newer = false;
|
||||
if (update && metadata && fstat(fd, &destination_stat) == 0 &&
|
||||
S_ISREG(destination_stat.st_mode)) {
|
||||
newer = stat_is_newer(&destination_stat, metadata);
|
||||
if (update && metadata && stat_is_newer(&destination_stat, metadata)) {
|
||||
newer = true;
|
||||
}
|
||||
if (newer) {
|
||||
ok = true;
|
||||
@@ -952,15 +1094,25 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
/* Normalize the mode: apply the metadata-derived safe mode when the
|
||||
sender supplied metadata (setuid/setgid/sticky are never honored);
|
||||
otherwise fall back to a safe default so dangerous bits on an
|
||||
existing destination cannot survive an overwrite. */
|
||||
existing destination cannot survive an overwrite. When the policy
|
||||
requests neither -p nor -E the source mode is deliberately ignored
|
||||
and the pre-existing destination mode (or 0644 for a new file) is
|
||||
restored instead. The exec-bits-only -E change is likewise applied
|
||||
on top of that destination-derived base, not the scratch file's
|
||||
0600. */
|
||||
if (ok) {
|
||||
if (metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
else if (fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0)
|
||||
if (metadata) {
|
||||
if (!policy.perms &&
|
||||
fchmod(fd, file_mode_base(metadata, existing_mode_known, existing_mode)) != 0)
|
||||
ok = false;
|
||||
if (ok)
|
||||
ok = file_restore_metadata_fd(fd, metadata, policy);
|
||||
} else if (fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0) {
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
if (ok)
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super);
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super, policy);
|
||||
if (ok && use_fsync)
|
||||
ok = fsync(fd) == 0;
|
||||
}
|
||||
@@ -973,25 +1125,31 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
install failure (partial data may exist, --partial may retain it) from a
|
||||
pre-write validation failure (nothing to retain). */
|
||||
bool write_attempted = false;
|
||||
if (update && metadata) {
|
||||
/* This check protects the normal atomic path as far as possible. A
|
||||
concurrent replacement can still occur before the final rename. */
|
||||
/* Probe the destination ONCE up front: it both drives the --update check
|
||||
and records the pre-existing mode the no--p/-E fallback preserves. */
|
||||
struct stat destination_stat;
|
||||
if (fstatat(dirfd, leaf, &destination_stat, AT_SYMLINK_NOFOLLOW) == 0 &&
|
||||
S_ISREG(destination_stat.st_mode) && stat_is_newer(&destination_stat, metadata)) {
|
||||
bool destination_is_regular =
|
||||
fstatat(dirfd, leaf, &destination_stat, AT_SYMLINK_NOFOLLOW) == 0 &&
|
||||
S_ISREG(destination_stat.st_mode);
|
||||
if (destination_is_regular) {
|
||||
existing_mode = destination_stat.st_mode & 0777;
|
||||
existing_mode_known = true;
|
||||
}
|
||||
if (update && metadata && destination_is_regular &&
|
||||
stat_is_newer(&destination_stat, metadata)) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
/* Scratch directory for the temporary working copy. When NULL the temp
|
||||
file is created in the destination directory, exactly as historically. */
|
||||
int scratch_dirfd = -1;
|
||||
if (temp_dir) {
|
||||
scratch_dirfd = file_open_private_dir(temp_dir);
|
||||
scratch_dirfd = file_open_temp_dir(temp_dir);
|
||||
if (scratch_dirfd < 0) {
|
||||
int saved_errno = errno;
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s",
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--temp-dir '%s' could not be opened (rsync requires it to already exist): %s",
|
||||
temp_dir, strerror(saved_errno));
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
@@ -1060,10 +1218,19 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
? file_store_write_sparse(fd, (const unsigned char*)data, data_size)
|
||||
: write_all(fd, data, data_size);
|
||||
}
|
||||
if (ok && metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
if (ok) {
|
||||
if (metadata) {
|
||||
if (!policy.perms &&
|
||||
fchmod(fd, file_mode_base(metadata, existing_mode_known, existing_mode)) != 0)
|
||||
ok = false;
|
||||
if (ok)
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super);
|
||||
ok = file_restore_metadata_fd(fd, metadata, policy);
|
||||
} else if (fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0) {
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
if (ok)
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super, policy);
|
||||
if (ok && use_fsync)
|
||||
ok = fsync(fd) == 0;
|
||||
}
|
||||
@@ -1080,17 +1247,16 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
errno != ENOENT)
|
||||
ok = false;
|
||||
} else {
|
||||
/* Cross-device (or otherwise impossible) link: rsync falls back to
|
||||
writing the file directly in the destination directory. Record
|
||||
it and retry below with no scratch dir. */
|
||||
if (scratch_dirfd >= 0 && errno == EXDEV)
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"temp dir is on a different filesystem than the destination; cannot "
|
||||
"link file into place (EXDEV); no fallback copy is attempted");
|
||||
cross_device_fallback = true;
|
||||
ok = false;
|
||||
}
|
||||
} else if (renameat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, dirfd, leaf) != 0) {
|
||||
if (scratch_dirfd >= 0 && errno == EXDEV)
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"temp dir is on a different filesystem than the destination; cannot "
|
||||
"atomically install file (EXDEV); no fallback copy is attempted");
|
||||
cross_device_fallback = true;
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
@@ -1123,43 +1289,49 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
close(fd);
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
if (cross_device_fallback) {
|
||||
/* rsync semantics: a --temp-dir on another filesystem must not abort the
|
||||
write. Retry once with no scratch dir so the file is written and
|
||||
installed non-atomically in the destination directory. */
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"temp dir is on a different filesystem than the destination; falling back to a "
|
||||
"non-atomic copy into the destination directory");
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
policy, update, no_replace, use_fsync, NULL, xattrs, fake_super,
|
||||
keep_partial);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, const char* temp_dir) {
|
||||
FileAttrPolicy policy, const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, false, false, false, temp_dir, NULL,
|
||||
false, false);
|
||||
policy, false, false, false, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, true, false, false, temp_dir, NULL, false,
|
||||
false);
|
||||
policy, true, false, false, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
const char* temp_dir) {
|
||||
FileAttrPolicy policy, bool use_fsync, const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, false, false, use_fsync, temp_dir, NULL,
|
||||
false, false);
|
||||
policy, false, false, use_fsync, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
unsigned long long data_size, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, false, sparse, preallocate, metadata,
|
||||
preserve_executability, false, true, false, temp_dir, NULL, false,
|
||||
false);
|
||||
policy, false, true, false, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
/* Receiver write-path variant that also applies the per-file xattrs (-X/-A)
|
||||
@@ -1169,13 +1341,12 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
* failed write's temp. See file_to_disk_secure_impl for the semantics. */
|
||||
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool update, bool no_replace, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super, bool keep_partial,
|
||||
const char* temp_dir) {
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
|
||||
bool fake_super, bool keep_partial, const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, update, no_replace, use_fsync, temp_dir,
|
||||
xattrs, fake_super, keep_partial);
|
||||
policy, update, no_replace, use_fsync, temp_dir, xattrs,
|
||||
fake_super, keep_partial);
|
||||
}
|
||||
|
||||
/* Atomic --link-dest install. The destination is replaced (via a temporary
|
||||
@@ -1196,7 +1367,7 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
|
||||
static bool file_to_disk_secure_link_impl(const char* path, const char* basis_path,
|
||||
const void* data, unsigned long long data_size,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
FileAttrPolicy policy, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir) {
|
||||
if (!path || !basis_path)
|
||||
@@ -1208,11 +1379,12 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
|
||||
int scratch_dirfd = -1;
|
||||
if (temp_dir) {
|
||||
scratch_dirfd = file_open_private_dir(temp_dir);
|
||||
scratch_dirfd = file_open_temp_dir(temp_dir);
|
||||
if (scratch_dirfd < 0) {
|
||||
int saved_errno = errno;
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s", temp_dir,
|
||||
strerror(saved_errno));
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--temp-dir '%s' could not be opened (rsync requires it to already exist): %s",
|
||||
temp_dir, strerror(saved_errno));
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
@@ -1247,10 +1419,18 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
if (linked) {
|
||||
int target_dirfd = scratch_dirfd >= 0 ? scratch_dirfd : dirfd;
|
||||
if (use_fsync) {
|
||||
int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (tfd < 0 || fsync(tfd) != 0) {
|
||||
/* O_NONBLOCK: the freshly linked temp is normally the basis's regular
|
||||
file, but a raced-in FIFO at the name must not block this reopen
|
||||
forever. With O_NONBLOCK such an open fails with ENXIO instead of
|
||||
blocking, which is treated as a benign fsync-skip (the link itself
|
||||
is still installed); any other open/fsync failure falls back to the
|
||||
byte-copy path as before. */
|
||||
int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC | O_NONBLOCK);
|
||||
if (tfd < 0) {
|
||||
if (errno != ENXIO)
|
||||
linked = false;
|
||||
} else if (fsync(tfd) != 0) {
|
||||
linked = false;
|
||||
if (tfd >= 0)
|
||||
close(tfd);
|
||||
} else {
|
||||
close(tfd);
|
||||
@@ -1277,8 +1457,8 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
/* The basis file could not be linked in (missing, cross-device, refused
|
||||
by the filesystem). Write a byte-identical local copy instead. */
|
||||
return file_to_disk_secure_attrs(path, data, data_size, false, false, preallocate, metadata,
|
||||
preserve_executability, false, false, use_fsync, xattrs,
|
||||
fake_super, false, temp_dir);
|
||||
policy, false, false, use_fsync, xattrs, fake_super, false,
|
||||
temp_dir);
|
||||
}
|
||||
|
||||
if (scratch_dirfd >= 0)
|
||||
@@ -1290,25 +1470,25 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
|
||||
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool use_fsync, const char* temp_dir) {
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata,
|
||||
preserve_executability, use_fsync, NULL, false, temp_dir);
|
||||
policy, use_fsync, NULL, false, temp_dir);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata,
|
||||
preserve_executability, use_fsync, xattrs, fake_super,
|
||||
temp_dir);
|
||||
policy, use_fsync, xattrs, fake_super, temp_dir);
|
||||
}
|
||||
|
||||
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse) {
|
||||
if (!path || (!data && data_size != 0) || has_path_traversal(path))
|
||||
return false;
|
||||
return file_to_disk_secure(path, data, data_size, inplace, sparse, false, NULL, false, NULL);
|
||||
FileAttrPolicy policy = {false, false, false, false};
|
||||
return file_to_disk_secure(path, data, data_size, inplace, sparse, false, NULL, policy, NULL);
|
||||
}
|
||||
+50
-32
@@ -28,17 +28,36 @@ void file_metadata_destroy(void* metadata);
|
||||
/* --open-noatime process-wide sender policy; see file.c. */
|
||||
void file_set_open_noatime(bool enable);
|
||||
bool file_get_open_noatime(void);
|
||||
/* Capture the process umask ONCE, before any threads are created. Call this at
|
||||
* the very top of main() in both entry points so the cached value is read while
|
||||
* the process is still single-threaded: reading the umask needs a get+set round
|
||||
* trip (umask(0); umask(old)), which would race against receiver threads
|
||||
* creating files if it happened during the first write. Idempotent and safe to
|
||||
* call more than once. */
|
||||
void file_umask_capture(void);
|
||||
/* Process-wide umask, captured once (thread-safe). Used to derive the mode of
|
||||
* a brand-new destination like rsync: source_mode & 0777 & ~umask. Falls back
|
||||
* to file_umask_capture() (behind pthread_once) if capture was never called. */
|
||||
unsigned file_process_umask(void);
|
||||
/* Open `path` read-only for transfer, honouring --open-noatime when set. */
|
||||
int file_open_for_read(const char* path);
|
||||
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse);
|
||||
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave). --munge-links
|
||||
* sender-side marker: every transmitted symlink target is prefixed with this
|
||||
* while the flag is on; the receiver strips it to restore the real target. */
|
||||
#define SYMLINK_MUNGE_PREFIX "#SYMLINK/"
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave; rsync parity).
|
||||
* --munge-links is a RECEIVER-side rewrite: rsync prefixes every stored symlink
|
||||
* target with this marker, making the link unusable while the referenced
|
||||
* directory does not exist. A SENDER receiving a munged source strips it back
|
||||
* off before transmitting (so a munged tree round-trips through the receiver's
|
||||
* re-munging). */
|
||||
#define SYMLINK_MUNGE_PREFIX "/rsyncd-munged/"
|
||||
|
||||
char* file_symlink_munge(const char* target);
|
||||
/* rsync 3.4.1 unsafe_symlink(): true when `target` escapes the transfer tree
|
||||
* rooted at `link_path` (the symlink's transfer-relative path incl. its name).
|
||||
* Absolute/empty targets and targets climbing above the transfer root (via
|
||||
* "..") are unsafe, as are internal "/../" components and trailing "/..". */
|
||||
bool file_symlink_unsafe(const char* target, const char* link_path);
|
||||
/* True when a lexical target is relative and contains no ".." component, so it
|
||||
* can never escape the receive root once created beneath it. */
|
||||
bool file_symlink_target_contained(const char* target);
|
||||
@@ -46,13 +65,13 @@ bool file_symlink_target_contained(const char* target);
|
||||
* returns true when a marker was removed. */
|
||||
bool file_symlink_unmunge(char* target);
|
||||
/* Create a symlink at `path` -> `target`, confined below the authorized root
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). Returns
|
||||
* false when a directory already occupies `path`. */
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). The link
|
||||
* value is copied verbatim (rsync -l); only the placement path is confined.
|
||||
* Returns false when a directory already occupies `path`. */
|
||||
bool file_symlink_at_secure(const char* path, const char* target);
|
||||
/* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing
|
||||
* symlink-to-directory to be followed as a directory. */
|
||||
void file_set_keep_dirlinks(bool enable);
|
||||
bool file_get_keep_dirlinks(void);
|
||||
|
||||
/* --trust-sender receiver process-wide policy (Phase 5). When set, the
|
||||
* receiver trusts that the sender already produced a clean file list and skips
|
||||
@@ -63,9 +82,6 @@ bool file_get_keep_dirlinks(void);
|
||||
void file_set_trust_sender(bool enable);
|
||||
bool file_get_trust_sender(void);
|
||||
|
||||
/* A configured fd without a canonical identity deliberately rejects paths. */
|
||||
bool file_set_authorized_root(int fd, const char* canonical_path);
|
||||
|
||||
/* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */
|
||||
bool file_path_exists_secure(const char* path);
|
||||
bool file_stat_secure(const char* path, struct stat* st);
|
||||
@@ -79,37 +95,40 @@ bool file_rename_secure(const char* old_path, const char* new_path);
|
||||
regular file. See the .c for the exact success semantics. */
|
||||
bool file_remove_tree_secure(const char* path);
|
||||
/* Open a private 0700 directory (creating it on demand) that must live below
|
||||
the authorized root. Used for the --temp-dir scratch directory and the
|
||||
--delay-updates staging directory. */
|
||||
the authorized root. Used for the --delay-updates staging directory. */
|
||||
int file_open_private_dir(const char* dir_path);
|
||||
|
||||
/* Open an existing --temp-dir scratch directory as-is (absolute or relative;
|
||||
no creation, no root confinement), matching rsync's --temp-dir handling. */
|
||||
int file_open_temp_dir(const char* dir_path);
|
||||
|
||||
/* The file_to_disk_secure* variants write a temporary copy in the destination
|
||||
directory and atomically rename it over `path`. temp_dir is an absolute,
|
||||
root-confined scratch directory (already validated by the caller): when it
|
||||
is non-NULL the temporary copy is instead created there (with a name unique
|
||||
across the whole scratch directory) and atomically renamed into the
|
||||
destination directory once fully written and fsynced. A rename across
|
||||
filesystems (EXDEV) fails the write with an error; the file is never
|
||||
silently copied into place. Pass NULL for the historical same-directory
|
||||
behavior. --inplace writes never use temp_dir. */
|
||||
directory and atomically rename it over `path`. temp_dir is a scratch
|
||||
directory (an absolute path, or one the caller already resolved against the
|
||||
destination root): when it is non-NULL the temporary copy is instead created
|
||||
there (with a name unique across the whole scratch directory) and atomically
|
||||
renamed into the destination directory once fully written and fsynced. When
|
||||
that rename/link fails with EXDEV (the scratch dir is on another filesystem)
|
||||
the write falls back to a non-atomic copy directly in the destination
|
||||
directory, matching rsync. Pass NULL for the same-directory behavior.
|
||||
--inplace writes never use temp_dir. */
|
||||
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, const char* temp_dir);
|
||||
FileAttrPolicy policy, const char* temp_dir);
|
||||
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
FileAttrPolicy policy, bool use_fsync, const char* temp_dir);
|
||||
/* With update enabled, an existing newer destination is left untouched. The
|
||||
check is descriptor-based for inplace writes; atomic replacement still has
|
||||
an unavoidable final rename race without filesystem locking. */
|
||||
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
unsigned long long data_size, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
/* Receiver write-path variant that also applies per-file xattrs (-X/-A) and the
|
||||
* --fake-super stat xattr fd-relative before the final rename. `update` /
|
||||
@@ -117,10 +136,9 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
* enables --partial best-effort retention of a failed write's temp. */
|
||||
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool update, bool no_replace, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super, bool keep_partial,
|
||||
const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
|
||||
bool fake_super, bool keep_partial, const char* temp_dir);
|
||||
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
|
||||
(via a temp name + rename); fall back to a byte-identical local copy from
|
||||
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
|
||||
@@ -129,15 +147,15 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
|
||||
never re-allocated). */
|
||||
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool use_fsync, const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
/* Like file_to_disk_secure_link, but the byte-copy fallback also applies the
|
||||
* per-file xattrs (-X/-A) and --fake-super stat xattr (fd-relative). On a
|
||||
* successful hard link no attributes are applied (the shared inode already
|
||||
* carries the basis's). */
|
||||
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
#ifndef FILE_ATTR_H
|
||||
#define FILE_ATTR_H
|
||||
|
||||
#include "config.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/*
|
||||
* Per-attribute receiver policy for applying a transmitted FileMetadata. This
|
||||
* is the split-out replacement for the former single use_metadata bundle: each
|
||||
* flag is applied independently, matching rsync's -p/-t/-o/-g/-E/-U semantics.
|
||||
* `use_metadata` remains the transport/presence gate (whether the metadata frame
|
||||
* travelled at all); this struct decides which attributes are ACTUALLY applied.
|
||||
*
|
||||
* It lives in its own header (rather than metadata.h) because xattr.h's
|
||||
* fake_super_restore_fd() takes one and metadata.h <-> file_types.h form an
|
||||
* include cycle that must not be entered from xattr.h.
|
||||
*
|
||||
* The mode leg is: perms wins over executability; an exec-bits-only change is
|
||||
* made only when perms is off; when neither is set the receiver deliberately
|
||||
* sets no source mode. file.c then substitutes the pre-existing destination
|
||||
* mode for a brand-new destination with metadata it uses the sanitized
|
||||
* source-mode-&-umask base (S_IWGRP|S_IWOTH cleared), and the fixed 0644
|
||||
* default only when no metadata is available at all, so a no--p overwrite
|
||||
* does not lose the destination's perms.
|
||||
*/
|
||||
typedef struct FileAttrPolicy {
|
||||
bool perms; /* config->preserve_perms: apply the source mode bits */
|
||||
bool times; /* config->preserve_times: apply the source mtime */
|
||||
bool atimes; /* config->preserve_atimes (-U): apply the source atime */
|
||||
bool executability; /* config->use_executability (-E): exec-bits-only mode */
|
||||
} FileAttrPolicy;
|
||||
|
||||
/* Build the per-attribute policy from a connection's Config. A NULL config
|
||||
* yields the all-off policy (no attribute application). */
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config);
|
||||
|
||||
#endif
|
||||
+99
-22
@@ -2,6 +2,7 @@
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -22,6 +23,8 @@ static void string_list_destroy(StringList* list) {
|
||||
|
||||
static bool string_list_add(StringList* list, const char* text) {
|
||||
if (list->count == list->capacity) {
|
||||
if (list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_cap = list->capacity > 0 ? list->capacity * 2 : 16;
|
||||
char** grown = realloc(list->items, (size_t)new_cap * sizeof(char*));
|
||||
if (!grown)
|
||||
@@ -51,11 +54,18 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings,
|
||||
if (len == 0)
|
||||
return 0;
|
||||
if (raw[0] == '/') {
|
||||
snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", (int)len, raw);
|
||||
int print_len = len > (size_t)INT_MAX ? INT_MAX : (int)len;
|
||||
snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", print_len, raw);
|
||||
return -1;
|
||||
}
|
||||
/* Reject NUL bytes inside a token defensively. In NUL-delimited mode the
|
||||
* delimiter itself is the final byte and is expected; in line mode any NUL is
|
||||
* embedded garbage (strlen-based parsing would otherwise silently truncate). */
|
||||
size_t scan_len = strip_line_endings ? len : len - 1;
|
||||
if (memchr(raw, '\0', scan_len)) {
|
||||
snprintf(err, err_size, "entry contains an embedded NUL byte");
|
||||
return -1;
|
||||
}
|
||||
/* Reject NUL bytes inside a token defensively (NUL-delimited mode splits on
|
||||
* them, so this only guards against embedded garbage). */
|
||||
char* dup = malloc(len + 1);
|
||||
if (!dup) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
@@ -102,8 +112,27 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings,
|
||||
return result;
|
||||
}
|
||||
|
||||
/* Build the membership index over the exact entries only. `file_list_affects`
|
||||
combines the exact/descendant lookups with a walk of the query's own ancestor
|
||||
prefixes, so no ancestor prefix is ever materialized as a copy and the index
|
||||
stays O(entry count) memory regardless of path depth. An empty entry (the
|
||||
source root) sets whole_tree and short-circuits every query. */
|
||||
static bool file_list_index_build(FileListSet* set, char* err, size_t err_size) {
|
||||
if (!path_index_build(&set->index, (const char* const*)set->entries, (size_t)set->count)) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < set->count; i++) {
|
||||
if (set->entries[i][0] == '\0') {
|
||||
set->whole_tree = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_size) {
|
||||
FileListSet* set = malloc(sizeof(FileListSet));
|
||||
FileListSet* set = calloc(1, sizeof(FileListSet));
|
||||
if (!set) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
@@ -112,6 +141,10 @@ static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_si
|
||||
set->entries = raw->items;
|
||||
raw->items = NULL;
|
||||
raw->count = 0;
|
||||
if (!file_list_index_build(set, err, err_size)) {
|
||||
file_list_destroy(set);
|
||||
return NULL;
|
||||
}
|
||||
return set;
|
||||
}
|
||||
|
||||
@@ -133,10 +166,20 @@ FileListSet* file_list_load(const char* path, bool null_separated, char* err, si
|
||||
StringList raw = {0};
|
||||
char* line = NULL;
|
||||
size_t line_cap = 0;
|
||||
ssize_t n;
|
||||
bool ok = true;
|
||||
char delim = null_separated ? '\0' : '\n';
|
||||
while (ok && (n = getdelim(&line, &line_cap, delim, fp)) != -1) {
|
||||
while (ok) {
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, delim, UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
if (errno == EFBIG)
|
||||
snprintf(err, err_size, "entry in file list exceeds %d bytes", (int)UTILS_MAX_LINE_LEN);
|
||||
else
|
||||
snprintf(err, err_size, "error reading file list: %s", strerror(errno));
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (n == 0)
|
||||
break;
|
||||
int r = normalize_entry(line, (size_t)n, !null_separated, &raw, err, err_size);
|
||||
if (r < 0) {
|
||||
ok = false;
|
||||
@@ -158,34 +201,68 @@ FileListSet* file_list_load(const char* path, bool null_separated, char* err, si
|
||||
void file_list_destroy(FileListSet* set) {
|
||||
if (!set)
|
||||
return;
|
||||
path_index_free(&set->index);
|
||||
for (int i = 0; i < set->count; i++)
|
||||
free(set->entries[i]);
|
||||
free(set->entries);
|
||||
free(set);
|
||||
}
|
||||
|
||||
static bool path_has_prefix(const char* path, const char* prefix) {
|
||||
size_t plen = strlen(prefix);
|
||||
if (strncmp(path, prefix, plen) != 0)
|
||||
return false;
|
||||
return path[plen] == '/' || path[plen] == '\0';
|
||||
}
|
||||
|
||||
bool file_list_affects(const FileListSet* set, const char* rel) {
|
||||
if (!set)
|
||||
return true;
|
||||
if (!rel)
|
||||
return false;
|
||||
for (int i = 0; i < set->count; i++) {
|
||||
const char* entry = set->entries[i];
|
||||
if (entry[0] == '\0')
|
||||
if (set->whole_tree)
|
||||
return true; /* whole tree listed */
|
||||
if (strcmp(rel, entry) == 0)
|
||||
return true; /* the entry itself is listed */
|
||||
if (path_has_prefix(rel, entry))
|
||||
return true; /* rel lives under a listed directory */
|
||||
if (path_has_prefix(entry, rel))
|
||||
return true; /* rel is an ancestor directory of a listed entry */
|
||||
/* An exact entry match means `rel` itself is listed. */
|
||||
if (path_index_contains(&set->index, rel))
|
||||
return true;
|
||||
/* Otherwise `rel` is affected when a listed entry is an ancestor directory of
|
||||
it; walk rel's own directory prefixes (which preserve path-boundary
|
||||
semantics) and test each for an exact entry. No prefixes are stored. */
|
||||
size_t len = strlen(rel);
|
||||
while (len > 0) {
|
||||
const char* slash = NULL;
|
||||
for (size_t i = len; i-- > 0;) {
|
||||
if (rel[i] == '/') {
|
||||
slash = rel + i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!slash)
|
||||
break;
|
||||
len = (size_t)(slash - rel);
|
||||
if (path_index_contains_n(&set->index, rel, len))
|
||||
return true;
|
||||
}
|
||||
/* Finally `rel` is affected when it is an ancestor directory of a listed
|
||||
entry (binary search for the first entry at or after `rel` + '/'). */
|
||||
return path_index_has_descendant(&set->index, rel);
|
||||
}
|
||||
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel) {
|
||||
if (!set || set->whole_tree)
|
||||
return true;
|
||||
if (!rel || rel[0] == '\0')
|
||||
return false;
|
||||
/* `rel` itself is listed, or one of its ancestor prefixes is an exact listed
|
||||
directory (a listed prefix of a directory path is necessarily a
|
||||
directory). */
|
||||
size_t len = strlen(rel);
|
||||
while (len > 0) {
|
||||
const char* slash = NULL;
|
||||
for (size_t i = len; i-- > 0;) {
|
||||
if (rel[i] == '/') {
|
||||
slash = rel + i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!slash)
|
||||
break;
|
||||
len = (size_t)(slash - rel);
|
||||
if (path_index_contains_n(&set->index, rel, len))
|
||||
return true;
|
||||
}
|
||||
return path_index_contains(&set->index, rel);
|
||||
}
|
||||
+20
-2
@@ -1,6 +1,7 @@
|
||||
#ifndef FILE_LIST_H
|
||||
#define FILE_LIST_H
|
||||
|
||||
#include "utils.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
|
||||
@@ -12,11 +13,18 @@
|
||||
* of "." means the whole tree, absolute entries and ".." traversal are
|
||||
* rejected at parse time. The set is immutable and shared read-only across
|
||||
* scanner worker threads.
|
||||
*/
|
||||
|
||||
*
|
||||
* Membership is answered from `index`, built once at load time over the exact
|
||||
* entries only: `index.exact` matches a listed path, the sorted view detects an
|
||||
* ancestor directory of a listed entry, and `rel`'s own directory prefixes are
|
||||
* matched against the exact set while descending. No ancestor prefix is stored
|
||||
* as a separate string, so the index is O(entry count) memory however deep the
|
||||
* paths are, and each query is O(path length) comparisons. */
|
||||
typedef struct {
|
||||
char** entries; /* normalized rel paths; "" means the whole tree */
|
||||
int count;
|
||||
PathIndex index;
|
||||
bool whole_tree; /* an entry of "" lists the source root */
|
||||
} FileListSet;
|
||||
|
||||
/* Load and validate a --files-from file. When `null_separated` (-0/--from0)
|
||||
@@ -32,4 +40,14 @@ void file_list_destroy(FileListSet* set);
|
||||
* this returns true, files are transferred only when it returns true. */
|
||||
bool file_list_affects(const FileListSet* set, const char* rel);
|
||||
|
||||
/* True when the DIRECTORY `rel` (path relative to the source root) is inside a
|
||||
* listed directory subtree: `rel` itself is a listed entry, or one of `rel`'s
|
||||
* ancestor directory prefixes is an exact listed entry. Unlike
|
||||
* file_list_affects this does NOT treat an ancestor of a listed entry as
|
||||
* affected, so an implied parent directory of a listed file is not synchronized
|
||||
* (rsync deletes nothing in it). With no set or a whole-tree set every
|
||||
* directory is in scope. This is the delete-walker's "synchronized directory"
|
||||
* predicate. */
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel);
|
||||
|
||||
#endif
|
||||
+909
-464
File diff suppressed because it is too large.
Load diff
+64
-14
@@ -7,6 +7,15 @@
|
||||
|
||||
/* Server-side file receive/save path. */
|
||||
|
||||
/* Cumulative caps for the deferred directory-time accumulator. The sender may
|
||||
* legitimately split a large tree across repeated STATUS_DIR_TIMES frames, so a
|
||||
* per-frame bound is not enough: the receiver must bound the TOTAL it retains
|
||||
* against a hostile sender. Mirror the delete-manifest limits
|
||||
* (MAX_MANIFEST_ENTRIES / MAX_MANIFEST_BYTES): the entry count bounds the
|
||||
* metadata array and the byte budget bounds the concatenated path strings. */
|
||||
#define MAX_DIR_TIME_ENTRIES (1024 * 1024)
|
||||
#define MAX_DIR_TIME_BYTES (16ULL * 1024 * 1024)
|
||||
|
||||
File* file_receive(const Config* config, int file_descriptor);
|
||||
File* file_receive_directory(int file_descriptor, const Config* config);
|
||||
File* file_receive_dir_time(int file_descriptor, const Config* config);
|
||||
@@ -15,6 +24,13 @@ File* file_receive_symlink(int file_descriptor, const Config* config);
|
||||
File* file_receive_special(int file_descriptor);
|
||||
bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode);
|
||||
File* receive_incremental_check(int fd, const Config* config, bool* skipped);
|
||||
/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set
|
||||
* true only on the server-contacting --dry-run path when the file is not up to
|
||||
* date: the receiver has already sent STATUS_DRY_RUN_TRANSFER and returns NULL
|
||||
* without storing anything. On that path `*skipped` is true for an up-to-date
|
||||
* (STATUS_OK) file and both flags are false for a genuine error. */
|
||||
File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped,
|
||||
bool* would_transfer);
|
||||
|
||||
/* P7 Wave D directory-time accumulator. The receiver collects the metadata of
|
||||
* every directory it creates/receives (STATUS_MKDIR with metadata and/or the
|
||||
@@ -26,21 +42,37 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped);
|
||||
typedef struct {
|
||||
char** paths; /* owned, destination-relative wire paths */
|
||||
FileMetadata* entries; /* owned, parallel to paths */
|
||||
FileXattrList** xattrs; /* owned, parallel to paths; NULL when none */
|
||||
size_t count;
|
||||
size_t capacity;
|
||||
size_t bytes; /* cumulative strlen of every retained path */
|
||||
} DirTimeList;
|
||||
|
||||
/* Capture gate shared by the sender-side and receiver-side sinks: directory
|
||||
* metadata is accumulated only when a directory attribute is requested
|
||||
* (-p/--perms for directory modes, or -t/--times for directory mtimes with
|
||||
* -O/--omit-dir-times not suppressing them) and metadata rides the wire. Kept
|
||||
* here, next to the accumulator it guards, so both call sites express the same
|
||||
* condition. */
|
||||
bool dir_metadata_should_capture(const Config* config);
|
||||
|
||||
void dir_time_list_init(DirTimeList* list);
|
||||
void dir_time_list_free(DirTimeList* list);
|
||||
/* Deep-copy one directory's path + metadata into the list. Returns false on
|
||||
* allocation failure (the caller fails the transfer). */
|
||||
bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata);
|
||||
/* Apply every accumulated directory's mtime (and atime when captured) beneath
|
||||
* `root_directory`, confined fd-relative. Best-effort per entry: an absent
|
||||
* directory (an empty/pruned source dir that was deliberately not created) or a
|
||||
* non-directory at the path is skipped QUIETLY, an unreachable one with a
|
||||
* warning, and never fatal. */
|
||||
void dir_time_list_apply(const DirTimeList* list, const char* root_directory);
|
||||
/* Deep-copy one directory's path + metadata (and, when non-NULL, its captured
|
||||
* xattr/ACL block) into the list. Returns false on allocation failure OR when
|
||||
* the cumulative entry/byte caps would be exceeded (the caller fails the
|
||||
* transfer). */
|
||||
bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata,
|
||||
const FileXattrList* xattrs);
|
||||
/* Apply every accumulated directory's metadata beneath `root_directory`,
|
||||
* confined fd-relative: ownership through the negotiated identity policy,
|
||||
* times (mtime, plus atime when -U captured one under -t), the mode (through
|
||||
* --chmod when configured, under -p), and the captured xattrs/ACLs (under
|
||||
* -X/-A). Best-effort per entry: an absent directory (an empty/pruned source
|
||||
* dir that was deliberately not created) or a non-directory at the path is
|
||||
* skipped QUIETLY, an unreachable one with a warning, and never fatal. */
|
||||
void dir_metadata_list_apply(const DirTimeList* list, const char* root_directory,
|
||||
const Config* config);
|
||||
|
||||
/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative
|
||||
paths the sender transferred/keeps) plus `protected`, destination-relative
|
||||
@@ -56,11 +88,18 @@ typedef struct DeleteManifest {
|
||||
ArrayList* keeps;
|
||||
ArrayList* protected;
|
||||
ArrayList* missing;
|
||||
/* Destination-relative paths of the directories the sender synchronized for
|
||||
this run. The extras walker only removes entries directly inside one of
|
||||
these (the receive root is the "." sentinel); `--files-from` runs therefore
|
||||
leave untransmitted directories and the unlisted parts of listed ones
|
||||
alone, matching rsync's "delete only in synchronized directories". */
|
||||
ArrayList* dirs;
|
||||
} DeleteManifest;
|
||||
|
||||
void delete_manifest_free(DeleteManifest* manifest);
|
||||
/* Read a delete-manifest frame: keep count + keeps, then protected count +
|
||||
protected prefixes, then missing count + missing paths (self-delimiting; the
|
||||
/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then
|
||||
protected count + protected prefixes, then missing count + missing paths,
|
||||
then synchronized-directory count + directory paths (self-delimiting; the
|
||||
leading STATUS_MANIFEST code has been consumed). Returns an owned
|
||||
DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */
|
||||
DeleteManifest* receive_manifest_entries(int fd);
|
||||
@@ -79,11 +118,22 @@ bool manifest_delete_extras(const Config* config, DeleteManifest* manifest);
|
||||
confinement or I/O error (the run then fails); tolerated per-path cases are
|
||||
reported and skipped. */
|
||||
bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest);
|
||||
/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's
|
||||
partial --max-delete result: the budget allowed some deletions and the rest
|
||||
were skipped (the run still stores all file data but the client exits 25). */
|
||||
typedef enum {
|
||||
DELETE_COMMIT_OK = 0,
|
||||
DELETE_COMMIT_LIMIT_REACHED,
|
||||
DELETE_COMMIT_ERROR
|
||||
} DeleteCommitResult;
|
||||
|
||||
/* Run every deletion family the manifest carries: the --delete-missing-args
|
||||
exact-path deletions first (user requests are not blocked by exclusion
|
||||
protection), then the ordinary extras walk when --delete is active. Returns
|
||||
true when nothing to do or everything committed. */
|
||||
bool manifest_delete_all(const Config* config, DeleteManifest* manifest);
|
||||
protection), then the ordinary extras walk when --delete is active. Both
|
||||
share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to
|
||||
do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget
|
||||
stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */
|
||||
DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest);
|
||||
|
||||
/* Outcome of a single file_save_to_disk operation. The receiver needs to
|
||||
distinguish "written" from "skipped" so --remove-source-files can be told
|
||||
|
||||
+10
-2
@@ -143,10 +143,17 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
|
||||
}
|
||||
|
||||
off_t offset = 0;
|
||||
/* A non-positive --timeout disables the deadline: poll blocks until the
|
||||
* socket is writable (rsync's --timeout=0 default). */
|
||||
int io_timeout_sec = protocol_get_io_timeout_sec();
|
||||
struct timespec deadline;
|
||||
if (io_timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += 60;
|
||||
deadline.tv_sec += io_timeout_sec;
|
||||
}
|
||||
while ((unsigned long long)offset < file_size) {
|
||||
int timeout = -1;
|
||||
if (io_timeout_sec > 0) {
|
||||
struct timespec now;
|
||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||
long long remaining = (long long)(deadline.tv_sec - now.tv_sec) * 1000LL +
|
||||
@@ -155,8 +162,9 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
timeout = remaining > INT_MAX ? INT_MAX : (int)remaining;
|
||||
}
|
||||
struct pollfd pfd = {.fd = file_descriptor, .events = POLLOUT};
|
||||
int timeout = remaining > INT_MAX ? INT_MAX : (int)remaining;
|
||||
int polled = poll(&pfd, 1, timeout);
|
||||
if (polled <= 0 || (pfd.revents & (POLLERR | POLLHUP | POLLNVAL))) {
|
||||
close(fd);
|
||||
|
||||
@@ -1,129 +1,7 @@
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "file_store.h"
|
||||
#include "metadata.h"
|
||||
#include "utils.h"
|
||||
|
||||
static int authorized_root_fd = -1;
|
||||
static char* authorized_root_path;
|
||||
|
||||
static bool path_is_within_root(const char* root, const char* path) {
|
||||
size_t root_length = strlen(root);
|
||||
return strncmp(root, path, root_length) == 0 &&
|
||||
(path[root_length] == '\0' || path[root_length] == '/');
|
||||
}
|
||||
|
||||
bool file_store_set_authorized_root(int fd, const char* canonical_path) {
|
||||
char* new_path = canonical_path ? str_dup(canonical_path) : NULL;
|
||||
if (canonical_path && !new_path) {
|
||||
authorized_root_fd = -1;
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = NULL;
|
||||
return false;
|
||||
}
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = new_path;
|
||||
authorized_root_fd = fd;
|
||||
return true;
|
||||
}
|
||||
|
||||
int file_store_open_secure_parent(const char* path, char** leaf_out) {
|
||||
char* copy = str_dup(path);
|
||||
if (!copy)
|
||||
return -1;
|
||||
char* parent = dirname(copy);
|
||||
const char* slash = strrchr(path, '/');
|
||||
char* leaf = str_dup(slash ? slash + 1 : path);
|
||||
if (!leaf) {
|
||||
free(copy);
|
||||
return -1;
|
||||
}
|
||||
int fd;
|
||||
if (authorized_root_fd >= 0) {
|
||||
if (!authorized_root_path || path[0] != '/' ||
|
||||
!path_is_within_root(authorized_root_path, path)) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
fd = dup(authorized_root_fd);
|
||||
if (fd < 0) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
size_t root_length = strlen(authorized_root_path);
|
||||
char* relative = str_dup(path + root_length);
|
||||
if (!relative) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
close(fd);
|
||||
return -1;
|
||||
}
|
||||
free(copy);
|
||||
copy = relative;
|
||||
parent = dirname(copy);
|
||||
} else {
|
||||
fd = (parent[0] == '/') ? open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC)
|
||||
: open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
|
||||
}
|
||||
if (fd < 0) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
char* save = NULL;
|
||||
char* component = strtok_r(parent, "/", &save);
|
||||
while (component) {
|
||||
if (strcmp(component, "..") == 0) {
|
||||
close(fd);
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
if (strcmp(component, ".") != 0) {
|
||||
int next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (next < 0 && errno == ENOENT) {
|
||||
if (mkdirat(fd, component, 0755) == 0 || errno == EEXIST)
|
||||
next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
}
|
||||
if (next < 0) {
|
||||
close(fd);
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
close(fd);
|
||||
fd = next;
|
||||
}
|
||||
component = strtok_r(NULL, "/", &save);
|
||||
}
|
||||
free(copy);
|
||||
*leaf_out = leaf;
|
||||
return fd;
|
||||
}
|
||||
|
||||
bool file_store_rename_secure(const char* old_path, const char* new_path) {
|
||||
char *old_leaf = NULL, *new_leaf = NULL;
|
||||
int old_parent = file_store_open_secure_parent(old_path, &old_leaf);
|
||||
int new_parent = file_store_open_secure_parent(new_path, &new_leaf);
|
||||
bool ok = old_parent >= 0 && new_parent >= 0 &&
|
||||
renameat(old_parent, old_leaf, new_parent, new_leaf) == 0;
|
||||
if (old_parent >= 0)
|
||||
close(old_parent);
|
||||
if (new_parent >= 0)
|
||||
close(new_parent);
|
||||
free(old_leaf);
|
||||
free(new_leaf);
|
||||
return ok;
|
||||
}
|
||||
|
||||
static bool write_all(int fd, const void* data, unsigned long long size) {
|
||||
const unsigned char* p = data;
|
||||
@@ -176,67 +54,3 @@ bool file_store_write_sparse(int fd, const unsigned char* data, unsigned long lo
|
||||
}
|
||||
return ftruncate(fd, (off_t)size) == 0;
|
||||
}
|
||||
|
||||
bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, const FileMetadata* metadata,
|
||||
bool preserve_executability) {
|
||||
char* leaf = NULL;
|
||||
int dirfd = file_store_open_secure_parent(path, &leaf);
|
||||
if (dirfd < 0)
|
||||
return false;
|
||||
int fd = -1;
|
||||
bool ok = false;
|
||||
if (inplace) {
|
||||
fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC | O_NOFOLLOW, 0644);
|
||||
if (fd >= 0) {
|
||||
if (sparse && data_size > 0) {
|
||||
if (ftruncate(fd, (off_t)data_size) == 0)
|
||||
ok = file_store_write_sparse(fd, data, data_size);
|
||||
} else {
|
||||
ok = write_all(fd, data, data_size);
|
||||
}
|
||||
if (ok && metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
}
|
||||
} else {
|
||||
int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%u", leaf, (long)getpid(), 99U);
|
||||
if (tmp_size < 0) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
char* tmp = malloc((size_t)tmp_size + 1);
|
||||
if (!tmp) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
for (unsigned int i = 0; i < 100 && !ok; ++i) {
|
||||
snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i);
|
||||
fd = openat(dirfd, tmp, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | O_NOFOLLOW, 0600);
|
||||
if (fd < 0)
|
||||
continue;
|
||||
if (sparse && data_size > 0)
|
||||
ok = ftruncate(fd, (off_t)data_size) == 0;
|
||||
if (ok || (!sparse || data_size == 0))
|
||||
ok = (sparse && data_size > 0)
|
||||
? file_store_write_sparse(fd, (const unsigned char*)data, data_size)
|
||||
: write_all(fd, data, data_size);
|
||||
if (ok && metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
if (close(fd) != 0)
|
||||
ok = false;
|
||||
fd = -1;
|
||||
if (ok && renameat(dirfd, tmp, dirfd, leaf) != 0)
|
||||
ok = false;
|
||||
if (!ok)
|
||||
unlinkat(dirfd, tmp, 0);
|
||||
}
|
||||
free(tmp);
|
||||
}
|
||||
if (fd >= 0)
|
||||
close(fd);
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return ok;
|
||||
}
|
||||
@@ -1,15 +1,8 @@
|
||||
#ifndef FILE_STORE_H
|
||||
#define FILE_STORE_H
|
||||
|
||||
#include "file.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
bool file_store_set_authorized_root(int fd, const char* canonical_path);
|
||||
int file_store_open_secure_parent(const char* path, char** leaf_out);
|
||||
bool file_store_rename_secure(const char* old_path, const char* new_path);
|
||||
bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, const FileMetadata* metadata,
|
||||
bool preserve_executability);
|
||||
/* Sparse-aware write (--sparse/-S): every all-zero run of at least
|
||||
* SPARSE_HOLE_MIN bytes is skipped with lseek(SEEK_CUR) so it becomes a real
|
||||
* hole; every other byte is written. The caller pre-sizes the file with
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#define FILE_TYPES_H
|
||||
|
||||
#include "data.h"
|
||||
#include "format.h"
|
||||
#include "xattr.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
@@ -86,6 +87,12 @@ typedef struct {
|
||||
* Receiver: parsed off the wire, attached here, and applied fd-relative on
|
||||
* the written file. NULL/0 == the file carries no xattrs. */
|
||||
FileXattrList* xattrs;
|
||||
/* Sender-side output-parity state (never serialized): the receiver-reported
|
||||
* pre-transfer destination snapshot for this entry, filled by the per-file
|
||||
* STATUS_CHECK exchange when report_dest_info is set. `known` is false when
|
||||
* no report was requested/received, in which case -i/--out-format treats the
|
||||
* entry conservatively as newly created. */
|
||||
OutputDestState dest_state;
|
||||
} File;
|
||||
|
||||
/* The path that should be sent on the wire and used for the receiver-side
|
||||
|
||||
+20
-4
@@ -2,6 +2,7 @@
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -176,6 +177,8 @@ bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) {
|
||||
if (!list || !rule)
|
||||
return false;
|
||||
if (list->count == list->capacity) {
|
||||
if (list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_cap = list->capacity > 0 ? list->capacity * 2 : 8;
|
||||
FilterRule** grown = realloc(list->items, (size_t)new_cap * sizeof(FilterRule*));
|
||||
if (!grown)
|
||||
@@ -322,8 +325,10 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
|
||||
if (!fp) {
|
||||
if (errno == ENOENT || errno == ENOTDIR)
|
||||
return filter_rule_list_create();
|
||||
log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s", dir_path,
|
||||
strerror(errno));
|
||||
char* escaped_dir = output_escape(dir_path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s",
|
||||
escaped_dir ? escaped_dir : "<allocation failed>", strerror(errno));
|
||||
free(escaped_dir);
|
||||
return filter_rule_list_create();
|
||||
}
|
||||
if (exists)
|
||||
@@ -336,9 +341,20 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
|
||||
}
|
||||
char* line = NULL;
|
||||
size_t line_cap = 0;
|
||||
ssize_t n;
|
||||
bool ok = true;
|
||||
while ((n = getline(&line, &line_cap, fp)) != -1) {
|
||||
while (true) {
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, '\n', UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
if (errno == EFBIG) {
|
||||
snprintf(err, err_size, "line in .rsync-filter exceeds %d bytes", (int)UTILS_MAX_LINE_LEN);
|
||||
} else {
|
||||
snprintf(err, err_size, "error reading .rsync-filter: %s", strerror(errno));
|
||||
}
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (n == 0)
|
||||
break;
|
||||
const char* p = line;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
#include "format.h"
|
||||
#include "protocol.h"
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
|
||||
bool format_human_size_decimal(unsigned long long bytes, char* buffer, size_t buffer_size) {
|
||||
if (!buffer || buffer_size == 0)
|
||||
return false;
|
||||
if (bytes < 1000ULL) {
|
||||
int written = snprintf(buffer, buffer_size, "%llu", bytes);
|
||||
return written >= 0 && (size_t)written < buffer_size;
|
||||
}
|
||||
static const char units[] = "KMGTPE";
|
||||
double value = (double)bytes;
|
||||
size_t divisions = 0;
|
||||
while (value >= 1000.0 && divisions < sizeof(units) - 1) {
|
||||
value /= 1000.0;
|
||||
divisions++;
|
||||
}
|
||||
int written = snprintf(buffer, buffer_size, "%.2f%c", value, units[divisions - 1]);
|
||||
return written >= 0 && (size_t)written < buffer_size;
|
||||
}
|
||||
|
||||
bool format_big_num(unsigned long long value, bool human_readable, char* buffer,
|
||||
size_t buffer_size) {
|
||||
if (human_readable)
|
||||
return format_human_size_decimal(value, buffer, buffer_size);
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
size_t len = (size_t)written;
|
||||
size_t separators = len > 1 ? (len - 1) / 3 : 0;
|
||||
size_t total = len + separators;
|
||||
if (total + 1 > buffer_size)
|
||||
return false;
|
||||
size_t out = total;
|
||||
buffer[out] = '\0';
|
||||
size_t digits_since_sep = 0;
|
||||
for (size_t i = len; i > 0; i--) {
|
||||
buffer[--out] = digits[i - 1];
|
||||
digits_since_sep++;
|
||||
if (digits_since_sep == 3 && i > 1) {
|
||||
buffer[--out] = ',';
|
||||
digits_since_sep = 0;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool format_rsync_datetime(time_t when, bool dash, char* buffer, size_t buffer_size) {
|
||||
if (!buffer || buffer_size == 0)
|
||||
return false;
|
||||
struct tm broken_down;
|
||||
if (localtime_r(&when, &broken_down) == NULL)
|
||||
return false;
|
||||
const char* format = dash ? "%Y/%m/%d-%H:%M:%S" : "%Y/%m/%d %H:%M:%S";
|
||||
return strftime(buffer, buffer_size, format, &broken_down) != 0;
|
||||
}
|
||||
|
||||
bool format_dest_state_send(int fd, const OutputDestState* state) {
|
||||
if (!state)
|
||||
return false;
|
||||
int32_t has_old = state->existed ? 1 : 0;
|
||||
uint64_t size = (uint64_t)state->size;
|
||||
int64_t mtime = (int64_t)state->mtime_sec;
|
||||
int64_t mtime_nsec = state->mtime_nsec;
|
||||
uint32_t mode = state->mode;
|
||||
int32_t uid = state->uid;
|
||||
int32_t gid = state->gid;
|
||||
return send_n_data(fd, &has_old, sizeof(has_old)) && send_n_data(fd, &size, sizeof(size)) &&
|
||||
send_n_data(fd, &mtime, sizeof(mtime)) &&
|
||||
send_n_data(fd, &mtime_nsec, sizeof(mtime_nsec)) && send_n_data(fd, &mode, sizeof(mode)) &&
|
||||
send_n_data(fd, &uid, sizeof(uid)) && send_n_data(fd, &gid, sizeof(gid));
|
||||
}
|
||||
|
||||
bool format_dest_state_receive(int fd, OutputDestState* state) {
|
||||
if (!state)
|
||||
return false;
|
||||
int32_t has_old = 0;
|
||||
uint64_t size = 0;
|
||||
int64_t mtime = 0;
|
||||
int64_t mtime_nsec = 0;
|
||||
uint32_t mode = 0;
|
||||
int32_t uid = 0;
|
||||
int32_t gid = 0;
|
||||
if (!receive_n_data(fd, &has_old, sizeof(has_old)) || !receive_n_data(fd, &size, sizeof(size)) ||
|
||||
!receive_n_data(fd, &mtime, sizeof(mtime)) ||
|
||||
!receive_n_data(fd, &mtime_nsec, sizeof(mtime_nsec)) ||
|
||||
!receive_n_data(fd, &mode, sizeof(mode)) || !receive_n_data(fd, &uid, sizeof(uid)) ||
|
||||
!receive_n_data(fd, &gid, sizeof(gid)))
|
||||
return false;
|
||||
memset(state, 0, sizeof(*state));
|
||||
state->known = true;
|
||||
state->existed = has_old != 0;
|
||||
state->size = size;
|
||||
state->mtime_sec = mtime;
|
||||
state->mtime_nsec = mtime_nsec;
|
||||
state->mode = mode;
|
||||
state->uid = uid;
|
||||
state->gid = gid;
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
#ifndef FORMAT_H
|
||||
#define FORMAT_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <time.h>
|
||||
|
||||
/* Low-level output-formatting primitives shared by the change-event model
|
||||
* (change_list.c) and the transfer driver (client_send.c).
|
||||
*
|
||||
* The functions here are pure/string-level except for the STATUS_DEST_INFO
|
||||
* codec, which lets the receiver report the pre-transfer destination entry so
|
||||
* the sender can render rsync-accurate --itemize-changes / --out-format
|
||||
* columns (see protocol.h). */
|
||||
|
||||
/* Pre-transfer destination snapshot, reported by the receiver when the wire
|
||||
* config carries report_dest_info. `known` distinguishes "no report was
|
||||
* requested/received" from "the destination did not exist" (`existed == false`
|
||||
* with `known == true`). */
|
||||
typedef struct {
|
||||
bool known;
|
||||
bool existed;
|
||||
unsigned long long size;
|
||||
long long mtime_sec;
|
||||
long long mtime_nsec;
|
||||
uint32_t mode;
|
||||
int32_t uid;
|
||||
int32_t gid;
|
||||
} OutputDestState;
|
||||
|
||||
/* rsync's -h/--human-readable size (decimal, base 1000): integers below 1000
|
||||
* print verbatim; larger values use the largest unit that keeps the value
|
||||
* below 1000 (K/M/G/T/P/E) with exactly two decimals, so 1500000 -> "1.50M"
|
||||
* and 999999 -> "1000.00K" (matching rsync's human_num). Returns false when
|
||||
* the buffer is too small (nothing is written). */
|
||||
bool format_human_size_decimal(unsigned long long bytes, char* buffer, size_t buffer_size);
|
||||
|
||||
/* rsync's general number formatting (big_num). When `human_readable` is true
|
||||
* this is format_human_size_decimal; otherwise the integer is rendered with a
|
||||
* ',' thousands separator every three digits (rsync's separator in the C
|
||||
* locale). Returns false on an undersized buffer. */
|
||||
bool format_big_num(unsigned long long value, bool human_readable, char* buffer,
|
||||
size_t buffer_size);
|
||||
|
||||
/* rsync's %M/%t timestamp. When `dash` is true the separator between the date
|
||||
* and the time is '-' (the %M form: "YYYY/MM/DD-HH:MM:SS"); otherwise it is a
|
||||
* space (the %t form: "YYYY/MM/DD HH:MM:SS"). Local time. Returns false on a
|
||||
* bad time or an undersized buffer. */
|
||||
bool format_rsync_datetime(time_t when, bool dash, char* buffer, size_t buffer_size);
|
||||
|
||||
/* Fixed-width STATUS_DEST_INFO record codec (int32 has_old, uint64 size,
|
||||
* int64 mtime, int64 mtime_nsec, uint32 mode, int32 uid, int32 gid). The
|
||||
* status frame itself is sent/received by the caller. Returns false on I/O
|
||||
* failure. */
|
||||
bool format_dest_state_send(int fd, const OutputDestState* state);
|
||||
bool format_dest_state_receive(int fd, OutputDestState* state);
|
||||
|
||||
#endif
|
||||
+407
-111
@@ -30,20 +30,40 @@ typedef struct {
|
||||
/* --super / --no-super tri-state (SUPER_MODE_AUTO when unset). Snapshotted
|
||||
* per connection so privilege_super_permitted() can gate super-user
|
||||
* activities without a Config argument. */
|
||||
int super_mode;
|
||||
SuperMode super_mode;
|
||||
/* --copy-as=USER[:GROUP]: snapshotted so the ownership resolver can force the
|
||||
* target ids without a Config argument. */
|
||||
bool copy_as_set;
|
||||
int32_t copy_as_uid;
|
||||
int32_t copy_as_gid;
|
||||
/* -o/--owner and -g/--group: preserve the source owner/group through the
|
||||
* normal name/identity resolution path. Split out of the former
|
||||
* use_metadata bundle; unlike --numeric-ids/--chown/--usermap/--groupmap/-a
|
||||
* these are a preserve-source request, not an arbitrary client-chosen owner,
|
||||
* so they are tracked separately from the explicit ownership gate. */
|
||||
bool preserve_owner;
|
||||
bool preserve_group;
|
||||
/* --fake-super: when active the receiver must only RECORD the (resolved)
|
||||
* ownership in the reserved xattr, never perform a real chown. Snapshotted
|
||||
* so the fd-relative ownership helpers can suppress the chown without a
|
||||
* Config argument. */
|
||||
bool fake_super;
|
||||
bool set;
|
||||
} IdentityActive;
|
||||
|
||||
static IdentityActive g_identity;
|
||||
|
||||
static void identity_active_reset(void) {
|
||||
if (g_identity.usermap) {
|
||||
for (int i = 0; i < g_identity.usermap_count; i++)
|
||||
free(g_identity.usermap[i].to_name);
|
||||
free(g_identity.usermap);
|
||||
}
|
||||
if (g_identity.groupmap) {
|
||||
for (int i = 0; i < g_identity.groupmap_count; i++)
|
||||
free(g_identity.groupmap[i].to_name);
|
||||
free(g_identity.groupmap);
|
||||
}
|
||||
g_identity.usermap = NULL;
|
||||
g_identity.groupmap = NULL;
|
||||
g_identity.usermap_count = 0;
|
||||
@@ -57,6 +77,9 @@ static void identity_active_reset(void) {
|
||||
g_identity.copy_as_set = false;
|
||||
g_identity.copy_as_uid = 0;
|
||||
g_identity.copy_as_gid = 0;
|
||||
g_identity.preserve_owner = false;
|
||||
g_identity.preserve_group = false;
|
||||
g_identity.fake_super = false;
|
||||
g_identity.set = false;
|
||||
}
|
||||
|
||||
@@ -77,32 +100,57 @@ bool identity_set_active(const Config* config) {
|
||||
g_identity.copy_as_set = config->copy_as_set;
|
||||
g_identity.copy_as_uid = config->copy_as_uid;
|
||||
g_identity.copy_as_gid = config->copy_as_gid;
|
||||
g_identity.preserve_owner = config->preserve_owner;
|
||||
g_identity.preserve_group = config->preserve_group;
|
||||
g_identity.fake_super = config->fake_super;
|
||||
if (config->usermap_count > 0) {
|
||||
g_identity.usermap = calloc((size_t)config->usermap_count, sizeof(IdentityMap));
|
||||
if (!g_identity.usermap)
|
||||
goto alloc_failed;
|
||||
memcpy(g_identity.usermap, config->usermap,
|
||||
(size_t)config->usermap_count * sizeof(IdentityMap));
|
||||
for (int i = 0; i < config->usermap_count; i++) {
|
||||
g_identity.usermap[i] = config->usermap[i];
|
||||
g_identity.usermap[i].to_name =
|
||||
config->usermap[i].to_name ? str_dup(config->usermap[i].to_name) : NULL;
|
||||
if (config->usermap[i].to_name && !g_identity.usermap[i].to_name) {
|
||||
g_identity.usermap_count = i; /* free only the entries already duplicated */
|
||||
goto alloc_failed;
|
||||
}
|
||||
}
|
||||
g_identity.usermap_count = config->usermap_count;
|
||||
}
|
||||
if (config->groupmap_count > 0) {
|
||||
g_identity.groupmap = calloc((size_t)config->groupmap_count, sizeof(IdentityMap));
|
||||
if (!g_identity.groupmap)
|
||||
goto alloc_failed;
|
||||
memcpy(g_identity.groupmap, config->groupmap,
|
||||
(size_t)config->groupmap_count * sizeof(IdentityMap));
|
||||
for (int i = 0; i < config->groupmap_count; i++) {
|
||||
g_identity.groupmap[i] = config->groupmap[i];
|
||||
g_identity.groupmap[i].to_name =
|
||||
config->groupmap[i].to_name ? str_dup(config->groupmap[i].to_name) : NULL;
|
||||
if (config->groupmap[i].to_name && !g_identity.groupmap[i].to_name) {
|
||||
g_identity.groupmap_count = i;
|
||||
goto alloc_failed;
|
||||
}
|
||||
}
|
||||
g_identity.groupmap_count = config->groupmap_count;
|
||||
}
|
||||
g_identity.set = true;
|
||||
/* A root receiver would honor any client-supplied ownership request (a
|
||||
--usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids).
|
||||
Surface that prominently; a privileged daemon applying arbitrary client
|
||||
ownership is a deliberate, opt-in choice the operator should be aware of. */
|
||||
if (geteuid() == 0)
|
||||
--usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids)
|
||||
ONLY when super-user activities are permitted. --no-super (or a daemon
|
||||
veto that forced SUPER_MODE_OFF) forbids the chown even for root, so do
|
||||
not claim the ownership will be honored in that case. */
|
||||
if (geteuid() == 0) {
|
||||
if (privilege_super_mode_permitted(g_identity.super_mode))
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"identity mapping active and running as root: client-supplied "
|
||||
"ownership (usermap/groupmap/chown/numeric-ids) will be honored; "
|
||||
"run the daemon as an unprivileged user unless intended");
|
||||
else
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"identity mapping active and running as root, but super-user activities are "
|
||||
"disabled (--no-super): requested ownership will NOT be applied; run the "
|
||||
"daemon as an unprivileged user unless intended");
|
||||
}
|
||||
/* --super explicitly requests super-user activities, but FastSync never
|
||||
elevates privileges: when the receiver is not already root the kernel will
|
||||
refuse those confined attempts and each is skipped per entry. Warn exactly
|
||||
@@ -128,7 +176,7 @@ bool privilege_super_permitted(void) {
|
||||
return privilege_super_mode_permitted(g_identity.super_mode);
|
||||
}
|
||||
|
||||
bool privilege_super_mode_permitted(int mode) {
|
||||
bool privilege_super_mode_permitted(SuperMode mode) {
|
||||
/* AUTO and ON both attempt the confined operation; OFF forbids it even for a
|
||||
* root receiver. AUTO is the historical FastSync behavior (always attempt
|
||||
* and let the kernel refuse an unprivileged call, which the caller skips), so
|
||||
@@ -138,25 +186,55 @@ bool privilege_super_mode_permitted(int mode) {
|
||||
}
|
||||
|
||||
bool identity_active_enabled(void) {
|
||||
/* numeric_ids is included: this set only gates identity_apply_ownership,
|
||||
which runs only when metadata is present (a -M/--preserve transfer). A
|
||||
standalone --numeric-ids (no ownership-affecting flag) carries no
|
||||
metadata, never reaches identity_apply_ownership, and therefore correctly
|
||||
stays inert; combined with -M it activates raw-id application. --super /
|
||||
--no-super does NOT enable ownership: it only permits or forbids the
|
||||
already-requested super-user activities, so a --super with no explicit
|
||||
identity flag must never silently apply client-chosen ownership. */
|
||||
/* --numeric-ids is deliberately NOT included: it is a mapping MODIFIER (use
|
||||
* the transmitted numeric id raw instead of a name lookup), not a request to
|
||||
* change ownership. rsync's --numeric-ids on its own never chowns anything;
|
||||
* it only changes how an already-requested -o/-g/map resolves. Ownership is
|
||||
* activated only by an explicit request: --chown/--usermap/--groupmap/
|
||||
* --copy-as or a preserve-source -o/--owner / -g/--group. --super/--no-super
|
||||
* likewise does NOT enable ownership: it only permits or forbids the
|
||||
* already-requested super-user activities. */
|
||||
return g_identity.set &&
|
||||
(g_identity.numeric_ids || g_identity.chown_uid_set || g_identity.chown_gid_set ||
|
||||
g_identity.usermap_count > 0 || g_identity.groupmap_count > 0 || g_identity.copy_as_set);
|
||||
(g_identity.chown_uid_set || g_identity.chown_gid_set || g_identity.usermap_count > 0 ||
|
||||
g_identity.groupmap_count > 0 || g_identity.copy_as_set || g_identity.preserve_owner ||
|
||||
g_identity.preserve_group);
|
||||
}
|
||||
|
||||
bool identity_owner_requested(void) {
|
||||
return g_identity.set && (g_identity.copy_as_set || g_identity.chown_uid_set ||
|
||||
g_identity.preserve_owner || g_identity.usermap_count > 0);
|
||||
}
|
||||
|
||||
bool identity_group_requested(void) {
|
||||
return g_identity.set && (g_identity.copy_as_set || g_identity.chown_gid_set ||
|
||||
g_identity.preserve_group || g_identity.groupmap_count > 0);
|
||||
}
|
||||
|
||||
bool identity_ownership_requested(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
/* Every value that makes the receiver act on a client-chosen owner, plus an
|
||||
* explicit --super (super-user device-node activities). Pure config, so the
|
||||
* daemon gate can evaluate it before identity_set_active(). */
|
||||
/* General-awareness predicate: every value that makes the receiver act on a
|
||||
* client-chosen owner, plus an explicit --super (super-user device-node
|
||||
* activities) and the preserve-source -o/-g requests. Pure config, so callers
|
||||
* can evaluate it before identity_set_active(). The daemon module gate uses
|
||||
* the narrower identity_explicit_ownership_requested() below, which treats a
|
||||
* plain -o/-g/-a as a preserve-source request rather than arbitrary
|
||||
* client-chosen ownership. */
|
||||
return config->numeric_ids || config->chown_uid_set || config->chown_gid_set ||
|
||||
config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set ||
|
||||
config->preserve_owner || config->preserve_group || config->fake_super ||
|
||||
config->super_mode == SUPER_MODE_ON;
|
||||
}
|
||||
|
||||
bool identity_explicit_ownership_requested(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
/* The narrow set the daemon gate refuses for a non-opted module: a request
|
||||
* that lets the CLIENT choose an arbitrary owner/group (rather than preserve
|
||||
* the source's own). Deliberately EXCLUDES preserve_owner/preserve_group so a
|
||||
* plain -a/-o/-g push is not refused; for those the gate instead forces
|
||||
* super-user ownership activity off (no chown happens) unless the module has
|
||||
* `client owner = yes`. */
|
||||
return config->numeric_ids || config->chown_uid_set || config->chown_gid_set ||
|
||||
config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set ||
|
||||
config->fake_super || config->super_mode == SUPER_MODE_ON;
|
||||
@@ -177,6 +255,29 @@ bool identity_copy_as_refused(const Config* config) {
|
||||
return geteuid() != 0 || config->super_mode == SUPER_MODE_OFF;
|
||||
}
|
||||
|
||||
/* Validate one received FROM:TO map rule. `from` is a single id, the LOW end
|
||||
* of an inclusive range, IDENTITY_MATCH_ANY, or IDENTITY_MATCH_UNNAMED; a
|
||||
* sentinel FROM must carry the same value in from_hi. `to` is a non-negative
|
||||
* id, IDENTITY_CURRENT, or ignored when a bounded receiver-resolved `to_name`
|
||||
* is present. */
|
||||
static bool identity_wire_map_valid(const IdentityMap* map) {
|
||||
if (!map)
|
||||
return false;
|
||||
if (map->from < IDENTITY_MATCH_UNNAMED)
|
||||
return false;
|
||||
if (map->from < 0) {
|
||||
if (map->from_hi != map->from)
|
||||
return false;
|
||||
} else if (map->from_hi < map->from) {
|
||||
return false;
|
||||
}
|
||||
if (map->to < IDENTITY_CURRENT)
|
||||
return false;
|
||||
if (map->to_name && strlen(map->to_name) > 255)
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool identity_wire_valid(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
@@ -188,11 +289,11 @@ bool identity_wire_valid(const Config* config) {
|
||||
if (config->chown_gid_set && config->chown_gid < IDENTITY_MATCH_ANY)
|
||||
return false;
|
||||
for (int i = 0; i < config->usermap_count; i++) {
|
||||
if (config->usermap[i].from < IDENTITY_MATCH_ANY || config->usermap[i].to < IDENTITY_CURRENT)
|
||||
if (!identity_wire_map_valid(&config->usermap[i]))
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < config->groupmap_count; i++) {
|
||||
if (config->groupmap[i].from < IDENTITY_MATCH_ANY || config->groupmap[i].to < IDENTITY_CURRENT)
|
||||
if (!identity_wire_map_valid(&config->groupmap[i]))
|
||||
return false;
|
||||
}
|
||||
/* Defense-in-depth: a --copy-as block must never carry a negative (sentinel)
|
||||
@@ -250,15 +351,137 @@ static int identity_resolve_token(const char* token, bool is_group, int32_t* out
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int identity_append_rule(IdentityMap** map, int* count, int32_t from, int32_t to) {
|
||||
static bool identity_all_digits(const char* token) {
|
||||
if (!token || *token == '\0')
|
||||
return false;
|
||||
for (const char* p = token; *p; p++)
|
||||
if (*p < '0' || *p > '9')
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool identity_token_has_glob(const char* token) {
|
||||
return token && (strchr(token, '*') || strchr(token, '?') || strchr(token, '['));
|
||||
}
|
||||
|
||||
/* Parse a --usermap/--groupmap FROM token into a matcher (from/from_hi). rsync
|
||||
* accepts a name, a numeric id, an inclusive LOW-HIGH range, '*' (any id), or an
|
||||
* empty token (ids with no name on the sender). Returns 0 on success, -1 on a
|
||||
* malformed token or an unresolvable sender-side name. */
|
||||
static int identity_parse_from(const char* token, bool is_group, int32_t* out_from,
|
||||
int32_t* out_hi) {
|
||||
if (token[0] == '\0') {
|
||||
*out_from = IDENTITY_MATCH_UNNAMED;
|
||||
*out_hi = IDENTITY_MATCH_UNNAMED;
|
||||
return 0;
|
||||
}
|
||||
if (strcmp(token, "*") == 0) {
|
||||
*out_from = IDENTITY_MATCH_ANY;
|
||||
*out_hi = IDENTITY_MATCH_ANY;
|
||||
return 0;
|
||||
}
|
||||
const char* num = token[0] == '@' ? token + 1 : token;
|
||||
if (identity_all_digits(num)) {
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_from = id;
|
||||
*out_hi = id;
|
||||
return 0;
|
||||
}
|
||||
/* An inclusive LOW-HIGH numeric range. */
|
||||
const char* dash = strchr(num, '-');
|
||||
if (dash && dash != num && dash[1] != '\0' && strchr(dash + 1, '-') == NULL) {
|
||||
size_t lo_len = (size_t)(dash - num);
|
||||
size_t hi_len = strlen(dash + 1);
|
||||
char low[16];
|
||||
char high[16];
|
||||
if (lo_len < sizeof(low) && hi_len < sizeof(high)) {
|
||||
memcpy(low, num, lo_len);
|
||||
low[lo_len] = '\0';
|
||||
memcpy(high, dash + 1, hi_len);
|
||||
high[hi_len] = '\0';
|
||||
if (identity_all_digits(low) && identity_all_digits(high)) {
|
||||
char* endptr = NULL;
|
||||
errno = 0;
|
||||
long lo = strtol(low, &endptr, 10);
|
||||
if (errno != 0 || !endptr || *endptr != '\0')
|
||||
return -1;
|
||||
errno = 0;
|
||||
long hi = strtol(high, &endptr, 10);
|
||||
if (errno != 0 || !endptr || *endptr != '\0' || hi < lo || hi > INT32_MAX)
|
||||
return -1;
|
||||
*out_from = (int32_t)lo;
|
||||
*out_hi = (int32_t)hi;
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
/* Not a numeric LOW-HIGH range: fall through and treat as a name (a
|
||||
* hyphenated account name like "wayne-smith" must still resolve). */
|
||||
}
|
||||
/* A sender-side name. A wildcard other than the bare '*' is matched by rsync
|
||||
* against the sender's names; because FastSync transmits numeric ids only, the
|
||||
* receiver cannot evaluate it, so reject rather than silently mis-match. */
|
||||
if (identity_token_has_glob(token)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%smap FROM '%s': name wildcards other than '*' are not supported "
|
||||
"(FastSync transmits numeric ids, so sender names are unavailable on the "
|
||||
"receiver)",
|
||||
is_group ? "--group" : "--user", token);
|
||||
return -1;
|
||||
}
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_from = id;
|
||||
*out_hi = id;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Parse a --usermap/--groupmap TO token. '*', a bare numeric id, or an @N id is
|
||||
* stored numerically; every other non-empty token is a NAME resolved on the
|
||||
* RECEIVER at apply time (rsync resolves TO names against the receiving side).
|
||||
* Returns 0 on success, -1 on an empty/malformed token. */
|
||||
static int identity_parse_to(const char* token, bool is_group, int32_t* out_to, char** out_name) {
|
||||
if (token[0] == '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "%smap TO value is missing", is_group ? "--group" : "--user");
|
||||
return -1;
|
||||
}
|
||||
if (strcmp(token, "*") == 0) {
|
||||
*out_to = IDENTITY_CURRENT;
|
||||
*out_name = NULL;
|
||||
return 0;
|
||||
}
|
||||
const char* num = token[0] == '@' ? token + 1 : token;
|
||||
if (identity_all_digits(num)) {
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_to = id;
|
||||
*out_name = NULL;
|
||||
return 0;
|
||||
}
|
||||
if (identity_token_has_glob(token)) {
|
||||
log_message(LOG_LEVEL_ERROR, "%smap TO '%s' may not contain a wildcard",
|
||||
is_group ? "--group" : "--user", token);
|
||||
return -1;
|
||||
}
|
||||
char* name = str_dup(token);
|
||||
if (!name)
|
||||
return -1;
|
||||
*out_to = 0;
|
||||
*out_name = name;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int identity_append_rule(IdentityMap** map, int* count, const IdentityMap* rule) {
|
||||
if (*count >= MAX_IDENTITY_MAP)
|
||||
return -1;
|
||||
IdentityMap* grown = realloc(*map, (size_t)(*count + 1) * sizeof(IdentityMap));
|
||||
if (!grown)
|
||||
return -1;
|
||||
*map = grown;
|
||||
(*map)[*count].from = from;
|
||||
(*map)[*count].to = to;
|
||||
(*map)[*count] = *rule;
|
||||
(*count)++;
|
||||
return 0;
|
||||
}
|
||||
@@ -275,7 +498,7 @@ int identity_parse_map(Config* config, const char* value, bool is_group) {
|
||||
char* saveptr = NULL;
|
||||
for (char* rule = strtok_r(list, ",", &saveptr); rule; rule = strtok_r(NULL, ",", &saveptr)) {
|
||||
char* colon = strchr(rule, ':');
|
||||
if (!colon || colon == rule) {
|
||||
if (!colon) {
|
||||
/* Log before freeing: `rule` points into the str_dup'd list. */
|
||||
log_message(LOG_LEVEL_ERROR, "%s rules must be FROM:TO (got '%s')", optname, rule);
|
||||
free(list);
|
||||
@@ -284,25 +507,25 @@ int identity_parse_map(Config* config, const char* value, bool is_group) {
|
||||
*colon = '\0';
|
||||
char* from_token = rule;
|
||||
char* to_token = colon + 1;
|
||||
if (*to_token == '\0') {
|
||||
IdentityMap parsed;
|
||||
memset(&parsed, 0, sizeof(parsed));
|
||||
if (identity_parse_from(from_token, is_group, &parsed.from, &parsed.from_hi) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%s could not resolve FROM '%s' in '%s' (a name must exist on the "
|
||||
"source; use @N for a numeric id)",
|
||||
optname, from_token, value);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR, "%s rule 'FROM:' is missing the TO value (got '%s')", optname,
|
||||
value);
|
||||
return -1;
|
||||
}
|
||||
int32_t from_id, to_id;
|
||||
if (identity_resolve_token(from_token, is_group, &from_id) != 0 ||
|
||||
identity_resolve_token(to_token, is_group, &to_id) != 0) {
|
||||
if (identity_parse_to(to_token, is_group, &parsed.to, &parsed.to_name) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s could not parse TO '%s' in '%s'", optname, to_token, value);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%s could not resolve '%s' (name must exist on the source; use "
|
||||
"@N for a numeric id)",
|
||||
optname, value);
|
||||
return -1;
|
||||
}
|
||||
if (identity_append_rule(is_group ? &config->groupmap : &config->usermap,
|
||||
is_group ? &config->groupmap_count : &config->usermap_count, from_id,
|
||||
to_id) != 0) {
|
||||
is_group ? &config->groupmap_count : &config->usermap_count,
|
||||
&parsed) != 0) {
|
||||
free(parsed.to_name);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR, "%s has too many rules (max %d)", optname, MAX_IDENTITY_MAP);
|
||||
return -1;
|
||||
@@ -567,9 +790,6 @@ int identity_parse_copy_as(Config* config, const char* value) {
|
||||
config->copy_as_set = true;
|
||||
config->copy_as_uid = uid;
|
||||
config->copy_as_gid = gid;
|
||||
/* Ownership application needs the metadata path (the source uid/gid must be
|
||||
* transmitted); imply it exactly like --chown/--usermap/--groupmap. */
|
||||
config->use_metadata = true;
|
||||
ret = 0;
|
||||
|
||||
done:
|
||||
@@ -580,34 +800,120 @@ done:
|
||||
|
||||
/* ---- Receiver-side ownership application ---- */
|
||||
|
||||
static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id,
|
||||
/* True when a map rule's FROM matcher accepts `id`. A sentinel FROM never
|
||||
* carries a range. IDENTITY_MATCH_UNNAMED mirrors rsync's empty FROM: it
|
||||
* matches only ids that have no name in the account database (rsync matches the
|
||||
* sender's names; FastSync transmits numeric ids only, so it approximates this
|
||||
* with the receiver's database -- documented in RSYNC_COMPAT.md). */
|
||||
static bool identity_map_from_matches(const IdentityMap* map, int32_t id, bool is_group) {
|
||||
if (map->from == IDENTITY_MATCH_ANY)
|
||||
return true;
|
||||
if (map->from == IDENTITY_MATCH_UNNAMED)
|
||||
return is_group ? (getgrgid((gid_t)id) == NULL) : (getpwuid((uid_t)id) == NULL);
|
||||
return id >= map->from && id <= map->from_hi;
|
||||
}
|
||||
|
||||
/* First matching rule wins. A rule whose TO is a receiver-side name resolves it
|
||||
* against the receiver's account database here; an unresolvable TO name is
|
||||
* skipped with a warning and the next rule is considered (rsync prints "Unknown
|
||||
* --usermap name on receiver" and leaves the id unmapped rather than aborting). */
|
||||
static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id, bool is_group,
|
||||
int32_t* out_to) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (map[i].from == IDENTITY_MATCH_ANY || map[i].from == source_id) {
|
||||
if (!identity_map_from_matches(&map[i], source_id, is_group))
|
||||
continue;
|
||||
if (map[i].to_name) {
|
||||
if (is_group) {
|
||||
struct group* gr = getgrnam(map[i].to_name);
|
||||
if (!gr) {
|
||||
log_message(LOG_LEVEL_WARNING, "Unknown --groupmap name on receiver: %s", map[i].to_name);
|
||||
continue;
|
||||
}
|
||||
*out_to = (int32_t)gr->gr_gid;
|
||||
} else {
|
||||
struct passwd* pw = getpwnam(map[i].to_name);
|
||||
if (!pw) {
|
||||
log_message(LOG_LEVEL_WARNING, "Unknown --usermap name on receiver: %s", map[i].to_name);
|
||||
continue;
|
||||
}
|
||||
*out_to = (int32_t)pw->pw_uid;
|
||||
}
|
||||
} else {
|
||||
*out_to = map[i].to;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Resolve the owner side from the negotiated policy. Sets *out and returns
|
||||
* true when an owner-affecting request is active (a usermap, --chown USER, or
|
||||
* -o/--owner); returns false (leaving *out untouched) when the owner side is
|
||||
* not requested, so callers can pass (uid_t)-1 to fchown and leave it as-is.
|
||||
* --numeric-ids only changes the RESOLUTION (raw id instead of a name lookup);
|
||||
* it never makes the side requested. */
|
||||
static bool identity_resolve_owner(int32_t source_uid, uid_t* out) {
|
||||
if (!(g_identity.chown_uid_set || g_identity.preserve_owner || g_identity.usermap_count > 0))
|
||||
return false;
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, false,
|
||||
&target)) {
|
||||
*out = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target;
|
||||
} else if (g_identity.chown_uid_set) {
|
||||
*out = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
*out = (uid_t)source_uid;
|
||||
} else {
|
||||
/* Best-effort name mapping against the receiver's own database. When the
|
||||
* transmitted (numeric) id has no name here, fall back to the raw numeric id
|
||||
* so -o still preserves the source owner. */
|
||||
struct passwd* pw = getpwuid((uid_t)source_uid);
|
||||
if (pw) {
|
||||
const struct passwd* mapped = getpwnam(pw->pw_name);
|
||||
*out = mapped ? mapped->pw_uid : (uid_t)source_uid;
|
||||
} else {
|
||||
*out = (uid_t)source_uid;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Group-side counterpart of identity_resolve_owner(). */
|
||||
static bool identity_resolve_group(int32_t source_gid, gid_t* out) {
|
||||
if (!(g_identity.chown_gid_set || g_identity.preserve_group || g_identity.groupmap_count > 0))
|
||||
return false;
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, true,
|
||||
&target)) {
|
||||
*out = target == IDENTITY_CURRENT ? getegid() : (gid_t)target;
|
||||
} else if (g_identity.chown_gid_set) {
|
||||
*out = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
*out = (gid_t)source_gid;
|
||||
} else {
|
||||
struct group* gr = getgrgid((gid_t)source_gid);
|
||||
if (gr) {
|
||||
const struct group* mapped = getgrnam(gr->gr_name);
|
||||
*out = mapped ? mapped->gr_gid : (gid_t)source_gid;
|
||||
} else {
|
||||
*out = (gid_t)source_gid;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Resolve the target ownership from the negotiated policy against the entry's
|
||||
* current stat. Shared by the fd (regular file) and no-follow (symlink) apply
|
||||
* paths. Returns false when no side is to be changed. */
|
||||
static bool identity_resolve_targets(const struct stat* st, int32_t source_uid, int32_t source_gid,
|
||||
uid_t* out_uid, gid_t* out_gid) {
|
||||
bool set_uid = false;
|
||||
bool set_gid = false;
|
||||
uid_t uid = 0;
|
||||
gid_t gid = 0;
|
||||
|
||||
/* --copy-as (P7 Wave E) has the highest priority: it forces BOTH the owner
|
||||
* and group of every written entry to the requested ids, beating usermap /
|
||||
* groupmap / --chown / --numeric-ids and the best-effort name lookup. Only
|
||||
* skip when the entry already carries exactly those ids. */
|
||||
if (g_identity.copy_as_set) {
|
||||
uid = (uid_t)g_identity.copy_as_uid;
|
||||
gid = (gid_t)g_identity.copy_as_gid;
|
||||
uid_t uid = (uid_t)g_identity.copy_as_uid;
|
||||
gid_t gid = (gid_t)g_identity.copy_as_gid;
|
||||
if (st->st_uid == uid && st->st_gid == gid)
|
||||
return false;
|
||||
*out_uid = uid;
|
||||
@@ -615,67 +921,52 @@ static bool identity_resolve_targets(const struct stat* st, int32_t source_uid,
|
||||
return true;
|
||||
}
|
||||
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, &target)) {
|
||||
uid = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target;
|
||||
set_uid = true;
|
||||
} else if (g_identity.chown_uid_set) {
|
||||
uid = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid;
|
||||
set_uid = true;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
uid = (uid_t)source_uid;
|
||||
set_uid = true;
|
||||
} else {
|
||||
/* Best-effort name mapping against the receiver's own database: if the
|
||||
* transmitted (numeric) id resolves to a name present on this machine,
|
||||
* re-resolve it. On a shared-account host this is the identity operation;
|
||||
* when the id has no name here, the user side is left alone. */
|
||||
struct passwd* pw = getpwuid((uid_t)source_uid);
|
||||
if (pw) {
|
||||
const struct passwd* mapped = getpwnam(pw->pw_name);
|
||||
if (mapped) {
|
||||
uid = mapped->pw_uid;
|
||||
set_uid = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, &target)) {
|
||||
gid = target == IDENTITY_CURRENT ? getegid() : (gid_t)target;
|
||||
set_gid = true;
|
||||
} else if (g_identity.chown_gid_set) {
|
||||
gid = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid;
|
||||
set_gid = true;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
gid = (gid_t)source_gid;
|
||||
set_gid = true;
|
||||
} else {
|
||||
struct group* gr = getgrgid((gid_t)source_gid);
|
||||
if (gr) {
|
||||
const struct group* mapped = getgrnam(gr->gr_name);
|
||||
if (mapped) {
|
||||
gid = mapped->gr_gid;
|
||||
set_gid = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!set_uid && !set_gid)
|
||||
/* Each side is resolved independently: -o/-g and the explicit identity flags
|
||||
* request the owner/group respectively, and a side that is NOT requested must
|
||||
* be left exactly as it is (`-1` to fchown on that side). This is what lets
|
||||
* plain -g change only the group, or -o only the owner. */
|
||||
uid_t uid = (uid_t)-1;
|
||||
gid_t gid = (gid_t)-1;
|
||||
bool owner_requested = identity_resolve_owner(source_uid, &uid);
|
||||
bool group_requested = identity_resolve_group(source_gid, &gid);
|
||||
if (!owner_requested && !group_requested)
|
||||
return false;
|
||||
/* An unset side keeps the file's current id so the other side can change. */
|
||||
if (!set_uid)
|
||||
uid = st->st_uid;
|
||||
if (!set_gid)
|
||||
gid = st->st_gid;
|
||||
/* Only change ownership when the target differs (avoid needless syscalls and
|
||||
* any chance of clearing setuid/setgid on an already-correct entry). */
|
||||
if (st->st_uid == uid && st->st_gid == gid)
|
||||
|
||||
/* Only change ownership when a requested side actually differs (avoid
|
||||
* needless syscalls and any chance of clearing setuid/setgid on an
|
||||
* already-correct entry). */
|
||||
bool changed = (owner_requested && uid != st->st_uid) || (group_requested && gid != st->st_gid);
|
||||
if (!changed)
|
||||
return false;
|
||||
*out_uid = uid;
|
||||
*out_gid = gid;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* --fake-super storage resolution: the receiver records the ownership it WOULD
|
||||
* have applied. A requested side uses the resolved mapping (--copy-as /
|
||||
* usermap / --chown / -o/-g, with --numeric-ids as the raw-id modifier); a side
|
||||
* that was not requested keeps the source's own id, so a plain --fake-super run
|
||||
* records the source owner untouched. */
|
||||
void identity_resolve_storage_ids(int32_t source_uid, int32_t source_gid, uint32_t* out_uid,
|
||||
uint32_t* out_gid) {
|
||||
if (g_identity.copy_as_set) {
|
||||
*out_uid = (uint32_t)g_identity.copy_as_uid;
|
||||
*out_gid = (uint32_t)g_identity.copy_as_gid;
|
||||
return;
|
||||
}
|
||||
uid_t uid = (uid_t)source_uid;
|
||||
gid_t gid = (gid_t)source_gid;
|
||||
uid_t resolved_uid;
|
||||
gid_t resolved_gid;
|
||||
if (identity_resolve_owner(source_uid, &resolved_uid))
|
||||
uid = resolved_uid;
|
||||
if (identity_resolve_group(source_gid, &resolved_gid))
|
||||
gid = resolved_gid;
|
||||
*out_uid = (uint32_t)uid;
|
||||
*out_gid = (uint32_t)gid;
|
||||
}
|
||||
|
||||
static void identity_log_chown_failure(const char* what, uid_t uid, gid_t gid) {
|
||||
/* EPERM/EACCES are expected when the receiver is not privileged (e.g. the CI
|
||||
* `nobody` user): warn and continue, never abort the transfer. Any other
|
||||
@@ -710,8 +1001,12 @@ bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) {
|
||||
/* Ownership application is OFF unless the client requested an identity flag.
|
||||
* This is the controlled gate: a default (or plain -M) transfer never changes
|
||||
* ownership, byte-for-byte preserving FastSync's existing behavior. --no-super
|
||||
* additionally forbids it even when the receiver is root. */
|
||||
if (!identity_active_enabled() || !privilege_super_permitted() || fd < 0)
|
||||
* additionally forbids it even when the receiver is root. --fake-super never
|
||||
* performs a REAL chown: that would defeat the point of the flag (record the
|
||||
* source ownership on an unprivileged receiver for a later privileged
|
||||
* restore); the resolved ownership is stored in the reserved xattr instead by
|
||||
* fake_super_store_fd(). */
|
||||
if (!identity_active_enabled() || g_identity.fake_super || !privilege_super_permitted() || fd < 0)
|
||||
return true;
|
||||
struct stat st;
|
||||
if (fstat(fd, &st) != 0)
|
||||
@@ -731,7 +1026,8 @@ bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) {
|
||||
|
||||
bool identity_apply_ownership_link(int parent_fd, const char* leaf, int32_t source_uid,
|
||||
int32_t source_gid) {
|
||||
if (!identity_active_enabled() || !privilege_super_permitted() || parent_fd < 0 || !leaf)
|
||||
if (!identity_active_enabled() || g_identity.fake_super || !privilege_super_permitted() ||
|
||||
parent_fd < 0 || !leaf)
|
||||
return true;
|
||||
struct stat st;
|
||||
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0)
|
||||
|
||||
+36
-7
@@ -80,16 +80,37 @@ void identity_clear_active(void);
|
||||
* snapshot. Ownership stays OFF ("do not apply") for every transfer that
|
||||
* requests none of them, preserving FastSync's existing behavior. --super /
|
||||
* --no-super alone does NOT enable ownership; an explicit identity flag
|
||||
* (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) is required. */
|
||||
* (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) or a
|
||||
* preserve-source -o/--owner / -g/--group request is required. */
|
||||
bool identity_active_enabled(void);
|
||||
|
||||
/* Pure, config-only predicate: true when the client requested ANY
|
||||
* client-chosen ownership or super-user activity (--numeric-ids, --chown,
|
||||
* --usermap/--groupmap, --copy-as, --fake-super, or an explicit --super). Used
|
||||
* by the daemon module gate to decide whether a module's per-module opt-in is
|
||||
* required; it never reads the per-connection snapshot. */
|
||||
/* Per-side predicates over the ACTIVE per-connection snapshot (call
|
||||
* identity_set_active() first). They mirror the owner_requested /
|
||||
* group_requested conditions inside identity_resolve_targets() exactly, so
|
||||
* callers that must apply only one side (e.g. the --fake-super owner replay)
|
||||
* can pass (uid_t)-1 / (gid_t)-1 for the side that was NOT requested and leave
|
||||
* it untouched. The owner side is requested by --copy-as, --chown USER,
|
||||
* --numeric-ids, -o/--owner, or a non-empty --usermap; the group side by
|
||||
* --copy-as, --chown :GROUP, --numeric-ids, -g/--group, or a non-empty
|
||||
* --groupmap. */
|
||||
bool identity_owner_requested(void);
|
||||
bool identity_group_requested(void);
|
||||
|
||||
/* Pure, config-only predicate: true when the client requested ANY client-chosen
|
||||
* ownership or super-user activity (--numeric-ids, --chown, --usermap/--groupmap,
|
||||
* --copy-as, --fake-super, an explicit --super, or a preserve-source -o/-g).
|
||||
* General awareness only; the daemon module gate uses the narrower
|
||||
* identity_explicit_ownership_requested() below. Never reads the snapshot. */
|
||||
bool identity_ownership_requested(const Config* config);
|
||||
|
||||
/* Pure, config-only predicate for the narrow set that lets the CLIENT choose an
|
||||
* arbitrary owner/group: --numeric-ids, --chown, --usermap/--groupmap,
|
||||
* --copy-as, --fake-super, or an explicit --super. Deliberately EXCLUDES a
|
||||
* plain -o/--owner / -g/--group (or -a) preserve-source request, which the
|
||||
* daemon gate handles by forcing super-user ownership activity off rather than
|
||||
* refusing the whole transfer. Never reads the snapshot. */
|
||||
bool identity_explicit_ownership_requested(const Config* config);
|
||||
|
||||
/* Apply the negotiated ownership to an already-written file descriptor.
|
||||
* source_uid/source_gid are the transmitted numeric ids. Resolution order:
|
||||
* --copy-as (highest priority, forces both ids), then a matching
|
||||
@@ -106,6 +127,14 @@ bool identity_ownership_requested(const Config* config);
|
||||
* is active returns true. */
|
||||
bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid);
|
||||
|
||||
/* Resolve the ownership that --fake-super should RECORD in the reserved xattr
|
||||
* (rather than chown for real). A requested side (--copy-as / usermap /
|
||||
* --chown / -o / -g, with --numeric-ids as the raw-id modifier) yields the
|
||||
* resolved target; a side that was not requested keeps the transmitted source
|
||||
* id. Must be called after identity_set_active(). */
|
||||
void identity_resolve_storage_ids(int32_t source_uid, int32_t source_gid, uint32_t* out_uid,
|
||||
uint32_t* out_gid);
|
||||
|
||||
/* P7 Wave D: the no-follow (symlink) counterpart. Resolves the same
|
||||
* usermap/groupmap/chown/numeric-ids/copy-as policy but applies it with
|
||||
* fchownat(..., AT_SYMLINK_NOFOLLOW) so a symlink's own ownership is changed
|
||||
@@ -128,6 +157,6 @@ bool identity_wire_valid(const Config* config);
|
||||
* best-effort behavior where an unprivileged attempt is refused by the kernel
|
||||
* and skipped. Neither EVER elevates privileges. */
|
||||
bool privilege_super_permitted(void);
|
||||
bool privilege_super_mode_permitted(int mode);
|
||||
bool privilege_super_mode_permitted(SuperMode mode);
|
||||
|
||||
#endif
|
||||
+72
-26
@@ -3,7 +3,9 @@
|
||||
#include <stdbool.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <threads.h>
|
||||
#include <time.h>
|
||||
|
||||
static const char* log_level_strings[] = {"DEBUG", "INFO", "WARN", "ERROR"};
|
||||
@@ -15,6 +17,18 @@ static FILE* log_fp = NULL;
|
||||
static _Thread_local bool eight_bit_output;
|
||||
static LogStderrMode stderr_mode = LOG_STDERR_ERRORS;
|
||||
|
||||
/* Serializes access to log_fp and makes each emitted line atomic: the
|
||||
* timestamp prefix, formatted body, and trailing newline are written as one
|
||||
* critical section so concurrent threads cannot interleave partial lines.
|
||||
* Initialized lazily (matching the protocol.c bw_mutex idiom) because logging
|
||||
* can happen before main() installs any synchronization. */
|
||||
static mtx_t log_mutex;
|
||||
static once_flag log_mutex_once = ONCE_FLAG_INIT;
|
||||
|
||||
static void log_mutex_init(void) {
|
||||
mtx_init(&log_mutex, mtx_plain);
|
||||
}
|
||||
|
||||
void set_log_level(LogLevel level) {
|
||||
current_log_level = level;
|
||||
}
|
||||
@@ -41,7 +55,10 @@ uint32_t get_log_info_flags(void) {
|
||||
}
|
||||
|
||||
void log_set_file(FILE* fp) {
|
||||
call_once(&log_mutex_once, log_mutex_init);
|
||||
mtx_lock(&log_mutex);
|
||||
log_fp = fp;
|
||||
mtx_unlock(&log_mutex);
|
||||
}
|
||||
|
||||
void log_set_8_bit_output(bool enabled) {
|
||||
@@ -60,13 +77,48 @@ LogStderrMode log_get_stderr_mode(void) {
|
||||
return stderr_mode;
|
||||
}
|
||||
|
||||
static inline void write_message(FILE* dest_io, LogLevel log_level, struct tm t, const char* format,
|
||||
/* Format one complete log line (timestamp prefix + body + newline) into a
|
||||
* freshly allocated buffer. This is pure CPU/malloc work and must happen
|
||||
* OUTSIDE the log mutex: the mutex only guards the log_fp pointer, so a
|
||||
* stalled stderr/stdout pipe cannot block every logging thread. Returns NULL
|
||||
* on allocation/formatting failure. */
|
||||
static char* format_log_line(LogLevel log_level, const struct tm* t, const char* format,
|
||||
va_list args) {
|
||||
fprintf(dest_io, "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t.tm_year + 1900, t.tm_mon + 1,
|
||||
t.tm_mday, t.tm_hour, t.tm_min, t.tm_sec, log_level_strings[log_level]);
|
||||
char prefix[64];
|
||||
int prefix_len = snprintf(
|
||||
prefix, sizeof(prefix), "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t->tm_year + 1900,
|
||||
t->tm_mon + 1, t->tm_mday, t->tm_hour, t->tm_min, t->tm_sec, log_level_strings[log_level]);
|
||||
if (prefix_len < 0 || prefix_len >= (int)sizeof(prefix))
|
||||
return NULL;
|
||||
va_list copy;
|
||||
va_copy(copy, args);
|
||||
int body_len = vsnprintf(NULL, 0, format, copy);
|
||||
va_end(copy);
|
||||
if (body_len < 0)
|
||||
return NULL;
|
||||
size_t total = (size_t)prefix_len + (size_t)body_len;
|
||||
char* line = malloc(total + 2); /* body bytes + '\n' + NUL */
|
||||
if (!line)
|
||||
return NULL;
|
||||
memcpy(line, prefix, (size_t)prefix_len);
|
||||
vsnprintf(line + prefix_len, (size_t)body_len + 1, format, args);
|
||||
line[total] = '\n';
|
||||
line[total + 1] = '\0';
|
||||
return line;
|
||||
}
|
||||
|
||||
vfprintf(dest_io, format, args);
|
||||
fprintf(dest_io, "\n");
|
||||
/* Write an already-formatted line to the console and, if configured, the log
|
||||
* file. Only the log_fp pointer is read under the mutex (so log_set_file /
|
||||
* config_delete cannot free it while it is in use); the single console fputs
|
||||
* runs unlocked but is internally atomic per stdio stream. */
|
||||
static void emit_log_line(FILE* console, const char* line) {
|
||||
fputs(line, console);
|
||||
call_once(&log_mutex_once, log_mutex_init);
|
||||
mtx_lock(&log_mutex);
|
||||
FILE* file = log_fp;
|
||||
if (file)
|
||||
fputs(line, file);
|
||||
mtx_unlock(&log_mutex);
|
||||
}
|
||||
|
||||
void log_message(LogLevel log_level, const char* format, ...) {
|
||||
@@ -86,14 +138,12 @@ void log_message(LogLevel log_level, const char* format, ...) {
|
||||
|
||||
va_list args;
|
||||
va_start(args, format);
|
||||
write_message(dest_io, log_level, t, format, args);
|
||||
char* line = format_log_line(log_level, &t, format, args);
|
||||
va_end(args);
|
||||
|
||||
if (log_fp) {
|
||||
va_start(args, format);
|
||||
write_message(log_fp, log_level, t, format, args);
|
||||
va_end(args);
|
||||
}
|
||||
if (!line)
|
||||
return;
|
||||
emit_log_line(dest_io, line);
|
||||
free(line);
|
||||
}
|
||||
|
||||
void log_debug_message(LogDebugFlag flag, const char* format, ...) {
|
||||
@@ -107,14 +157,12 @@ void log_debug_message(LogDebugFlag flag, const char* format, ...) {
|
||||
|
||||
va_list args;
|
||||
va_start(args, format);
|
||||
write_message(stdout, LOG_LEVEL_DEBUG, t, format, args);
|
||||
char* line = format_log_line(LOG_LEVEL_DEBUG, &t, format, args);
|
||||
va_end(args);
|
||||
|
||||
if (log_fp) {
|
||||
va_start(args, format);
|
||||
write_message(log_fp, LOG_LEVEL_DEBUG, t, format, args);
|
||||
va_end(args);
|
||||
}
|
||||
if (!line)
|
||||
return;
|
||||
emit_log_line(stdout, line);
|
||||
free(line);
|
||||
}
|
||||
|
||||
void log_info_message(LogInfoFlag flag, const char* format, ...) {
|
||||
@@ -129,14 +177,12 @@ void log_info_message(LogInfoFlag flag, const char* format, ...) {
|
||||
|
||||
va_list args;
|
||||
va_start(args, format);
|
||||
write_message(stdout, LOG_LEVEL_INFO, t, format, args);
|
||||
char* line = format_log_line(LOG_LEVEL_INFO, &t, format, args);
|
||||
va_end(args);
|
||||
|
||||
if (log_fp) {
|
||||
va_start(args, format);
|
||||
write_message(log_fp, LOG_LEVEL_INFO, t, format, args);
|
||||
va_end(args);
|
||||
}
|
||||
if (!line)
|
||||
return;
|
||||
emit_log_line(stdout, line);
|
||||
free(line);
|
||||
}
|
||||
|
||||
void log_perror(const char* context) {
|
||||
|
||||
+149
-194
@@ -90,63 +90,65 @@ void metadata_to_buf(char** buf, const FileMetadata* m) {
|
||||
*buf += sizeof(crtime_nsec);
|
||||
}
|
||||
|
||||
FileMetadata* metadata_from_buf(char** buf) {
|
||||
FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len) {
|
||||
if (buf == NULL || len < sizeof(int32_t))
|
||||
return NULL;
|
||||
int32_t present;
|
||||
memcpy(&present, *buf, sizeof(present));
|
||||
*buf += sizeof(present);
|
||||
if (present != 0 && present != 1)
|
||||
memcpy(&present, buf, sizeof(present));
|
||||
if (present != 1)
|
||||
return NULL;
|
||||
if (!present)
|
||||
if (len < sizeof(int32_t) + FILE_METADATA_WIRE_SIZE)
|
||||
return NULL;
|
||||
const uint8_t* cursor = buf + sizeof(int32_t);
|
||||
FileMetadata* m = protocol_alloc(sizeof(FileMetadata));
|
||||
if (m == NULL)
|
||||
return NULL;
|
||||
int32_t mode;
|
||||
memcpy(&mode, *buf, sizeof(mode));
|
||||
*buf += sizeof(mode);
|
||||
memcpy(&mode, cursor, sizeof(mode));
|
||||
cursor += sizeof(mode);
|
||||
m->mode = (mode_t)mode;
|
||||
int32_t uid;
|
||||
memcpy(&uid, *buf, sizeof(uid));
|
||||
*buf += sizeof(uid);
|
||||
memcpy(&uid, cursor, sizeof(uid));
|
||||
cursor += sizeof(uid);
|
||||
m->uid = (uid_t)uid;
|
||||
int32_t gid;
|
||||
memcpy(&gid, *buf, sizeof(gid));
|
||||
*buf += sizeof(gid);
|
||||
memcpy(&gid, cursor, sizeof(gid));
|
||||
cursor += sizeof(gid);
|
||||
m->gid = (gid_t)gid;
|
||||
int64_t mtime_sec;
|
||||
memcpy(&mtime_sec, *buf, sizeof(mtime_sec));
|
||||
*buf += sizeof(mtime_sec);
|
||||
memcpy(&mtime_sec, cursor, sizeof(mtime_sec));
|
||||
cursor += sizeof(mtime_sec);
|
||||
m->mtime_sec = (time_t)mtime_sec;
|
||||
int64_t mtime_nsec;
|
||||
memcpy(&mtime_nsec, *buf, sizeof(mtime_nsec));
|
||||
*buf += sizeof(mtime_nsec);
|
||||
memcpy(&mtime_nsec, cursor, sizeof(mtime_nsec));
|
||||
cursor += sizeof(mtime_nsec);
|
||||
m->mtime_nsec = (long)mtime_nsec;
|
||||
int32_t atime_valid;
|
||||
memcpy(&atime_valid, *buf, sizeof(atime_valid));
|
||||
*buf += sizeof(atime_valid);
|
||||
memcpy(&atime_valid, cursor, sizeof(atime_valid));
|
||||
cursor += sizeof(atime_valid);
|
||||
int64_t atime_sec;
|
||||
memcpy(&atime_sec, *buf, sizeof(atime_sec));
|
||||
*buf += sizeof(atime_sec);
|
||||
memcpy(&atime_sec, cursor, sizeof(atime_sec));
|
||||
cursor += sizeof(atime_sec);
|
||||
int64_t atime_nsec;
|
||||
memcpy(&atime_nsec, *buf, sizeof(atime_nsec));
|
||||
*buf += sizeof(atime_nsec);
|
||||
memcpy(&atime_nsec, cursor, sizeof(atime_nsec));
|
||||
cursor += sizeof(atime_nsec);
|
||||
int32_t crtime_valid;
|
||||
memcpy(&crtime_valid, *buf, sizeof(crtime_valid));
|
||||
*buf += sizeof(crtime_valid);
|
||||
memcpy(&crtime_valid, cursor, sizeof(crtime_valid));
|
||||
cursor += sizeof(crtime_valid);
|
||||
int64_t crtime_sec;
|
||||
memcpy(&crtime_sec, *buf, sizeof(crtime_sec));
|
||||
*buf += sizeof(crtime_sec);
|
||||
memcpy(&crtime_sec, cursor, sizeof(crtime_sec));
|
||||
cursor += sizeof(crtime_sec);
|
||||
int64_t crtime_nsec;
|
||||
memcpy(&crtime_nsec, *buf, sizeof(crtime_nsec));
|
||||
*buf += sizeof(crtime_nsec);
|
||||
memcpy(&crtime_nsec, cursor, sizeof(crtime_nsec));
|
||||
cursor += sizeof(crtime_nsec);
|
||||
m->atime_valid = atime_valid != 0;
|
||||
m->atime_sec = (time_t)atime_sec;
|
||||
m->atime_nsec = (long)atime_nsec;
|
||||
m->crtime_valid = crtime_valid != 0;
|
||||
m->crtime_sec = (time_t)crtime_sec;
|
||||
m->crtime_nsec = (long)crtime_nsec;
|
||||
if (present != 1 || mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 ||
|
||||
gid < 0 || atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 ||
|
||||
if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 ||
|
||||
atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 ||
|
||||
(atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) ||
|
||||
(crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) {
|
||||
free(m);
|
||||
@@ -157,33 +159,17 @@ FileMetadata* metadata_from_buf(char** buf) {
|
||||
|
||||
bool metadata_send(int file_descriptor, const FileMetadata* m) {
|
||||
if (m == NULL) {
|
||||
int32_t zero = 0;
|
||||
return send_n_data(file_descriptor, &zero, sizeof(zero));
|
||||
int32_t absent = 0;
|
||||
return send_n_data(file_descriptor, &absent, sizeof(absent));
|
||||
}
|
||||
int32_t present = 1;
|
||||
int32_t mode = (int32_t)m->mode;
|
||||
int32_t uid = (int32_t)m->uid;
|
||||
int32_t gid = (int32_t)m->gid;
|
||||
int64_t mtime_sec = (int64_t)m->mtime_sec;
|
||||
int64_t mtime_nsec = (int64_t)m->mtime_nsec;
|
||||
int32_t atime_valid = m->atime_valid ? 1 : 0;
|
||||
int64_t atime_sec = (int64_t)m->atime_sec;
|
||||
int64_t atime_nsec = (int64_t)m->atime_nsec;
|
||||
int32_t crtime_valid = m->crtime_valid ? 1 : 0;
|
||||
int64_t crtime_sec = (int64_t)m->crtime_sec;
|
||||
int64_t crtime_nsec = (int64_t)m->crtime_nsec;
|
||||
return send_n_data(file_descriptor, &present, sizeof(present)) &&
|
||||
send_n_data(file_descriptor, &mode, sizeof(mode)) &&
|
||||
send_n_data(file_descriptor, &uid, sizeof(uid)) &&
|
||||
send_n_data(file_descriptor, &gid, sizeof(gid)) &&
|
||||
send_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec)) &&
|
||||
send_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec)) &&
|
||||
send_n_data(file_descriptor, &atime_valid, sizeof(atime_valid)) &&
|
||||
send_n_data(file_descriptor, &atime_sec, sizeof(atime_sec)) &&
|
||||
send_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec)) &&
|
||||
send_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid)) &&
|
||||
send_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec)) &&
|
||||
send_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec));
|
||||
/* One packed frame (protocol 2.20.0): the int32 present flag followed by the
|
||||
fixed FILE_METADATA_WIRE_SIZE-byte field record. metadata_to_buf() emits
|
||||
exactly that layout (present + fields), so build it once and write the
|
||||
whole record in a single call instead of one frame per field. */
|
||||
char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE];
|
||||
char* cursor = packed;
|
||||
metadata_to_buf(&cursor, m);
|
||||
return send_n_data(file_descriptor, packed, sizeof(packed));
|
||||
}
|
||||
|
||||
FileMetadata* metadata_receive(int file_descriptor, int* ok) {
|
||||
@@ -203,130 +189,78 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) {
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
FileMetadata* m = protocol_alloc(sizeof(FileMetadata));
|
||||
/* Rebuild the packed record metadata_from_buf() expects: the present flag we
|
||||
just read, followed by exactly FILE_METADATA_WIRE_SIZE field bytes. */
|
||||
char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE];
|
||||
memcpy(packed, &present, sizeof(present));
|
||||
if (!receive_n_data(file_descriptor, packed + sizeof(present), FILE_METADATA_WIRE_SIZE)) {
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
FileMetadata* m = metadata_from_buf((const uint8_t*)packed, sizeof(packed));
|
||||
if (m == NULL) {
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int32_t mode;
|
||||
if (!receive_n_data(file_descriptor, &mode, sizeof(mode))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->mode = (mode_t)mode;
|
||||
int32_t uid;
|
||||
if (!receive_n_data(file_descriptor, &uid, sizeof(uid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->uid = (uid_t)uid;
|
||||
int32_t gid;
|
||||
if (!receive_n_data(file_descriptor, &gid, sizeof(gid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->gid = (gid_t)gid;
|
||||
int64_t mtime_sec;
|
||||
if (!receive_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->mtime_sec = (time_t)mtime_sec;
|
||||
int64_t mtime_nsec;
|
||||
if (!receive_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->mtime_nsec = (long)mtime_nsec;
|
||||
int32_t atime_valid;
|
||||
if (!receive_n_data(file_descriptor, &atime_valid, sizeof(atime_valid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t atime_sec;
|
||||
if (!receive_n_data(file_descriptor, &atime_sec, sizeof(atime_sec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t atime_nsec;
|
||||
if (!receive_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int32_t crtime_valid;
|
||||
if (!receive_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t crtime_sec;
|
||||
if (!receive_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t crtime_nsec;
|
||||
if (!receive_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->atime_valid = atime_valid != 0;
|
||||
m->atime_sec = (time_t)atime_sec;
|
||||
m->atime_nsec = (long)atime_nsec;
|
||||
m->crtime_valid = crtime_valid != 0;
|
||||
m->crtime_sec = (time_t)crtime_sec;
|
||||
m->crtime_nsec = (long)crtime_nsec;
|
||||
if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 ||
|
||||
atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 ||
|
||||
(atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) ||
|
||||
(crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
if (ok)
|
||||
*ok = 1;
|
||||
return m;
|
||||
}
|
||||
|
||||
static mode_t metadata_mode(const FileMetadata* metadata, mode_t current_mode,
|
||||
bool preserve_executability) {
|
||||
bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrPolicy policy,
|
||||
mode_t* out_mode) {
|
||||
const mode_t execute_bits = S_IXUSR | S_IXGRP | S_IXOTH;
|
||||
if (preserve_executability)
|
||||
return (current_mode & 0777 & ~execute_bits) | (metadata->mode & execute_bits);
|
||||
return metadata->mode & 0777 & ~(S_IWGRP | S_IWOTH);
|
||||
if (policy.perms) {
|
||||
/* rsync --perms copies the source's permission and special bits exactly,
|
||||
* including group/other write and setuid/setgid/sticky. The kernel may
|
||||
* still clear setgid when the receiver is not in the file's group; the
|
||||
* caller logs a failed chmod rather than silently masking the bits here. */
|
||||
*out_mode = source_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
return true;
|
||||
}
|
||||
if (policy.executability) {
|
||||
/* -E/--executability (rsync 3.4 rule): do NOT copy the source's execute
|
||||
* bits per class. If the source is executable at all, derive the execute
|
||||
* bits from the DESTINATION's own read bits (so a class that can read may
|
||||
* execute); otherwise clear every execute bit. This runs on the
|
||||
* destination-derived base (pre-existing dest mode, or source&~umask for a
|
||||
* new file), and leaves the special bits untouched. --perms wins when both
|
||||
* are set (handled above). */
|
||||
mode_t base = current_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
if (source_mode & 0111)
|
||||
*out_mode = base | ((base & 0444) >> 2);
|
||||
else
|
||||
*out_mode = base & ~execute_bits;
|
||||
return true;
|
||||
}
|
||||
/* Neither requested: no source mode is applied at all. */
|
||||
return false;
|
||||
}
|
||||
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool preserve_executability) {
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config) {
|
||||
FileAttrPolicy policy = {false, false, false, false};
|
||||
if (config) {
|
||||
policy.perms = config->preserve_perms;
|
||||
policy.times = config->preserve_times;
|
||||
policy.atimes = config->preserve_atimes;
|
||||
policy.executability = config->use_executability;
|
||||
}
|
||||
return policy;
|
||||
}
|
||||
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata, FileAttrPolicy policy) {
|
||||
if (metadata == NULL)
|
||||
return;
|
||||
bool apply_mode = false;
|
||||
mode_t safe_mode = 0;
|
||||
if (policy.perms || policy.executability) {
|
||||
struct stat current;
|
||||
mode_t current_mode = stat(path, ¤t) == 0 ? current.st_mode : 0;
|
||||
mode_t safe_mode = metadata_mode(metadata, current_mode, preserve_executability);
|
||||
if (chmod(path, safe_mode) != 0) {
|
||||
apply_mode = metadata_mode_for_policy(metadata->mode, current_mode, policy, &safe_mode);
|
||||
}
|
||||
if (apply_mode && chmod(path, safe_mode) != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Failed to chmod %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
|
||||
@@ -334,21 +268,17 @@ void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
}
|
||||
/* Never apply client-supplied ownership. The descriptor API below is the
|
||||
receiver write path; retain this legacy API only for compatibility. */
|
||||
struct timespec times[2];
|
||||
times[0].tv_sec = 0;
|
||||
times[0].tv_nsec = UTIME_OMIT;
|
||||
if (policy.times || (policy.atimes && metadata->atime_valid)) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
if (metadata->atime_valid) {
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
if (metadata->crtime_valid) {
|
||||
log_message(LOG_LEVEL_DEBUG,
|
||||
"crtime (birth time) %lld.%09ld transmitted for %s but not applied: no portable "
|
||||
"setter exists",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec, path);
|
||||
}
|
||||
if (utimensat(AT_FDCWD, path, times, 0) != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s",
|
||||
@@ -356,9 +286,16 @@ void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
free(escaped_path);
|
||||
}
|
||||
}
|
||||
if (metadata->crtime_valid) {
|
||||
log_message(LOG_LEVEL_DEBUG,
|
||||
"crtime (birth time) %lld.%09ld transmitted for %s but not applied: no portable "
|
||||
"setter exists",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec, path);
|
||||
}
|
||||
}
|
||||
|
||||
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool omit_link_times) {
|
||||
FileAttrPolicy policy, bool omit_link_times) {
|
||||
if (path == NULL || metadata == NULL)
|
||||
return !identity_copy_as_active();
|
||||
char* leaf = NULL;
|
||||
@@ -371,18 +308,25 @@ bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadat
|
||||
best-effort. */
|
||||
bool owned = identity_apply_ownership_link(parent_fd, leaf, (int32_t)metadata->uid,
|
||||
(int32_t)metadata->gid);
|
||||
/* Symlink mode: not settable on Linux (fchmodat AT_SYMLINK_NOFOLLOW returns
|
||||
EOPNOTSUPP/ENOTSUP); attempt it for platforms that support it and quietly
|
||||
ignore the unsupported case so the transfer never fails over it. */
|
||||
mode_t link_mode = metadata->mode & 0777;
|
||||
/* Symlink mode: only when -p is in effect. It is not settable on Linux
|
||||
(fchmodat AT_SYMLINK_NOFOLLOW returns EOPNOTSUPP/ENOTSUP); attempt it for
|
||||
platforms that support it and quietly ignore the unsupported case so the
|
||||
transfer never fails over it. */
|
||||
if (policy.perms) {
|
||||
mode_t link_mode = metadata->mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
if (fchmodat(parent_fd, leaf, link_mode, AT_SYMLINK_NOFOLLOW) != 0 && errno != EOPNOTSUPP &&
|
||||
errno != ENOTSUP && errno != ENOSYS) {
|
||||
log_message(LOG_LEVEL_DEBUG, "Could not set symlink mode on %s: %s", path, strerror(errno));
|
||||
}
|
||||
if (!omit_link_times) {
|
||||
}
|
||||
if (!omit_link_times && (policy.times || (policy.atimes && metadata->atime_valid))) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
|
||||
if (metadata->atime_valid) {
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
@@ -398,19 +342,13 @@ bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadat
|
||||
return owned;
|
||||
}
|
||||
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability) {
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, FileAttrPolicy policy) {
|
||||
if (fd < 0 || metadata == NULL)
|
||||
return metadata == NULL;
|
||||
bool ok = true;
|
||||
struct stat current;
|
||||
if (fstat(fd, ¤t) != 0)
|
||||
return false;
|
||||
mode_t safe_mode = metadata_mode(metadata, current.st_mode, preserve_executability);
|
||||
if (fchmod(fd, safe_mode) != 0)
|
||||
ok = false;
|
||||
/* Client uid/gid values are deliberately not authoritative UNLESS the client
|
||||
explicitly opted in with an identity flag (--numeric-ids / --usermap /
|
||||
--groupmap / --chown). identity_apply_ownership is the controlled,
|
||||
--groupmap / --chown / -o/-g). identity_apply_ownership is the controlled,
|
||||
privilege-gated path: it consults the negotiated policy, resolves the
|
||||
target ids, and applies them via an fd-relative fchown() that is confined
|
||||
to the just-written file (EPERM/EACCES are logged, never fatal) -- EXCEPT
|
||||
@@ -418,14 +356,19 @@ bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserv
|
||||
marks this entry as failed instead of reporting a wrong-owner write as
|
||||
success. With no identity flag set it is a no-op, so a default or plain -M
|
||||
transfer keeps FastSync's existing behavior of never applying client
|
||||
ownership. */
|
||||
ownership. Ownership runs BEFORE the mode because a chown clears
|
||||
setuid/setgid; rsync likewise chowns first and then restores the source
|
||||
mode (including its special bits). */
|
||||
if (!identity_apply_ownership(fd, (int32_t)metadata->uid, (int32_t)metadata->gid))
|
||||
ok = false;
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
|
||||
if (metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
if (policy.perms || policy.executability) {
|
||||
struct stat current;
|
||||
if (fstat(fd, ¤t) != 0)
|
||||
return false;
|
||||
mode_t safe_mode = 0;
|
||||
bool apply_mode = metadata_mode_for_policy(metadata->mode, current.st_mode, policy, &safe_mode);
|
||||
if (apply_mode && fchmod(fd, safe_mode) != 0)
|
||||
ok = false;
|
||||
}
|
||||
/* --crtimes captures and transmits the source birth time, but there is no
|
||||
* portable way to set a birth time (utimensat can only set atime/mtime), so
|
||||
@@ -437,7 +380,19 @@ bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserv
|
||||
"crtime (birth time) %lld.%09ld transmitted but not applied: no portable setter",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec);
|
||||
}
|
||||
if (policy.times || (policy.atimes && metadata->atime_valid)) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
if (futimens(fd, times) != 0)
|
||||
ok = false;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
+35
-9
@@ -2,7 +2,9 @@
|
||||
#define METADATA_H
|
||||
|
||||
#include "file.h"
|
||||
#include "file_attr.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
@@ -29,26 +31,50 @@
|
||||
|
||||
/* Size of metadata fields on wire, excluding the int32_t `present` field that
|
||||
* is always sent first. The total wire size for present metadata is
|
||||
* sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms). */
|
||||
* sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms).
|
||||
*
|
||||
* metadata_send()/metadata_receive() (protocol 2.20.0) frame the metadata as a
|
||||
* single packed record: one int32 present flag (0 = absent) followed, when
|
||||
* present, by exactly FILE_METADATA_WIRE_SIZE bytes of field data. This is the
|
||||
* same present+fields byte layout metadata_to_buf()/metadata_from_buf() use, so
|
||||
* the wire metadata is now one frame instead of one frame per field. */
|
||||
#define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 5 + sizeof(int64_t) * 6)
|
||||
|
||||
void metadata_to_buf(char** buf, const FileMetadata* m);
|
||||
FileMetadata* metadata_from_buf(char** buf);
|
||||
/* Decode one packed metadata record (an int32 present flag followed, when
|
||||
* present, by FILE_METADATA_WIRE_SIZE field bytes) from `buf`, which has `len`
|
||||
* readable bytes. Every read is bounds-checked against `len`, so the function
|
||||
* can never over-read the caller's buffer: a too-short record, an absent
|
||||
* (present == 0) record and a malformed record all return NULL. A successful
|
||||
* decode returns a heap-allocated FileMetadata owned by the caller. */
|
||||
FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len);
|
||||
bool metadata_send(int file_descriptor, const FileMetadata* m);
|
||||
FileMetadata* metadata_receive(int file_descriptor, int* ok);
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool preserve_executability);
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability);
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata, FileAttrPolicy policy);
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, FileAttrPolicy policy);
|
||||
|
||||
/* Shared mode-policy helper: the single source of truth for the receiver's
|
||||
* mode rule. Given a source mode and the destination's CURRENT mode, returns
|
||||
* true and stores the exact mode to apply in *out_mode when `policy` requests
|
||||
* a change, or false when it requests neither --perms nor --executability (the
|
||||
* caller then leaves the destination mode alone). --perms wins over -E; the
|
||||
* -E rule derives exec bits from the destination's read bits (rsync 3.4);
|
||||
* group/other write is never granted from a client-supplied mode. Shared by
|
||||
* file_restore_metadata_fd() and the --fake-super replay so the two cannot
|
||||
* diverge. */
|
||||
bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrPolicy policy,
|
||||
mode_t* out_mode);
|
||||
/* P7 Wave D: apply a SYMLINK's own metadata using no-follow primitives only
|
||||
* (utimensat/lchown/fchmodat with AT_SYMLINK_NOFOLLOW), confined fd-relative
|
||||
* under the authorized root. `omit_link_times` (-J/--omit-link-times)
|
||||
* suppresses the timestamps; the link's mode/ownership are still attempted
|
||||
* (ownership stays gated by the identity policy and by default is not applied).
|
||||
* under the authorized root. The link's mode is applied only when policy.perms;
|
||||
* policy.times (further suppressed by `omit_link_times` for -J) applies the
|
||||
* mtime with policy.atimes controlling the atime slot; ownership stays gated by
|
||||
* the identity policy and by default is not applied.
|
||||
* A null metadata or an unfollowable parent is a harmless no-op. Returns false
|
||||
* only when a REQUIRED --copy-as ownership application failed, so the caller can
|
||||
* report the entry as failed instead of claiming a wrong-owner success. */
|
||||
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool omit_link_times);
|
||||
FileAttrPolicy policy, bool omit_link_times);
|
||||
|
||||
/* Compare timestamps using rsync's whole-second modification window. */
|
||||
bool metadata_mtime_matches(time_t left_sec, long left_nsec, time_t right_sec, long right_nsec,
|
||||
|
||||
+92
-247
@@ -1,15 +1,16 @@
|
||||
#include "multiprocessing.h"
|
||||
#include "receiver.h"
|
||||
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "data.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -25,8 +26,12 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que
|
||||
context->queue_loader = queue_loader;
|
||||
context->scanner_done = false;
|
||||
context->loader_done = false;
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->manifest = NULL;
|
||||
context->excluded_paths = NULL;
|
||||
context->size_skipped_paths = NULL;
|
||||
context->synced_dirs = NULL;
|
||||
context->missing_args = NULL;
|
||||
context->scan_had_io_error = false;
|
||||
context->remove_source_files = NULL;
|
||||
@@ -41,6 +46,7 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que
|
||||
protocol_session_set_max_alloc(&context->allocation_session, config->max_alloc);
|
||||
context->dir_entries = NULL;
|
||||
context->dir_entries_mutex_init = false;
|
||||
context->delete_limit = false;
|
||||
int init = 0;
|
||||
if (config->use_metadata) {
|
||||
context->dir_entries = array_list_create(file_destroy);
|
||||
@@ -96,12 +102,97 @@ fail:
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void pipeline_context_sender_set_queue_byte_limit(PipelineContextSender* context,
|
||||
size_t max_bytes) {
|
||||
if (context == NULL)
|
||||
return;
|
||||
mtx_lock(&context->mutex_loader);
|
||||
context->max_queue_bytes = max_bytes;
|
||||
context->queued_bytes = 0;
|
||||
cnd_broadcast(&context->condition_not_full_loader);
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
}
|
||||
|
||||
size_t pipeline_context_sender_chunk_bytes(const Chunk* chunk) {
|
||||
if (chunk == NULL || chunk->items == NULL)
|
||||
return 0;
|
||||
size_t total = 0;
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
const File* file = chunk->items[i];
|
||||
if (file == NULL || file->data == NULL || file->data->data == NULL)
|
||||
continue;
|
||||
if (file->data->size > SIZE_MAX - total)
|
||||
return SIZE_MAX;
|
||||
total += file->data->size;
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
void pipeline_context_sender_note_bytes_released(PipelineContextSender* context,
|
||||
size_t released_bytes) {
|
||||
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
|
||||
return;
|
||||
mtx_lock(&context->mutex_loader);
|
||||
if (released_bytes >= context->queued_bytes)
|
||||
context->queued_bytes = 0;
|
||||
else
|
||||
context->queued_bytes -= released_bytes;
|
||||
cnd_signal(&context->condition_not_full_loader);
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
}
|
||||
|
||||
bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk* chunk) {
|
||||
if (context == NULL || chunk == NULL)
|
||||
return false;
|
||||
size_t chunk_bytes = pipeline_context_sender_chunk_bytes(chunk);
|
||||
mtx_lock(&context->mutex_loader);
|
||||
while (!atomic_load(&context->cancelled)) {
|
||||
bool blocked_by_count = queue_is_full(context->queue_loader);
|
||||
bool blocked_by_budget = false;
|
||||
if (context->max_queue_bytes > 0) {
|
||||
size_t budget = context->max_queue_bytes;
|
||||
size_t used = context->queued_bytes;
|
||||
if (used >= budget) {
|
||||
blocked_by_budget = true;
|
||||
} else if (chunk_bytes > budget - used) {
|
||||
/* A single payload larger than the whole budget is only admitted to an
|
||||
empty pipeline so the wait can never deadlock. */
|
||||
blocked_by_budget = used != 0;
|
||||
}
|
||||
}
|
||||
if (!blocked_by_count && !blocked_by_budget)
|
||||
break;
|
||||
cnd_wait(&context->condition_not_full_loader, &context->mutex_loader);
|
||||
}
|
||||
if (atomic_load(&context->cancelled)) {
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
chunk_destroy(chunk);
|
||||
return false;
|
||||
}
|
||||
if (!queue_enqueue(context->queue_loader, chunk)) {
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
chunk_destroy(chunk);
|
||||
return false;
|
||||
}
|
||||
context->queued_bytes += chunk_bytes;
|
||||
cnd_signal(&context->condition_not_empty_loader);
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
return true;
|
||||
}
|
||||
|
||||
void pipeline_context_sender_destroy(PipelineContextSender* context) {
|
||||
/* `config` is borrowed: the caller retains ownership and frees it after the
|
||||
pipeline has been destroyed (the worker threads are already joined, so no
|
||||
config access can outlive this call). */
|
||||
if (context->manifest) {
|
||||
array_list_delete(context->manifest);
|
||||
}
|
||||
if (context->excluded_paths)
|
||||
array_list_delete(context->excluded_paths);
|
||||
if (context->size_skipped_paths)
|
||||
array_list_delete(context->size_skipped_paths);
|
||||
if (context->synced_dirs)
|
||||
array_list_delete(context->synced_dirs);
|
||||
if (context->missing_args)
|
||||
array_list_delete(context->missing_args);
|
||||
if (context->remove_source_files)
|
||||
@@ -110,7 +201,6 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) {
|
||||
array_list_delete(context->dir_entries);
|
||||
if (context->dir_entries_mutex_init)
|
||||
mtx_destroy(&context->dir_entries_mutex);
|
||||
config_delete(context->config);
|
||||
queue_destroy(context->queue_scanner);
|
||||
queue_destroy(context->queue_loader);
|
||||
mtx_destroy(&context->mutex_scanner);
|
||||
@@ -122,248 +212,3 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) {
|
||||
mtx_destroy(&context->mutex_progress);
|
||||
free(context);
|
||||
}
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue,
|
||||
int file_descriptor, SSL* ssl) {
|
||||
PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver));
|
||||
if (context == NULL)
|
||||
return NULL;
|
||||
context->config = config;
|
||||
context->queue = queue;
|
||||
context->file_descriptor = file_descriptor;
|
||||
context->ssl = ssl;
|
||||
context->outcomes.entries = NULL;
|
||||
context->outcomes.count = 0;
|
||||
context->outcomes.capacity = 0;
|
||||
dir_time_list_init(&context->dir_times);
|
||||
protocol_session_init(&context->session, file_descriptor, file_descriptor);
|
||||
protocol_session_set_ssl(&context->session, ssl);
|
||||
context->receiver_done = false;
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->deferred_manifest = NULL;
|
||||
atomic_init(&context->cancelled, false);
|
||||
int init = 0;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_full) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_empty) != thrd_success)
|
||||
goto fail;
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
return context;
|
||||
|
||||
fail:
|
||||
log_perror("Error initializing synchronization objects");
|
||||
if (init >= 3)
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
if (init >= 2)
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&context->mutex);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
|
||||
config_delete(context->config);
|
||||
if (context->deferred_manifest)
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
queue_destroy(context->queue);
|
||||
receiver_outcomes_destroy(&context->outcomes);
|
||||
dir_time_list_free(&context->dir_times);
|
||||
mtx_destroy(&context->mutex);
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
free(context);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes) {
|
||||
if (context == NULL)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
context->max_queue_bytes = max_bytes;
|
||||
context->queued_bytes = 0;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes) {
|
||||
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
if (released_bytes >= context->queued_bytes)
|
||||
context->queued_bytes = 0;
|
||||
else
|
||||
context->queued_bytes -= released_bytes;
|
||||
cnd_signal(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) {
|
||||
if (context == NULL || file == NULL)
|
||||
return false;
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
mtx_lock(&context->mutex);
|
||||
while (!atomic_load(&context->cancelled)) {
|
||||
bool blocked_by_count = queue_is_full(context->queue);
|
||||
bool blocked_by_budget = false;
|
||||
if (context->max_queue_bytes > 0) {
|
||||
size_t budget = context->max_queue_bytes;
|
||||
size_t used = context->queued_bytes;
|
||||
if (used >= budget) {
|
||||
blocked_by_budget = true;
|
||||
} else if (file_bytes > budget - used) {
|
||||
/* A single payload larger than the whole budget (not possible with
|
||||
the per-file receive cap) is only admitted to an empty pipeline so
|
||||
the wait can never deadlock. */
|
||||
blocked_by_budget = used != 0;
|
||||
}
|
||||
}
|
||||
if (!blocked_by_count && !blocked_by_budget)
|
||||
break;
|
||||
cnd_wait(&context->condition_not_full, &context->mutex);
|
||||
}
|
||||
if (atomic_load(&context->cancelled)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (!queue_enqueue(context->queue, file)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
context->queued_bytes += file_bytes;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool receiver_enqueue_file(File* file, void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
return pipeline_context_receiver_enqueue_file(context, file);
|
||||
}
|
||||
|
||||
static void receiver_thread_fail(PipelineContextReceiver* context) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
int receive_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
int file_descriptor = context->file_descriptor;
|
||||
const Config* config = context->config;
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL};
|
||||
if (receiver_process_pending((Config*)config, file_descriptor, &sink,
|
||||
&context->deferred_manifest) != 0) {
|
||||
receiver_thread_fail(context);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
mtx_lock(&context->mutex);
|
||||
context->receiver_done = true;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
|
||||
int write_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
bool save_to_disk = context->config->save_to_disk;
|
||||
char* root_directory = str_dup(context->config->receive_root_directory);
|
||||
mtx_unlock(&context->mutex);
|
||||
if (save_to_disk && !root_directory) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
File* file =
|
||||
queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty,
|
||||
&context->condition_not_full, &context->receiver_done);
|
||||
if (file == NULL) {
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
FileSaveResult result = FILE_SAVE_SKIPPED;
|
||||
if (save_to_disk) {
|
||||
result = file_save_to_disk_full(root_directory, file, context->config);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
}
|
||||
/* P7 Wave D: a directory's times are never applied inline (a later child
|
||||
write would clobber them); accumulate the metadata here and let the
|
||||
caller apply it once every writer has drained. */
|
||||
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
context->config->use_metadata && !context->config->omit_dir_times &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
/* Record the per-file outcome so a --remove-source-files sender learns
|
||||
which sources were actually written versus skipped on the receiver.
|
||||
Explicit directory entries and recreated device/special nodes have no
|
||||
source and are never acknowledged (mirrors receiver.c). */
|
||||
if (context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip &&
|
||||
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
}
|
||||
}
|
||||
@@ -5,11 +5,11 @@
|
||||
#include <stdatomic.h>
|
||||
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "file.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "receiver.h"
|
||||
#include "stop_condition.h"
|
||||
#include <openssl/ssl.h>
|
||||
|
||||
@@ -25,6 +25,15 @@ typedef struct {
|
||||
cnd_t condition_not_full_loader;
|
||||
cnd_t condition_not_empty_loader;
|
||||
bool loader_done;
|
||||
/* Aggregate loaded payload bytes queued on queue_loader but not yet released
|
||||
by the sender. Guarded by `mutex_loader`. When `max_queue_bytes` is
|
||||
non-zero the loader blocks before enqueueing a chunk that would push this
|
||||
total over it, so the sender buffers a bounded number of bytes rather than
|
||||
an unbounded count of chunks that may each be up to chunk_size (or a single
|
||||
file) in size. Files streamed straight from disk by sendfile hold no
|
||||
payload, so only in-memory (`data->data`) payloads are counted. */
|
||||
size_t queued_bytes;
|
||||
size_t max_queue_bytes;
|
||||
ArrayList* manifest;
|
||||
/* Protected prefixes (paths the source scan excluded by user rules) sent
|
||||
with the keep-set manifest so --delete leaves them alone unless
|
||||
@@ -33,6 +42,18 @@ typedef struct {
|
||||
scanner's exclusion sink) or, in the early modes, by the path-only pre-scan
|
||||
on the calling thread before the pipeline starts. */
|
||||
ArrayList* excluded_paths;
|
||||
/* --max-size/--min-size pruned source paths. These are ALWAYS sent as
|
||||
protected prefixes (even with --delete-excluded), so the destination
|
||||
mirrors of size-skipped files survive --delete like rsync. Populated by
|
||||
the scanner thread (workers append under mutex_scanner) or, in the early
|
||||
modes, by the path-only pre-scan on the calling thread. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Destination-relative paths of the directories the source scan synchronized
|
||||
for this run (the receive root is the "." sentinel). Sent with the
|
||||
manifest so the receiver confines its extras walk to them, matching rsync's
|
||||
"delete only in synchronized directories" (notably for --files-from).
|
||||
Populated by the scanner thread or the early pre-scan. */
|
||||
ArrayList* synced_dirs;
|
||||
/* --delete-missing-args: the destination-relative mirrors of the --files-from
|
||||
entries that are missing under the source. Computed by the preflight on
|
||||
the calling thread before the pipeline starts; the sender thread transmits
|
||||
@@ -74,60 +95,31 @@ typedef struct {
|
||||
ArrayList* dir_entries;
|
||||
mtx_t dir_entries_mutex;
|
||||
bool dir_entries_mutex_init;
|
||||
/* Set by the sender thread when the receiver reported a --max-delete-capped
|
||||
deletion (STATUS_DELETE_LIMIT): the transfer succeeded and the process must
|
||||
exit 25 like rsync. Read by the caller after the sender thread is joined. */
|
||||
bool delete_limit;
|
||||
} PipelineContextSender;
|
||||
|
||||
typedef struct PipelineContextReceiver {
|
||||
Queue* queue;
|
||||
Config* config;
|
||||
int file_descriptor;
|
||||
SSL* ssl;
|
||||
ProtocolSession session;
|
||||
ReceiverOutcomes outcomes;
|
||||
mtx_t mutex;
|
||||
cnd_t condition_not_full;
|
||||
cnd_t condition_not_empty;
|
||||
bool receiver_done;
|
||||
atomic_bool cancelled;
|
||||
/* Aggregate payload bytes that have been received but not yet released by
|
||||
the disk writer (queued or in the writer's hand). Guarded by `mutex`.
|
||||
When `max_queue_bytes` is non-zero the receiver blocks before enqueuing
|
||||
once this total would exceed it, so decompressed/copied file payloads
|
||||
buffered ahead of a slow disk writer respect the per-connection memory
|
||||
budget instead of growing without bound. */
|
||||
size_t queued_bytes;
|
||||
size_t max_queue_bytes;
|
||||
/* Keep-set manifest for the commit-style (late) deletion
|
||||
(--delete/--delete-after/--delete-delay). receive_thread parses the whole
|
||||
protocol stream but hands the manifest here instead of deleting while the
|
||||
disk writer may still be draining; the caller (server.c) commits the
|
||||
deletion after both threads have joined, so no extra is removed unless the
|
||||
transfer truly succeeded. NULL in the early delete modes (which delete at
|
||||
the manifest). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* P7 Wave D: directory metadata collected by write_thread from received
|
||||
directory entries. Only write_thread mutates it (before it joins); the
|
||||
caller (server.c) applies it after the delete/delay-updates phase. */
|
||||
DirTimeList dir_times;
|
||||
} PipelineContextReceiver;
|
||||
|
||||
/* `config` is borrowed and must outlive the context: destroy does NOT free it,
|
||||
so the caller owns it and frees it with config_delete() afterwards. */
|
||||
PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* queue_scanner,
|
||||
Queue* queue_loader);
|
||||
void pipeline_context_sender_destroy(PipelineContextSender* context);
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
|
||||
int file_descriptor, SSL* ssl);
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context);
|
||||
/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes);
|
||||
/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue
|
||||
is full by element count or when adding `file` would push queued_bytes over
|
||||
the configured byte limit; waits until the disk writer releases bytes.
|
||||
Takes ownership of `file` on success and destroys it on failure/cancel. */
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file);
|
||||
/* Account for `released_bytes` of payload memory that has been freed by the
|
||||
disk writer, unblocking a receiver that is waiting on the byte limit. */
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
/* Bound the loaded payload bytes the sender may buffer ahead of the network
|
||||
writer (see max_queue_bytes). */
|
||||
void pipeline_context_sender_set_queue_byte_limit(PipelineContextSender* context, size_t max_bytes);
|
||||
/* Total payload bytes a chunk currently holds in memory (loaded file data
|
||||
only; zero for entries with no payload or data streamed from disk). */
|
||||
size_t pipeline_context_sender_chunk_bytes(const Chunk* chunk);
|
||||
/* Blocking enqueue used by the sender's loader stage. Blocks while
|
||||
queue_loader is full by element count or when adding `chunk` would push the
|
||||
queued payload bytes over the configured byte limit; waits until the sender
|
||||
releases bytes. Takes ownership of `chunk` on success and destroys it on
|
||||
failure/cancel. */
|
||||
bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk* chunk);
|
||||
/* Account for `released_bytes` of payload memory that the sender freed after
|
||||
destroying a chunk, unblocking a loader waiting on the byte limit. */
|
||||
void pipeline_context_sender_note_bytes_released(PipelineContextSender* context,
|
||||
size_t released_bytes);
|
||||
int receive_thread(void* pipeline_context);
|
||||
int write_thread(void* pipeline_context);
|
||||
#endif
|
||||
+368
-27
@@ -12,8 +12,7 @@
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define RECEIVE_TIMEOUT_SEC 60 /* 60 second per-message timeout */
|
||||
#define SEND_TIMEOUT_SEC 60
|
||||
#define RECEIVE_TIMEOUT_SEC 60 /* built-in fallback for explicit -timed calls only */
|
||||
|
||||
static __thread int io_read_fd = -1;
|
||||
static __thread int io_write_fd = -1;
|
||||
@@ -22,6 +21,10 @@ static __thread ProtocolSession* bound_session;
|
||||
static __thread ProtocolSession legacy_io_session = {
|
||||
.read_fd = -1, .write_fd = -1, .max_alloc = DEFAULT_MAX_ALLOC};
|
||||
|
||||
/* Last STATUS_ERROR_DETAIL reason received on this thread (protocol 2.21.0).
|
||||
* Empty when the last status read carried no detail. */
|
||||
static __thread char io_error_detail[MAX_ERROR_DETAIL_BYTES + 1];
|
||||
|
||||
static unsigned long long io_bwlimit = 0;
|
||||
static mtx_t bw_mutex;
|
||||
static once_flag bw_mutex_once = ONCE_FLAG_INIT;
|
||||
@@ -40,7 +43,9 @@ static bool protocol_reserve_memory(ProtocolSession* session, size_t charge) {
|
||||
}
|
||||
}
|
||||
|
||||
static void protocol_release_memory_for_session(ProtocolSession* session, size_t charge) {
|
||||
void protocol_release_memory_for_session(ProtocolSession* session, size_t charge) {
|
||||
if (!session)
|
||||
return;
|
||||
unsigned long long allocated = atomic_load(&session->total_allocated_bytes);
|
||||
while (true) {
|
||||
unsigned long long remaining = (unsigned long long)charge >= allocated ? 0 : allocated - charge;
|
||||
@@ -57,6 +62,10 @@ void io_set_fds(int read_fd, int write_fd) {
|
||||
bound_session = NULL;
|
||||
io_read_fd = read_fd;
|
||||
io_write_fd = write_fd;
|
||||
/* A descriptor switch starts a new connection on this thread: a stale
|
||||
rejection detail captured from the previous transport must not leak into
|
||||
the new one. */
|
||||
io_error_detail[0] = '\0';
|
||||
/* A descriptor switch starts a new transport; never reuse a TLS object
|
||||
belonging to a previous connection or test pipe. */
|
||||
io_ssl = NULL;
|
||||
@@ -76,10 +85,29 @@ void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd)
|
||||
session->read_fd = read_fd;
|
||||
session->write_fd = write_fd;
|
||||
session->max_alloc = DEFAULT_MAX_ALLOC;
|
||||
session->io_timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
atomic_init(&session->total_allocated_bytes, 0);
|
||||
protocol_session_set_bwlimit(session, global_bwlimit());
|
||||
}
|
||||
|
||||
void protocol_session_set_io_timeout(ProtocolSession* session, int sec) {
|
||||
if (!session)
|
||||
return;
|
||||
session->io_timeout_sec = sec;
|
||||
}
|
||||
|
||||
int protocol_get_io_timeout_sec(void) {
|
||||
const ProtocolSession* session = bound_session ? bound_session : &legacy_io_session;
|
||||
/* 0 (or negative) means the session timeout is disabled, matching rsync's
|
||||
* --timeout=0 default. Callers must treat a non-positive result as "wait
|
||||
* without a deadline" instead of substituting a built-in window. */
|
||||
return session->io_timeout_sec > 0 ? session->io_timeout_sec : 0;
|
||||
}
|
||||
|
||||
int protocol_server_io_timeout_sec(int client_timeout) {
|
||||
return client_timeout > 0 ? client_timeout : SERVER_IO_TIMEOUT_SEC;
|
||||
}
|
||||
|
||||
void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc) {
|
||||
if (!session)
|
||||
session = bound_session ? bound_session : &legacy_io_session;
|
||||
@@ -87,7 +115,8 @@ void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long
|
||||
}
|
||||
|
||||
static bool allocation_allowed(const ProtocolSession* session, size_t size) {
|
||||
return (unsigned long long)size <= session->max_alloc;
|
||||
/* max_alloc == 0 is rsync's --max-alloc=0 "no limit". */
|
||||
return session->max_alloc == 0 || (unsigned long long)size <= session->max_alloc;
|
||||
}
|
||||
|
||||
static void* protocol_alloc_for_session(const ProtocolSession* session, size_t size) {
|
||||
@@ -257,10 +286,15 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
log_debug_message(LOG_DEBUG_IO, " Sending n Data: %zu", data_size);
|
||||
if (!session)
|
||||
return false;
|
||||
/* A non-positive session timeout disables the deadline entirely (rsync's
|
||||
* --timeout=0 default); poll then blocks until the socket becomes writable. */
|
||||
int timeout_sec = session->io_timeout_sec > 0 ? session->io_timeout_sec : 0;
|
||||
int fd = session->write_fd;
|
||||
struct timespec deadline;
|
||||
if (timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += SEND_TIMEOUT_SEC;
|
||||
deadline.tv_sec += timeout_sec;
|
||||
}
|
||||
short wait_events = POLLOUT;
|
||||
ssize_t total_bytes_send = 0;
|
||||
while ((size_t)total_bytes_send < data_size) {
|
||||
@@ -268,7 +302,7 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
if (session->bwlimit > 0 && chunk > 65536)
|
||||
chunk = 65536;
|
||||
struct pollfd pfd = {.fd = fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, deadline_remaining_ms(&deadline));
|
||||
int poll_result = poll(&pfd, 1, timeout_sec > 0 ? deadline_remaining_ms(&deadline) : -1);
|
||||
if (poll_result == 0 || (poll_result < 0 && errno != EINTR)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Send timeout or poll failure");
|
||||
return false;
|
||||
@@ -278,10 +312,14 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
if (pfd.revents & (POLLERR | POLLNVAL))
|
||||
return false;
|
||||
ssize_t bytes_send;
|
||||
if (session->ssl)
|
||||
bytes_send = SSL_write(session->ssl, (const char*)data + total_bytes_send, chunk);
|
||||
else
|
||||
if (session->ssl) {
|
||||
/* SSL_write takes an int length; clamp a >INT_MAX request into chunks so
|
||||
* the size_t downcast can never truncate into a negative/partial write. */
|
||||
size_t ssl_chunk = chunk > (size_t)INT_MAX ? (size_t)INT_MAX : chunk;
|
||||
bytes_send = SSL_write(session->ssl, (const char*)data + total_bytes_send, (int)ssl_chunk);
|
||||
} else {
|
||||
bytes_send = write(fd, (const char*)data + total_bytes_send, chunk);
|
||||
}
|
||||
if (bytes_send <= 0) {
|
||||
if (session->ssl) {
|
||||
int ssl_err = SSL_get_error(session->ssl, (int)bytes_send);
|
||||
@@ -289,6 +327,12 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN;
|
||||
continue;
|
||||
}
|
||||
/* A signal (e.g. Ctrl-C) interrupts the blocking TLS write: retry so
|
||||
the send loop can observe the abort flag at the next checkpoint. */
|
||||
if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR)
|
||||
continue;
|
||||
} else if (errno == EINTR) {
|
||||
continue;
|
||||
}
|
||||
log_message(LOG_LEVEL_ERROR, "Could not send data");
|
||||
return false;
|
||||
@@ -304,32 +348,43 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
|
||||
bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size,
|
||||
int timeout_sec);
|
||||
static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, size_t data_size,
|
||||
const struct timespec* deadline);
|
||||
|
||||
bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_size) {
|
||||
return protocol_receive_n_data_timed(session, data, data_size, RECEIVE_TIMEOUT_SEC);
|
||||
/* Honor the session's configured deadline. A non-positive value disables the
|
||||
* deadline (rsync's --timeout=0 default): wait without a poll timeout. The
|
||||
* explicit _timed variants keep their own 0 -> built-in-default contract. */
|
||||
if (!session)
|
||||
return false;
|
||||
if (session->io_timeout_sec <= 0)
|
||||
return protocol_receive_n_data_until(session, data, data_size, NULL);
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += session->io_timeout_sec;
|
||||
return protocol_receive_n_data_until(session, data, data_size, &deadline);
|
||||
}
|
||||
|
||||
bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size,
|
||||
int timeout_sec) {
|
||||
/* Read exactly `data_size` bytes from `session` before `deadline` elapses
|
||||
* (CLOCK_MONOTONIC). Shared by the ordinary timed primitive and the error-detail
|
||||
* body reader so the latter can clamp itself to whatever deadline its caller
|
||||
* already established instead of always applying the session's 60 s window. */
|
||||
static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, size_t data_size,
|
||||
const struct timespec* deadline) {
|
||||
log_debug_message(LOG_DEBUG_IO, " Receiving n Data: %zu", data_size);
|
||||
if (!session)
|
||||
return false;
|
||||
int fd = session->read_fd;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
|
||||
size_t total_bytes_received = 0;
|
||||
short wait_events = POLLIN;
|
||||
while (total_bytes_received < data_size) {
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
struct pollfd pfd = {.fd = fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, deadline_remaining_ms(&deadline));
|
||||
/* A NULL deadline means "wait indefinitely" (timeout disabled). */
|
||||
int poll_result = poll(&pfd, 1, deadline ? deadline_remaining_ms(deadline) : -1);
|
||||
if (poll_result == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", timeout_sec);
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout");
|
||||
return false;
|
||||
}
|
||||
if (poll_result < 0) {
|
||||
@@ -356,6 +411,13 @@ bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t
|
||||
wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN;
|
||||
continue;
|
||||
}
|
||||
/* A signal interrupts the blocking TLS read: retry (mirrors the send
|
||||
path and protocol_read_status_until) so the loop reaches its next
|
||||
abort/deadline checkpoint instead of failing spuriously. */
|
||||
if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR)
|
||||
continue;
|
||||
} else if (errno == EINTR) {
|
||||
continue;
|
||||
}
|
||||
if (bytes_received == 0)
|
||||
log_message(LOG_LEVEL_ERROR, "Connection closed while receiving data");
|
||||
@@ -371,6 +433,18 @@ bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t
|
||||
return true;
|
||||
}
|
||||
|
||||
bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size,
|
||||
int timeout_sec) {
|
||||
if (!session)
|
||||
return false;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
return protocol_receive_n_data_until(session, data, data_size, &deadline);
|
||||
}
|
||||
|
||||
static const char* status_to_string(Status status) {
|
||||
switch (status) {
|
||||
case STATUS_OK:
|
||||
@@ -421,6 +495,14 @@ static const char* status_to_string(Status status) {
|
||||
return "AUTH_OK";
|
||||
case STATUS_AUTH_FAILED:
|
||||
return "AUTH_FAILED";
|
||||
case STATUS_ERROR_DETAIL:
|
||||
return "ERROR_DETAIL";
|
||||
case STATUS_DRY_RUN_TRANSFER:
|
||||
return "DRY_RUN_TRANSFER";
|
||||
case STATUS_DELETE_LIMIT:
|
||||
return "DELETE_LIMIT";
|
||||
case STATUS_DEST_INFO:
|
||||
return "DEST_INFO";
|
||||
default:
|
||||
return "UNKNOWN";
|
||||
}
|
||||
@@ -550,13 +632,10 @@ Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long
|
||||
return NULL;
|
||||
}
|
||||
result->protocol_charge = allocation_size;
|
||||
result->owner = session;
|
||||
return result;
|
||||
}
|
||||
|
||||
Data* protocol_receive_data(ProtocolSession* session) {
|
||||
return protocol_receive_data_limited(session, MAX_DATA_PAYLOAD_SIZE);
|
||||
}
|
||||
|
||||
bool protocol_send_int(ProtocolSession* session, int data) {
|
||||
if (!protocol_send_n_data(session, &data, sizeof(int)))
|
||||
return false;
|
||||
@@ -578,8 +657,85 @@ bool protocol_send_status(ProtocolSession* session, Status status) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Read the bounded, length-prefixed body of a STATUS_ERROR_DETAIL frame within
|
||||
* `deadline` (CLOCK_MONOTONIC), polling `abort_check` (may be NULL) between
|
||||
* drain chunks. The declared length is validated BEFORE any allocation:
|
||||
*
|
||||
* - `size > MAX_STRING_SIZE`: an absurd framing error. Reading/draining that
|
||||
* many bytes could never finish, so it is fatal (the caller tears the
|
||||
* connection down) rather than drained.
|
||||
* - `MAX_ERROR_DETAIL_BYTES < size <= MAX_STRING_SIZE`: drain exactly `size`
|
||||
* bytes through a small fixed scratch buffer so the stream stays in sync,
|
||||
* leaving the captured detail empty. No allocation happens.
|
||||
* - `size <= MAX_ERROR_DETAIL_BYTES`: read straight into the thread-local
|
||||
* `io_error_detail` buffer (size+1 capacity, already reserved), so the
|
||||
* session's --max-alloc / MAX_CONNECTION_MEMORY budgets are never touched.
|
||||
*
|
||||
* Returns false on a fatal framing problem or any I/O failure; the terminal
|
||||
* detail is then empty. The body is consumed on every non-fatal path even when
|
||||
* the caller ignores protocol_last_error(), so the stream never desyncs. */
|
||||
static bool protocol_receive_error_detail_until(ProtocolSession* session,
|
||||
const struct timespec* deadline,
|
||||
ProtocolWaitAbort abort_check) {
|
||||
io_error_detail[0] = '\0';
|
||||
size_t size = 0;
|
||||
if (!protocol_receive_n_data_until(session, &size, sizeof(size), deadline))
|
||||
return false;
|
||||
if (size > MAX_STRING_SIZE) {
|
||||
log_message(LOG_LEVEL_ERROR, "Error detail length %zu exceeds maximum %llu", size,
|
||||
(unsigned long long)MAX_STRING_SIZE);
|
||||
return false;
|
||||
}
|
||||
if (size > MAX_ERROR_DETAIL_BYTES) {
|
||||
char scratch[256];
|
||||
size_t remaining = size;
|
||||
while (remaining > 0) {
|
||||
if (abort_check && abort_check())
|
||||
return false;
|
||||
size_t chunk = remaining < sizeof(scratch) ? remaining : sizeof(scratch);
|
||||
if (!protocol_receive_n_data_until(session, scratch, chunk, deadline))
|
||||
return false;
|
||||
remaining -= chunk;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (!protocol_receive_n_data_until(session, io_error_detail, size, deadline))
|
||||
return false;
|
||||
io_error_detail[size] = '\0';
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Consume the optional detail body of a STATUS_ERROR_DETAIL frame and map the
|
||||
* status back to STATUS_ERROR for existing callers. Invoked for EVERY status
|
||||
* read so a stale detail from an earlier exchange is never reported for a later
|
||||
* one -- except for STATUS_KEEPALIVE, which carries no body and whose drain
|
||||
* (protocol_receive_status_keepalive) must NOT erase the terminal detail that
|
||||
* arrived just before it. Returns false on a fatal framing error. */
|
||||
static bool protocol_capture_error_detail(ProtocolSession* session, Status* status,
|
||||
const struct timespec* deadline,
|
||||
ProtocolWaitAbort abort_check) {
|
||||
if (*status == STATUS_KEEPALIVE)
|
||||
return true;
|
||||
io_error_detail[0] = '\0';
|
||||
if (*status != STATUS_ERROR_DETAIL)
|
||||
return true;
|
||||
*status = STATUS_ERROR;
|
||||
return protocol_receive_error_detail_until(session, deadline, abort_check);
|
||||
}
|
||||
|
||||
bool protocol_receive_status(ProtocolSession* session, Status* status) {
|
||||
if (!protocol_receive_n_data(session, status, sizeof(Status)))
|
||||
if (!session || !status)
|
||||
return false;
|
||||
struct timespec deadline;
|
||||
const struct timespec* deadline_ptr = NULL;
|
||||
if (session->io_timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += session->io_timeout_sec;
|
||||
deadline_ptr = &deadline;
|
||||
}
|
||||
if (!protocol_receive_n_data_until(session, status, sizeof(Status), deadline_ptr))
|
||||
return false;
|
||||
if (!protocol_capture_error_detail(session, status, deadline_ptr, NULL))
|
||||
return false;
|
||||
log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status));
|
||||
return true;
|
||||
@@ -588,10 +744,169 @@ bool protocol_receive_status(ProtocolSession* session, Status* status) {
|
||||
/* protocol_receive_status with an explicit per-message deadline (seconds).
|
||||
Used where a single reply may legitimately take far longer than the default
|
||||
60 s receive window - e.g. the sender waiting for the early-delete ACK after
|
||||
the receiver committed a large (up to MAX_SERVER_DELETE_COUNT) deletion. */
|
||||
the receiver committed a large (up to MAX_SERVER_DELETE_COUNT) deletion. The
|
||||
error-detail body shares the same deadline as the status header. */
|
||||
bool protocol_receive_status_timed(ProtocolSession* session, Status* status, int timeout_sec) {
|
||||
if (!protocol_receive_n_data_timed(session, status, sizeof(Status), timeout_sec))
|
||||
if (!session || !status)
|
||||
return false;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
if (!protocol_receive_n_data_until(session, status, sizeof(Status), &deadline))
|
||||
return false;
|
||||
if (!protocol_capture_error_detail(session, status, &deadline, NULL))
|
||||
return false;
|
||||
log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status));
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Read exactly one Status frame within `deadline` (CLOCK_MONOTONIC). Unlike
|
||||
* protocol_receive_status_keepalive this never emits a keepalive: it is used
|
||||
* to consume the first byte(s) of an already-signalled frame and to drain the
|
||||
* peer's outstanding keepalive replies, where injecting a write could split a
|
||||
* reply across a frame boundary. Returns false on timeout/EOF/error. */
|
||||
static bool protocol_read_status_until(ProtocolSession* session, Status* status,
|
||||
const struct timespec* deadline) {
|
||||
Status received = STATUS_ERROR;
|
||||
size_t got = 0;
|
||||
short wait_events = POLLIN;
|
||||
while (got < sizeof(Status)) {
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
int remaining_ms = deadline ? deadline_remaining_ms(deadline) : -1;
|
||||
if (remaining_ms == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status");
|
||||
return false;
|
||||
}
|
||||
struct pollfd pfd = {.fd = session->read_fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, remaining_ms);
|
||||
if (poll_result == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status");
|
||||
return false;
|
||||
}
|
||||
if (poll_result < 0) {
|
||||
if (errno == EINTR)
|
||||
continue;
|
||||
return false;
|
||||
}
|
||||
if (pfd.revents & (POLLERR | POLLNVAL))
|
||||
return false;
|
||||
}
|
||||
ssize_t bytes_received;
|
||||
if (session->ssl)
|
||||
bytes_received = SSL_read(session->ssl, (char*)&received + got, sizeof(Status) - got);
|
||||
else
|
||||
bytes_received = read(session->read_fd, (char*)&received + got, sizeof(Status) - got);
|
||||
if (bytes_received <= 0) {
|
||||
if (session->ssl) {
|
||||
int ssl_err = SSL_get_error(session->ssl, (int)bytes_received);
|
||||
if (ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) {
|
||||
wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if (bytes_received < 0 && errno == EINTR)
|
||||
continue;
|
||||
log_message(LOG_LEVEL_ERROR, "Connection closed while receiving status");
|
||||
return false;
|
||||
}
|
||||
got += (size_t)bytes_received;
|
||||
}
|
||||
*status = received;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, int timeout_sec,
|
||||
int keepalive_interval_sec, ProtocolWaitAbort abort_check) {
|
||||
if (!session || !status)
|
||||
return false;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
if (keepalive_interval_sec <= 0)
|
||||
keepalive_interval_sec = timeout_sec;
|
||||
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
|
||||
unsigned long keepalives_sent = 0;
|
||||
unsigned long replies_seen = 0;
|
||||
Status final = STATUS_ERROR;
|
||||
while (true) {
|
||||
if (abort_check && abort_check())
|
||||
return false;
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
int remaining_ms = deadline_remaining_ms(&deadline);
|
||||
if (remaining_ms <= 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", timeout_sec);
|
||||
return false;
|
||||
}
|
||||
/* Only interleave a keepalive while waiting for the FIRST byte of a
|
||||
* frame; once part of a frame is buffered a write could race the peer's
|
||||
* reply into the middle of it. */
|
||||
long long interval_ms_ll = (long long)keepalive_interval_sec * 1000LL;
|
||||
int interval_ms = interval_ms_ll > INT_MAX ? INT_MAX : (int)interval_ms_ll;
|
||||
int wait_ms = interval_ms < remaining_ms ? interval_ms : remaining_ms;
|
||||
struct pollfd pfd = {.fd = session->read_fd, .events = POLLIN};
|
||||
int poll_result = poll(&pfd, 1, wait_ms);
|
||||
if (poll_result == 0) {
|
||||
if (abort_check && abort_check())
|
||||
return false;
|
||||
if (!protocol_send_status(session, STATUS_KEEPALIVE))
|
||||
return false;
|
||||
keepalives_sent++;
|
||||
continue;
|
||||
}
|
||||
if (poll_result < 0) {
|
||||
if (errno == EINTR)
|
||||
continue;
|
||||
return false;
|
||||
}
|
||||
if (pfd.revents & (POLLERR | POLLNVAL))
|
||||
return false;
|
||||
}
|
||||
Status received;
|
||||
if (!protocol_read_status_until(session, &received, &deadline))
|
||||
return false;
|
||||
if (!protocol_capture_error_detail(session, &received, &deadline, abort_check))
|
||||
return false;
|
||||
if (received == STATUS_KEEPALIVE) {
|
||||
/* The receiver's answer to one of our keepalives. */
|
||||
replies_seen++;
|
||||
continue;
|
||||
}
|
||||
final = received;
|
||||
break;
|
||||
}
|
||||
/* Drain the replies the receiver still owes for keepalives we sent while it
|
||||
* was busy. It answers them only after the real status, so leaving them
|
||||
* unread would put stale KEEPALIVE frames ahead of the next exchange and
|
||||
* desynchronize the protocol. */
|
||||
if (replies_seen < keepalives_sent) {
|
||||
/* A short separate grace, not the (possibly exhausted) main deadline: the
|
||||
terminal status already arrived, so a peer that never answers its owed
|
||||
keepalives must not turn a successful ack into a reported failure. */
|
||||
struct timespec drain_deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &drain_deadline);
|
||||
drain_deadline.tv_sec += 1;
|
||||
while (replies_seen < keepalives_sent) {
|
||||
Status drained;
|
||||
if (!protocol_read_status_until(session, &drained, &drain_deadline)) {
|
||||
log_message(LOG_LEVEL_WARNING, "peer did not answer %lu keepalive(s); continuing",
|
||||
keepalives_sent - replies_seen);
|
||||
break;
|
||||
}
|
||||
if (!protocol_capture_error_detail(session, &drained, &drain_deadline, abort_check))
|
||||
return false;
|
||||
if (drained != STATUS_KEEPALIVE) {
|
||||
log_message(LOG_LEVEL_ERROR, "Unexpected status while draining keepalive replies");
|
||||
return false;
|
||||
}
|
||||
replies_seen++;
|
||||
}
|
||||
}
|
||||
*status = final;
|
||||
log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status));
|
||||
return true;
|
||||
}
|
||||
@@ -635,3 +950,29 @@ bool receive_status(int fd, Status* status) {
|
||||
bool receive_status_timed(int fd, Status* status, int timeout_sec) {
|
||||
return protocol_receive_status_timed(legacy_session(fd, -1), status, timeout_sec);
|
||||
}
|
||||
bool receive_status_keepalive(int fd, Status* status, int timeout_sec, int keepalive_interval_sec,
|
||||
ProtocolWaitAbort abort_check) {
|
||||
return protocol_receive_status_keepalive(legacy_session(fd, -1), status, timeout_sec,
|
||||
keepalive_interval_sec, abort_check);
|
||||
}
|
||||
|
||||
bool send_error_detail(int fd, const char* message) {
|
||||
if (!message)
|
||||
message = "";
|
||||
char bounded[MAX_ERROR_DETAIL_BYTES + 1];
|
||||
size_t len = strlen(message);
|
||||
if (len > MAX_ERROR_DETAIL_BYTES) {
|
||||
memcpy(bounded, message, MAX_ERROR_DETAIL_BYTES);
|
||||
bounded[MAX_ERROR_DETAIL_BYTES] = '\0';
|
||||
message = bounded;
|
||||
}
|
||||
return send_status(fd, STATUS_ERROR_DETAIL) && send_str(fd, message);
|
||||
}
|
||||
|
||||
const char* protocol_last_error(void) {
|
||||
return io_error_detail;
|
||||
}
|
||||
|
||||
void protocol_clear_last_error(void) {
|
||||
io_error_detail[0] = '\0';
|
||||
}
|
||||
+103
-2
@@ -9,6 +9,12 @@
|
||||
/* Maximum allowed string size for receive_str (64 KB) */
|
||||
#define MAX_STRING_SIZE (64 * 1024)
|
||||
|
||||
/* Hard cap on the optional server->client rejection detail carried by
|
||||
* STATUS_ERROR_DETAIL (protocol 2.21.0). A longer message is sliced to this
|
||||
* many bytes before it is sent, so a peer can never be made to retain more than
|
||||
* this for a rejection and the detail frame stays a small, fixed bound. */
|
||||
#define MAX_ERROR_DETAIL_BYTES 4096
|
||||
|
||||
/* Maximum uncompressed file payload accepted by the receiver's whole-file
|
||||
* paths. A single whole file is charged against the per-connection memory
|
||||
* reservation (MAX_CONNECTION_MEMORY) and against the server allocation
|
||||
@@ -28,6 +34,11 @@
|
||||
#define DEFAULT_MAX_ALLOC (1ULL * 1024 * 1024 * 1024)
|
||||
/* Server policy ceiling for a client-provided allocation limit. */
|
||||
#define MAX_SERVER_ALLOC (256ULL * 1024 * 1024)
|
||||
/* Server-owned floor for the per-message I/O deadline. A client --timeout=0
|
||||
(rsync's default) disables the client's own deadlines, but a server session
|
||||
must never be held open forever by a silent peer (slow-loris), so the server
|
||||
floors the effective deadline at this value. */
|
||||
#define SERVER_IO_TIMEOUT_SEC 60
|
||||
/* Bounded cumulative per-connection receive budget. In-flight wire buffers,
|
||||
decompression buffers and queued (not yet written) file payloads for a
|
||||
connection must stay within this ceiling. */
|
||||
@@ -52,6 +63,14 @@ typedef struct ProtocolSession {
|
||||
atomic_ullong total_allocated_bytes;
|
||||
bool eight_bit_output;
|
||||
unsigned long long max_alloc;
|
||||
/* Per-session deadline (seconds) applied to every protocol send/receive by
|
||||
* protocol_send_n_data / protocol_receive_n_data. The initialized default is
|
||||
* the built-in 60 s window; a value <= 0 disables the deadline (rsync's
|
||||
* --timeout=0). Set from the negotiated Config->timeout so --timeout is
|
||||
* honored by the poll()-driven protocol I/O, not just the socket
|
||||
* SO_RCVTIMEO/SO_SNDTIMEO. The server does not propagate a client 0 here: it
|
||||
* installs protocol_server_io_timeout_sec() so its sessions keep a floor. */
|
||||
int io_timeout_sec;
|
||||
} ProtocolSession;
|
||||
|
||||
typedef int Status;
|
||||
@@ -126,7 +145,43 @@ enum NET_STATUS {
|
||||
STATUS_AUTH_CHALLENGE,
|
||||
STATUS_AUTH_RESPONSE,
|
||||
STATUS_AUTH_OK,
|
||||
STATUS_AUTH_FAILED
|
||||
STATUS_AUTH_FAILED,
|
||||
/* Optional server->client rejection detail (protocol 2.21.0). When the
|
||||
* server refuses a transfer for a concrete reason it may send
|
||||
* STATUS_ERROR_DETAIL followed by a length-prefixed, bounded string instead
|
||||
* of a bare STATUS_ERROR. receive_status() consumes the string and maps the
|
||||
* status back to STATUS_ERROR, so every pre-2.21 call site keeps working;
|
||||
* callers that want the human-readable reason consult protocol_last_error().
|
||||
* Appended immediately after STATUS_AUTH_FAILED so the existing wire values
|
||||
* never move. */
|
||||
STATUS_ERROR_DETAIL,
|
||||
/* Server-contacting --dry-run (protocol 2.21.0). Sent by the receiver in
|
||||
* response to a per-file STATUS_CHECK when the wire config carries
|
||||
* dry_run=true and the file is NOT already up to date: it tells the sender
|
||||
* the file WOULD be transferred, and the sender must NOT transmit any data
|
||||
* (the receiver reads none in dry-run). STATUS_OK keeps its meaning in this
|
||||
* path ("already up to date / nothing to do"). Appended after
|
||||
* STATUS_ERROR_DETAIL so no existing status is renumbered. */
|
||||
STATUS_DRY_RUN_TRANSFER,
|
||||
/* --max-delete budget exhausted (protocol 2.23.0). Sent by the receiver as
|
||||
* the terminal success status INSTEAD of STATUS_OK when a --delete/
|
||||
* --delete-missing-args commit removed up to the --max-delete bound but had
|
||||
* to skip further extras. The transfer itself succeeded and all file data is
|
||||
* stored; the sender maps this to rsync's exit code 25 ("the --max-delete
|
||||
* limit stopped deletions"). Appended after STATUS_DRY_RUN_TRANSFER so no
|
||||
* existing status is renumbered. */
|
||||
STATUS_DELETE_LIMIT,
|
||||
/* Destination-state report for output parity (protocol 2.23.0). When the
|
||||
* wire config carries report_dest_info=true, the receiver answers every
|
||||
* per-file STATUS_CHECK request with STATUS_DEST_INFO FIRST, followed by a
|
||||
* fixed record describing the pre-transfer destination entry
|
||||
* (int32 has_old; uint64 size; int64 mtime; int64 mtime_nsec; uint32 mode;
|
||||
* int32 uid; int32 gid). The ordinary STATUS_OK/STATUS_NEXT/... verdict
|
||||
* follows, so the sender can render rsync-accurate -i/--out-format columns
|
||||
* (new vs modified, and which of size/time/perms/owner/group differ) without
|
||||
* changing the transfer decision itself. Appended after
|
||||
* STATUS_DELETE_LIMIT so no existing status is renumbered. */
|
||||
STATUS_DEST_INFO
|
||||
};
|
||||
|
||||
void io_set_fds(int read_fd, int write_fd);
|
||||
@@ -141,6 +196,20 @@ void protocol_session_unbind(void);
|
||||
void protocol_session_set_ssl(ProtocolSession* session, SSL* ssl);
|
||||
void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec);
|
||||
void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc);
|
||||
/* Override the per-message send/receive deadline for this session. The value
|
||||
* is stored verbatim: a positive value sets the deadline, `sec` <= 0 disables
|
||||
* it (rsync's --timeout=0). An explicit long deadline (e.g. the delete-ack
|
||||
* wait) is applied per-call by protocol_receive_status_timed and is unaffected
|
||||
* by this setter. */
|
||||
void protocol_session_set_io_timeout(ProtocolSession* session, int sec);
|
||||
/* Effective per-message I/O deadline (seconds) for the currently-bound session.
|
||||
* Zero means the deadline is disabled (rsync's --timeout=0). Used by the
|
||||
* plaintext sendfile path which bypasses the protocol send primitive. */
|
||||
int protocol_get_io_timeout_sec(void);
|
||||
/* The server-side effective deadline for a client-requested timeout: a positive
|
||||
* client value is honored, otherwise the SERVER_IO_TIMEOUT_SEC floor applies so
|
||||
* a silent peer can never hold a session open forever. */
|
||||
int protocol_server_io_timeout_sec(int client_timeout);
|
||||
void* protocol_alloc(size_t size);
|
||||
void* protocol_realloc(void* ptr, size_t size);
|
||||
void protocol_session_set_8_bit_output(ProtocolSession* session, bool enabled);
|
||||
@@ -157,7 +226,6 @@ char* protocol_receive_str(ProtocolSession* session);
|
||||
bool protocol_send_str_redacted(ProtocolSession* session, const char* data);
|
||||
char* protocol_receive_str_redacted(ProtocolSession* session);
|
||||
bool protocol_send_data(ProtocolSession* session, const Data* data);
|
||||
Data* protocol_receive_data(ProtocolSession* session);
|
||||
Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long maximum_size);
|
||||
bool protocol_send_int(ProtocolSession* session, int data);
|
||||
bool protocol_receive_int(ProtocolSession* session, int* data);
|
||||
@@ -178,10 +246,43 @@ bool send_int(int file_descriptor, int data);
|
||||
bool receive_int(int file_descriptor, int* data);
|
||||
bool send_status(int file_descriptor, Status status);
|
||||
bool receive_status(int file_descriptor, Status* status);
|
||||
/* Send STATUS_ERROR_DETAIL followed by a bounded (<= MAX_ERROR_DETAIL_BYTES)
|
||||
* length-prefixed string. Over-long messages are sliced and NULL is treated
|
||||
* as "". Returns false if the status or the string could not be sent. */
|
||||
bool send_error_detail(int file_descriptor, const char* message);
|
||||
/* Human-readable reason captured from the most recent STATUS_ERROR_DETAIL
|
||||
* received on this thread, or "" when the last status was a bare STATUS_ERROR
|
||||
* (or no detail was seen). Thread-local, and valid until the next non-keepalive
|
||||
* status read on the same thread; a later STATUS_KEEPALIVE does NOT clear it.
|
||||
* The detail body is bounded by MAX_ERROR_DETAIL_BYTES: an over-cap declared
|
||||
* length is drained and yields "" (so the stream never desyncs), while an
|
||||
* absurd length is a fatal framing error that fails the status read. */
|
||||
const char* protocol_last_error(void);
|
||||
/* Clear the thread-local last-error buffer. */
|
||||
void protocol_clear_last_error(void);
|
||||
/* receive_status with an explicit per-message deadline in seconds, instead of
|
||||
the default RECEIVE_TIMEOUT_SEC. A reply that may legitimately take longer
|
||||
(e.g. the early-delete ACK after a large receiver-side deletion) must use
|
||||
this so the sender does not abort after the deletion already committed. */
|
||||
bool receive_status_timed(int file_descriptor, Status* status, int timeout_sec);
|
||||
|
||||
/* Callback polled by protocol_receive_status_keepalive once per keepalive
|
||||
interval. Return true to stop waiting (e.g. a SIGINT/SIGTERM abort flag was
|
||||
set). Kept as a function pointer so the protocol layer does not depend on
|
||||
client signal state. */
|
||||
typedef bool (*ProtocolWaitAbort)(void);
|
||||
|
||||
/* Like receive_status_timed, but while the peer is silent it emits
|
||||
STATUS_KEEPALIVE every keepalive_interval_sec (the receiver answers each with
|
||||
STATUS_KEEPALIVE, which this function consumes and skips) so a long
|
||||
server-side operation does not look like a dead connection. The total wait
|
||||
is still bounded by timeout_sec; abort_check (may be NULL) is polled every
|
||||
interval and, when it returns true, ends the wait immediately with false.
|
||||
Runs entirely on the calling thread: the protocol send path is NOT safe for
|
||||
concurrent writers, so this must not be paired with a helper thread. */
|
||||
bool receive_status_keepalive(int file_descriptor, Status* status, int timeout_sec,
|
||||
int keepalive_interval_sec, ProtocolWaitAbort abort_check);
|
||||
bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, int timeout_sec,
|
||||
int keepalive_interval_sec, ProtocolWaitAbort abort_check);
|
||||
|
||||
#endif
|
||||
+181
-21
@@ -44,6 +44,181 @@ static bool parse_two_digits(const char* s, int* out) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* True when the current character of the cursor is a decimal digit. */
|
||||
static bool is_digit(const char* cp) {
|
||||
return *cp >= '0' && *cp <= '9';
|
||||
}
|
||||
|
||||
/* rsync 3.4.1's flexible --stop-at date parser (ported from
|
||||
* options.c:parse_time). Returns a time_t, or (time_t)-1 on a malformed value.
|
||||
* Accepted forms include Y-M-DTh:m, Y/M/DTh:m, Y-M-D, M-D, D, h:m, :m and
|
||||
* "T h:m"; a 1- or 2-digit year and omitted fields are resolved to the next
|
||||
* matching point in time in the local timezone. Seconds are NOT accepted
|
||||
* (rsync rejects them too); FastSync keeps its own HH:MM:SS spelling as an
|
||||
* extension handled by the caller. `now` is passed in so tests are
|
||||
* deterministic; production passes time(NULL). */
|
||||
static time_t parse_time_rsync(const char* value, time_t now) {
|
||||
const char* cp;
|
||||
time_t val;
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return (time_t)-1;
|
||||
struct tm t;
|
||||
int in_date, old_mday, n;
|
||||
|
||||
memset(&t, 0, sizeof t);
|
||||
t.tm_year = t.tm_mon = t.tm_mday = -1;
|
||||
t.tm_hour = t.tm_min = t.tm_isdst = -1;
|
||||
cp = value;
|
||||
if (*cp == 'T' || *cp == 't' || *cp == ':') {
|
||||
in_date = *cp == ':' ? 0 : -1;
|
||||
cp++;
|
||||
} else
|
||||
in_date = 1;
|
||||
for (;; cp++) {
|
||||
if (!is_digit(cp))
|
||||
return (time_t)-1;
|
||||
n = 0;
|
||||
do {
|
||||
n = n * 10 + *cp++ - '0';
|
||||
} while (is_digit(cp));
|
||||
if (*cp == ':')
|
||||
in_date = 0;
|
||||
if (in_date > 0) {
|
||||
if (t.tm_year != -1)
|
||||
return (time_t)-1;
|
||||
t.tm_year = t.tm_mon;
|
||||
t.tm_mon = t.tm_mday;
|
||||
t.tm_mday = n;
|
||||
if (!*cp)
|
||||
break;
|
||||
if (*cp == 'T' || *cp == 't') {
|
||||
if (!cp[1])
|
||||
break;
|
||||
in_date = -1;
|
||||
} else if (*cp != '-' && *cp != '/')
|
||||
return (time_t)-1;
|
||||
continue;
|
||||
}
|
||||
if (t.tm_hour != -1)
|
||||
return (time_t)-1;
|
||||
t.tm_hour = t.tm_min;
|
||||
t.tm_min = n;
|
||||
if (!*cp) {
|
||||
if (in_date < 0)
|
||||
return (time_t)-1;
|
||||
break;
|
||||
}
|
||||
if (*cp != ':')
|
||||
return (time_t)-1;
|
||||
in_date = 0;
|
||||
}
|
||||
|
||||
in_date = 0;
|
||||
if (t.tm_year < 0) {
|
||||
t.tm_year = today.tm_year;
|
||||
in_date = 1;
|
||||
} else if (t.tm_year < 100) {
|
||||
while (t.tm_year < today.tm_year)
|
||||
t.tm_year += 100;
|
||||
} else
|
||||
t.tm_year -= 1900;
|
||||
if (t.tm_mon < 0) {
|
||||
t.tm_mon = today.tm_mon;
|
||||
in_date = 2;
|
||||
} else
|
||||
t.tm_mon--;
|
||||
if (t.tm_mday < 0) {
|
||||
t.tm_mday = today.tm_mday;
|
||||
in_date = 3;
|
||||
}
|
||||
|
||||
n = 0;
|
||||
if (t.tm_min < 0) {
|
||||
t.tm_hour = t.tm_min = 0;
|
||||
} else if (t.tm_hour < 0) {
|
||||
if (in_date != 3)
|
||||
return (time_t)-1;
|
||||
in_date = 0;
|
||||
t.tm_hour = today.tm_hour;
|
||||
n = 60 * 60;
|
||||
}
|
||||
|
||||
/* mktime() may roll a too-large tm_mday into the following month; undo that
|
||||
* in the "next match" loop below. */
|
||||
old_mday = t.tm_mday;
|
||||
if (t.tm_hour > 23 || t.tm_min > 59 || t.tm_mon < 0 || t.tm_mon >= 12 || t.tm_mday < 1 ||
|
||||
t.tm_mday > 31 || (val = mktime(&t)) == (time_t)-1)
|
||||
return (time_t)-1;
|
||||
|
||||
while (in_date && (val <= now || t.tm_mday < old_mday)) {
|
||||
switch (in_date) {
|
||||
case 3:
|
||||
old_mday = ++t.tm_mday;
|
||||
break;
|
||||
case 2:
|
||||
if (t.tm_mday < old_mday)
|
||||
t.tm_mday = old_mday; /* the month already got bumped forward */
|
||||
else if (++t.tm_mon == 12) {
|
||||
t.tm_mon = 0;
|
||||
t.tm_year++;
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
if (t.tm_mday < old_mday) {
|
||||
/* mon==1 mday==29 got bumped to mon==2 */
|
||||
if (t.tm_mon != 2 || old_mday != 29)
|
||||
return (time_t)-1;
|
||||
t.tm_mon = 1;
|
||||
t.tm_mday = 29;
|
||||
}
|
||||
t.tm_year++;
|
||||
break;
|
||||
}
|
||||
if ((val = mktime(&t)) == (time_t)-1) {
|
||||
if (in_date != 3 || t.tm_mday <= 28)
|
||||
return (time_t)-1;
|
||||
t.tm_mday = old_mday = 1;
|
||||
in_date = 2;
|
||||
}
|
||||
}
|
||||
if (n) {
|
||||
while (val <= now)
|
||||
val += n;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
/* FastSync's HH:MM or HH:MM:SS spelling on the current local day. rsync's own
|
||||
* --stop-at accepts only HH:MM, so this is a strict superset extension. */
|
||||
static bool parse_clock_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
size_t len = strlen(value);
|
||||
if (len != 5 && len != 8)
|
||||
return false;
|
||||
if (value[2] != ':' || (len == 8 && value[5] != ':'))
|
||||
return false;
|
||||
int hh, mm, ss = 0;
|
||||
if (!parse_two_digits(value, &hh) || !parse_two_digits(value + 3, &mm))
|
||||
return false;
|
||||
if (len == 8 && !parse_two_digits(value + 6, &ss))
|
||||
return false;
|
||||
if (hh > 23 || mm > 59 || ss > 59)
|
||||
return false;
|
||||
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return false;
|
||||
today.tm_hour = hh;
|
||||
today.tm_min = mm;
|
||||
today.tm_sec = ss;
|
||||
today.tm_isdst = -1;
|
||||
time_t deadline = mktime(&today);
|
||||
if (deadline == (time_t)-1)
|
||||
return false;
|
||||
*out_deadline = deadline;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
if (!value || !out_deadline)
|
||||
return false;
|
||||
@@ -91,28 +266,13 @@ bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* HH:MM or HH:MM:SS on the current local day. */
|
||||
size_t len = strlen(value);
|
||||
if (len != 5 && len != 8)
|
||||
return false;
|
||||
if (value[2] != ':' || (len == 8 && value[5] != ':'))
|
||||
return false;
|
||||
int hh, mm, ss = 0;
|
||||
if (!parse_two_digits(value, &hh) || !parse_two_digits(value + 3, &mm))
|
||||
return false;
|
||||
if (len == 8 && !parse_two_digits(value + 6, &ss))
|
||||
return false;
|
||||
if (hh > 23 || mm > 59 || ss > 59)
|
||||
return false;
|
||||
/* HH:MM or HH:MM:SS on the current local day (FastSync extension). */
|
||||
if (parse_clock_time(value, now, out_deadline))
|
||||
return true;
|
||||
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return false;
|
||||
today.tm_hour = hh;
|
||||
today.tm_min = mm;
|
||||
today.tm_sec = ss;
|
||||
today.tm_isdst = -1;
|
||||
time_t deadline = mktime(&today);
|
||||
/* rsync's full/partial date-and-time form (e.g. 2000-12-31T23:59, 12-31,
|
||||
* 14:00, :59, 1, 1-30). */
|
||||
time_t deadline = parse_time_rsync(value, now);
|
||||
if (deadline == (time_t)-1)
|
||||
return false;
|
||||
*out_deadline = deadline;
|
||||
|
||||
@@ -75,6 +75,15 @@ static int parse_remote_dest(const char* dest, RemoteDest* r) {
|
||||
memcpy(r->host, dest, host_len);
|
||||
r->host[host_len] = '\0';
|
||||
}
|
||||
/* The user@host token is handed to ssh in option position. Reject anything
|
||||
* that ssh would consume as an option (a leading '-') or an empty host, so a
|
||||
* crafted destination can never inject an ssh option such as
|
||||
* -oProxyCommand=... . This mirrors config_parse_ssh_dest's validation and
|
||||
* is defense-in-depth for callers that bypass it. */
|
||||
if (r->host[0] == '\0' || r->host[0] == '-' || r->user[0] == '-') {
|
||||
remote_dest_destroy(r);
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -128,7 +137,7 @@ char* ssh_build_remote_command(const char* server_path, bool old_args, char* con
|
||||
q++;
|
||||
len++;
|
||||
}
|
||||
if (len > SIZE_MAX - q * 3 || len + q * 3 + 3 > SIZE_MAX - command_len)
|
||||
if (q > (SIZE_MAX - len) / 3 || len + q * 3 + 3 > SIZE_MAX - command_len)
|
||||
return NULL;
|
||||
command_len += len + q * 3 + 3;
|
||||
}
|
||||
@@ -216,10 +225,10 @@ char** ssh_build_client_argv(const char* rsh_command, int port, const char* user
|
||||
nwords = 1;
|
||||
}
|
||||
|
||||
/* Fixed tail: three -o pairs (6) + optional -p/value (2) + user@host +
|
||||
* remote command + terminating NULL. */
|
||||
/* Fixed tail: three -o pairs (6) + optional -p/value (2) + the "--" end of
|
||||
* options marker + user@host + remote command + terminating NULL. */
|
||||
int port_extra = (port > 0 && port != 22) ? 2 : 0;
|
||||
size_t total = (size_t)nwords + 6 + (size_t)port_extra + 3;
|
||||
size_t total = (size_t)nwords + 6 + (size_t)port_extra + 4;
|
||||
char** argv = calloc(total, sizeof(char*));
|
||||
if (!argv) {
|
||||
for (int i = 0; i < nwords; i++)
|
||||
@@ -253,6 +262,13 @@ char** ssh_build_client_argv(const char* rsh_command, int port, const char* user
|
||||
goto fail_argv;
|
||||
ac++;
|
||||
}
|
||||
/* End of options: guarantees the user@host token that follows is treated as
|
||||
* the destination and never re-interpreted as an ssh option, even if every
|
||||
* caller-side validation were bypassed. */
|
||||
argv[ac] = str_dup("--");
|
||||
if (!argv[ac])
|
||||
goto fail_argv;
|
||||
ac++;
|
||||
argv[ac] = str_dup(userhost);
|
||||
if (!argv[ac])
|
||||
goto fail_argv;
|
||||
|
||||
+141
-17
@@ -1,4 +1,5 @@
|
||||
#include "transport_tcp.h"
|
||||
#include "daemon_limits.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
@@ -8,6 +9,7 @@
|
||||
#include <netinet/in.h>
|
||||
#include <netinet/tcp.h>
|
||||
#include <openssl/ssl.h>
|
||||
#include <pthread.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
@@ -18,18 +20,45 @@
|
||||
|
||||
static volatile sig_atomic_t g_active_connections = 0;
|
||||
|
||||
/* Shared registry installed on the active server; the SIGCHLD handler needs a
|
||||
* file-scope pointer so it can reclaim the dead child's slot. Set once by
|
||||
* accept_loop before the fork loop (single-threaded parent). */
|
||||
static DaemonLimitRegistry* g_limit_registry = NULL;
|
||||
/* Slot reserved by the parent for the connection child currently being forked.
|
||||
* Written before fork(), read by the child (which inherits the value). */
|
||||
static int g_current_slot = DAEMON_LIMITS_NO_SLOT;
|
||||
|
||||
static void tcp_apply_socket_timeout(int fd);
|
||||
static void tcp_enable_nodelay_default(int fd, int family);
|
||||
|
||||
static void sigchld_handler(int sig) {
|
||||
(void)sig;
|
||||
int saved_errno = errno;
|
||||
while (waitpid(-1, NULL, WNOHANG) > 0) {
|
||||
pid_t pid;
|
||||
while ((pid = waitpid(-1, NULL, WNOHANG)) > 0) {
|
||||
if (g_active_connections > 0)
|
||||
g_active_connections--;
|
||||
daemon_limits_reclaim_pid(g_limit_registry, (long)pid);
|
||||
}
|
||||
/* Re-derive the occupancy counters once for the whole reap batch. The slot
|
||||
* table is the source of truth, so this self-heals any count leaked by a child
|
||||
* SIGKILLed mid-registration. Atomics only: async-signal-safe. */
|
||||
if (g_limit_registry)
|
||||
daemon_limits_recompute(g_limit_registry);
|
||||
errno = saved_errno;
|
||||
}
|
||||
|
||||
/* Reset a signal to its default action with sigaction (preferred over
|
||||
* signal(3), whose semantics are implementation-defined). Used in the forked
|
||||
* child before it can spawn any thread. */
|
||||
static void reset_signal_default(int sig) {
|
||||
struct sigaction action;
|
||||
memset(&action, 0, sizeof(action));
|
||||
action.sa_handler = SIG_DFL;
|
||||
sigemptyset(&action.sa_mask);
|
||||
sigaction(sig, &action, NULL);
|
||||
}
|
||||
|
||||
/* Map a listen socket's address to its numeric port for logging, independent
|
||||
* of whether it is an IPv4 or IPv6 sockaddr. */
|
||||
static unsigned short server_address_port(const struct sockaddr_storage* addr) {
|
||||
@@ -107,6 +136,7 @@ Server* server_create_ex(int port, const ServerBindOptions* bind_opts) {
|
||||
server->ssl_ctx = NULL;
|
||||
server->max_connections = 100;
|
||||
server->active_connections = 0;
|
||||
server->limit_registry = NULL;
|
||||
|
||||
return server;
|
||||
}
|
||||
@@ -115,6 +145,20 @@ Server* server_create(int port) {
|
||||
return server_create_ex(port, NULL);
|
||||
}
|
||||
|
||||
void server_set_max_connections(Server* server, unsigned int max_connections) {
|
||||
if (server && max_connections > 0)
|
||||
server->max_connections = max_connections;
|
||||
}
|
||||
|
||||
void server_set_limit_registry(Server* server, struct DaemonLimitRegistry* registry) {
|
||||
if (server)
|
||||
server->limit_registry = registry;
|
||||
}
|
||||
|
||||
int transport_tcp_current_slot(void) {
|
||||
return g_current_slot;
|
||||
}
|
||||
|
||||
void server_delete(Server** server) {
|
||||
if (server == NULL || *server == NULL)
|
||||
return;
|
||||
@@ -133,9 +177,19 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
log_perror("Could not listen on port!");
|
||||
return;
|
||||
}
|
||||
signal(SIGCHLD, sigchld_handler);
|
||||
/* SIGCHLD via sigaction (not signal(3)); SA_RESTART keeps accept(2) from
|
||||
* failing with EINTR, and SA_NOCLDSTOP only notifies on child exit. The
|
||||
* accept loop is single-threaded at this point, so installing here cannot race
|
||||
* a worker thread. */
|
||||
struct sigaction chld_action;
|
||||
memset(&chld_action, 0, sizeof(chld_action));
|
||||
chld_action.sa_handler = sigchld_handler;
|
||||
sigemptyset(&chld_action.sa_mask);
|
||||
chld_action.sa_flags = SA_RESTART | SA_NOCLDSTOP;
|
||||
sigaction(SIGCHLD, &chld_action, NULL);
|
||||
g_limit_registry = server->limit_registry;
|
||||
while (1) {
|
||||
struct sockaddr_in client_addr;
|
||||
struct sockaddr_storage client_addr;
|
||||
socklen_t client_len = sizeof(client_addr);
|
||||
int fd = accept(server->file_descriptor, (struct sockaddr*)&client_addr, &client_len);
|
||||
if (fd < 0) {
|
||||
@@ -143,22 +197,69 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
continue;
|
||||
}
|
||||
tcp_apply_socket_timeout(fd);
|
||||
tcp_enable_nodelay_default(fd, client_addr.ss_family);
|
||||
char peer[128];
|
||||
if (!utils_sockaddr_to_string((const struct sockaddr*)&client_addr, peer, sizeof(peer)))
|
||||
snprintf(peer, sizeof(peer), "unknown");
|
||||
if ((unsigned int)g_active_connections >= server->max_connections) {
|
||||
log_message(LOG_LEVEL_WARNING, "Max connections (%u) reached, rejecting",
|
||||
server->max_connections);
|
||||
log_message(LOG_LEVEL_WARNING, "Max connections (%u) reached, rejecting %s",
|
||||
server->max_connections, peer);
|
||||
close(fd);
|
||||
continue;
|
||||
}
|
||||
log_message(LOG_LEVEL_INFO, "%s", log_fmt);
|
||||
int slot = DAEMON_LIMITS_NO_SLOT;
|
||||
if (server->limit_registry) {
|
||||
slot = daemon_limits_claim_slot(server->limit_registry);
|
||||
if (slot == DAEMON_LIMITS_NO_SLOT) {
|
||||
/* The global cap bounds live children, so this only happens when the
|
||||
* fixed registry is smaller than the configured cap; fail closed. */
|
||||
log_message(LOG_LEVEL_WARNING, "Connection registry slots exhausted (max %u), rejecting %s",
|
||||
server->max_connections, peer);
|
||||
close(fd);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
log_message(LOG_LEVEL_INFO, "%s from %s", log_fmt, peer);
|
||||
g_current_slot = slot;
|
||||
/* Block SIGCHLD across fork() and the parent's pid publication: a child
|
||||
* that exits immediately must not be reaped before its slot records its
|
||||
* pid, which would leak the slot and its module/source counts. Use
|
||||
* pthread_sigmask rather than sigprocmask so the behavior is well defined
|
||||
* even if this process ever gains threads: the mask is per-thread, the fork
|
||||
* copies only the calling thread, and the child inherits this thread's
|
||||
* blocked mask until it restores `previous` below. No thread exists yet at
|
||||
* this point, and none is created before the mask is restored, so the
|
||||
* critical window is race-free. */
|
||||
sigset_t blocked;
|
||||
sigset_t previous;
|
||||
sigemptyset(&blocked);
|
||||
sigaddset(&blocked, SIGCHLD);
|
||||
pthread_sigmask(SIG_BLOCK, &blocked, &previous);
|
||||
pid_t pid = fork();
|
||||
if (pid == 0) {
|
||||
pthread_sigmask(SIG_SETMASK, &previous, NULL);
|
||||
/* Connection children must not run the parent's global cleanup(): it
|
||||
* frees state (credentials / daemon conf) that the child's worker
|
||||
* threads may still be reading and closes fd numbers the child could
|
||||
* already have reused. Reset the inherited handlers so a signal
|
||||
* terminates the child directly; SIGCHLD is reset too since a child
|
||||
* must never reap the parent's children. This runs before the child
|
||||
* spawns any thread, so it cannot race one. */
|
||||
reset_signal_default(SIGINT);
|
||||
reset_signal_default(SIGTERM);
|
||||
reset_signal_default(SIGCHLD);
|
||||
close(server->file_descriptor);
|
||||
child_fn(fd, child_ctx);
|
||||
close(fd);
|
||||
_exit(0);
|
||||
} else if (pid > 0) {
|
||||
g_active_connections++;
|
||||
if (server->limit_registry)
|
||||
daemon_limits_set_slot_pid(server->limit_registry, slot, (long)pid);
|
||||
} else if (server->limit_registry) {
|
||||
/* fork() failed: release the reservation so the slot is not leaked. */
|
||||
daemon_limits_reclaim_slot(server->limit_registry, slot);
|
||||
}
|
||||
pthread_sigmask(SIG_SETMASK, &previous, NULL);
|
||||
close(fd);
|
||||
}
|
||||
}
|
||||
@@ -169,6 +270,9 @@ struct plain_ctx {
|
||||
|
||||
static void plain_child_fn(int fd, void* ctx) {
|
||||
((struct plain_ctx*)ctx)->handler(fd);
|
||||
/* handler() never closes the connection fd; the child owns its single
|
||||
* close here after the handler has fully torn down. */
|
||||
close(fd);
|
||||
}
|
||||
|
||||
bool server_listen(Server* server, void (*handler)(int file_descriptor)) {
|
||||
@@ -185,14 +289,14 @@ void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
accept_loop(server, child_fn, child_ctx, log_fmt);
|
||||
}
|
||||
|
||||
static int g_timeout_sec = 30;
|
||||
static int g_contimeout_sec = 10;
|
||||
/* rsync defaults: --timeout=0 (disabled) and --contimeout=60. A non-positive
|
||||
* value means "no timeout" rather than "leave the built-in value in place". */
|
||||
static int g_timeout_sec = 0;
|
||||
static int g_contimeout_sec = 60;
|
||||
|
||||
void tcp_set_timeouts(int timeout_sec, int contimeout_sec) {
|
||||
if (timeout_sec > 0)
|
||||
g_timeout_sec = timeout_sec;
|
||||
if (contimeout_sec > 0)
|
||||
g_contimeout_sec = contimeout_sec;
|
||||
g_timeout_sec = timeout_sec > 0 ? timeout_sec : 0;
|
||||
g_contimeout_sec = contimeout_sec > 0 ? contimeout_sec : 0;
|
||||
}
|
||||
|
||||
int tcp_get_contimeout_sec(void) {
|
||||
@@ -204,6 +308,10 @@ int tcp_get_timeout_sec(void) {
|
||||
}
|
||||
|
||||
static void tcp_apply_socket_timeout(int fd) {
|
||||
/* timeout 0 means no timeout: leave the socket in its default (blocking)
|
||||
* mode instead of installing a zero SO_RCVTIMEO/SO_SNDTIMEO. */
|
||||
if (g_timeout_sec <= 0)
|
||||
return;
|
||||
struct timeval tv;
|
||||
tv.tv_sec = g_timeout_sec;
|
||||
tv.tv_usec = 0;
|
||||
@@ -211,6 +319,19 @@ static void tcp_apply_socket_timeout(int fd) {
|
||||
setsockopt(fd, SOL_SOCKET, SO_SNDTIMEO, &tv, sizeof(tv));
|
||||
}
|
||||
|
||||
/* Enable TCP_NODELAY by default on a transfer socket: the protocol emits many
|
||||
* small messages and Nagle's algorithm would otherwise coalesce/delay them.
|
||||
* Best-effort only: the family guard keeps this to IP/TCP sockets, and a
|
||||
* setsockopt failure is ignored. A caller-provided --sockopts TCP_NODELAY=0
|
||||
* is applied afterwards on the connect path, so an explicit user choice still
|
||||
* wins. */
|
||||
static void tcp_enable_nodelay_default(int fd, int family) {
|
||||
if (family != AF_INET && family != AF_INET6)
|
||||
return;
|
||||
int value = 1;
|
||||
setsockopt(fd, IPPROTO_TCP, TCP_NODELAY, &value, sizeof(value));
|
||||
}
|
||||
|
||||
Client* client_create() {
|
||||
Client* client = (Client*)malloc(sizeof(Client));
|
||||
if (client == NULL) {
|
||||
@@ -356,6 +477,9 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
if (client->file_descriptor < 0)
|
||||
continue;
|
||||
|
||||
/* Default first; a user --sockopts TCP_NODELAY=0 applied below overrides. */
|
||||
tcp_enable_nodelay_default(client->file_descriptor, rp->ai_family);
|
||||
|
||||
if (opts && opts->sockopt_count > 0 &&
|
||||
!tcp_apply_sockopts(client->file_descriptor, opts->sockopts, opts->sockopt_count)) {
|
||||
close(client->file_descriptor);
|
||||
@@ -363,11 +487,15 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
break;
|
||||
}
|
||||
|
||||
/* --contimeout=0 disables the connect timeout: skip the pre-connect socket
|
||||
* timeouts entirely. */
|
||||
if (g_contimeout_sec > 0) {
|
||||
struct timeval ct;
|
||||
ct.tv_sec = g_contimeout_sec;
|
||||
ct.tv_usec = 0;
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &ct, sizeof(ct));
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &ct, sizeof(ct));
|
||||
}
|
||||
|
||||
if (bind_addr_family != 0) {
|
||||
if (rp->ai_family != bind_addr_family) {
|
||||
@@ -402,10 +530,6 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
return true;
|
||||
}
|
||||
|
||||
bool tcp_connect_socket(Client* client, const char* host, int port) {
|
||||
return tcp_connect_socket_ex(client, host, port, NULL);
|
||||
}
|
||||
|
||||
bool client_connect_ex(Client* client, const char* host, int port, const TcpConnectOptions* opts) {
|
||||
if (!tcp_connect_socket_ex(client, host, port, opts))
|
||||
return false;
|
||||
|
||||
@@ -7,6 +7,10 @@
|
||||
#include <stdbool.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
/* Cross-process daemon registry (daemon_limits.c). Only an opaque pointer is
|
||||
* stored here so the transport layer does not depend on daemon config. */
|
||||
struct DaemonLimitRegistry;
|
||||
|
||||
typedef struct Server {
|
||||
struct sockaddr_storage address;
|
||||
unsigned int address_length;
|
||||
@@ -14,6 +18,7 @@ typedef struct Server {
|
||||
void* ssl_ctx;
|
||||
unsigned int max_connections;
|
||||
volatile unsigned int active_connections;
|
||||
struct DaemonLimitRegistry* limit_registry;
|
||||
} Server;
|
||||
|
||||
typedef struct Client {
|
||||
@@ -45,6 +50,17 @@ typedef struct {
|
||||
|
||||
Server* server_create_ex(int port, const ServerBindOptions* bind_opts);
|
||||
Server* server_create(int port);
|
||||
/* Override the listener's connection cap (the global daemon `max connections`
|
||||
* value). A non-positive value is ignored so the default cap stands. */
|
||||
void server_set_max_connections(Server* server, unsigned int max_connections);
|
||||
/* Install the shared per-module / per-source registry used by the accept loop
|
||||
* to reserve a slot for each forked child. NULL disables the accounting (the
|
||||
* global cap and ACLs still apply). */
|
||||
void server_set_limit_registry(Server* server, struct DaemonLimitRegistry* registry);
|
||||
/* Slot reserved for the connection child currently running (set by the parent
|
||||
* before fork, inherited by the child). Returns DAEMON_LIMITS_NO_SLOT (-1)
|
||||
* outside the accept-loop child path. */
|
||||
int transport_tcp_current_slot(void);
|
||||
bool server_listen(Server* server, void (*handler)(int file_descriptor));
|
||||
void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* child_ctx,
|
||||
const char* log_fmt);
|
||||
@@ -54,7 +70,6 @@ bool client_connect_ex(Client* client, const char* host, int port, const TcpConn
|
||||
bool client_connect(Client* client, const char* host, int port);
|
||||
bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
const TcpConnectOptions* opts);
|
||||
bool tcp_connect_socket(Client* client, const char* host, int port);
|
||||
void client_disconnect(Client* client);
|
||||
void client_delete(Client* client);
|
||||
void tcp_set_timeouts(int timeout_sec, int contimeout_sec);
|
||||
|
||||
+93
-15
@@ -4,7 +4,9 @@
|
||||
#include "transport_tcp.h"
|
||||
#include "utils.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <fcntl.h>
|
||||
#include <openssl/err.h>
|
||||
#include <openssl/pem.h>
|
||||
#include <openssl/ssl.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
@@ -34,6 +36,59 @@ static void log_ssl_errors(void) {
|
||||
}
|
||||
}
|
||||
|
||||
/* Load the TLS private key through an already-opened, no-follow descriptor so
|
||||
* the owner/mode policy is checked on the SAME file object that is loaded: an
|
||||
* attacker cannot swap the path between a stat() and a later open() (TOCTOU).
|
||||
* The exact-owner / 0600 policy is preserved and group/other execute bits are
|
||||
* rejected as well. Ownership of the descriptor passes to the BIO and is
|
||||
* released exactly once by BIO_free() (BIO_CLOSE). */
|
||||
static bool load_private_key_secure(SSL_CTX* ctx, const char* key) {
|
||||
int fd = open(key, O_RDONLY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (fd < 0) {
|
||||
char* escaped = output_escape(key, false);
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to open private key: %s",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
struct stat key_stat;
|
||||
if (fstat(fd, &key_stat) != 0 || !S_ISREG(key_stat.st_mode) || key_stat.st_uid != geteuid() ||
|
||||
(key_stat.st_mode & (S_IRGRP | S_IWGRP | S_IROTH | S_IWOTH | S_IXGRP | S_IXOTH))) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"TLS private key must be a regular file owned by the current user and private "
|
||||
"(mode 0600)");
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
BIO* bio = BIO_new_fd(fd, BIO_CLOSE);
|
||||
if (!bio) {
|
||||
close(fd);
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to read private key");
|
||||
return false;
|
||||
}
|
||||
EVP_PKEY* pkey = PEM_read_bio_PrivateKey(bio, NULL, NULL, NULL);
|
||||
BIO_free(bio); /* releases fd via BIO_CLOSE */
|
||||
if (!pkey) {
|
||||
char* escaped = output_escape(key, false);
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to load private key: %s",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
log_ssl_errors();
|
||||
return false;
|
||||
}
|
||||
int use_ok = SSL_CTX_use_PrivateKey(ctx, pkey);
|
||||
EVP_PKEY_free(pkey);
|
||||
if (use_ok != 1) {
|
||||
char* escaped = output_escape(key, false);
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to use private key: %s",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
log_ssl_errors();
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static SSL_CTX* create_ssl_ctx(bool is_server, const char* cert, const char* key,
|
||||
const char* ca_path) {
|
||||
if (!is_server && !ca_path) {
|
||||
@@ -56,24 +111,38 @@ static SSL_CTX* create_ssl_ctx(bool is_server, const char* cert, const char* key
|
||||
#ifdef SSL_OP_NO_RENEGOTIATION
|
||||
SSL_CTX_set_options(ctx, SSL_OP_NO_RENEGOTIATION);
|
||||
#endif
|
||||
/* Let the server's own preference order decide the negotiated cipher rather
|
||||
* than the client's, so a client cannot steer both peers into a weaker (but
|
||||
* still offered) suite. */
|
||||
SSL_CTX_set_options(ctx, SSL_OP_CIPHER_SERVER_PREFERENCE);
|
||||
|
||||
if (SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION) != 1) {
|
||||
SSL_CTX_free(ctx);
|
||||
return NULL;
|
||||
}
|
||||
if (SSL_CTX_set_cipher_list(ctx, "HIGH:!aNULL:!eNULL:!MD5:!RC4:!3DES") != 1) {
|
||||
/* TLS 1.2 and below: an AEAD-only suite list. "HIGH" still includes CBC
|
||||
* suites (Lucky13/POODLE-adjacent MAC-then-encrypt constructions), so restrict
|
||||
* the list to ECDHE key agreement with an AEAD record cipher (AES-GCM or
|
||||
* ChaCha20-Poly1305). A NULL/weak/3DES cipher is never selectable. */
|
||||
if (SSL_CTX_set_cipher_list(ctx, "ECDHE+AESGCM:ECDHE+CHACHA20:!aNULL:!eNULL:!MD5:!RC4:!3DES") !=
|
||||
1) {
|
||||
SSL_CTX_free(ctx);
|
||||
return NULL;
|
||||
}
|
||||
/* TLS 1.3 ciphersuites are configured separately from the TLS 1.2 and below
|
||||
* cipher list above. Pin the three AEAD suites OpenSSL offers, dropping
|
||||
* TLS_AES_128_CCM_SHA256 and the CCM_8 variant, and fail closed if the
|
||||
* library rejects the policy. SSL_CTX_set_ciphersuites needs OpenSSL 1.1.1;
|
||||
* earlier versions have no TLS 1.3, so the call is compile-guarded. */
|
||||
#if OPENSSL_VERSION_NUMBER >= 0x10101000L
|
||||
if (SSL_CTX_set_ciphersuites(
|
||||
ctx, "TLS_AES_256_GCM_SHA384:TLS_CHACHA20_POLY1305_SHA256:TLS_AES_128_GCM_SHA256") != 1) {
|
||||
SSL_CTX_free(ctx);
|
||||
return NULL;
|
||||
}
|
||||
#endif
|
||||
|
||||
if (cert && key) {
|
||||
struct stat key_stat;
|
||||
if (stat(key, &key_stat) != 0 || !S_ISREG(key_stat.st_mode) || key_stat.st_uid != geteuid() ||
|
||||
(key_stat.st_mode & (S_IRGRP | S_IWGRP | S_IROTH | S_IWOTH))) {
|
||||
log_message(LOG_LEVEL_ERROR, "TLS private key must be owned by the current user and private");
|
||||
SSL_CTX_free(ctx);
|
||||
return NULL;
|
||||
}
|
||||
if (SSL_CTX_use_certificate_file(ctx, cert, SSL_FILETYPE_PEM) <= 0) {
|
||||
char* escaped = output_escape(cert, false);
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to load certificate: %s",
|
||||
@@ -83,12 +152,7 @@ static SSL_CTX* create_ssl_ctx(bool is_server, const char* cert, const char* key
|
||||
SSL_CTX_free(ctx);
|
||||
return NULL;
|
||||
}
|
||||
if (SSL_CTX_use_PrivateKey_file(ctx, key, SSL_FILETYPE_PEM) <= 0) {
|
||||
char* escaped = output_escape(key, false);
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to load private key: %s",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
log_ssl_errors();
|
||||
if (!load_private_key_secure(ctx, key)) {
|
||||
SSL_CTX_free(ctx);
|
||||
return NULL;
|
||||
}
|
||||
@@ -132,7 +196,16 @@ static SSL* wrap_fd_with_ssl(int fd, SSL_CTX* ctx, bool is_server, const char* h
|
||||
// Enable hostname verification for client connections when a hostname is provided.
|
||||
// Must be done before SSL_connect to take effect during the handshake.
|
||||
if (!is_server && hostname) {
|
||||
if (SSL_set1_host(ssl, hostname) != 1) {
|
||||
/* An IP-literal host must be verified against the certificate's IP SAN
|
||||
* (X509_check_ip_asc), not as a DNS name: SSL_set1_host would look for a
|
||||
* DNS SAN that a legitimate IP-SAN certificate never carries. */
|
||||
struct in_addr ipv4;
|
||||
struct in6_addr ipv6;
|
||||
bool is_ip_literal =
|
||||
inet_pton(AF_INET, hostname, &ipv4) == 1 || inet_pton(AF_INET6, hostname, &ipv6) == 1;
|
||||
int set_ok = is_ip_literal ? X509_VERIFY_PARAM_set1_ip_asc(SSL_get0_param(ssl), hostname)
|
||||
: SSL_set1_host(ssl, hostname);
|
||||
if (set_ok != 1) {
|
||||
SSL_free(ssl);
|
||||
return NULL;
|
||||
}
|
||||
@@ -180,13 +253,18 @@ static void tls_child_fn(int fd, void* arg) {
|
||||
SSL* ssl = wrap_fd_with_ssl(fd, ctx->ssl_ctx, true, NULL);
|
||||
if (!ssl) {
|
||||
io_set_ssl(NULL);
|
||||
close(fd);
|
||||
return;
|
||||
}
|
||||
io_set_ssl(ssl);
|
||||
ctx->handler(fd);
|
||||
/* Shut the TLS layer down before releasing the fd: handler() no longer
|
||||
* closes it, so SSL_shutdown still has a valid socket. The child owns the
|
||||
* single fd close, performed last. */
|
||||
SSL_shutdown(ssl);
|
||||
SSL_free(ssl);
|
||||
io_set_ssl(NULL);
|
||||
close(fd);
|
||||
}
|
||||
|
||||
bool server_listen_tls(Server* server, void (*handler)(int file_descriptor)) {
|
||||
|
||||
+592
-224
@@ -13,6 +13,7 @@
|
||||
#include <sys/socket.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
#include <xxhash.h>
|
||||
|
||||
static int authorized_root_fd = -1;
|
||||
static char* authorized_root_path;
|
||||
@@ -35,21 +36,41 @@ void utils_set_authorized_root_fd(int fd) {
|
||||
(void)utils_set_authorized_root(fd, NULL);
|
||||
}
|
||||
|
||||
static bool path_is_within_root(const char* root, const char* path) {
|
||||
/* Accessors for the process-global authorized root. The path pointer is
|
||||
* borrowed and valid until the next setter call; the root is a single-threaded,
|
||||
* set-before-worker-threads value (see server.c), so these carry no locking. */
|
||||
int utils_get_authorized_root_fd(void) {
|
||||
return authorized_root_fd;
|
||||
}
|
||||
|
||||
const char* utils_get_authorized_root_path(void) {
|
||||
return authorized_root_path;
|
||||
}
|
||||
|
||||
bool path_is_within_root(const char* root, const char* path) {
|
||||
size_t root_len = strlen(root);
|
||||
return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/');
|
||||
}
|
||||
|
||||
/* Open the destination root directory itself, confined to the authorized root.
|
||||
* NOTE (do not merge with file_open_secure_parent): this walk opens dest_root
|
||||
* (a directory that must already exist) and returns its fd, whereas
|
||||
* file_open_secure_parent resolves the PARENT of a file path, optionally
|
||||
* creating missing components and honouring --keep-dirlinks / --copy-as. The
|
||||
* two differ in create-vs-no-create, in what path component they stop at, and
|
||||
* in the extra receiver policies they apply, so they are intentionally kept
|
||||
* separate. Both rely on the shared lexical path_is_within_root check. */
|
||||
static int open_authorized_destination(const char* dest_root) {
|
||||
if (authorized_root_fd < 0 || !authorized_root_path || !dest_root ||
|
||||
!path_is_within_root(authorized_root_path, dest_root))
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
if (root_fd < 0 || !root_path || !dest_root || !path_is_within_root(root_path, dest_root))
|
||||
return -1;
|
||||
|
||||
int dirfd = dup(authorized_root_fd);
|
||||
int dirfd = dup(root_fd);
|
||||
if (dirfd < 0)
|
||||
return -1;
|
||||
|
||||
const char* relative_path = dest_root + strlen(authorized_root_path);
|
||||
const char* relative_path = dest_root + strlen(root_path);
|
||||
while (*relative_path == '/')
|
||||
relative_path++;
|
||||
char* relative = str_dup(*relative_path ? relative_path : ".");
|
||||
@@ -96,6 +117,241 @@ char* str_dup(const char* string) {
|
||||
return new_string;
|
||||
}
|
||||
|
||||
#define STR_HASH_SET_MIN_CAPACITY 16
|
||||
|
||||
static size_t str_hash_set_hash(const char* key, size_t len) {
|
||||
return (size_t)XXH64(key, len, 0);
|
||||
}
|
||||
|
||||
/* Store a borrowed key. Returns 1 when a new slot was filled and 0 for a
|
||||
* duplicate. */
|
||||
static int str_hash_set_put(StrHashSet* set, const char* key, size_t len) {
|
||||
size_t mask = set->capacity - 1;
|
||||
size_t index = str_hash_set_hash(key, len) & mask;
|
||||
while (true) {
|
||||
StrHashSetSlot* slot = &set->slots[index];
|
||||
if (!slot->key) {
|
||||
slot->key = key;
|
||||
set->size++;
|
||||
return 1;
|
||||
}
|
||||
if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0)
|
||||
return 0;
|
||||
index = (index + 1) & mask;
|
||||
}
|
||||
}
|
||||
|
||||
static bool str_hash_set_resize(StrHashSet* set, size_t new_capacity) {
|
||||
StrHashSetSlot* old_slots = set->slots;
|
||||
size_t old_capacity = set->capacity;
|
||||
StrHashSetSlot* slots = calloc(new_capacity, sizeof(StrHashSetSlot));
|
||||
if (!slots)
|
||||
return false;
|
||||
set->slots = slots;
|
||||
set->capacity = new_capacity;
|
||||
set->size = 0;
|
||||
for (size_t i = 0; i < old_capacity; i++) {
|
||||
if (old_slots[i].key)
|
||||
(void)str_hash_set_put(set, old_slots[i].key, strlen(old_slots[i].key));
|
||||
}
|
||||
free(old_slots);
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool str_hash_set_grow(StrHashSet* set) {
|
||||
if (set->capacity != 0 && (set->size + 1) * 4 <= set->capacity * 3)
|
||||
return true;
|
||||
size_t new_capacity = set->capacity ? set->capacity * 2 : STR_HASH_SET_MIN_CAPACITY;
|
||||
return str_hash_set_resize(set, new_capacity);
|
||||
}
|
||||
|
||||
bool str_hash_set_init(StrHashSet* set, size_t hint) {
|
||||
if (!set)
|
||||
return false;
|
||||
set->slots = NULL;
|
||||
set->capacity = 0;
|
||||
set->size = 0;
|
||||
size_t capacity = STR_HASH_SET_MIN_CAPACITY;
|
||||
while (capacity < (hint + 1) * 2 && capacity <= SIZE_MAX / 2)
|
||||
capacity *= 2;
|
||||
set->slots = calloc(capacity, sizeof(StrHashSetSlot));
|
||||
if (!set->slots)
|
||||
return false;
|
||||
set->capacity = capacity;
|
||||
return true;
|
||||
}
|
||||
|
||||
void str_hash_set_free(StrHashSet* set) {
|
||||
if (!set)
|
||||
return;
|
||||
free(set->slots);
|
||||
set->slots = NULL;
|
||||
set->capacity = 0;
|
||||
set->size = 0;
|
||||
}
|
||||
|
||||
bool str_hash_set_insert_ref(StrHashSet* set, const char* key) {
|
||||
if (!set || !key)
|
||||
return false;
|
||||
if (!str_hash_set_grow(set))
|
||||
return false;
|
||||
/* put() returns 1 for a new slot and 0 for a duplicate; both are success. */
|
||||
(void)str_hash_set_put(set, key, strlen(key));
|
||||
return true;
|
||||
}
|
||||
|
||||
static const StrHashSetSlot* str_hash_set_find_n(const StrHashSet* set, const char* key,
|
||||
size_t len) {
|
||||
if (!set || set->capacity == 0 || !key)
|
||||
return NULL;
|
||||
size_t mask = set->capacity - 1;
|
||||
size_t index = str_hash_set_hash(key, len) & mask;
|
||||
while (true) {
|
||||
const StrHashSetSlot* slot = &set->slots[index];
|
||||
if (!slot->key)
|
||||
return NULL;
|
||||
if (strlen(slot->key) == len && memcmp(slot->key, key, len) == 0)
|
||||
return slot;
|
||||
index = (index + 1) & mask;
|
||||
}
|
||||
}
|
||||
|
||||
bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len) {
|
||||
return str_hash_set_find_n(set, key, len) != NULL;
|
||||
}
|
||||
|
||||
bool str_hash_set_lookup(const StrHashSet* set, const char* key) {
|
||||
if (!key)
|
||||
return false;
|
||||
return str_hash_set_lookup_n(set, key, strlen(key));
|
||||
}
|
||||
|
||||
static int str_sorted_array_compare(const void* left, const void* right) {
|
||||
const char* const* left_key = left;
|
||||
const char* const* right_key = right;
|
||||
return strcmp(*left_key, *right_key);
|
||||
}
|
||||
|
||||
bool str_sorted_array_build(StrSortedArray* array, const char* const* items, size_t count) {
|
||||
if (!array)
|
||||
return false;
|
||||
array->items = NULL;
|
||||
array->count = 0;
|
||||
if (count == 0)
|
||||
return true;
|
||||
if (!items || count > SIZE_MAX / sizeof(const char*))
|
||||
return false;
|
||||
const char** sorted = malloc(count * sizeof(*sorted));
|
||||
if (!sorted)
|
||||
return false;
|
||||
for (size_t i = 0; i < count; i++)
|
||||
sorted[i] = items[i];
|
||||
qsort(sorted, count, sizeof(*sorted), str_sorted_array_compare);
|
||||
array->items = sorted;
|
||||
array->count = count;
|
||||
return true;
|
||||
}
|
||||
|
||||
void str_sorted_array_free(StrSortedArray* array) {
|
||||
if (!array)
|
||||
return;
|
||||
free(array->items);
|
||||
array->items = NULL;
|
||||
array->count = 0;
|
||||
}
|
||||
|
||||
bool str_sorted_array_contains(const StrSortedArray* array, const char* key) {
|
||||
if (!array || !key || array->count == 0)
|
||||
return false;
|
||||
size_t lo = 0;
|
||||
size_t hi = array->count;
|
||||
while (lo < hi) {
|
||||
size_t mid = lo + (hi - lo) / 2;
|
||||
int cmp = strcmp(array->items[mid], key);
|
||||
if (cmp < 0)
|
||||
lo = mid + 1;
|
||||
else if (cmp > 0)
|
||||
hi = mid;
|
||||
else
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Compare `entry` against the virtual key `key` + '/' without allocating the
|
||||
* concatenation. Returns <0, 0 or >0 as `entry` sorts before, equal to, or
|
||||
* after that virtual key. */
|
||||
static int str_sorted_array_compare_prefix(const char* entry, const char* key, size_t key_len) {
|
||||
int cmp = strncmp(entry, key, key_len);
|
||||
if (cmp != 0)
|
||||
return cmp;
|
||||
unsigned char next = (unsigned char)entry[key_len];
|
||||
if (next == '\0')
|
||||
return -1; /* entry == key sorts before key + '/' */
|
||||
return (int)next - (int)'/';
|
||||
}
|
||||
|
||||
bool str_sorted_array_has_child_prefix(const StrSortedArray* array, const char* key) {
|
||||
if (!array || !key || array->count == 0 || key[0] == '\0')
|
||||
return false;
|
||||
size_t key_len = strlen(key);
|
||||
size_t lo = 0;
|
||||
size_t hi = array->count;
|
||||
while (lo < hi) {
|
||||
size_t mid = lo + (hi - lo) / 2;
|
||||
if (str_sorted_array_compare_prefix(array->items[mid], key, key_len) < 0)
|
||||
lo = mid + 1;
|
||||
else
|
||||
hi = mid;
|
||||
}
|
||||
if (lo >= array->count)
|
||||
return false;
|
||||
const char* entry = array->items[lo];
|
||||
return strncmp(entry, key, key_len) == 0 && entry[key_len] == '/';
|
||||
}
|
||||
|
||||
bool path_index_build(PathIndex* index, const char* const* entries, size_t count) {
|
||||
if (!index)
|
||||
return false;
|
||||
index->exact.slots = NULL;
|
||||
index->exact.capacity = 0;
|
||||
index->exact.size = 0;
|
||||
index->sorted.items = NULL;
|
||||
index->sorted.count = 0;
|
||||
if (!str_hash_set_init(&index->exact, count))
|
||||
return false;
|
||||
if (!str_sorted_array_build(&index->sorted, entries, count)) {
|
||||
str_hash_set_free(&index->exact);
|
||||
return false;
|
||||
}
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
if (!str_hash_set_insert_ref(&index->exact, entries[i])) {
|
||||
path_index_free(index);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void path_index_free(PathIndex* index) {
|
||||
if (!index)
|
||||
return;
|
||||
str_hash_set_free(&index->exact);
|
||||
str_sorted_array_free(&index->sorted);
|
||||
}
|
||||
|
||||
bool path_index_contains(const PathIndex* index, const char* path) {
|
||||
return index && str_hash_set_lookup(&index->exact, path);
|
||||
}
|
||||
|
||||
bool path_index_contains_n(const PathIndex* index, const char* path, size_t len) {
|
||||
return index && str_hash_set_lookup_n(&index->exact, path, len);
|
||||
}
|
||||
|
||||
bool path_index_has_descendant(const PathIndex* index, const char* path) {
|
||||
return index && str_sorted_array_has_child_prefix(&index->sorted, path);
|
||||
}
|
||||
|
||||
char* output_escape(const char* string, bool eight_bit_output) {
|
||||
if (!string)
|
||||
return NULL;
|
||||
@@ -122,59 +378,155 @@ char* output_escape(const char* string, bool eight_bit_output) {
|
||||
return escaped;
|
||||
}
|
||||
|
||||
ssize_t utils_getdelim_bounded(FILE* stream, char** line, size_t* cap, int delim, size_t max_len) {
|
||||
if (!stream || !line || !cap || max_len == 0) {
|
||||
errno = EINVAL;
|
||||
return -1;
|
||||
}
|
||||
size_t limit = max_len + 1; /* content bytes plus the terminating NUL */
|
||||
if (*line == NULL || *cap < 2) {
|
||||
size_t initial = limit < 256 ? limit : 256;
|
||||
char* buf = malloc(initial);
|
||||
if (!buf)
|
||||
return -1;
|
||||
free(*line);
|
||||
*line = buf;
|
||||
*cap = initial;
|
||||
}
|
||||
size_t len = 0;
|
||||
int c;
|
||||
while ((c = getc_unlocked(stream)) != EOF) {
|
||||
if (len >= max_len) {
|
||||
errno = EFBIG;
|
||||
return -1;
|
||||
}
|
||||
if (len + 2 > *cap) {
|
||||
size_t new_cap = *cap * 2;
|
||||
if (new_cap < len + 2)
|
||||
new_cap = len + 2;
|
||||
if (new_cap > limit)
|
||||
new_cap = limit;
|
||||
char* grown = realloc(*line, new_cap);
|
||||
if (!grown)
|
||||
return -1;
|
||||
*line = grown;
|
||||
*cap = new_cap;
|
||||
}
|
||||
(*line)[len++] = (char)c;
|
||||
if (c == delim)
|
||||
break;
|
||||
}
|
||||
if (c == EOF && len == 0)
|
||||
return 0;
|
||||
(*line)[len] = '\0';
|
||||
return (ssize_t)len;
|
||||
}
|
||||
|
||||
/* Match a glob pattern against a string. Supported wildcards:
|
||||
* ? matches any single character except '/'.
|
||||
* * matches any sequence of characters within one path component (no '/').
|
||||
* ** matches any sequence of characters, including '/' (cross-directory).
|
||||
* slash-star-star-slash is treated as a cross-directory wildcard when it appears between
|
||||
* literals.
|
||||
*/
|
||||
*
|
||||
* The matcher is an iterative O(pattern * string) dynamic program rather than the
|
||||
* original backtracking recursion: overlapping `*`/`**` wildcards made a pattern
|
||||
* like `*a*a*a*...*b` run in exponential time against a long run of `a`, a CPU
|
||||
* denial-of-service vector reachable from a hostile --exclude/--include pattern
|
||||
* or `.rsync-filter`. The DP reasons over (pattern position, string position)
|
||||
* so every state is visited once; the transitions below mirror the original
|
||||
* recursion exactly. */
|
||||
bool glob_match(const char* pattern, const char* str) {
|
||||
while (*pattern) {
|
||||
if (*pattern == '*') {
|
||||
if (*(pattern + 1) == '*') {
|
||||
/* globstar: match across directories */
|
||||
pattern += 2;
|
||||
if (*pattern == '\0')
|
||||
return true;
|
||||
if (*pattern == '/')
|
||||
pattern++;
|
||||
while (*str) {
|
||||
if (glob_match(pattern, str))
|
||||
return true;
|
||||
str++;
|
||||
}
|
||||
return glob_match(pattern, str);
|
||||
}
|
||||
/* single *: match within one path component */
|
||||
pattern++;
|
||||
while (*str && *str != '/') {
|
||||
if (glob_match(pattern, str))
|
||||
return true;
|
||||
str++;
|
||||
}
|
||||
return glob_match(pattern, str);
|
||||
} else if (*pattern == '?') {
|
||||
if (!*str || *str == '/')
|
||||
if (!pattern || !str)
|
||||
return false;
|
||||
pattern++;
|
||||
str++;
|
||||
} else {
|
||||
if (*pattern != *str) {
|
||||
/* allow literal / ** / rest to match any number of directories */
|
||||
if (*pattern == '/' && *(pattern + 1) == '*' && *(pattern + 2) == '*') {
|
||||
const char* rest = pattern + 3;
|
||||
if (*rest == '/')
|
||||
size_t pattern_len = strlen(pattern);
|
||||
size_t str_len = strlen(str);
|
||||
if (pattern_len == 0)
|
||||
return str_len == 0;
|
||||
/* Defensive work cap: the DP is bounded by pattern*string states, but a
|
||||
* 64 KiB pattern against a 64 KiB path would still cost billions of steps.
|
||||
* Treat the pattern as non-matching above the cap instead of burning CPU. */
|
||||
if (str_len > (SIZE_MAX / (pattern_len + 1)) - 1)
|
||||
return false;
|
||||
if ((pattern_len + 1) * (str_len + 1) > 64u * 1024u * 1024u)
|
||||
return false;
|
||||
|
||||
size_t row_bytes = str_len + 1;
|
||||
/* Rows for pattern positions i, i+1, i+2 and i+3 are live at once (the
|
||||
* globstar transition can skip up to three pattern bytes). Four rotating
|
||||
* rows keep memory at O(string length); a stack buffer avoids an allocation
|
||||
* for the common short-leaf case. */
|
||||
enum { STACK_ROW = 257 };
|
||||
uint8_t stack_rows[4 * STACK_ROW];
|
||||
uint8_t* rows = stack_rows;
|
||||
if (row_bytes > STACK_ROW) {
|
||||
rows = malloc(4 * row_bytes);
|
||||
if (!rows)
|
||||
return false;
|
||||
}
|
||||
|
||||
#define GLOB_ROW(i) (rows + ((pattern_len - (i)) & 3) * row_bytes)
|
||||
|
||||
/* Base row: pattern position `pattern_len` matches only the string's end. */
|
||||
for (size_t j = 0; j <= str_len; j++)
|
||||
GLOB_ROW(pattern_len)[j] = (j == str_len) ? 1 : 0;
|
||||
|
||||
for (size_t i = pattern_len; i-- > 0;) {
|
||||
const char pc = pattern[i];
|
||||
uint8_t* cur = GLOB_ROW(i);
|
||||
const uint8_t* next = GLOB_ROW(i + 1);
|
||||
if (pc == '*') {
|
||||
if (i + 1 < pattern_len && pattern[i + 1] == '*') {
|
||||
/* Globstar: skip `**` and an optional following '/', then consume any
|
||||
* (possibly empty) run of characters -- including '/'. */
|
||||
size_t rest = i + 2;
|
||||
if (rest < pattern_len && pattern[rest] == '/')
|
||||
rest++;
|
||||
return glob_match(rest, str);
|
||||
const uint8_t* rest_row = GLOB_ROW(rest);
|
||||
for (size_t j = str_len + 1; j-- > 0;) {
|
||||
bool v = rest_row[j] != 0;
|
||||
if (!v && j < str_len)
|
||||
v = cur[j + 1] != 0;
|
||||
cur[j] = v ? 1 : 0;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
pattern++;
|
||||
str++;
|
||||
} else {
|
||||
/* Single `*`: zero characters, or one non-'/' character. */
|
||||
for (size_t j = str_len + 1; j-- > 0;) {
|
||||
bool v = next[j] != 0;
|
||||
if (!v && j < str_len && str[j] != '/')
|
||||
v = cur[j + 1] != 0;
|
||||
cur[j] = v ? 1 : 0;
|
||||
}
|
||||
}
|
||||
return *str == '\0';
|
||||
} else if (pc == '?') {
|
||||
for (size_t j = str_len + 1; j-- > 0;) {
|
||||
bool v = j < str_len && str[j] != '/' && next[j + 1] != 0;
|
||||
cur[j] = v ? 1 : 0;
|
||||
}
|
||||
} else {
|
||||
/* Literal: consume an equal character, or -- for a '/' immediately before
|
||||
* a globstar -- let the '/' match zero directories and continue at `**`. */
|
||||
for (size_t j = str_len + 1; j-- > 0;) {
|
||||
bool v = false;
|
||||
if (j < str_len && str[j] == pc) {
|
||||
v = next[j + 1] != 0;
|
||||
} else if (pc == '/' && i + 2 < pattern_len && pattern[i + 1] == '*' &&
|
||||
pattern[i + 2] == '*') {
|
||||
size_t rest = i + 3;
|
||||
if (rest < pattern_len && pattern[rest] == '/')
|
||||
rest++;
|
||||
v = GLOB_ROW(rest)[j] != 0;
|
||||
}
|
||||
cur[j] = v ? 1 : 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
bool matched = GLOB_ROW(0)[0] != 0;
|
||||
#undef GLOB_ROW
|
||||
if (rows != stack_rows)
|
||||
free(rows);
|
||||
return matched;
|
||||
}
|
||||
|
||||
bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_size) {
|
||||
@@ -196,15 +548,23 @@ bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_si
|
||||
return written >= 0 && (size_t)written < buffer_size;
|
||||
}
|
||||
|
||||
static bool is_dir_in_manifest(const char* rel_path, ArrayList* manifest) {
|
||||
size_t len = strlen(rel_path);
|
||||
for (int i = 0; i < manifest->size; i++) {
|
||||
const char* entry = (const char*)manifest->items[i];
|
||||
// Check if entry starts with rel_path + '/' or matches exactly
|
||||
if (strncmp(entry, rel_path, len) == 0 && (entry[len] == '/' || entry[len] == '\0'))
|
||||
return true;
|
||||
/* Build the keep-set index from the exact manifest entries only. A lookup of
|
||||
`rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor
|
||||
directory of kept content (the old is_dir_in_manifest predicate); the sorted
|
||||
view answers "is an ancestor of kept content" without materializing any
|
||||
per-component prefix copy, so the index is O(manifest size) memory. */
|
||||
static bool build_keep_index(const ArrayList* manifest, PathIndex* index) {
|
||||
if (!manifest || manifest->size <= 0)
|
||||
return path_index_build(index, NULL, 0);
|
||||
return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size);
|
||||
}
|
||||
return false;
|
||||
|
||||
static bool keep_is_dir(const PathIndex* index, const char* rel_path) {
|
||||
return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path);
|
||||
}
|
||||
|
||||
static bool keep_is_file(const PathIndex* index, const char* rel_path) {
|
||||
return path_index_contains(index, rel_path);
|
||||
}
|
||||
|
||||
/* True when child_rel is, or lies below, a protected entry. A prefix "a"
|
||||
@@ -224,22 +584,41 @@ bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSki
|
||||
return false;
|
||||
}
|
||||
|
||||
/* All-or-nothing max-delete needs to know BEFORE any unlink whether the run
|
||||
would delete more than max_delete entries. This rehearsal pass walks the
|
||||
destination with the same decisions as the delete pass but never touches the
|
||||
filesystem: it counts every regular file the delete pass would unlink and
|
||||
every directory it would rmdir (a directory is removed only once every entry
|
||||
below it has been removed and nothing the walker leaves in place survives).
|
||||
Entries the walker never removes (symlinks, manifest-listed files, protected
|
||||
prefixes) mark the enclosing directory as surviving, exactly as they would
|
||||
make a real rmdir fail with ENOTEMPTY. Stops early once *count reaches the
|
||||
cap (sets *exceeds). Returns false on a traversal error. */
|
||||
static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest, size_t cap,
|
||||
size_t* count, bool* exceeds, const DeleteSkipEntry* skips,
|
||||
int skip_count, bool* survives) {
|
||||
/* Per-run deletion budget and tallies. `max_delete` is the cap on the number
|
||||
of entries the walker may remove (SIZE_MAX = unlimited); once it is reached
|
||||
the remaining extras are counted in `skipped` and left in place, matching
|
||||
rsync's partial --max-delete behavior. */
|
||||
typedef struct {
|
||||
size_t max_delete;
|
||||
size_t deleted;
|
||||
size_t skipped;
|
||||
bool limit_hit;
|
||||
} DeleteBudget;
|
||||
|
||||
/* True when direct children of the directory named by `rel` may be removed.
|
||||
With no synchronization info (dirs == NULL) the whole tree is deletable; when
|
||||
a dirs index is supplied only its exact entries are (the receive root is the
|
||||
"." sentinel). */
|
||||
static bool is_synced_dir(const PathIndex* dirs, const char* rel) {
|
||||
if (!dirs)
|
||||
return true;
|
||||
return path_index_contains(dirs, rel[0] == '\0' ? "." : rel);
|
||||
}
|
||||
|
||||
/* Remove the extras directly inside the directory open on `dirfd`, recursing
|
||||
into every child directory so kept content below a synchronized prefix is
|
||||
reached. `all_removed` reports whether every child entry was removed (so the
|
||||
caller may rmdir this directory). A child directory is never removed when it
|
||||
is itself a synchronized directory or holds kept content; with a dirs index
|
||||
supplied, direct children of a non-synchronized directory are never extras at
|
||||
all (they are left in place but still descended into). Symlinks are unlinked
|
||||
like any other non-directory extra (never followed). */
|
||||
static bool delete_extras_fd(int dirfd, const char* rel_path, const PathIndex* keep,
|
||||
const PathIndex* dirs, DeleteBudget* budget,
|
||||
const DeleteSkipEntry* skips, int skip_count, bool parent_deletable,
|
||||
bool* all_removed) {
|
||||
/* openat(dirfd, ".") opens an independent file description: a dup() would
|
||||
share dirfd's file offset, and a prior rehearsal pass must not have drained
|
||||
this directory's stream before the delete pass reads it again. */
|
||||
share dirfd's file offset and a prior pass could leave the stream drained. */
|
||||
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
@@ -250,99 +629,10 @@ static bool count_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest
|
||||
}
|
||||
bool operation_ok = true;
|
||||
bool local_survives = false;
|
||||
bool at_root = rel_path[0] == '\0';
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
if (*exceeds)
|
||||
break;
|
||||
char* child_rel = path_cat((char*)rel_path, entry->d_name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (path_under_skip_prefix(child_rel, at_root, skips, skip_count)) {
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT)
|
||||
operation_ok = false;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (S_ISLNK(st.st_mode)) {
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (S_ISDIR(st.st_mode)) {
|
||||
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_ok = true;
|
||||
bool child_survives = true;
|
||||
if (childfd >= 0) {
|
||||
child_ok = count_extras_fd(childfd, child_rel, manifest, cap, count, exceeds, skips,
|
||||
skip_count, &child_survives);
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
if (!child_ok)
|
||||
operation_ok = false;
|
||||
if (is_dir_in_manifest(child_rel, manifest)) {
|
||||
/* A directory with kept content below it is never removed. */
|
||||
local_survives = true;
|
||||
} else if (child_survives) {
|
||||
/* The directory still holds entries the walker leaves in place, so an
|
||||
rmdir would fail with ENOTEMPTY; the delete pass leaves it behind
|
||||
rather than reporting an error (matching rsync). */
|
||||
local_survives = true;
|
||||
} else {
|
||||
if (*count >= cap) {
|
||||
*exceeds = true;
|
||||
} else {
|
||||
(*count)++;
|
||||
}
|
||||
}
|
||||
} else {
|
||||
bool found = false;
|
||||
for (int i = 0; i < manifest->size; i++) {
|
||||
if (strcmp((char*)manifest->items[i], child_rel) == 0) {
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!found) {
|
||||
if (*count >= cap) {
|
||||
*exceeds = true;
|
||||
} else {
|
||||
(*count)++;
|
||||
}
|
||||
}
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
closedir(dir);
|
||||
*survives = local_survives;
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifest,
|
||||
size_t max_delete, size_t* deleted_count, const DeleteSkipEntry* skips,
|
||||
int skip_count) {
|
||||
/* Independent file description (see count_extras_fd). */
|
||||
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
DIR* dir = fdopendir(scanfd);
|
||||
if (!dir) {
|
||||
close(scanfd);
|
||||
return false;
|
||||
}
|
||||
bool operation_ok = true;
|
||||
/* A directory is deletable when it or ANY ancestor is synchronized; the
|
||||
`parent_deletable` flag carries that down the recursion so dest-only
|
||||
directories below a synchronized root are removed wholesale. */
|
||||
bool deletable = parent_deletable || is_synced_dir(dirs, rel_path);
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
@@ -360,6 +650,7 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes
|
||||
destination directory that happens to be called .fastsync-stage is
|
||||
ordinary content. */
|
||||
if (path_under_skip_prefix(child_rel, rel_path[0] == '\0', skips, skip_count)) {
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
@@ -370,28 +661,28 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
// Skip symlinks to prevent following them outside the destination tree
|
||||
if (S_ISLNK(st.st_mode)) {
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (S_ISDIR(st.st_mode)) {
|
||||
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_removed = false;
|
||||
bool child_all_removed = false;
|
||||
if (childfd >= 0) {
|
||||
child_removed = delete_extras_fd(childfd, child_rel, manifest, max_delete, deleted_count,
|
||||
skips, skip_count);
|
||||
if (!child_removed)
|
||||
if (!delete_extras_fd(childfd, child_rel, keep, dirs, budget, skips, skip_count, deletable,
|
||||
&child_all_removed))
|
||||
operation_ok = false;
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
if (child_removed && !is_dir_in_manifest(child_rel, manifest)) {
|
||||
if (*deleted_count >= max_delete) {
|
||||
operation_ok = false;
|
||||
} else {
|
||||
if (unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0) {
|
||||
bool child_synced = dirs && path_index_contains(dirs, child_rel);
|
||||
if (child_synced || keep_is_dir(keep, child_rel)) {
|
||||
/* A synchronized directory and a directory holding kept content are
|
||||
never removed. */
|
||||
local_survives = true;
|
||||
} else if (child_all_removed && deletable) {
|
||||
if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
local_survives = true;
|
||||
} else if (unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0) {
|
||||
/* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory
|
||||
still holds entries the walker leaves in place (a protected
|
||||
excluded prefix, a kept file the manifest protects, a symlink);
|
||||
@@ -399,32 +690,29 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes
|
||||
Only genuine I/O failures abort the deletion. */
|
||||
if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST)
|
||||
operation_ok = false;
|
||||
local_survives = true;
|
||||
} else {
|
||||
(*deleted_count)++;
|
||||
}
|
||||
}
|
||||
budget->deleted++;
|
||||
}
|
||||
} else {
|
||||
// Check if relative path is in manifest
|
||||
bool found = false;
|
||||
for (int i = 0; i < manifest->size; i++) {
|
||||
if (strcmp((char*)manifest->items[i], child_rel) == 0) {
|
||||
found = true;
|
||||
break;
|
||||
local_survives = true;
|
||||
}
|
||||
}
|
||||
if (!found) {
|
||||
if (*deleted_count >= max_delete) {
|
||||
operation_ok = false;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
if (unlinkat(dirfd, entry->d_name, 0) != 0) {
|
||||
} else {
|
||||
bool found = keep_is_file(keep, child_rel);
|
||||
if (found || !deletable) {
|
||||
/* Kept file, or a child of a directory that is not synchronized: never
|
||||
an extra for this run. */
|
||||
local_survives = true;
|
||||
} else if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
local_survives = true;
|
||||
} else if (unlinkat(dirfd, entry->d_name, 0) != 0) {
|
||||
if (errno != ENOENT)
|
||||
operation_ok = false;
|
||||
local_survives = true;
|
||||
} else {
|
||||
(*deleted_count)++;
|
||||
}
|
||||
budget->deleted++;
|
||||
char* escaped_path = output_escape(child_rel, log_get_8_bit_output());
|
||||
fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : "<allocation failed>");
|
||||
free(escaped_path);
|
||||
@@ -433,57 +721,72 @@ static bool delete_extras_fd(int dirfd, const char* rel_path, ArrayList* manifes
|
||||
free(child_rel);
|
||||
}
|
||||
closedir(dir);
|
||||
*all_removed = !local_survives;
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifest,
|
||||
size_t max_delete, const DeleteSkipEntry* skips,
|
||||
int skip_count, size_t* deleted_out) {
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
size_t* deleted_out, size_t* skipped_out) {
|
||||
if (deleted_out)
|
||||
*deleted_out = 0;
|
||||
if (skipped_out)
|
||||
*skipped_out = 0;
|
||||
if (!manifest)
|
||||
return DELETE_WALK_ERROR;
|
||||
/* Index the keep-set (and the synchronized-dir set, when supplied) once so
|
||||
membership is answered in O(path length) instead of scanning every entry
|
||||
for every destination entry. */
|
||||
PathIndex keep;
|
||||
if (!build_keep_index(manifest, &keep))
|
||||
return DELETE_WALK_ERROR;
|
||||
PathIndex dirs;
|
||||
bool have_dirs = synced_dirs != NULL;
|
||||
if (have_dirs &&
|
||||
!path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) {
|
||||
path_index_free(&keep);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
int rootfd;
|
||||
if (authorized_root_fd >= 0) {
|
||||
if (authorized_root_path)
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
if (root_fd >= 0) {
|
||||
if (utils_get_authorized_root_path())
|
||||
rootfd = open_authorized_destination(dest_root);
|
||||
else if (dest_root == NULL)
|
||||
rootfd = dup(authorized_root_fd);
|
||||
rootfd = dup(root_fd);
|
||||
else
|
||||
rootfd = -1;
|
||||
} else {
|
||||
rootfd = open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
}
|
||||
if (rootfd < 0)
|
||||
return DELETE_WALK_ERROR;
|
||||
if (max_delete != SIZE_MAX) {
|
||||
/* Rehearse the deletion first so a run that would exceed the cap removes
|
||||
nothing (rsync's all-or-nothing --max-delete contract). */
|
||||
size_t count = 0;
|
||||
bool exceeds = false;
|
||||
bool survives = false;
|
||||
bool counted_ok = count_extras_fd(rootfd, "", manifest, max_delete, &count, &exceeds, skips,
|
||||
skip_count, &survives);
|
||||
if (!counted_ok) {
|
||||
close(rootfd);
|
||||
if (rootfd < 0) {
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
if (exceeds) {
|
||||
close(rootfd);
|
||||
return DELETE_WALK_LIMIT_EXCEEDED;
|
||||
}
|
||||
}
|
||||
size_t deleted_count = 0;
|
||||
bool ok = delete_extras_fd(rootfd, "", manifest, max_delete, &deleted_count, skips, skip_count);
|
||||
DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
bool all_removed = false;
|
||||
bool ok = delete_extras_fd(rootfd, "", &keep, have_dirs ? &dirs : NULL, &budget, skips,
|
||||
skip_count, false, &all_removed);
|
||||
if (close(rootfd) != 0)
|
||||
ok = false;
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
if (deleted_out)
|
||||
*deleted_out = deleted_count;
|
||||
return ok ? DELETE_WALK_OK : DELETE_WALK_ERROR;
|
||||
*deleted_out = budget.deleted;
|
||||
if (skipped_out)
|
||||
*skipped_out = budget.skipped;
|
||||
if (!ok)
|
||||
return DELETE_WALK_ERROR;
|
||||
return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
bool delete_extras(const char* dest_root, ArrayList* manifest) {
|
||||
return delete_extras_limited(dest_root, manifest, SIZE_MAX, NULL, 0, NULL) == DELETE_WALK_OK;
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest) {
|
||||
return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL) ==
|
||||
DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
bool has_path_traversal(const char* path) {
|
||||
@@ -587,6 +890,71 @@ bool utils_fd_peer_is_local(int fd) {
|
||||
return utils_sockaddr_is_loopback((const struct sockaddr*)&peer);
|
||||
}
|
||||
|
||||
/* Numeric peer address of a connected fd. Only AF_INET/AF_INET6 peers are
|
||||
formatted; every other descriptor/family (pipe, AF_UNIX socketpair, ...) or a
|
||||
getpeername failure returns false with buf emptied. The caller must treat
|
||||
that as "cannot tell". */
|
||||
bool utils_fd_peer_ip(int fd, char* buf, size_t len) {
|
||||
if (!buf || len == 0)
|
||||
return false;
|
||||
buf[0] = '\0';
|
||||
if (fd < 0)
|
||||
return false;
|
||||
struct sockaddr_storage peer;
|
||||
socklen_t peer_len = sizeof(peer);
|
||||
if (getpeername(fd, (struct sockaddr*)&peer, &peer_len) != 0)
|
||||
return false;
|
||||
const void* src = NULL;
|
||||
int family = peer.ss_family;
|
||||
if (family == AF_INET) {
|
||||
src = &((const struct sockaddr_in*)&peer)->sin_addr;
|
||||
} else if (family == AF_INET6) {
|
||||
const struct sockaddr_in6* peer6 = (const struct sockaddr_in6*)&peer;
|
||||
/* A dual-stack IPv6 listener reports IPv4 peers as ::ffff:a.b.c.d. Emit
|
||||
* the IPv4 form so IPv4 ACL patterns (and logs) see the real address. */
|
||||
if (IN6_IS_ADDR_V4MAPPED(&peer6->sin6_addr)) {
|
||||
struct in_addr v4;
|
||||
memcpy(&v4, &peer6->sin6_addr.s6_addr[12], sizeof(v4));
|
||||
return inet_ntop(AF_INET, &v4, buf, (socklen_t)len) != NULL;
|
||||
}
|
||||
src = &peer6->sin6_addr;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
return inet_ntop(family, src, buf, (socklen_t)len) != NULL;
|
||||
}
|
||||
|
||||
/* "ip:port" / "[ip]:port" for a connected peer, used to log the connecting
|
||||
address in the accept loop. Returns false for a non-INET family. */
|
||||
bool utils_sockaddr_to_string(const struct sockaddr* addr, char* buf, size_t len) {
|
||||
if (!addr || !buf || len == 0)
|
||||
return false;
|
||||
buf[0] = '\0';
|
||||
char ip[INET6_ADDRSTRLEN];
|
||||
unsigned short port;
|
||||
int written;
|
||||
if (addr->sa_family == AF_INET) {
|
||||
const struct sockaddr_in* v4 = (const struct sockaddr_in*)addr;
|
||||
if (!inet_ntop(AF_INET, &v4->sin_addr, ip, sizeof(ip)))
|
||||
return false;
|
||||
port = ntohs(v4->sin_port);
|
||||
written = snprintf(buf, len, "%s:%u", ip, port);
|
||||
} else if (addr->sa_family == AF_INET6) {
|
||||
const struct sockaddr_in6* v6 = (const struct sockaddr_in6*)addr;
|
||||
if (!inet_ntop(AF_INET6, &v6->sin6_addr, ip, sizeof(ip)))
|
||||
return false;
|
||||
port = ntohs(v6->sin6_port);
|
||||
written = snprintf(buf, len, "[%s]:%u", ip, port);
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
if (written < 0 || (size_t)written >= len) {
|
||||
buf[0] = '\0';
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* True when a client-supplied host string names a loopback destination:
|
||||
"localhost", any 127.0.0.0/8 literal, "::1", or "[::1]". */
|
||||
bool utils_host_is_loopback(const char* host) {
|
||||
|
||||
+140
-18
@@ -4,20 +4,104 @@
|
||||
#include "array_list.h"
|
||||
#include <stddef.h>
|
||||
#include <stdbool.h>
|
||||
#include <stdio.h>
|
||||
#include <sys/socket.h>
|
||||
#include <sys/types.h>
|
||||
|
||||
/* Small open-addressing string hash set used to turn quadratic membership
|
||||
* scans into O(path length) exact-match lookups (the --delete keep-set and the
|
||||
* --files-from allow-set). Keys are hashed with xxHash64 (seed 0); collisions
|
||||
* are resolved by linear probing over a power-of-two table that grows at 75%
|
||||
* load. Keys are always borrowed from the caller and must outlive the set; the
|
||||
* set never copies or owns keys, so indexing M entries costs O(M) memory. The
|
||||
* set is not thread-safe for mutation, but a fully built set supports
|
||||
* concurrent read-only lookups. */
|
||||
typedef struct {
|
||||
const char* key; /* NULL marks an empty slot */
|
||||
} StrHashSetSlot;
|
||||
|
||||
typedef struct {
|
||||
StrHashSetSlot* slots;
|
||||
size_t capacity; /* power of two, zero before init */
|
||||
size_t size;
|
||||
} StrHashSet;
|
||||
|
||||
/* Initialize an empty set sized for roughly `hint` entries. Returns false on
|
||||
* allocation failure. */
|
||||
bool str_hash_set_init(StrHashSet* set, size_t hint);
|
||||
void str_hash_set_free(StrHashSet* set);
|
||||
/* Insert a borrowed key (must outlive the set). A duplicate is ignored.
|
||||
* Returns false on allocation failure. */
|
||||
bool str_hash_set_insert_ref(StrHashSet* set, const char* key);
|
||||
/* Look up a NUL-terminated key / a key of `len` bytes. */
|
||||
bool str_hash_set_lookup(const StrHashSet* set, const char* key);
|
||||
bool str_hash_set_lookup_n(const StrHashSet* set, const char* key, size_t len);
|
||||
|
||||
/* Sorted, non-owning view of NUL-terminated strings. Built from borrowed
|
||||
* pointers (qsort), so indexing M entries costs O(M) memory and O(M log M)
|
||||
* time; exact membership and ancestor-prefix existence are binary searches
|
||||
* that never materialize a prefix copy. */
|
||||
typedef struct {
|
||||
const char** items; /* sorted with strcmp; borrowed, never freed */
|
||||
size_t count;
|
||||
} StrSortedArray;
|
||||
|
||||
/* Build `array` over the borrowed `items`. Only the pointer array is copied,
|
||||
* never the strings. Returns false on allocation failure. */
|
||||
bool str_sorted_array_build(StrSortedArray* array, const char* const* items, size_t count);
|
||||
void str_sorted_array_free(StrSortedArray* array);
|
||||
/* True when some item equals `key`. */
|
||||
bool str_sorted_array_contains(const StrSortedArray* array, const char* key);
|
||||
/* True when some item starts with `key` followed by '/' (i.e. `key` is a proper
|
||||
* ancestor directory of an item). Allocates nothing. */
|
||||
bool str_sorted_array_has_child_prefix(const StrSortedArray* array, const char* key);
|
||||
|
||||
/* Read-only membership index over exact relative paths. `exact` answers
|
||||
* O(path length) equality; `sorted` answers whether any indexed path lies
|
||||
* strictly below a query directory. Both borrow their keys from the caller and
|
||||
* no ancestor prefix is stored as a separate string, so an index over M entries
|
||||
* is O(M) memory regardless of path depth. Not thread-safe to build, but safe
|
||||
* for concurrent read-only queries once built. */
|
||||
typedef struct {
|
||||
StrHashSet exact;
|
||||
StrSortedArray sorted;
|
||||
} PathIndex;
|
||||
|
||||
/* Build an index borrowing `entries` (which must outlive the index). Returns
|
||||
* false on allocation failure, freeing any partial state. */
|
||||
bool path_index_build(PathIndex* index, const char* const* entries, size_t count);
|
||||
void path_index_free(PathIndex* index);
|
||||
/* True when `path` is an indexed entry. */
|
||||
bool path_index_contains(const PathIndex* index, const char* path);
|
||||
/* Length-bounded form of path_index_contains (`path` need not be terminated). */
|
||||
bool path_index_contains_n(const PathIndex* index, const char* path, size_t len);
|
||||
/* True when some indexed entry lies strictly below `path` (starts with
|
||||
* `path` + '/'). */
|
||||
bool path_index_has_descendant(const PathIndex* index, const char* path);
|
||||
|
||||
char* str_dup(const char* string);
|
||||
char* output_escape(const char* string, bool eight_bit_output);
|
||||
/* Upper bound on one line/token read from a local list file (--files-from,
|
||||
* --exclude-from/--include-from, .rsync-filter). Mirrors MAX_STRING_SIZE and
|
||||
* stops a hostile multi-gigabyte line from forcing unbounded allocation. */
|
||||
#define UTILS_MAX_LINE_LEN (64 * 1024)
|
||||
/* Read one `delim`-terminated record from `stream` into *line (grown as needed
|
||||
* and NUL-terminated), refusing to consume/allocate more than `max_len` bytes
|
||||
* of content. Returns the number of bytes stored (delimiter included, matching
|
||||
* getdelim), 0 at end of file, or -1 on error (errno is EFBIG when the record
|
||||
* exceeds `max_len`, ENOMEM on allocation failure). *line and *cap are updated
|
||||
* as the buffer grows and the caller owns *line. */
|
||||
ssize_t utils_getdelim_bounded(FILE* stream, char** line, size_t* cap, int delim, size_t max_len);
|
||||
char* path_cat(const char* path1, const char* path2);
|
||||
bool glob_match(const char* pattern, const char* str);
|
||||
/* Result of a bounded extra-file deletion run. */
|
||||
typedef enum {
|
||||
/* Every extra entry was removed (or there were none). */
|
||||
DELETE_WALK_OK = 0,
|
||||
/* The destination holds more extras than the numeric cap for this run. With
|
||||
the all-or-nothing max-delete semantics NOTHING was removed (the walker
|
||||
counts first and refuses to start when the run would exceed the limit). */
|
||||
DELETE_WALK_LIMIT_EXCEEDED,
|
||||
/* The numeric cap for this run was reached before every extra was removed.
|
||||
The walker removed exactly the entries the cap allowed and skipped (without
|
||||
removing) the rest, matching rsync's partial --max-delete behavior. */
|
||||
DELETE_WALK_LIMIT_REACHED,
|
||||
/* A traversal or unlink failure aborted the deletion (partial removal is
|
||||
possible, mirroring the delete pass). */
|
||||
DELETE_WALK_ERROR
|
||||
@@ -38,24 +122,54 @@ typedef struct {
|
||||
only DIRECT children of the destination root, i.e. child_rel has no '/'). */
|
||||
bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips,
|
||||
int skip_count);
|
||||
/* Remove files/dirs under dest_root that are not listed in manifest without
|
||||
ever descending into a protected prefix (see DeleteSkipEntry). When
|
||||
max_delete is not SIZE_MAX the run is all-or-nothing: extras are counted
|
||||
first and DELETE_WALK_LIMIT_EXCEEDED is returned (with nothing removed) when
|
||||
the count would exceed the cap. `deleted_out` optionally receives the number
|
||||
of entries actually removed. The all-or-nothing guarantee holds only while
|
||||
the destination tree is not being concurrently modified: the rehearsal pass
|
||||
and the delete pass are two separate walks, so a concurrent change between
|
||||
them (another process adding/removing entries) can make the second pass
|
||||
delete a different set than the first one counted. */
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, ArrayList* manifest,
|
||||
size_t max_delete, const DeleteSkipEntry* skips,
|
||||
int skip_count, size_t* deleted_out);
|
||||
bool delete_extras(const char* dest_root, ArrayList* manifest);
|
||||
/* Remove files/dirs/symlinks under dest_root that are not listed in manifest
|
||||
without ever descending into a protected prefix (see DeleteSkipEntry). When
|
||||
`synced_dirs` is non-NULL, extras are only removed directly inside a directory
|
||||
whose destination-relative path is an exact entry in that list (the receive
|
||||
root is the "." sentinel); directories outside the synchronized set are still
|
||||
descended into so kept content below a listed directory is preserved, but
|
||||
nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of
|
||||
treating the whole destination tree as deletable. `max_delete` caps the
|
||||
number of removed entries (SIZE_MAX = unlimited): the walker removes up to the
|
||||
cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained.
|
||||
`deleted_out`/`skipped_out` optionally receive the number of entries removed
|
||||
and the number skipped because of the cap. */
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
size_t* deleted_out, size_t* skipped_out);
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest);
|
||||
bool utils_set_authorized_root(int fd, const char* canonical_path);
|
||||
/* The fd-only compatibility form is fail-closed for path-based operations;
|
||||
* callers should use utils_set_authorized_root with the canonical identity. */
|
||||
void utils_set_authorized_root_fd(int fd);
|
||||
/* Read accessors for the process-wide authorized root, so every secure-walk
|
||||
* site consumes the single shared state instead of keeping its own copy. The
|
||||
* fd is caller-owned (see the setters): it is returned verbatim, never dup'd,
|
||||
* and the caller that opened it is responsible for closing it. With no root
|
||||
* configured the fd accessor returns -1 and the path accessor returns NULL.
|
||||
*
|
||||
* The pointer returned by utils_get_authorized_root_path() is borrowed into
|
||||
* process-global state and is invalidated by the next
|
||||
* utils_set_authorized_root() / utils_set_authorized_root_fd() call. The fd
|
||||
* and path are stored separately and read independently, so the pair is NOT
|
||||
* observed atomically together; the accessors are non-reentrant and callers
|
||||
* must serialize configuration (the server installs the root before any worker
|
||||
* threads spawn; see utils.c). */
|
||||
int utils_get_authorized_root_fd(void);
|
||||
const char* utils_get_authorized_root_path(void);
|
||||
/* True when `path` is `root` itself or lies directly beneath it: a lexical
|
||||
* prefix test requiring the byte after `root` to be '\0' or '/'. Both `root`
|
||||
* and `path` must be absolute canonical paths free of "."/".." components (the
|
||||
* callers guarantee this); this is containment by string, not by resolved
|
||||
* symlinks. Shared by the utils and file secure-walk root confinement. */
|
||||
bool path_is_within_root(const char* root, const char* path);
|
||||
/* True when `path` contains a ".." component. This is a purely lexical
|
||||
* dot-dot check: an absolute path is NOT rejected here, because default
|
||||
* (non-relative) transfers legitimately put the sender's absolute source path
|
||||
* on the wire and the receiver re-roots it under the destination with
|
||||
* path_cat(). Callers that accept a strictly relative path (e.g. batch paths)
|
||||
* must reject a leading '/' themselves (see utils_valid_batch_path). */
|
||||
bool has_path_traversal(const char* path);
|
||||
bool utils_valid_batch_path(const char* path);
|
||||
bool format_human_bytes(unsigned long long bytes, char* buffer, size_t buffer_size);
|
||||
@@ -77,5 +191,13 @@ bool append_tail_length(unsigned long long old_size, unsigned long long check_si
|
||||
bool utils_sockaddr_is_loopback(const struct sockaddr* addr);
|
||||
bool utils_fd_peer_is_local(int fd);
|
||||
bool utils_host_is_loopback(const char* host);
|
||||
/* Numeric peer address of a connected fd (INET6_ADDRSTRLEN is always enough).
|
||||
* Returns false and leaves buf empty when the fd is not a connected INET socket
|
||||
* or getpeername/inet_ntop fails. Used by the daemon host-access gate; a false
|
||||
* return is "cannot tell" and must be treated as fail-closed when ACLs apply. */
|
||||
bool utils_fd_peer_ip(int fd, char* buf, size_t len);
|
||||
/* Format a sockaddr as "ip:port" (IPv4) or "[ip]:port" (IPv6) for logging.
|
||||
* Returns false (buf emptied) for a non-INET family or a formatting failure. */
|
||||
bool utils_sockaddr_to_string(const struct sockaddr* addr, char* buf, size_t len);
|
||||
|
||||
#endif
|
||||
Loaded 100 of 160 files, more files were not shown because too many files have changed in this diff.
Show more
Reference in new issue
Block a user