Compare commits
3
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
d86029e570 | ||
|
|
5607ec6a88 | ||
|
|
0494116999 |
+14
-69
@@ -4,15 +4,14 @@ on:
|
||||
push:
|
||||
branches: [main, dev]
|
||||
pull_request:
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v9
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: clang-format check
|
||||
run: find src/ tests/ -name '*.c' -o -name '*.h' | xargs clang-format --dry-run --Werror
|
||||
@@ -20,17 +19,13 @@ jobs:
|
||||
- name: cppcheck
|
||||
run: cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/
|
||||
|
||||
# Fast PR gate: build + unit tests + a representative subset of integration
|
||||
# tests (marked `ci`), parallelized with pytest-xdist. Only the full coverage
|
||||
# jobs below (sanitizers/fuzz/coverage/valgrind and the FULL integration
|
||||
# suite) run on merge to dev/main, so PR CI stays well under ~3 minutes.
|
||||
build-and-test:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v9
|
||||
needs: lint
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
@@ -41,66 +36,19 @@ jobs:
|
||||
- name: Unit Tests
|
||||
run: ctest --test-dir build --output-on-failure -j$(nproc)
|
||||
|
||||
- name: Integration Tests (PR smoke subset)
|
||||
if: github.event_name == 'pull_request'
|
||||
run: python3 -m pytest tests/integration/ -n 4 --dist=load -m ci --durations=25 --tb=short -q
|
||||
|
||||
- name: Integration Tests (full suite)
|
||||
if: github.event_name == 'push'
|
||||
run: python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" --durations=25 --tb=short -q
|
||||
|
||||
# Differential rsync-parity gate: runs real rsync 3.4.1 and FastSync over the
|
||||
# same corpora and compares destinations + normalized output. The fast subset
|
||||
# guards the ✅ surface on every PR; the full set (with FASTSYNC_PARITY_STRICT
|
||||
# so a fixed caveat must be removed from the allowlist) burns the documented
|
||||
# ⚠️/❌ residuals down on push. See tests/integration/README.md.
|
||||
parity-fast:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'pull_request'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build -j$(nproc)
|
||||
|
||||
- name: Differential parity (fast subset)
|
||||
run: python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci -q
|
||||
|
||||
parity-full:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
- name: Build
|
||||
run: cmake --build build -j$(nproc)
|
||||
|
||||
- name: Differential parity (full set)
|
||||
run: FASTSYNC_PARITY_STRICT=1 python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity -q
|
||||
- name: Integration Tests
|
||||
run: python3 -m pytest tests/integration/ -v --tb=short
|
||||
|
||||
sanitizers:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v9
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
strategy:
|
||||
matrix:
|
||||
sanitizer: [address, undefined]
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build-${{ matrix.sanitizer }} -S . -DSANITIZER=${{ matrix.sanitizer }}
|
||||
@@ -113,12 +61,11 @@ jobs:
|
||||
|
||||
fuzz-build:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v9
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Configure (clang + fuzz)
|
||||
run: CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
|
||||
@@ -135,12 +82,11 @@ jobs:
|
||||
|
||||
coverage:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v9
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DENABLE_COVERAGE=ON
|
||||
@@ -159,12 +105,11 @@ jobs:
|
||||
|
||||
valgrind:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v9
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
-13
@@ -8,16 +8,3 @@ build-*/
|
||||
build2/
|
||||
build3/
|
||||
build_docker2/
|
||||
|
||||
# Test/run artifacts
|
||||
root/
|
||||
test_partial_install_tmp/
|
||||
|
||||
# Editor/tooling + test caches/artifacts
|
||||
.pytest_cache/
|
||||
*.gcda
|
||||
*.gcno
|
||||
*.gcov
|
||||
di/
|
||||
test_data-manual/
|
||||
*.log
|
||||
|
||||
@@ -128,7 +128,7 @@ Do not wait for the user to tell you CI failed — check proactively. The user s
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -92,7 +92,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,19 +27,16 @@ FetchContent_Declare(xxhash GIT_REPOSITORY https://github.com/Cyan4973/xxHash GI
|
||||
FetchContent_MakeAvailable(xxhash)
|
||||
|
||||
# Sanitizer option
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, undefined, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread undefined none)
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread none)
|
||||
if(SANITIZER STREQUAL "address")
|
||||
add_compile_options(-fsanitize=address -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=address)
|
||||
elseif(SANITIZER STREQUAL "thread")
|
||||
add_compile_options(-fsanitize=thread -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=thread)
|
||||
elseif(SANITIZER STREQUAL "undefined")
|
||||
add_compile_options(-fsanitize=undefined -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=undefined)
|
||||
elseif(NOT SANITIZER STREQUAL "none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, undefined, none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, none")
|
||||
endif()
|
||||
|
||||
option(STRICT_WARNINGS "Enable strict warnings" OFF)
|
||||
@@ -55,16 +52,6 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
file(GLOB SHARED_SRCS "src/shared/*.c")
|
||||
@@ -74,15 +61,15 @@ file(GLOB TEST_SRCS "tests/*.c")
|
||||
|
||||
add_executable(server ${SERVER_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(server PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(client PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} src/client/scanner.c)
|
||||
target_include_directories(tests PRIVATE tests src/shared src/server src/client)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
```
|
||||
|
||||
### Source Layout
|
||||
@@ -95,25 +82,19 @@ tests/integration/ — Python pytest integration tests
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)` (default compression codec)
|
||||
- **zlib** — found via `find_library(ZLIB_LIBRARY z)` (the `zlib`/`zlibx` codecs)
|
||||
- **lz4** — found via `find_library(LZ4_LIBRARY lz4)` (the `lz4` codec)
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)`
|
||||
- **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport)
|
||||
- **xxHash** — fetched via `FetchContent` from the upstream repository (delta transfer hashing, v0.8.3)
|
||||
- **xxHash** — fetched via `FetchContent` from GitHub (delta transfer hashing, v0.8.3)
|
||||
- **pthreads** — found via `find_package(Threads REQUIRED)`
|
||||
- **C11 standard** — required
|
||||
- **CMake 3.22+** — minimum version
|
||||
|
||||
The codec matrix (protocol 2.26.0) uses zstd/zlib/lz4 for compression and
|
||||
xxHash/OpenSSL for the `xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1` checksums
|
||||
(`none` needs no library); both codec families are negotiated per transfer.
|
||||
|
||||
## Conventions
|
||||
|
||||
- Use `file(GLOB ...)` for source collection (existing pattern).
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `${ZLIB_LIBRARY}`, `${LZ4_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- Include directories: `src/shared`, `src/server`, `src/client`, `tests` (for test target).
|
||||
- Sanitizer support: pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (live option in CMakeLists.txt).
|
||||
- Sanitizer support: pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (live option in CMakeLists.txt).
|
||||
- Build with `cmake -B build -S . && cmake --build build -j$(nproc)`.
|
||||
- For CI, dependencies are provided by the project's custom Docker image (repo-root `Dockerfile`, same image CI uses). For local development, use `nix-shell`. Never add `apt-get install` / `pip install` to CI workflows. See `AGENTS.md`.
|
||||
|
||||
@@ -124,7 +105,7 @@ xxHash/OpenSSL for the `xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1` checksums
|
||||
3. Add new dependencies with `find_package` or `find_library`.
|
||||
4. When adding a new executable target, follow the pattern of existing targets.
|
||||
5. When adding a new library (static/shared), use `add_library` and follow the project's naming.
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (matching CI's matrix strategy).
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (matching CI's matrix strategy).
|
||||
7. Always verify the build compiles after changes.
|
||||
|
||||
## Sanitizer Configurations
|
||||
@@ -138,9 +119,11 @@ cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (race conditions)
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
UndefinedBehaviorSanitizer uses the same built-in option:
|
||||
For UndefinedBehaviorSanitizer (no `-DSANITIZER=undefined` option in CMakeLists.txt yet), use the manual flag approach:
|
||||
```bash
|
||||
cmake -B build -S . -DSANITIZER=undefined
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=undefined -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=undefined"
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
@@ -176,7 +159,7 @@ cmake -B build -S . -DCMAKE_BUILD_TYPE=RelWithDebInfo
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
./build/server
|
||||
./build/client
|
||||
./build/tests
|
||||
```
|
||||
@@ -204,7 +187,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ FastSync is a file synchronization tool (like rsync, but faster). It transfers f
|
||||
cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Server (TCP mode)
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
./build/server
|
||||
|
||||
# Client (TCP mode)
|
||||
./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk
|
||||
@@ -37,13 +37,13 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Run tests
|
||||
./build/tests # unit tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # integration tests
|
||||
python3 test.py # integration tests
|
||||
```
|
||||
|
||||
## Code Walkthrough
|
||||
|
||||
### Client Entry Point (`src/client/client_cli.c`)
|
||||
- Parses CLI arguments using a custom option-table parser (`OPTION_TABLE` in `src/client/client_cli.c`); there is no `getopt*` usage
|
||||
- Parses CLI arguments using `getopt_long`
|
||||
- Creates `Config` struct with all options
|
||||
- Detects SSH destinations (contains `:`)
|
||||
- Calls into `client_send.c` for the actual transfer
|
||||
@@ -109,7 +109,7 @@ Collection of files for batch transfer. Serialized with file count, then per-fil
|
||||
zstd streaming compression via `ZSTD_compressStream2`/`ZSTD_decompressStream`. Compression happens per-chunk in the sender stage. Level 1-22 (default 5). Streaming means memory usage stays bounded regardless of file size.
|
||||
|
||||
### "How does sendfile() work?"
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `--sendfile` (long form only). Only works with TCP (not SSH, not compression).
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `-f` flag. Only works with TCP (not SSH, not compression).
|
||||
|
||||
### "How does incremental sync work?"
|
||||
Client sends file metadata (path, size, mtime) to server. Server checks if destination file has same size+mtime. If match, server responds `STATUS_OK` (skip). If mismatch, server responds `STATUS_NEXT` (send).
|
||||
@@ -138,7 +138,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -316,7 +316,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -14,9 +14,10 @@ Diagnose crashes, memory errors, hangs, and logic bugs. You use structured debug
|
||||
### Memory Errors
|
||||
```bash
|
||||
# AddressSanitizer (fast, recommended first)
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/client # or ./build-asan/server -p 8080 --allow-unauthenticated
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/client # or ./build/server
|
||||
|
||||
# Valgrind (slower, more thorough)
|
||||
valgrind --leak-check=full --show-leak-kinds=all --track-origins=yes \
|
||||
@@ -31,9 +32,10 @@ valgrind --tool=drd ./build/client ...
|
||||
|
||||
### Thread Sanitizer
|
||||
```bash
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=thread" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
```
|
||||
|
||||
### GDB
|
||||
@@ -141,7 +143,7 @@ gprof ./build/client gmon.out
|
||||
|
||||
### Step 5: Verify
|
||||
- Run `./build/tests` (unit tests)
|
||||
- Run `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` (integration tests)
|
||||
- Run `python3 test.py` (integration tests)
|
||||
- Run under valgrind again to confirm clean
|
||||
- Test under ASan again
|
||||
|
||||
@@ -160,7 +162,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -96,7 +96,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -16,18 +16,12 @@ Scan the codebase for patterns that suggest new feature opportunities. You ident
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, OPTION_TABLE parser, config setup
|
||||
usage.c Usage/help text (authoritative CLI flag list)
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
change_list.c File change-list bookkeeping
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
server_cli.c Server option-table CLI parsing
|
||||
receiver.c Receiver-side file handling
|
||||
receiver_pipeline.c Receiver worker pipeline
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
@@ -38,63 +32,40 @@ src/shared/ Shared libraries (used by both client and server)
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_send.c/h Sender-side file transfer
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_list.c/h File list model
|
||||
file_store.c/h Destination file store
|
||||
array_list.c/h Dynamic array
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
batch.c/h Batch files (--write-batch/--read-batch)
|
||||
charset.c/h Filename charset conversion (--iconv)
|
||||
chmod.c/h Permission modification (--chmod)
|
||||
xattr.c/h Extended attributes
|
||||
hardlink.c/h Hard-link handling
|
||||
identity.c/h uid/gid mapping (--usermap/--groupmap/--chown)
|
||||
credentials.c/h Daemon credentials
|
||||
daemon_conf.c/h Daemon module configuration
|
||||
motd.c/h Daemon MOTD
|
||||
delay_updates.c/h Delayed update staging
|
||||
stop_condition.c/h Stop-after/stop-at handling
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
file_types.h Shared file type definitions
|
||||
```
|
||||
|
||||
### Existing CLI Flags (authoritative source: `src/client/usage.c`)
|
||||
### Existing CLI Flags (from client_cli.c)
|
||||
```
|
||||
--source-dir <dir> Source directory
|
||||
--dest-dir <dir> Destination directory on server
|
||||
--server-host <ip> Server IP address (default: 127.0.0.1)
|
||||
--server-port <n> Server port (default: 8080); --port is an alias
|
||||
-c, --checksum Verify content by checksum instead of size+mtime
|
||||
-z, --compress [level] Enable compression (level 1-22, default 5)
|
||||
-j, --threads[=N] Enable multithreaded scanner/loader/sender pipeline
|
||||
--chunk-serialization Enable chunk serialization (long form only)
|
||||
--sendfile sendfile() zero-copy (TCP only; long form only)
|
||||
-s, --secluded-args Protect-args compatibility option (no effect)
|
||||
--tls Enable TLS encryption; --cert/--key/--ca give PEMs
|
||||
--bwlimit <KB/s> Bandwidth limit in kilobytes per second
|
||||
--delete Delete files on receiver not in source
|
||||
--incremental Skip files unchanged since last transfer
|
||||
--delta Delta transfer for changed files (needs --incremental)
|
||||
-f, --filter=RULE rsync-style filter rule (+/- include/exclude)
|
||||
--exclude <pattern> Exclude files matching pattern
|
||||
--include <pattern> Only include files matching pattern
|
||||
-m, --prune-empty-dirs Do not transfer empty directory entries
|
||||
-n, --dry-run Show what would be transferred
|
||||
--save-to-disk Write received files to disk
|
||||
--source-dir <dir> Source directory to sync (required)
|
||||
--dest-dir <dir> Destination directory on server (required)
|
||||
--host <host> Server hostname/IP (required)
|
||||
--port <port> Server TCP port
|
||||
--server-mode Listen as server
|
||||
--use-compression, -c Enable zstd compression
|
||||
--use-multithreading, -m Enable multithreaded transfer
|
||||
--use-sendfile, -s Use sendfile() zero-copy TCP
|
||||
--use-ssh, -S Use SSH transport
|
||||
--use-tls, -T Enable TLS encryption
|
||||
--cert <file> TLS certificate file
|
||||
--key <file> TLS key file
|
||||
--ca <file> TLS CA certificate file
|
||||
--insecure Skip TLS verification
|
||||
--bwlimit <bytes/s> Bandwidth limit
|
||||
--delete Delete files not in source
|
||||
--include <pattern> Include filter pattern
|
||||
--exclude <pattern> Exclude filter pattern
|
||||
--dry-run Print what would be transferred
|
||||
--save-to-disk Save transferred files to disk (for server tests)
|
||||
--version Print version and exit
|
||||
--help Show help
|
||||
--help Print help
|
||||
```
|
||||
> Always confirm the current flags with `./build/client --help`; the table above
|
||||
> is a representative subset. `src/client/usage.c` is the authoritative list and
|
||||
> `OPTION_TABLE` in `src/client/client_cli.c` is the parser (there is no `getopt*`).
|
||||
|
||||
## Feature Scout Checklist
|
||||
|
||||
@@ -317,7 +288,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ Design integration tests that verify the full transfer pipeline works end-to-end
|
||||
- Multiple configurations (TCP, SSH, TLS, compression, multithreading)
|
||||
- Network shaping (LAN, WAN profiles)
|
||||
- Feature tests (dry run, archive, exclude, delete, incremental, bandwidth limit)
|
||||
- Run: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
- Run: `python3 -m pytest tests/ -v --tb=short`
|
||||
|
||||
### 3. New: Focused Integration Tests
|
||||
When adding new features or fixing bugs, write targeted integration tests.
|
||||
@@ -35,14 +35,13 @@ mkdir -p /tmp/fastsync_test/src
|
||||
echo "test content" > /tmp/fastsync_test/src/file.txt
|
||||
|
||||
# Start server
|
||||
./build/server -p 8080 --allow-unauthenticated &
|
||||
./build/server &
|
||||
SERVER_PID=$!
|
||||
sleep 0.5
|
||||
|
||||
# Run client
|
||||
./build/client --source-dir /tmp/fastsync_test/src \
|
||||
--dest-dir /tmp/fastsync_test/dst \
|
||||
--server-port 8080 \
|
||||
--save-to-disk
|
||||
|
||||
# Verify
|
||||
@@ -77,7 +76,7 @@ openssl req -x509 -newkey rsa:2048 -keyout /tmp/key.pem -out /tmp/cert.pem \
|
||||
### Pattern 4: Incremental Sync
|
||||
```bash
|
||||
# First sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
|
||||
# Modify source
|
||||
echo "updated" >> /tmp/src/file.txt
|
||||
@@ -90,14 +89,14 @@ echo "updated" >> /tmp/src/file.txt
|
||||
### Pattern 5: Delete Verification
|
||||
```bash
|
||||
# Initial sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
|
||||
# Add extra file to dest
|
||||
echo "extra" > /tmp/dst/.../extra.txt
|
||||
|
||||
# Sync with --delete
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst \
|
||||
--save-to-disk --delete
|
||||
--save-to-disk --delete -M
|
||||
|
||||
# Verify extra.txt is gone
|
||||
test ! -f /tmp/dst/.../extra.txt
|
||||
@@ -116,7 +115,7 @@ The project uses Gitea Actions. Key jobs:
|
||||
jobs:
|
||||
new-job:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v7
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Configure
|
||||
@@ -128,7 +127,7 @@ jobs:
|
||||
- name: Unit Tests
|
||||
run: ./build-${{ matrix.sanitizer }}/tests
|
||||
- name: Integration Tests
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/ -v --tb=short
|
||||
```
|
||||
The symlink step is required because `tests/conftest.py` expects `./build` to exist.
|
||||
|
||||
@@ -136,7 +135,7 @@ The symlink step is required because `tests/conftest.py` expects `./build` to ex
|
||||
|
||||
After any code change:
|
||||
- [ ] Unit tests pass: `./build/tests`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/ -v --tb=short`
|
||||
- [ ] Build clean: no warnings with `-Wall`
|
||||
- [ ] No memory errors: ASan clean
|
||||
- [ ] No thread errors: TSan clean (if threading involved)
|
||||
@@ -157,7 +156,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
---
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates Gitea issues from their findings.
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates GitHub issues from their findings.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
@@ -12,7 +12,7 @@ You are the primary orchestrator agent. Your job is to:
|
||||
2. Decide which specialized sub-agents to dispatch for analysis
|
||||
3. Delegate analysis work using the task tool
|
||||
4. Receive structured findings from sub-agents
|
||||
5. Create Gitea issues from those findings using `tea issues create`
|
||||
5. Create GitHub issues from those findings using `gh issue create`
|
||||
6. Coordinate the overall analysis workflow end-to-end
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
@@ -97,7 +97,7 @@ First, read the repository structure to understand what exists:
|
||||
### Phase 2: Determine Analysis Scope
|
||||
Based on what the user requests or what needs attention:
|
||||
- **New features wanted?** → Dispatch `feature-scout` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-auditor` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-screener` sub-agent
|
||||
- **Code quality review?** → Dispatch `code-quality-guardian` sub-agent
|
||||
- **All of the above?** → Run all three in parallel
|
||||
|
||||
@@ -110,7 +110,7 @@ Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
```
|
||||
Task: Ask the security-auditor agent to analyze the codebase.
|
||||
Task: Ask the security-screener agent to analyze the codebase.
|
||||
Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
@@ -138,14 +138,14 @@ Each sub-agent returns findings in this structured format:
|
||||
- **Labels**: comma-separated labels for the issue
|
||||
```
|
||||
|
||||
### Phase 5: Create Gitea Issues
|
||||
For each finding, create a Gitea issue:
|
||||
### Phase 5: Create GitHub Issues
|
||||
For each finding, create a GitHub issue:
|
||||
|
||||
```bash
|
||||
tea issues create --repo TapTap/FastSync \
|
||||
gh issue create \
|
||||
--title "<Finding Title>" \
|
||||
--labels "<labels>" \
|
||||
--description "## Description
|
||||
--label "<labels>" \
|
||||
--body "## Description
|
||||
<description>
|
||||
|
||||
## Location
|
||||
@@ -175,13 +175,11 @@ _This issue was automatically generated by the issue-creator agent._"
|
||||
|
||||
### Duplicate Detection
|
||||
Before creating an issue:
|
||||
1. Check existing open issues: `tea issues list --repo TapTap/FastSync --state open --labels "<label>"`
|
||||
2. Search for similar titles using `tea issues list --repo TapTap/FastSync --keyword "<keywords>"`
|
||||
1. Check existing open issues: `gh issue list --state open --label "<label>"`
|
||||
2. Search for similar titles using `gh issue list --search "<keywords>"`
|
||||
3. If a similar issue exists, add a comment instead of creating a duplicate:
|
||||
```bash
|
||||
tea comment --repo TapTap/FastSync <issue-number> "Additional finding from automated analysis: <details>"
|
||||
# or POST to the Gitea API:
|
||||
# POST https://gitea.tap-tap.win/api/v1/repos/TapTap/FastSync/issues/<n>/comments
|
||||
gh issue comment <issue-number> --body "Additional finding from automated analysis: <details>"
|
||||
```
|
||||
|
||||
## Sub-Agent Reference
|
||||
@@ -191,12 +189,13 @@ Before creating an issue:
|
||||
| Agent | File | Purpose |
|
||||
|---|---|---|
|
||||
| feature-scout | `.opencode/agents/feature-scout.md` | Scans for feature opportunities |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits and vulnerability scans |
|
||||
| security-screener | `.opencode/agents/security-screener.md` | Scans for security vulnerabilities |
|
||||
| code-quality-guardian | `.opencode/agents/code-quality-guardian.md` | Scans for code quality improvements |
|
||||
| architect | `.opencode/agents/architect.md` | Architecture reviews |
|
||||
| c-reviewer | `.opencode/agents/c-reviewer.md` | C code correctness reviews |
|
||||
| debugger | `.opencode/agents/debugger.md` | Bug diagnosis |
|
||||
| refactorer | `.opencode/agents/refactorer.md` | Code refactoring |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits |
|
||||
| test-writer | `.opencode/agents/test-writer.md` | Test development |
|
||||
| perf-analyst | `.opencode/agents/perf-analyst.md` | Performance analysis |
|
||||
| protocol-designer | `.opencode/agents/protocol-designer.md` | Protocol design |
|
||||
@@ -243,7 +242,7 @@ tests/test_file.c — File tests
|
||||
tests/test_transport_tcp.c — TCP transport tests
|
||||
tests/test_transport_tls.c — TLS transport tests
|
||||
tests/test_array_list.c — Array list tests
|
||||
tests/integration/ — Python pytest integration tests
|
||||
tests/pytest/ — Python integration tests
|
||||
```
|
||||
|
||||
### Build & Config Files
|
||||
@@ -260,7 +259,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -56,11 +56,9 @@ DirectoryScanner → Queue(Scanner→Loader) → ChunkBuilder → Queue(Loader
|
||||
|
||||
### Benchmark Context
|
||||
|
||||
Use the maintained benchmark tool — do not cite stale README numbers:
|
||||
- `python3 benchmark/bench.py` runs the repeatable throughput benchmark.
|
||||
- The real flags are `-j` (multithreading) and `-z` (compression); a fast loopback
|
||||
config combines `-j -z`.
|
||||
- `sendfile()` (via `--sendfile`) bypasses userspace → ~2× faster on localhost
|
||||
From README benchmarks (25MB mixed files, localhost):
|
||||
- Best config: `-m -c` (multithread + compression) → 0.20s, 11.2× faster than rsync
|
||||
- `sendfile()` bypasses userspace → ~2× faster on localhost
|
||||
- Compression reduces wire data enough that transfer becomes latency-bound on WAN
|
||||
|
||||
## Output Format
|
||||
@@ -122,7 +120,6 @@ time ./build/client [args...]
|
||||
|
||||
# High precision
|
||||
perf stat -e task-clock ./build/client [args...]
|
||||
```
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
@@ -130,8 +127,9 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
```
|
||||
|
||||
@@ -91,7 +91,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -160,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -3,64 +3,25 @@ description: Audits FastSync for security vulnerabilities — TLS config, input
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are the security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport. This is the single canonical security agent.
|
||||
You are a security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
|
||||
## Your Role
|
||||
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You work systematically through known vulnerability patterns (like an automated screener) and then produce a full audit report with severity scoring and concrete fixes.
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices.
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
## Attack Surface
|
||||
|
||||
## Project Architecture
|
||||
### Network Input Points
|
||||
1. **TCP server** (`src/server/server.c`) — accepts connections from any client
|
||||
2. **SSH transport** (`src/shared/transport_ssh.c`) — receives data via stdio pipe
|
||||
3. **Protocol parsing** (`src/shared/protocol.c`) — deserializes all incoming data
|
||||
4. **Config deserialization** (`src/shared/config.c`) — receives remote config
|
||||
5. **Chunk deserialization** (`src/shared/chunk.c`) — receives file batches
|
||||
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
receiver.c Receiver-side file handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_store.c/h Destination file store
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
xattr.c/h Extended attributes
|
||||
identity.c/h uid/gid mapping
|
||||
credentials.c/h Daemon credentials
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` / `receiver.c` | Writes received files to disk |
|
||||
### TLS Configuration
|
||||
- OpenSSL TLS 1.2+ via `src/shared/transport_tls.c`
|
||||
- Certificate/key loading, CA verification
|
||||
- SSL context setup, cipher suite selection
|
||||
|
||||
## Security Audit Checklist
|
||||
|
||||
@@ -72,182 +33,51 @@ src/shared/ Shared libraries (used by both client and server)
|
||||
- [ ] Chunk count and file count validated before allocation
|
||||
- [ ] Config field lengths bounded
|
||||
|
||||
### 2. Buffer Overflow Risks
|
||||
### 2. Buffer Safety
|
||||
- [ ] No `strcpy` — use `snprintf` or `strncpy` with null termination
|
||||
- [ ] `malloc` size calculations don't overflow (e.g., `count * sizeof(...)`)
|
||||
- [ ] No fixed-size stack buffers for unbounded input
|
||||
- [ ] `receive_n_data` always checks return value
|
||||
- [ ] Off-by-one in path concatenation
|
||||
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
### 3. Memory Safety in Error Paths
|
||||
- [ ] All error paths free allocated resources
|
||||
- [ ] No use-after-free on error paths
|
||||
- [ ] No double-free on error paths
|
||||
- [ ] Partial reads handled (don't use incomplete data)
|
||||
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
### 4. TLS/SSL Security
|
||||
- [ ] TLS 1.2 minimum enforced (no SSLv3, TLS 1.0, TLS 1.1)
|
||||
- [ ] Certificate verification enabled when CA provided
|
||||
- [ ] Certificate verification disabled only with explicit warning
|
||||
- [ ] Private key file permissions checked
|
||||
- [ ] No hardcoded certificates or keys
|
||||
- [ ] Cipher suites restricted to strong algorithms
|
||||
- [ ] SSL error codes checked after `SSL_read`/`SSL_write`
|
||||
|
||||
### 3. Path Traversal in File Operations
|
||||
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 4. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 5. TLS / SSL Security
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--ca`/warning
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
- [ ] **No hardcoded certificates or keys**
|
||||
- [ ] **SSL error codes checked after `SSL_read`/`SSL_write`**
|
||||
|
||||
### 6. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
- [ ] **All error paths free allocated resources** (no leaks / UAF / double-free)
|
||||
- [ ] **Partial reads handled** (don't use incomplete data)
|
||||
|
||||
### 7. Integer Overflow in Allocation
|
||||
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 8. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 9. Authentication & Authorization
|
||||
### 5. Authentication & Authorization
|
||||
- [ ] SSH transport relies on SSH authentication (not custom auth)
|
||||
- [ ] No password/credential storage in plaintext
|
||||
- [ ] Server doesn't trust client-supplied paths blindly
|
||||
- [ ] Destination directory validated before writing
|
||||
|
||||
### 10. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
### 6. Denial of Service
|
||||
- [ ] Bounded memory allocation (can't OOM server with huge chunk)
|
||||
- [ ] Timeout on connections (no indefinite blocking)
|
||||
- [ ] Maximum connection limit or rate limiting
|
||||
- [ ] Malformed protocol messages handled gracefully (no crash)
|
||||
|
||||
### 11. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
### 7. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
|
||||
### 12. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 13. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 14. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
### 15. File System Security
|
||||
### 8. File System Security
|
||||
- [ ] Received file permissions validated (no SUID/SGID injection)
|
||||
- [ ] Symlink attack prevention (don't follow symlinks in destination)
|
||||
- [ ] Race conditions in file creation (TOCTOU)
|
||||
- [ ] Temporary file security (if any)
|
||||
|
||||
### 16. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
|
||||
## Common Vulnerability Patterns
|
||||
|
||||
### Format String Bugs
|
||||
@@ -288,59 +118,9 @@ receive_n_data(fd, buffer, expected_size);
|
||||
if (!receive_n_data(fd, buffer, expected_size)) { /* handle error */ }
|
||||
```
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Detailed Finding Fields
|
||||
|
||||
For each vulnerability found, also be prepared to report:
|
||||
For each vulnerability found:
|
||||
1. **Location** — file:line
|
||||
2. **Severity** — critical / high / medium / low / informational
|
||||
3. **Category** — input-validation / buffer / memory / tls / auth / dos / crypto / fs
|
||||
@@ -349,31 +129,6 @@ For each vulnerability found, also be prepared to report:
|
||||
6. **Fix** — concrete code change
|
||||
7. **CVSS estimate** — rough severity score if exploitable
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### Audit Summary
|
||||
|
||||
Also provide a summary:
|
||||
```
|
||||
=== SECURITY AUDIT SUMMARY ===
|
||||
@@ -385,30 +140,13 @@ Low: <count>
|
||||
Informational: <count>
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -0,0 +1,310 @@
|
||||
---
|
||||
description: Scans the FastSync codebase for security vulnerabilities — buffer overflows, path traversal, TLS issues, memory safety, and cryptographic hygiene.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are a security screener for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
|
||||
## Your Role
|
||||
|
||||
Scan the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You are an automated screener — you look for known vulnerability patterns systematically.
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
|
||||
## Project Architecture
|
||||
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
array_list.c/h Dynamic array
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` | Writes received files to disk |
|
||||
|
||||
## Security Screener Checklist
|
||||
|
||||
### 1. Buffer Overflow Risks
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
|
||||
### 2. Path Traversal in File Operations
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 3. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 4. TLS / SSL Misconfiguration
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--insecure` flag
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
|
||||
### 5. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
|
||||
### 6. Integer Overflow in Allocation
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 7. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 8. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
|
||||
### 9. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
|
||||
### 10. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 11. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 12. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
@@ -138,9 +138,11 @@ int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
|
||||
|
||||
Build for fuzzing:
|
||||
```bash
|
||||
CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
|
||||
cmake -B build-fuzz -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=fuzzer,address,undefined -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=fuzzer,address,undefined"
|
||||
cmake --build build-fuzz -j$(nproc)
|
||||
./build-fuzz/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
./build-fuzz/tests/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
```
|
||||
|
||||
### AFL++ Harness
|
||||
@@ -214,7 +216,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -38,20 +38,16 @@ dd if=/dev/urandom of=/tmp/fastsync_bench/src/large.bin bs=1M count=10 2>/dev/nu
|
||||
Test each configuration 3 times, record median:
|
||||
|
||||
```bash
|
||||
# Real FastSync flags: -z=compression, -j=multithreading,
|
||||
# --chunk-serialization, --sendfile (long form only). The old rsync-style
|
||||
# spellings -c/-m/-s/-f are NOT the same options (-c=--checksum,
|
||||
# -m=--prune-empty-dirs, -s=--secluded-args, -f=--filter) and must not be used.
|
||||
CONFIGS=(
|
||||
"Standard|"
|
||||
"Compression|-z"
|
||||
"Multithreading|-j"
|
||||
"MT+Compression|-j -z"
|
||||
"Chunk Serialization|-j -z --chunk-serialization"
|
||||
"Sendfile|--sendfile"
|
||||
"Compression|-c"
|
||||
"Multithreading|-m"
|
||||
"MT+Compression|-m -c"
|
||||
"Chunk Serialization|-s"
|
||||
"MT+Compression+Chunk|-m -c -s"
|
||||
"Sendfile|-f"
|
||||
)
|
||||
|
||||
PORT=18080
|
||||
for config in "${CONFIGS[@]}"; do
|
||||
IFS='|' read -r name flags <<< "$config"
|
||||
echo "=== $name ==="
|
||||
@@ -59,14 +55,13 @@ for config in "${CONFIGS[@]}"; do
|
||||
rm -rf /tmp/fastsync_bench/dst
|
||||
mkdir -p /tmp/fastsync_bench/dst
|
||||
|
||||
./build/server -p "$PORT" --allow-unauthenticated &
|
||||
./build/server &
|
||||
SERVER_PID=$!
|
||||
sleep 0.5
|
||||
|
||||
START=$(date +%s%N)
|
||||
./build/client --source-dir /tmp/fastsync_bench/src \
|
||||
--dest-dir /tmp/fastsync_bench/dst \
|
||||
--server-port "$PORT" \
|
||||
--save-to-disk $flags
|
||||
END=$(date +%s%N)
|
||||
|
||||
@@ -79,21 +74,14 @@ for config in "${CONFIGS[@]}"; do
|
||||
done
|
||||
```
|
||||
|
||||
### Step 4: Full Benchmark Tool (Preferred)
|
||||
|
||||
The maintained benchmark tool is `benchmark/bench.py`. It handles building,
|
||||
data generation, network shaping (LAN/WAN profiles or custom `--delay`/`--jitter`/
|
||||
`--throughput`/`--loss`), rsync comparison, and JSON/table reporting:
|
||||
### Step 4: Full Integration Benchmark (Optional)
|
||||
|
||||
For comprehensive benchmarking with network shaping:
|
||||
```bash
|
||||
python3 benchmark/bench.py --help
|
||||
python3 benchmark/bench.py --runs 5 --profiles unlimited
|
||||
python3 benchmark/bench.py --size-mb 100 --random-ratio 0.5 --output json
|
||||
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
|
||||
python3 test.py --full
|
||||
```
|
||||
|
||||
Network shaping needs root (`tc`/`netem` on `lo`). SSH and TLS coverage lives in
|
||||
the pytest integration suite, not the benchmark tool.
|
||||
This tests LAN/WAN profiles, SSH, TLS, and compares against rsync.
|
||||
|
||||
### Step 5: Report Results
|
||||
|
||||
@@ -105,13 +93,12 @@ Platform: <OS, CPU, network>
|
||||
Configuration | Run 1 | Run 2 | Run 3 | Median
|
||||
-----------------------|---------|---------|---------|--------
|
||||
Standard | 0.12s | 0.11s | 0.12s | 0.12s
|
||||
Compression (-z) | 0.09s | 0.08s | 0.09s | 0.09s
|
||||
Multithreading (-j) | 0.07s | 0.07s | 0.08s | 0.07s
|
||||
MT+Compression (-j -z) | 0.05s | 0.05s | 0.06s | 0.05s
|
||||
Chunk Serialization (--chunk-serialization) | 0.05s | 0.04s | 0.05s | 0.05s
|
||||
Sendfile (--sendfile) | 0.04s | 0.04s | 0.04s | 0.04s
|
||||
Compression (-c) | 0.09s | 0.08s | 0.09s | 0.09s
|
||||
Multithreading (-m) | 0.07s | 0.07s | 0.08s | 0.07s
|
||||
MT+Compression (-m -c) | 0.05s | 0.05s | 0.06s | 0.05s
|
||||
Sendfile (-f) | 0.04s | 0.04s | 0.04s | 0.04s
|
||||
|
||||
Best configuration: Sendfile (--sendfile)
|
||||
Best configuration: MT+Compression (-m -c)
|
||||
Throughput: <X> MB/s
|
||||
```
|
||||
|
||||
|
||||
@@ -32,19 +32,23 @@ Try to reproduce the issue with the exact command the user provides.
|
||||
|
||||
**Memory errors (first priority):**
|
||||
```bash
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
# or run the failing command
|
||||
```
|
||||
|
||||
**Thread errors:**
|
||||
```bash
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
```
|
||||
|
||||
**Valgrind (if ASan doesn't find it):**
|
||||
@@ -104,12 +108,13 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
|
||||
# If integration test needed
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
python3 test.py
|
||||
|
||||
# Re-run under sanitizer to confirm fix
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
rm -rf build
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
# reproduce the original failing command
|
||||
```
|
||||
|
||||
|
||||
@@ -19,7 +19,7 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log dev..HEAD --oneline
|
||||
git log main..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Clean build
|
||||
@@ -39,13 +39,17 @@ If the PR touches threading, memory management, or network code, also build with
|
||||
```bash
|
||||
# AddressSanitizer
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake -B build-asan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
|
||||
# ThreadSanitizer (if threading changes)
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake -B build-tsan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
@@ -87,10 +91,10 @@ If tests fail:
|
||||
### Step 6: Run integration tests (optional)
|
||||
|
||||
```bash
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
python3 test.py
|
||||
```
|
||||
|
||||
This runs the integration suite (benchmarking is `benchmark/bench.py`). It takes longer — only run if the user asks or if unit tests pass.
|
||||
This runs the integration + benchmark suite. It takes longer — only run if the user asks or if unit tests pass.
|
||||
|
||||
### Step 7: Fix and commit
|
||||
|
||||
|
||||
@@ -19,13 +19,13 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log dev..HEAD --oneline
|
||||
git log main..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Get changed files
|
||||
|
||||
```bash
|
||||
git diff dev --name-only -- '*.c' '*.h'
|
||||
git diff main --name-only -- '*.c' '*.h'
|
||||
```
|
||||
|
||||
This gives the list of C source and header files changed in the PR.
|
||||
@@ -125,7 +125,7 @@ STYLE: <count>
|
||||
|
||||
If the user wants to post the review as a PR comment:
|
||||
```bash
|
||||
tea comment --repo TapTap/FastSync <number> "<review report>"
|
||||
tea pr comment <number> --comment "<review report>"
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -16,7 +16,7 @@ Ask the user or determine from context:
|
||||
- **Minor** (x.Y.0) — new features, backward compatible
|
||||
- **Patch** (x.y.Z) — bug fixes, no protocol changes
|
||||
|
||||
Current version: `PROTOCOL_VERSION "2.29.0"` in `src/shared/config.h`
|
||||
Current version: `PROTOCOL_VERSION "1.1.0"` in `src/shared/config.h`
|
||||
|
||||
### Step 2: Check Protocol Version
|
||||
|
||||
@@ -37,7 +37,7 @@ rm -rf build
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
python3 test.py
|
||||
```
|
||||
|
||||
ALL tests must pass before release.
|
||||
@@ -46,10 +46,12 @@ ALL tests must pass before release.
|
||||
|
||||
```bash
|
||||
# ASan
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
```
|
||||
|
||||
### Step 5: Update README (If Needed)
|
||||
@@ -77,23 +79,12 @@ git commit -m "Release vX.Y.Z
|
||||
git tag -a vX.Y.Z -m "Release vX.Y.Z"
|
||||
```
|
||||
|
||||
### Step 8: Push and Open dev → main PR
|
||||
|
||||
`main` is protected and only receives changes via `dev` → `main` PRs (see AGENTS.md). Never push directly to `main`.
|
||||
### Step 8: Push
|
||||
|
||||
```bash
|
||||
# Push the release commit and tag to dev
|
||||
git push origin dev
|
||||
git push origin vX.Y.Z
|
||||
|
||||
# Open the dev → main release PR for review + CI
|
||||
tea pr create --repo TapTap/FastSync --head dev --base main \
|
||||
--title "Release vX.Y.Z" \
|
||||
--description "Release vX.Y.Z"
|
||||
git push origin main --tags
|
||||
```
|
||||
|
||||
Then wait for the full CI to pass and request review before the PR is merged to `main`.
|
||||
|
||||
### Step 9: Report
|
||||
|
||||
```
|
||||
|
||||
@@ -102,9 +102,9 @@ Informational: <count>
|
||||
...
|
||||
|
||||
=== VERDICT ===
|
||||
[PASS] No critical/high-severity issues found
|
||||
[PASS] No critical/high issues found
|
||||
— or —
|
||||
[FAIL] <N> critical/high-severity issues must be fixed
|
||||
[FAIL] <N> critical/high issues must be fixed
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -4,32 +4,31 @@ FastSync is a high-performance file synchronization system written in C11. It su
|
||||
|
||||
## Dependency installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v11`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, zlib1g-dev, liblz4-dev, libxxhash-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, Node.js, plus `rsync` 3.4.1 (with zstd/xxhash/lz4), `acl` and `attr` (setfacl/getfacl, setfattr/getfattr) for drop-in parity tests. (CMake hard-requires zstd, zlib, and lz4; xxHash is fetched via `FetchContent`.)
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v7`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest, openssh-client, and Node.js.
|
||||
|
||||
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, zlib, lz4, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
|
||||
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
|
||||
|
||||
```bash
|
||||
# Use the prebuilt CI image directly (faster, guaranteed CI parity)
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v11 fastsync-ci:local
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v7
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v7 fastsync-ci:local
|
||||
|
||||
# Or build the image from the repo-root Dockerfile
|
||||
# (Note: the prebuilt :v11 image is built from the current Dockerfile and
|
||||
# includes rsync 3.4.1 plus acl/attr; rebuild from source after changing
|
||||
# the Dockerfile.)
|
||||
# (Note: the prebuilt :v7 image reflects the previous Dockerfile state;
|
||||
# rebuild from source to pick up any newly added packages like lcov/valgrind.)
|
||||
docker build -t fastsync-ci:local .
|
||||
|
||||
# Build, run unit tests, and run integration tests inside the container
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace fastsync-ci:local \
|
||||
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load'
|
||||
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/'
|
||||
|
||||
# Avoid root-owned build/ artifacts by matching your host UID/GID
|
||||
docker run --rm --user "$(id -u):$(id -g)" -v "$PWD:/workspace" \
|
||||
-w /workspace fastsync-ci:local \
|
||||
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load'
|
||||
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/'
|
||||
```
|
||||
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash from GitHub via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
|
||||
If a dependency is missing from the CI image, add it to the `Dockerfile` (and rebuild) rather than adding an install step to the CI workflow.
|
||||
|
||||
@@ -38,12 +37,11 @@ If a dependency is missing from the CI image, add it to the `Dockerfile` (and re
|
||||
When configuring for CI parity, use:
|
||||
```bash
|
||||
cmake -B build -S . -DSTRICT_WARNINGS=ON # -Wextra -Wpedantic -Werror
|
||||
cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan); in the CI matrix
|
||||
cmake -B build -S . -DSANITIZER=undefined # UndefinedBehaviorSanitizer (UBSan); in the CI matrix
|
||||
cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan); local-only, NOT in CI
|
||||
cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan)
|
||||
cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan)
|
||||
```
|
||||
|
||||
The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, the `address`+`undefined` sanitizer matrix, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. TSan is not part of the CI matrix and is a local-only configuration. The `setpriv`-marked privilege tests (four decorated functions, collecting to eight instances because two are parametrized) are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions.
|
||||
The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), build + test (unit + integration), and sanitizer (currently only `address`) jobs sequentially.
|
||||
|
||||
## Build
|
||||
|
||||
@@ -55,49 +53,33 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
```bash
|
||||
./build/tests # unit tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # full integration suite (CI excludes env-dependent privilege tests)
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset only
|
||||
|
||||
# Differential rsync-parity gate (real rsync 3.4.1 vs FastSync)
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity_ci # fast PR subset
|
||||
python3 -m pytest tests/integration/test_differential_parity.py -n 4 --dist=load -m parity # full set
|
||||
```
|
||||
|
||||
See `tests/integration/README.md` for the differential parity gate and its
|
||||
`parity_caveats.py` allowlist (the residual burn-down mechanism).
|
||||
|
||||
Unit tests under valgrind must set `FASTSYNC_UNDER_VALGRIND=1` (CI does): the
|
||||
tests use it to skip fork-based tests, because valgrind 3.22 does not expose
|
||||
`vgpreload` in the guest's `/proc/self/maps`.
|
||||
|
||||
```bash
|
||||
FASTSYNC_UNDER_VALGRIND=1 valgrind --leak-check=full --show-leak-kinds=definite --error-exitcode=1 ./build/tests
|
||||
python3 -m pytest tests/ # integration tests
|
||||
```
|
||||
|
||||
## CI Workflow — Waiting for Results
|
||||
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Monitor CI status via the Gitea API (see below) or `tea actions`, then inspect logs on failure.
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Use `gh run watch` or similar to monitor CI status, then inspect logs on failure.
|
||||
|
||||
## CI Troubleshooting
|
||||
|
||||
### If lint (clang-format) fails
|
||||
Run clang-format in the CI Docker image to match the exact CI version:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v9 \
|
||||
sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i'
|
||||
```
|
||||
|
||||
### If cppcheck fails
|
||||
Fix reported issues locally, then verify with:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v9 \
|
||||
sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/'
|
||||
```
|
||||
|
||||
### If integration tests fail
|
||||
Run locally before pushing:
|
||||
```bash
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
python3 -m pytest tests/ -v --tb=short
|
||||
```
|
||||
|
||||
## Branch Strategy
|
||||
@@ -106,7 +88,7 @@ Two main branches: `dev` (integration) and `main` (stable releases).
|
||||
|
||||
### Rules
|
||||
- **All PRs target `dev`** — never target `main` directly
|
||||
- **`dev` is intended to be the default branch** in Gitea repo settings — verify in the repo settings, since this clone's `origin/HEAD` still points at `main`
|
||||
- **`dev` is the default branch** in Gitea repo settings
|
||||
- **`main` is protected** — only merged from `dev` via PR with 2 approvals + full CI pass
|
||||
- **Feature/bug branches** branch from `dev`, PR back to `dev`
|
||||
- **`dev` → `main` merges** happen on-demand or weekly, requiring full CI + review
|
||||
@@ -182,7 +164,7 @@ This can be cron'd locally if desired (e.g., `crontab -e` with `opencode run`).
|
||||
## Is opencode a good option?
|
||||
|
||||
**Yes, for FastSync's needs.** The hybrid model works well:
|
||||
- opencode's 16 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- opencode's 17 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- The assistant orchestrates subagents, merges branches, and iterates on CI
|
||||
- You only review the final output
|
||||
|
||||
|
||||
-695
@@ -1,695 +0,0 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to FastSync are documented here. Versions match
|
||||
`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must
|
||||
run the same version because the handshake is strict.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
Wire backlog cycle (protocol 2.29.0 → 2.30.0; config-frame layout unchanged).
|
||||
|
||||
- **`--stderr=client` client-message channel (#313):** the client now accepts
|
||||
`--stderr=client` (and maps the deprecated `--no-msgs2stderr` to it), routing
|
||||
its own diagnostics over the new bounded `STATUS_CLIENT_MSG` client->server
|
||||
frame instead of writing them locally; the server writes each received
|
||||
message to its stderr (respecting the server log destination). `errors`/`all`
|
||||
behavior is unchanged.
|
||||
- **Receiver partial failures exit 23 (#320):** a per-entry receiver failure
|
||||
that does not abort the stream (e.g. an unprivileged `--devices` mknod) now
|
||||
sends the terminal `STATUS_PARTIAL`; the client exits 23 like rsync and, under
|
||||
`--remove-source-files`, still removes the sources it successfully
|
||||
transferred. A clean run stays 0 and a fatal/connection error stays non-23.
|
||||
- **Directory/symlink destination-state itemize (#314):** when `report_dest_info`
|
||||
is negotiated (now also for `--progress`), the receiver answers `STATUS_MKDIR`
|
||||
and `STATUS_SYMLINK` with the entry's pre-transfer destination snapshot
|
||||
(existence, type, perms/owner/group/time, and whether an existing symlink's
|
||||
target already matches), and the sender probes every ancestor directory before
|
||||
the receiver creates it implicitly. A re-run over an unchanged tree no longer
|
||||
emits per-directory `cd+++++++++` or unchanged-symlink lines, a changed
|
||||
directory renders rsync's `.d..t......`, and a changed symlink renders
|
||||
`cLc........` / `.L..t......`. Directory/symlink time comparison uses whole
|
||||
seconds (rsync's `cmp_time`). The `STATUS_MKDIR` body gains a probe flag and
|
||||
the `STATUS_DEST_INFO` record gains a symlink-target-match field; the
|
||||
config-frame layout is unchanged. Differential-tested against rsync 3.4.1.
|
||||
- **`--stats` deleted per-type breakdown (#316):** `STATUS_STATS` gains
|
||||
`deleted_reg/dir/link/special`, tallied by the delete observers and rendered
|
||||
as rsync's `Number of deleted files: X (reg: A, dir: B, link: C, special: D)`.
|
||||
Differential-tested against rsync 3.4.1 for a mixed-type `--delete` tree.
|
||||
|
||||
## [2.29.0] - 2026-09-23
|
||||
|
||||
The rsync-parity cycle 2.29 (no wire change; `PROTOCOL_VERSION` stays 2.28.0).
|
||||
`RSYNC_COMPAT.md` moves from **116 ✅ / 14 ⚠️ / 27 ❌** to
|
||||
**120 ✅ / 10 ⚠️ / 27 ❌** of 157 rows.
|
||||
|
||||
An audit cycle follows on the same wire version (`PROTOCOL_VERSION` stays
|
||||
2.28.0): a security-and-correctness pass over the parity-2.29 baseline, plus a
|
||||
set of audit follow-ups (filter merge modifiers, the `--inplace`/`--partial-dir`
|
||||
conflict, credential-file hardening, and small leak/log/test fixes). It fixes
|
||||
a `--temp-dir` symlink escape, gates client-controlled special permission bits,
|
||||
corrects `--partial-dir`/`--bwlimit`/`-z` behavior, handles unsupported filter
|
||||
modifiers, and tightens client and wire validation. The only parity
|
||||
reclassification is `--filter=RULE` moving ✅ → ⚠️, because its merge-only
|
||||
`e`/`n`/`w`/`-` modifiers are now accepted and consumed but their semantics
|
||||
remain unimplemented (accepted-but-ignored); the matrix is therefore **119 ✅ /
|
||||
11 ⚠️ / 27 ❌** of 157 rows. The affected rows' notes and the summary tally in
|
||||
`RSYNC_COMPAT.md` were updated. A following triage-fix cycle (see **Triage
|
||||
fixes** below) moves `-F` and `-i` to ⚠️, for a final **117 ✅ / 13 ⚠️ / 27 ❌**
|
||||
of 157 rows.
|
||||
|
||||
A no-wire parity burn-down cycle follows on 2.28.0: it accepts
|
||||
`--inc-recursive`/`--no-inc-recursive` as inert no-ops, accepts an absolute
|
||||
`--temp-dir` that canonicalizes inside the receive root, closes the
|
||||
`--delete-before` phase-0 divergence (both the single-threaded and `--threads`
|
||||
data passes replay the pre-scan list), makes `--fake-super` interoperable with
|
||||
rsync's `user.rsync.%stat` key/grammar (regular files and char/block devices
|
||||
faked as regular files), turns a failed device `mknod` into a continuing
|
||||
per-entry failure, and accepts a practical subset of rsync's `rsyncd.conf`
|
||||
grammar (modules are read-only by default, and accepted-but-unenforced
|
||||
access-control keys emit a startup warning). The matrix moves to **119 ✅ /
|
||||
14 ⚠️ / 24 ❌** of 157 rows.
|
||||
|
||||
A structural cycle then lands a transport I/O vtable over TCP/TLS (fixing the
|
||||
TLS-multithreaded sendfile path and making the per-thread SSL resolution
|
||||
explicit) and bumps the wire to **2.29.0**: the `STATUS_SYMLINK` frame grows an
|
||||
optional symlink-xattr block (captured no-follow with `llistxattr`/`lgetxattr`,
|
||||
applied no-follow with `lsetxattr`). Because the handshake is strict, 2.28.0 and
|
||||
2.29.0 peers are incompatible. Note: Linux refuses to associate xattrs with a
|
||||
symlink at all, so the symlink-xattr block is a no-op on Linux and is carried
|
||||
for correctness on platforms/filesystems that do support it; the config-frame
|
||||
layout is unchanged (golden length still 886).
|
||||
|
||||
### Changed
|
||||
|
||||
- **rsync-exact traversal order.** The sequential scanner now walks each
|
||||
directory's entries in rsync 3.4.1's flist order (non-directories ascending,
|
||||
then directories ascending, depth-first), so `--info=name`, the
|
||||
`--delete-during`/`--delete-delay`/`-n` would-delete order and the partial
|
||||
`--max-delete` survivor set match rsync byte-for-byte. `--threads` has no
|
||||
rsync analogue and stays unordered.
|
||||
- **Delete timing.** The complete `--delete-during`/`--delete-delay`
|
||||
per-directory plan set is transmitted before the first data frame, so a
|
||||
mid-transfer abort has already removed every planned extra like rsync's
|
||||
generator; `-d/--dirs` uses per-directory plans (shielded untraversed
|
||||
subdirectories) instead of the end-of-transfer commit. `-n`, `--delete`,
|
||||
`--del`/`--delete-during` and `--delete-delay` are now ✅ Parity.
|
||||
- **Basis directories.** A relative `--compare-dest`/`--copy-dest`/`--link-dest`
|
||||
DIR resolves against the destination directory with the transfer-relative
|
||||
name appended, exactly like rsync 3.4.1.
|
||||
- **`-y`/`--fuzzy`.** The candidate search no longer inherits the ordinary delta
|
||||
engine's 16 KiB minimum or 10× size-ratio bound, so an oversized or
|
||||
sub-16-KiB sibling is reused exactly as rsync reuses it.
|
||||
- `--info=mount` prints rsync's mount-point skip line (repeated `-xx` drops the
|
||||
mount-point directory); `--info=stats` enables the `--stats` block; `-x` is
|
||||
repeatable. `--stats` counts traversed directories for the `Number of files`
|
||||
breakdown under a plain `-r` scan. `--debug` emits real output for
|
||||
`flist`/`del`/`hash`/`deltasum`/`recv`/`filter`/`send`.
|
||||
|
||||
### Known residuals
|
||||
|
||||
- `--progress` and `--info` still need a receiver→sender event channel for the
|
||||
root `./` line, ancestor-directory suppression, receiver-side `skip`/`backup`
|
||||
wording, and symlink/empty-directory quick-checks.
|
||||
- `--delete-before`'s phase-0 late-file divergence remains (rsync's pre-scan
|
||||
fixes the file list before the data pass).
|
||||
- A whole-file sender that cannot stream its codec (lz4's one-shot block
|
||||
format) or a `--append`/delta source above the bound still buffers; the
|
||||
default zstd/zlib and the uncompressed paths stream (see #318).
|
||||
- `--stats` byte totals and `--msgs2stderr` stay documented divergences.
|
||||
|
||||
### Security
|
||||
|
||||
- **`--temp-dir` symlink escape fixed.** The receiver's scratch directory was
|
||||
opened with a bare `open()`, so a symlink planted under the receive root could
|
||||
redirect receiver scratch files outside the authorized root. The opened
|
||||
directory is now judged by the real path of its fd (`/proc/self/fd` via
|
||||
`realpath`) and an escaping target is refused (`EACCES`, logged); an in-root
|
||||
link to another filesystem (the `EXDEV` fallback case) still works.
|
||||
- **Client-controlled special bits masked when super-user activities are not
|
||||
permitted.** Setuid/setgid/sticky bits (`--perms`, `--chmod`, the symlink and
|
||||
special-node paths, and deferred directory modes) are now stripped when the
|
||||
connection forbids super activities (`--no-super`, a non-opted daemon module,
|
||||
a privileged listener without `--allow-super`); exact rsync semantics are
|
||||
preserved wherever super activities are permitted.
|
||||
- **Daemon umask no longer forced to `0`.** `daemonize()` now sets the
|
||||
conventional `022`, so implied parent directories created without `-p` are no
|
||||
longer world-writable `0777`.
|
||||
- **Daemon modules are read-only by default.** A `--daemon` module is now
|
||||
served read-only unless it sets `read only = no` (or rsync's `write only =
|
||||
yes`), matching rsync: a real `rsyncd.conf` that omits `read only` is no
|
||||
longer silently writable. A global `read only` still sets the default for
|
||||
later modules, and an explicit module value wins. This is a behavior change
|
||||
for existing FastSync-native configs that relied on the old writable default;
|
||||
add `read only = no` to keep them writable. An rsync `write only = yes` is
|
||||
mapped to writability (FastSync is push-only, so a module can never be read
|
||||
from the network).
|
||||
- **Accepted-but-unenforced rsync security keys now warn at startup.** The
|
||||
rsync keys FastSync recognizes but does not implement — `secrets file`,
|
||||
`refuse options`, `exclude`/`include`/`filter`, `max size`/`min size`,
|
||||
`pre-xfer exec`/`post-xfer exec`, `incoming chmod`/`outgoing chmod`,
|
||||
`name converter`, `use chroot`, `uid`/`gid`, and the rest of the
|
||||
access-control set — load for migration compatibility but now emit a
|
||||
`WARN` naming the key (and module) so an operator does not believe the
|
||||
restriction is enforced. `auth users`/`secrets file` stay fail-closed: a
|
||||
module declaring `auth users` still requires a FastSync credential store.
|
||||
- **Credentials and signal handling hardened.** Secret files are opened with
|
||||
`O_NOFOLLOW|O_NONBLOCK` (while allowing fd-backed store paths and bound-waiting
|
||||
a FIFO read for ~3 s so a slow process substitution works but a connected-but-
|
||||
silent FIFO cannot hang), and signal handlers use `sigaction` with
|
||||
async-signal-safe bodies.
|
||||
|
||||
### Fixed
|
||||
|
||||
- **`-z` on 100–256 MiB files.** The decompressor's internal ceiling was 100 MiB
|
||||
while the receiver advertises and the sender compresses whole files up to
|
||||
`MAX_RECEIVE_WHOLE_FILE_SIZE` (256 MiB), so `-z` on a 100–256 MiB regular file
|
||||
failed with `Declared decompressed size exceeds 104857600 bytes`. The ceiling
|
||||
is now defined in terms of the protocol whole-file bound (still an
|
||||
allocation-clamped bomb guard).
|
||||
- **`--bwlimit` now paces `--sendfile`.** The plaintext-TCP `--sendfile` fast
|
||||
path bypassed the protocol's token bucket, so the limit was ignored there. It
|
||||
now throttles through the same per-session leaky bucket as the TLS path.
|
||||
- **`--partial-dir` implies `--partial`.** Matching rsync 3.4.1 (which sets
|
||||
`keep_partial` after option parsing), `--partial-dir=DIR` alone retains an
|
||||
interrupted transfer's partial and wins over an explicit `--no-partial`;
|
||||
`--inplace` still bypasses the partial machinery, and combining `--inplace`
|
||||
with `--partial-dir` is now rejected up front with rsync's message
|
||||
(`--inplace cannot be used with --partial-dir`).
|
||||
- **Filter modifiers handled.** The `x` xattr-name modifier is rejected with a
|
||||
clear error everywhere. The merge-only `e`/`n`/`w` and `-` modifiers are now
|
||||
accepted and consumed on `merge`/`dir-merge` rules (so they no longer leak
|
||||
into the merge filename) while still being rejected on non-merge rules,
|
||||
matching rsync; their semantics remain unimplemented (accepted-but-ignored).
|
||||
Glued patterns (`-newfile`, `-e2e`) and mixed tokens (`H,!secret`) keep their
|
||||
historical parsing.
|
||||
- **Credential-file reads hardened.** Secret files (`--password-file`/
|
||||
`--early-input`/`--hash-credentials` input) are opened with `O_NOFOLLOW`, so a
|
||||
symlinked credential path now fails closed (`ELOOP`) instead of being followed
|
||||
before the owner/mode gate; literal fd-backed paths (`/dev/fd/<digits>`,
|
||||
`/proc/self/fd/<digits>`) are exempt so process substitution still works. A
|
||||
FIFO/process-substitution read now waits under a bounded ~3 s deadline for its
|
||||
writer, so a slow producer works while a connected-but-silent FIFO fails
|
||||
instead of hanging.
|
||||
- **Miscellaneous correctness fixes:** `--filter` rule count is checked
|
||||
client-side against `MAX_FILTER_RULES` before any network I/O (the receiver
|
||||
still re-checks the expanded count); unknown wire `Status` values are rejected
|
||||
as protocol errors; a mutex leak on an init-failure path, an `errno` read
|
||||
after `free()` in deferred delete application, `log_perror` misuse for
|
||||
non-`errno` conditions, and a `NULL` `server_host`/`ssh_destination`
|
||||
allocation path were fixed (the `config_create` failure now releases through
|
||||
`config_delete`); the decompression-limit log now prints the effective bound
|
||||
rather than the compile-time ceiling; the daemon umask and root test fixtures
|
||||
were hardened; `SSL_read` length is clamped and `sendfile` `poll()` retries on
|
||||
`EINTR`.
|
||||
|
||||
### Refactored / Docs
|
||||
|
||||
- Dropped dead `filter_rules_apply` and dead `--old-args` plumbing, unified
|
||||
`set_error`, deduplicated `path_is_within` and shared constants, and added
|
||||
printf format attributes (fixing format mismatches). `RSYNC_COMPAT.md`,
|
||||
`CHANGELOG.md` and `HANDOFF.md` were updated for the audit cycle; the
|
||||
`RSYNC_COMPAT.md` summary tally was corrected to match the rows.
|
||||
|
||||
### Triage fixes
|
||||
|
||||
- **`--dirs` directory xattrs applied inline.** A `-d/--dirs` transfer now
|
||||
applies captured directory `-X`/`-A` xattrs fd-relative on the directory entry
|
||||
instead of dropping them, so directory xattrs survive the non-recursive path
|
||||
(`src/shared/file_save.c`, `tests/test_xattr.c`).
|
||||
- **Directory/root itemize and `--out-format` lines.** `-i`/`--itemize-changes`
|
||||
and `--out-format` now emit the transfer-root `./` line and per-directory
|
||||
`cd...`/`.d..t...` lines, rendered by the shared itemize code. This matches
|
||||
rsync's fresh-transfer output; because the root line is unconditional and an
|
||||
incremental re-run may itemize directories/symlinks that rsync's quick-check
|
||||
leaves silent, `-i` is now a ⚠️ Caveat row.
|
||||
- **FROM name globs for identity maps.** `--usermap`/`--groupmap` `FROM` tokens
|
||||
now accept `*`/`?`/`[...]` globs, expanded sender-side against the passwd/group
|
||||
database and collapsed into bounded numeric ranges (`MAX_IDENTITY_MAP`),
|
||||
matching rsync.
|
||||
- **Transport fallback unit tests.** Added unit coverage for the TCP/TLS
|
||||
transport fallback paths (`tests/test_transport_tcp.c`,
|
||||
`tests/test_transport_tls.c`).
|
||||
- **Docs corrections.** `RSYNC_COMPAT.md`/`README.md` corrected stale parity
|
||||
claims for issues #286–#297: the `-F` and `-i` reclassifications, the
|
||||
`--munge-links` direction, the accepted checksum/compression name sets,
|
||||
`--bwlimit` parsing, `--stop-at` grammar, `--trust-sender`, symlink xattrs, and
|
||||
the native/non-interoperable batch and credential notes. The summary tally is
|
||||
now **117 ✅ / 13 ⚠️ / 27 ❌** of 157 rows.
|
||||
|
||||
## [2.28.0] - 2026-09-20
|
||||
|
||||
The rsync-parity cycle. `PROTOCOL_VERSION` moves `2.26.0 → 2.27.0 → 2.28.0`;
|
||||
client and server must run the same version (the handshake is strict). See
|
||||
`RSYNC_COMPAT.md` for the per-option matrix, now **116 ✅ / 14 ⚠️ / 27 ❌** of
|
||||
157 rows.
|
||||
|
||||
### Added
|
||||
|
||||
- **Differential rsync 3.4.1 parity gate** (`tests/integration/
|
||||
test_differential_parity.py`, `parity_harness.py`, `parity_caveats.py`): runs
|
||||
real `rsync` and FastSync over generated corpora and diffs the destination
|
||||
tree, normalized stdout and exit code. A fast subset runs on pull requests and
|
||||
the full strict set on push; the residual allowlist is empty.
|
||||
- FastSync-only long option **`--verify-basis`**: require a
|
||||
`--compare-dest`/`--copy-dest`/`--link-dest` hit to match the source by
|
||||
whole-file digest instead of trusting the size+mtime quick-check.
|
||||
- FastSync-only long option **`--delete-commit`** (implies `--delete`): the old
|
||||
atomic late whole-tree commit.
|
||||
- `--bwlimit` now parses rsync's units exactly and paces like rsync's leaky
|
||||
bucket; `--ignore-errors` reproduces rsync's skip-unreadable-subdir and
|
||||
IO-error-suppressed deletion (exit 23).
|
||||
- `--info=name/flist/del/remove/nonreg/progress` emit rsync's line format,
|
||||
including real-run `deleting`/`*deleting` lines carried by a new
|
||||
`report_deletes` wire bool.
|
||||
- Receiver-observed `--stats` counters: `Number of created files` now carries
|
||||
rsync's `(reg/dir/link/special)` breakdown and `Literal data` is exact for a
|
||||
delta transfer (extended `STATUS_STATS`).
|
||||
- `--progress` uses an opt-in paths-only pre-count so the `to-chk` denominator
|
||||
counts every entry like rsync, and emits per-directory/symlink/special names.
|
||||
- Receiver-side `protect`/`risk` filter engine (new bounded filter-rule wire
|
||||
block): `--filter='P ...'` now shields a destination-only entry like rsync.
|
||||
- `auto` for `--compress-choice`/`--checksum-choice` honors
|
||||
`RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST`, and per-codec compression-level
|
||||
defaults match rsync.
|
||||
- Empty source directories are recreated recursively; `-R --no-implied-dirs
|
||||
--files-from` places listed files under missing implied parents; `--iconv`
|
||||
matches rsync's push direction; `--delete-delay` reports actual removals and
|
||||
recursively removes a refilled deferred directory.
|
||||
|
||||
### Changed
|
||||
|
||||
- **`--delete` now defaults to delete-during (rsync `--del`) timing.** With no
|
||||
explicit timing flag, a plain `--delete` removes each directory's extras as
|
||||
that directory is processed instead of committing one whole-tree deletion only
|
||||
after the entire transfer succeeds. This matches rsync, frees destination
|
||||
space progressively, and avoids the whole-old+new-tree peak that could
|
||||
`ENOSPC` a tight destination. The client maps the default onto the existing
|
||||
`delete_during` wire boolean, so `PROTOCOL_VERSION` stays `2.28.0`.
|
||||
- Basis directories (`--compare-dest`/`--copy-dest`/`--link-dest`) now default
|
||||
to rsync's metadata quick-check (equal size and mtime; `--size-only` drops the
|
||||
mtime leg) instead of FastSync's historical always-verify content hash.
|
||||
`--copy-dest` re-applies the source attributes, and basis materialization is
|
||||
streamed so the 256 MiB whole-file cap no longer applies to a basis hit.
|
||||
- The per-directory `STATUS_DELETE_PLAN` frame gained a one-int `apply` flag:
|
||||
the one-shot per-run config block (protected prefixes, size-pruned mirrors,
|
||||
`--delete-missing-args` exact paths) is now always transmitted first on a
|
||||
config-only carrier (`apply=false`), fixing a latent bug where a
|
||||
`--delete-missing-args` run whose `--files-from` list synchronized no directory
|
||||
never sent its exact deletions.
|
||||
|
||||
### Notes
|
||||
|
||||
- `--delete`/`--delete-during` remain caveats for the mid-transfer abort
|
||||
boundary (rsync's generator removes all planned extras ahead of its throttled
|
||||
sender; FastSync removes only reached directories — final trees agree).
|
||||
`--delete-before`, `--progress`, `--stats`, `--fuzzy` and the basis rows keep
|
||||
their documented residuals in `RSYNC_COMPAT.md`; `--filter` and
|
||||
`--delete-excluded` are now parity, including protection of a destination-only
|
||||
excluded entry under default `--delete`.
|
||||
|
||||
### Migration
|
||||
|
||||
- Scripts that relied on plain `--delete` deleting nothing until the transfer
|
||||
fully succeeded must pass **`--delete-commit`** (or `--delete-after`) to keep
|
||||
that behavior. Plain `--delete` now removes reached directories' extras during
|
||||
the transfer, exactly like rsync's default; on a completed run the final tree
|
||||
is unchanged.
|
||||
- Deployments that relied on FastSync's stricter basis verification should pass
|
||||
**`--verify-basis`**; the default now trusts the size+mtime quick-check like
|
||||
rsync.
|
||||
|
||||
## [2.26.0] - 2026-09-17
|
||||
|
||||
|
||||
### Added
|
||||
|
||||
- **Parity-completion wave.** Closed the remaining rsync-parity gaps against
|
||||
rsync 3.4.1 and reclassified the inherently non-rsync rows. It moved the wire
|
||||
protocol three times (`2.23.0 → 2.24.0 → 2.25.0 → 2.26.0`).
|
||||
- **Delete timing (2.24.0):** per-directory delete plans
|
||||
(`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`. An interrupted
|
||||
during-transfer has already removed the reached directories' extras, while a
|
||||
delayed transfer commits per directory only after the whole transfer
|
||||
succeeds (a late-created extra survives `--delete-delay` but not
|
||||
`--delete-after`). `-R --delete` is scoped to the transferred prefix; empty
|
||||
in-scope source directories survive; dry-run never deletes.
|
||||
- **Wire stats (2.25.0):** `STATUS_STATS` carries the receiver counters
|
||||
(matched data, deleted files) and the dry-run would-delete list. `--stats`
|
||||
prints rsync's protocol-independent lines; `--progress`/`-P` print per-file
|
||||
blocks; `--out-format` gains `%b` (wire bytes), `%c` (block-sum bytes) and
|
||||
`%C` (whole-file digest); `-n --delete` prints escaped `*deleting` lines in
|
||||
the sequential and `--threads` paths.
|
||||
- **Codecs (2.26.0):** `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/
|
||||
`none` checksums, with rsync-style `auto` negotiation (default `xxh128` +
|
||||
`zstd`) and exit-4 rejection of unknown names; the resolved `compression_algo`
|
||||
crosses the wire.
|
||||
- General `-R`/`--relative` (including the `/./` cut) and `--no-implied-dirs`;
|
||||
one-level `-d`/`--dirs` listing for `dir`, `dir/` and `.`; the full filter
|
||||
grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` and
|
||||
modifiers) with `-f` bound to `--filter`; a single `-F` transfers
|
||||
`.rsync-filter` and `-FF` excludes it.
|
||||
- Receiver-side `--chown`/`--usermap`/`--groupmap` TO-name resolution; absolute
|
||||
basis directories and a `--link-dest` relink of an up-to-date destination;
|
||||
a receiver-side `--ignore-existing` short-circuit before any payload;
|
||||
`--preallocate` now wins over `--sparse` via `fallocate(2)`.
|
||||
- Client quick wins: `--iconv=.`/`-`/`--no-iconv`, a lone `-h` prints help, an
|
||||
empty `--files-from` succeeds (exit 0), a broken referent under
|
||||
`-L`/`--copy-unsafe-links` exits 23, the full `--info`/`--debug`
|
||||
vocabularies, and the aliases `--ignore-non-existing`, `--protect-args`,
|
||||
`--msgs2stderr`.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.23.0 → 2.24.0` (delete plans),
|
||||
`2.24.0 → 2.25.0` (`STATUS_STATS` + `report_stats`), and
|
||||
`2.25.0 → 2.26.0` (codec negotiation + `md4`/`sha1`/`none`).
|
||||
- `--checksum-choice`/`--cc` now accepts `md4`, `sha1`, `none` and the two-name
|
||||
form; the negotiated whole-file default is `xxh128`.
|
||||
- `--compress-choice`/`--zc` now accepts `lz4`, `zlib`, `zlibx`.
|
||||
- `RSYNC_COMPAT.md` reclassifies the matrix: 9 already-parity rows to ✅, 17
|
||||
inherently non-rsync rows to ❌ (native daemon config/auth, batch, privileged
|
||||
xattr namespaces, and the safe-subset device/privilege flags), and the genuine
|
||||
fixes to ✅; new rows cover `--bwlimit`, `--partial`, `--partial-dir`,
|
||||
`--no-whole-file`, `--inc-recursive`/`--no-inc-recursive`, `--protect-args`
|
||||
and `--msgs2stderr`.
|
||||
- The client `--help` `--max-delete` text now describes the implemented partial
|
||||
semantics (delete up to N, skip the rest, exit 25).
|
||||
|
||||
### Notes
|
||||
|
||||
- Remaining documented divergences include the `--stats` per-type file-count
|
||||
breakdown, `%b`/`%c` being FastSync wire counts, `-n --delete` line ordering,
|
||||
the default `--delete` timing (delete-after, not rsync's delete-during),
|
||||
destination-only exclude protection (still sender-derived), `--temp-dir`
|
||||
absolute paths, basis-dir attribute re-application and the 256 MiB whole-file
|
||||
cap, `--fuzzy` tie-breaking, `--bwlimit=0`/decimal rates, `zlibx`==`zlib`, and
|
||||
recursive empty-directory creation.
|
||||
- Build: adds zlib and lz4 as link dependencies.
|
||||
|
||||
## [2.23.0] - 2026-09-16
|
||||
|
||||
### Added
|
||||
|
||||
- **Rsync-parity wave.** Closed the remaining CLI, filesystem, ownership,
|
||||
deletion, and output gaps against rsync 3.4.1.
|
||||
- Short options `-r` (`--recursive`), `-b` (`--backup`), `-L`
|
||||
(`--copy-links`), and `-B` (`--block-size`/`--delta-block`); rsync
|
||||
short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached/inline values
|
||||
(`--opt=value`, `-B1000`, `-essh`, `-MOPT`). A value that starts with `-`
|
||||
is not mistaken for a cluster.
|
||||
- `-c`/`--checksum` now implies the incremental checksum quick-check (and,
|
||||
like rsync, does not imply `-t`).
|
||||
- `--checksum-choice`/`--cc` accepts `xxh64`/`xxhash`/`xxh3`/`xxh128`/`md5`/
|
||||
`auto` and rejects `md4`/`sha1`/`none` and the two-name form by name;
|
||||
`--checksum-seed=0` (the default) is randomized per transfer and the chosen
|
||||
seed is sent to the receiver.
|
||||
- `--compress-choice`/`--zc` accepts `zstd`/`none`/`auto` and rejects
|
||||
`lz4`/`zlib`/`zlibx` by name; `--skip-compress` defaults to rsync 3.4.1's
|
||||
built-in suffix list; `--no-whole-file` is accepted.
|
||||
- `--timeout` defaults to 0 (disabled) and `--contimeout` to 60 s (both `0`
|
||||
disables), matching rsync; `--max-alloc=0` means no local limit.
|
||||
- `--temp-dir` is confined to the receive root (absolute/`..` rejected by the
|
||||
receiver) and an `EXDEV` install falls back to a non-atomic copy.
|
||||
- `--numeric-ids` is documented as a mapping modifier only;
|
||||
`--usermap`/`--groupmap` support inclusive `LOW-HIGH` ranges, `*`,
|
||||
empty-`FROM` (unnamed ids), and receiver-resolved `TO` names; `--chown`
|
||||
conflicts with a map on the same side are rejected.
|
||||
- `--fake-super` records the *resolved* owner (never a real chown) and replays
|
||||
mode/time; directory ownership and directory xattrs/ACLs are preserved.
|
||||
- `-l`/`--links` stores symlink targets verbatim (absolute and `..`-bearing
|
||||
included), matching rsync; `--safe-links`/`--copy-unsafe-links` are applied
|
||||
sender-side and `--munge-links` uses rsync's `/rsyncd-munged/` marker;
|
||||
`--trust-sender` no longer affects symlink targets.
|
||||
- `--specials` recreates unix sockets with `mknod(S_IFSOCK)` (so `-D` covers
|
||||
the full rsync node set).
|
||||
- Deletion: the manifest carries a synchronized-directory section so
|
||||
`--files-from` subsets no longer delete untransmitted paths;
|
||||
`--delete-excluded` leaves size-pruned mirrors protected; extraneous
|
||||
destination symlinks are unlinked (never followed); `--max-delete=N` is
|
||||
partial (delete up to N, skip the rest, exit 25) and `--delete-missing-args`
|
||||
removals draw from the same budget; `--force` is honored during
|
||||
`--delay-updates` publication.
|
||||
- `-x`/`--one-file-system` emits the mount-point directory entry; the
|
||||
`--include`/`--exclude` layers are an ordered first-match rule list.
|
||||
- `--chmod` is a faithful port of rsync 3.4.1 (numeric/symbolic, `D`/`F`/`X`,
|
||||
`s`/`t`, append semantics, no `-p` implication, no sanitization).
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.22.0 → 2.23.0`: the delete manifest gains a
|
||||
synchronized-directory section and the terminal status gains
|
||||
`STATUS_DELETE_LIMIT` (client exit 25 on a `--max-delete`-capped commit).
|
||||
- **The 2.22.0 mode-masking divergence is removed.** Under `-p` the source mode
|
||||
is copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky;
|
||||
`--chmod` no longer implies `-p`. New files without `-p` still use
|
||||
`source_mode & ~umask` when metadata is present (else `0644`), and new
|
||||
directories without `-p` still use the `0755` creation default.
|
||||
- `--protocol=NUM` accepts only the current `2.23.0` version string.
|
||||
|
||||
### Notes
|
||||
|
||||
- The rsync-compatibility matrix (`RSYNC_COMPAT.md`) now classifies every row
|
||||
as **parity**, **caveat** (works with a documented divergence), or
|
||||
**divergent** (not supported/no-op/impossible), replacing the previous
|
||||
misleading "N implemented / 0 divergence" summary. Durable documented
|
||||
divergences remain: receiver-side symlink target containment is not enforced
|
||||
by default (verbatim storage is rsync parity; use `--safe-links`),
|
||||
`--temp-dir` rejects absolute/foreign-filesystem paths, `--copy-devices`
|
||||
reads a bounded `st_size`, a broken referent under `--copy-links` exits 0,
|
||||
new directories without `-p` use `0755`, `--stats` receiver-only counters are
|
||||
0, and `--password-file`/`--early-input`/`--hash-credentials`/`--iterations`
|
||||
and the batch format are FastSync-native.
|
||||
|
||||
## [2.22.0] - 2026-09-15
|
||||
|
||||
### Added
|
||||
|
||||
- **Per-attribute metadata preservation (protocol 2.22.0).** The former single
|
||||
metadata bundle is split into four independent, rsync-compatible flags:
|
||||
`-p/--perms`, `-t/--times`, `-o/--owner`, and `-g/--group`, each applied
|
||||
independently on the receiver, with negations `--no-perms`/`--no-times`/
|
||||
`--no-owner`/`--no-group` (short `--no-p`/`--no-t`/`--no-o`/`--no-g`) and
|
||||
`--no-preserve` clearing all four. `-a/--archive` is now full rsync
|
||||
`-rlptgoD` (owner and group included; their application stays
|
||||
privilege-gated). `-A/--acls` and `--chmod` imply `-p`, `-X/--xattrs` does
|
||||
not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply
|
||||
`-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the
|
||||
user explicitly negated them.
|
||||
- Receiver applies directory modes under `-p` (at the end of the transfer,
|
||||
alongside the deferred directory times) and symlink mode under `-p`; `-O`
|
||||
suppresses directory times only.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.21.0 → 2.22.0`: the binary config frame gains
|
||||
four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/
|
||||
`preserve_group`) after `omit_link_times`. The fixed-width `FileMetadata`
|
||||
layout is unchanged; the receiver derives the metadata-frame gate
|
||||
(`use_metadata`) from the four attributes.
|
||||
|
||||
### Notes
|
||||
|
||||
- Documented divergences from rsync: a client-supplied mode never grants
|
||||
group/other write (`S_IWGRP|S_IWOTH` are stripped for files, directories,
|
||||
symlinks, and specials; rsync's `-p` preserves them exactly); a brand-new file
|
||||
without `-p` gets `source_mode & ~umask` (sanitized) when metadata is present,
|
||||
else the historical fixed `0644`; `--chmod` implies `-p` (rsync does not);
|
||||
`-o`/`-g` map by name on the receiver with a raw-numeric fallback (only
|
||||
numeric ids cross the wire); and a daemon module without `client owner = yes`
|
||||
does not refuse a plain `-a`/`-o`/`-g` but forces super-user activities off,
|
||||
applies no ownership, and logs a warning (explicit `--chown`/`--usermap`/
|
||||
`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused).
|
||||
|
||||
## [2.21.0] - 2026-09-14
|
||||
|
||||
### Added
|
||||
|
||||
- Optional server→client rejection detail (protocol 2.21.0). A rejected
|
||||
operation may now carry a bounded human-readable reason via
|
||||
`STATUS_ERROR_DETAIL` instead of a bare `STATUS_ERROR`, so the client can
|
||||
report *why* the server refused (daemon module gate, config validation,
|
||||
receiver-side path/node validation). `receive_status()` transparently maps the
|
||||
new status back to `STATUS_ERROR` for every existing call site and captures
|
||||
the reason into a thread-local buffer exposed by `protocol_last_error()`. The
|
||||
detail body is always consumed, so the stream cannot desynchronize, and
|
||||
messages are sliced to `MAX_ERROR_DETAIL_BYTES` (4096) on send.
|
||||
- **Server-contacting `--dry-run` (protocol 2.21.0).** `--dry-run` now performs
|
||||
a real handshake with a remote/daemon receiver and reports exactly what WOULD
|
||||
change based on receiver state (existing destination files, mtimes, checksums,
|
||||
basis dirs). The wire config carries the dry-run intent (`Config.dry_run`) and
|
||||
the receiver answers each per-file check with `STATUS_DRY_RUN_TRANSFER` (would
|
||||
transfer) or `STATUS_OK` (already up to date); the sender prints the
|
||||
would-transfer set and its trailer without sending any file data. The receiver
|
||||
performs the normal read-only incremental decision but mutates nothing: no temp
|
||||
files, writes, renames, deletes, metadata/xattr/chown, or directory creation.
|
||||
A plain local destination (no explicit `--server-port`/remote) keeps the
|
||||
original client-side dry-run. Would-delete reporting for `--delete*` is
|
||||
deferred to a follow-up; dry-run never deletes.
|
||||
- Daemon `max connections per host` (per-source-IP concurrent cap, default 0 =
|
||||
unlimited), `auth lockout threshold` (default 10; 0 disables) and
|
||||
`auth lockout duration` (default 300 s) config keys.
|
||||
- `fastsync-server --allow-super` opt-in for a privileged standalone TCP server;
|
||||
without it a root standalone receiver forces super-user activities off (device
|
||||
nodes, `--write-devices`, ownership). The `--stdio` SSH argv is client-composed,
|
||||
so super activities always stay off there.
|
||||
|
||||
### Changed
|
||||
|
||||
- Config wire fields are now declared once in an X-macro table
|
||||
(`CONFIG_WIRE_FIELDS` in `src/shared/config.h`) that generates the struct
|
||||
members, defaults, and the send/receive sequence, removing the manual
|
||||
six-site field sync. Wire bytes and `PROTOCOL_VERSION` are unchanged.
|
||||
- `receive_incremental_check()` (the per-file `STATUS_CHECK` fast path) is split
|
||||
into small static helpers with a short linear orchestrator. Pure refactor: the
|
||||
wire byte stream and all cleanup are unchanged.
|
||||
- `authorized_root` state has a single owner (`utils.c`) with read accessors; the
|
||||
duplicated statics in `file.c` and the server were removed.
|
||||
- `Data` records its owning `ProtocolSession` so its memory charge is returned to
|
||||
the session that reserved it, regardless of the destroying thread.
|
||||
- The receiver pipeline moved out of `shared` into `server/receiver_pipeline.[ch]`;
|
||||
the build now uses explicit `fastsync_shared` / `fastsync_client_core` /
|
||||
`fastsync_server_core` targets instead of a GLOB, and the client no longer links
|
||||
server code.
|
||||
- The benchmark tool generates the requested random/compressible data mix
|
||||
accurately, verifies each transfer before recording it, computes correct
|
||||
percentiles, adds a MB/s column, handles `tc`/netem without requiring `sudo`
|
||||
when already root, builds into a dedicated `build-bench/` directory, and adds a
|
||||
`--warm` incremental-transfer mode.
|
||||
- The `nix-shell` dev environment provides the full toolchain (clang-format,
|
||||
cppcheck, pytest-xdist, OpenSSH, rsync, iproute2, valgrind, lcov) and no longer
|
||||
builds on entry.
|
||||
|
||||
### Security
|
||||
|
||||
- Enforce the daemon's per-module `max connections` cap (0 = unlimited) and add
|
||||
the shared per-source `max connections per host` cap plus a cross-process
|
||||
`auth lockout`. Because the listener forks one child per connection, the
|
||||
counters live in an anonymous shared mapping created before the accept loop and
|
||||
reclaimed by the parent's `SIGCHLD` handler, so the per-module, per-source and
|
||||
auth-failure state is shared across every child (including after `SIGKILL`). The
|
||||
per-source table has a bounded lifetime (expired/idle entries are reclaimed,
|
||||
with a rate-limited warning when genuinely full), and the occupancy counters are
|
||||
re-derived from the shared slot table on every child exit. Trusted loopback
|
||||
peers are exempt (they share one address); clients behind a shared NAT/proxy
|
||||
share a single per-host budget and lockout, which is documented.
|
||||
- Hardening from a full security audit:
|
||||
- Fail a truncated zstd frame instead of spinning forever (remote DoS).
|
||||
- Open receiver destination/basis/hard-link entries `O_NONBLOCK` so a
|
||||
client-planted FIFO cannot block a worker indefinitely.
|
||||
- Require a regular file before `--inplace` writes, closing a FIFO-hang and a
|
||||
raw-device write that bypassed the `--write-devices` gate.
|
||||
- Reject SSH destinations whose user/host begins with `-` and insert `--` before
|
||||
the host token, closing `-o ProxyCommand=…` argument injection (RCE).
|
||||
- Gate client `--force` recursive removal behind the server `--allow-delete`
|
||||
policy.
|
||||
- Reject empty `hosts allow`/`hosts deny`/`auth users` values instead of
|
||||
silently meaning "unrestricted".
|
||||
- Restrict TLS 1.2 to AEAD suites and set server cipher preference; load the
|
||||
private key TOCTOU-safely from an `O_NOFOLLOW` fd; verify IP literals against
|
||||
IP SANs; guard client-cert CN truncation.
|
||||
- Make `--dry-run` content-blind: it neither reads destination files nor
|
||||
hashes basis files, removing a 1-bit content oracle against `read only`
|
||||
modules.
|
||||
- Bound glob matching (iterative DP, no exponential backtracking) and bound
|
||||
line reads for filter/`--files-from`/pattern files.
|
||||
- Gate `system.posix_acl_*` xattrs on `--acls` and charge decompression/chunk
|
||||
allocations against the per-connection memory budget.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Pre-auth NULL dereference in `config_delete()` when an over-long
|
||||
`basis_count` (and the analogous count fields) was received and then failed
|
||||
validation; received counts are now validated before being published.
|
||||
- Leaked inherited `Data` in the forked compression-truncation unit test
|
||||
(valgrind definite leak).
|
||||
- `receive_status()` no longer loses a captured rejection reason when owed
|
||||
keepalives are drained.
|
||||
|
||||
## [2.20.0] - 2026-09-13
|
||||
|
||||
### Security
|
||||
|
||||
- Cap cumulative `DirTimeList` growth and bound pre-auth config-string memory
|
||||
(remote memory-exhaustion DoS).
|
||||
- Daemon host access control (`hosts allow`/`hosts deny`, IPv4/IPv6/CIDR),
|
||||
configurable global `max connections`, connection audit logging, and a
|
||||
bounded `auth failure delay` throttle. IPv4-mapped peers are normalized and
|
||||
invalid patterns are rejected at parse time (no silent fail-open).
|
||||
- Honor `--timeout` for protocol I/O and bound idle/session time to defeat
|
||||
keepalive slowloris; child-safe signal handling in the forked daemon.
|
||||
- Compiler/linker hardening (`_FORTIFY_SOURCE`, stack protector, PIE, RELRO)
|
||||
and pinned build dependencies.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Use-after-free in the basis-dir oversize preflight.
|
||||
- Placeholder `Data` leaks, `missing_args` leak, scanner chunk leak.
|
||||
- Thread-safe logging; single fd owner and cleanup epilogue in the server
|
||||
handler.
|
||||
|
||||
### Performance
|
||||
|
||||
- Metadata now crosses the wire as one packed frame (protocol 2.20.0).
|
||||
- Delete keep-set and `--files-from` lookups indexed (O(n*m) → O(n)).
|
||||
- Reused per-thread zstd contexts; `TCP_NODELAY` by default.
|
||||
- Byte-bounded sender queues; removed a redundant scanner `stat()`.
|
||||
|
||||
## [2.19.0] - 2026-09-12
|
||||
|
||||
### Security
|
||||
|
||||
- **Daemon authentication rewritten as SCRAM-SHA-256 challenge/response**
|
||||
(`STATUS_AUTH_CHALLENGE` → `STATUS_AUTH_RESPONSE` → `STATUS_AUTH_OK`/`STATUS_AUTH_FAILED`),
|
||||
replacing the old replayable static `SHA-256(password)` bearer credential.
|
||||
Each proof is bound to a fresh per-connection server nonce plus a client
|
||||
nonce, so a captured response can never be reused.
|
||||
- **Salted verifier store.** `--password-file`/`--early-input` now hold
|
||||
`user:$fastsync$1$pbkdf2-sha256$<iters>$<salt>$<stored_key>$<server_key>`
|
||||
(PBKDF2-HMAC-SHA256, default 600000 iterations, range 100000–10000000). The
|
||||
legacy `user:SHA256HEX` form is hard-rejected; there is no auto-upgrade.
|
||||
Generate stores offline with `fastsync-server --hash-credentials FILE
|
||||
[--iterations N]`.
|
||||
- **Username-enumeration hardening.** Unknown/off-list users are answered with a
|
||||
dummy verifier whose salt is a deterministic per-username value
|
||||
(`HMAC-SHA256(dummy_key, username)`), using the store-wide uniform iteration
|
||||
count and a constant-time full-length membership scan. The dummy key is
|
||||
persisted in an owner-only `<store>.dummykey` sidecar (atomic publish, exact
|
||||
mode 0600) so challenges are stable across restarts.
|
||||
- **Verified transport for auth-required modules.** A module with `auth users`
|
||||
accepts credentials only over verified TLS whose client certificate matches
|
||||
`--client-cn`, or — when `--allow-unauthenticated` is explicitly set —
|
||||
plaintext from a loopback peer. Remote plaintext is refused before any
|
||||
challenge. Clients must use `--tls` to send `--password-file` credentials to a
|
||||
non-loopback daemon; `--client-cn` is mandatory with `--tls`.
|
||||
- **Secret hygiene.** The plaintext password, derived keys, nonces/proofs and
|
||||
the dummy key are wiped from memory on every path and never logged.
|
||||
- Carried-over hardening: `-K` TOCTOU-safe directory walk
|
||||
(`openat(O_NOFOLLOW)` per component), always shell-quoted SSH remote path,
|
||||
TLS compression/renegotiation disabled, race-free (open-then-`fstat`)
|
||||
`--password-file`/`--early-input` checks, log-injection escaping, and lazy
|
||||
protocol debug escaping.
|
||||
|
||||
### Added
|
||||
|
||||
- `fastsync-server --hash-credentials FILE [--iterations N]` offline tool.
|
||||
- `<store>.dummykey` sidecar (auto-created, owner-only, 0600).
|
||||
- Integration tests for auth replay rejection, malformed frames, legacy-store
|
||||
refusal, and the loopback/TLS transport policy; fuzz targets for config
|
||||
receive and daemon-auth parsing.
|
||||
|
||||
### Changed
|
||||
|
||||
- **Protocol version 2.18.0 → 2.19.0 (breaking).** The config-frame auth block
|
||||
is now `[present][username]` (digest removed) and the auth challenge/response
|
||||
frames are interleaved between the config frame and its `STATUS_OK`. A 2.19.0
|
||||
client and a 2.18.0 server (or vice versa) fail cleanly at the handshake.
|
||||
- Daemon modules declaring `auth users` require a configured credential store at
|
||||
startup (fail closed); operators regenerate stores from plaintext with
|
||||
`--hash-credentials`.
|
||||
|
||||
### Notes
|
||||
|
||||
- First tagged release. FastSync implements rsync-compatible file
|
||||
synchronization over TCP and SSH with TLS (OpenSSL), streaming zstd
|
||||
compression, multithreaded transfers, and incremental sync. See
|
||||
[RSYNC_COMPAT.md](RSYNC_COMPAT.md) for the flag-parity matrix.
|
||||
+28
-225
@@ -1,6 +1,6 @@
|
||||
cmake_minimum_required(VERSION 3.22)
|
||||
|
||||
project(FastFileTransfer VERSION 2.30.0)
|
||||
project(FastFileTransfer)
|
||||
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_C_STANDARD 11)
|
||||
@@ -26,9 +26,9 @@ elseif(NOT SANITIZER STREQUAL "none")
|
||||
endif()
|
||||
|
||||
# --- Strict warnings option ---
|
||||
option(STRICT_WARNINGS "Enable strict warnings (Wextra, Wpedantic, Wformat-signedness, Werror)" OFF)
|
||||
option(STRICT_WARNINGS "Enable strict warnings (Wextra, Wpedantic, Werror)" OFF)
|
||||
if(STRICT_WARNINGS)
|
||||
add_compile_options(-Wextra -Wpedantic -Wformat-signedness -Werror)
|
||||
add_compile_options(-Wextra -Wpedantic -Werror)
|
||||
endif()
|
||||
|
||||
# --- Coverage option ---
|
||||
@@ -38,24 +38,11 @@ if(ENABLE_COVERAGE)
|
||||
add_link_options(--coverage)
|
||||
endif()
|
||||
|
||||
# --- Build hardening option ---
|
||||
# Production hardening is applied to the shipping server/client binaries only,
|
||||
# and only when no sanitizer or coverage instrumentation is active: sanitizers
|
||||
# carry their own instrumentation, and _FORTIFY_SOURCE requires an optimising
|
||||
# build (never the -O0 used for coverage).
|
||||
option(ENABLE_HARDENING "Enable compiler/linker hardening for production targets" ON)
|
||||
set(HARDENING_ACTIVE OFF)
|
||||
if(ENABLE_HARDENING AND SANITIZER STREQUAL "none" AND NOT ENABLE_COVERAGE)
|
||||
set(HARDENING_ACTIVE ON)
|
||||
endif()
|
||||
|
||||
include(FetchContent)
|
||||
FetchContent_Declare(
|
||||
xxhash
|
||||
GIT_REPOSITORY https://github.com/Cyan4973/xxHash
|
||||
# v0.8.3 is a lightweight tag pointing at this exact commit (no ^{} peel
|
||||
# entry); pin the commit SHA instead of the mutable tag.
|
||||
GIT_TAG e626a72bc2321cd320e953a0ccf1584cad60f363 # v0.8.3
|
||||
GIT_TAG v0.8.3
|
||||
SOURCE_SUBDIR cmake_unofficial
|
||||
)
|
||||
FetchContent_MakeAvailable(xxhash)
|
||||
@@ -68,205 +55,37 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
# --- Explicit source lists ---
|
||||
# The shared library is self-contained: it must never depend on the client or
|
||||
# server modules. In particular, the receiver pipeline (receive_thread /
|
||||
# write_thread) lives under src/server, not here, so the client executable can
|
||||
# link the shared library without pulling in any server code.
|
||||
set(SHARED_SRCS
|
||||
src/shared/array_list.c
|
||||
src/shared/batch.c
|
||||
src/shared/charset.c
|
||||
src/shared/checksum.c
|
||||
src/shared/chmod.c
|
||||
src/shared/chunk.c
|
||||
src/shared/compression.c
|
||||
src/shared/config.c
|
||||
src/shared/credentials.c
|
||||
src/shared/daemon_conf.c
|
||||
src/shared/daemon_limits.c
|
||||
src/shared/data.c
|
||||
src/shared/delay_updates.c
|
||||
src/shared/delete.c
|
||||
src/shared/delete_commit.c
|
||||
src/shared/delete_plan.c
|
||||
src/shared/delta.c
|
||||
src/shared/file.c
|
||||
src/shared/file_list.c
|
||||
src/shared/file_receive.c
|
||||
src/shared/file_save.c
|
||||
src/shared/file_send.c
|
||||
src/shared/file_store.c
|
||||
src/shared/filter.c
|
||||
src/shared/format.c
|
||||
src/shared/hardlink.c
|
||||
src/shared/identity.c
|
||||
src/shared/incremental_check.c
|
||||
src/shared/log.c
|
||||
src/shared/metadata.c
|
||||
src/shared/motd.c
|
||||
src/shared/multiprocessing.c
|
||||
src/shared/protocol.c
|
||||
src/shared/queue.c
|
||||
src/shared/stop_condition.c
|
||||
src/shared/transport_ssh.c
|
||||
src/shared/transport_tcp.c
|
||||
src/shared/transport_tls.c
|
||||
src/shared/utils.c
|
||||
src/shared/xattr.c
|
||||
)
|
||||
|
||||
# Server implementation (no main): the receiver read/write pipeline plus the
|
||||
# CLI parser. The server executable adds its own main (server.c).
|
||||
set(SERVER_CORE_SRCS
|
||||
src/server/receiver.c
|
||||
src/server/receiver_pipeline.c
|
||||
src/server/server_cli.c
|
||||
)
|
||||
set(SERVER_MAIN_SRCS src/server/server.c)
|
||||
|
||||
# Client implementation (no main): everything except the CLI entry point.
|
||||
set(CLIENT_CORE_SRCS
|
||||
src/client/change_list.c
|
||||
src/client/client_manifest.c
|
||||
src/client/client_report.c
|
||||
src/client/client_scan.c
|
||||
src/client/client_send.c
|
||||
src/client/client_validation.c
|
||||
src/client/scanner.c
|
||||
src/client/scanner_filter.c
|
||||
src/client/scanner_parallel.c
|
||||
src/client/usage.c
|
||||
)
|
||||
set(CLIENT_MAIN_SRCS src/client/client_cli.c)
|
||||
|
||||
# --- Library targets ---
|
||||
add_library(fastsync_shared STATIC ${SHARED_SRCS})
|
||||
target_include_directories(fastsync_shared PUBLIC src/shared)
|
||||
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_library(fastsync_client_core STATIC ${CLIENT_CORE_SRCS})
|
||||
target_include_directories(fastsync_client_core PUBLIC src/client)
|
||||
target_link_libraries(fastsync_client_core PUBLIC fastsync_shared)
|
||||
|
||||
add_library(fastsync_server_core STATIC ${SERVER_CORE_SRCS})
|
||||
target_include_directories(fastsync_server_core PUBLIC src/server)
|
||||
target_link_libraries(fastsync_server_core PUBLIC fastsync_shared)
|
||||
file(GLOB SHARED_SRCS "src/shared/*.c")
|
||||
set(FILE_STORE_SRCS "${CMAKE_CURRENT_SOURCE_DIR}/src/shared/file_store.c")
|
||||
list(REMOVE_ITEM SHARED_SRCS ${FILE_STORE_SRCS})
|
||||
file(GLOB SERVER_SRCS "src/server/*.c")
|
||||
set(SERVER_RECEIVER_SRCS src/server/receiver.c)
|
||||
file(GLOB CLIENT_SRCS "src/client/*.c")
|
||||
|
||||
# --- Main executables ---
|
||||
# The client links only the shared library and its own core; it deliberately
|
||||
# does NOT get src/server on its include path nor compile receiver.c.
|
||||
add_executable(server ${SERVER_MAIN_SRCS})
|
||||
target_link_libraries(server PRIVATE fastsync_server_core)
|
||||
add_executable(server ${SERVER_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS})
|
||||
target_include_directories(server PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(client ${CLIENT_MAIN_SRCS})
|
||||
target_link_libraries(client PRIVATE fastsync_client_core)
|
||||
|
||||
# --- Production hardening ---
|
||||
# Each compile flag is probed so a compiler/architecture that lacks it still
|
||||
# configures cleanly. _FORTIFY_SOURCE is guarded separately because it only
|
||||
# works in an optimising build. xxHash is a static archive built by
|
||||
# FetchContent, so it must be position-independent for the -pie link; the same
|
||||
# applies to the first-party static libraries linked into the -pie binaries.
|
||||
if(HARDENING_ACTIVE)
|
||||
set_target_properties(xxhash fastsync_shared fastsync_server_core fastsync_client_core
|
||||
PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
include(CheckCCompilerFlag)
|
||||
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
|
||||
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
|
||||
check_c_compiler_flag("${flag}" ${_harden_var})
|
||||
endforeach()
|
||||
check_c_compiler_flag("-D_FORTIFY_SOURCE=2" HARDEN_FORTIFY_SOURCE)
|
||||
foreach(target fastsync_shared fastsync_server_core fastsync_client_core server client)
|
||||
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
|
||||
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
|
||||
if(${_harden_var})
|
||||
target_compile_options(${target} PRIVATE ${flag})
|
||||
endif()
|
||||
endforeach()
|
||||
if(HARDEN_FORTIFY_SOURCE)
|
||||
target_compile_options(${target} PRIVATE -D_FORTIFY_SOURCE=2)
|
||||
endif()
|
||||
endforeach()
|
||||
foreach(target server client)
|
||||
target_link_options(${target} PRIVATE -pie -Wl,-z,relro -Wl,-z,now -Wl,-z,noexecstack)
|
||||
endforeach()
|
||||
endif()
|
||||
add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
|
||||
target_include_directories(client PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
# --- Testing ---
|
||||
enable_testing()
|
||||
|
||||
# --- Unit tests ---
|
||||
# The monolithic test binary exercises both client and server code, so it is
|
||||
# the one place that legitimately sees both include directories and links both
|
||||
# core libraries. client_cli.c is compiled here directly (with the test build
|
||||
# define) rather than linked from fastsync_client_core so its test-only shims
|
||||
# and the absence of main() are preserved.
|
||||
set(TEST_SRCS
|
||||
tests/runner.c
|
||||
tests/test_array_list.c
|
||||
tests/test_batch.c
|
||||
tests/test_change_list.c
|
||||
tests/test_checksum.c
|
||||
tests/test_chunk.c
|
||||
tests/test_client_cli.c
|
||||
tests/test_compression.c
|
||||
tests/test_config.c
|
||||
tests/test_credentials.c
|
||||
tests/test_daemon_conf.c
|
||||
tests/test_daemon_limits.c
|
||||
tests/test_data.c
|
||||
tests/test_delay_updates.c
|
||||
tests/test_delete_plan.c
|
||||
tests/test_delta.c
|
||||
tests/test_file.c
|
||||
tests/test_file_list.c
|
||||
tests/test_file_sendfile.c
|
||||
tests/test_filter.c
|
||||
tests/test_format.c
|
||||
tests/test_fuzz_smoke.c
|
||||
tests/test_glob.c
|
||||
tests/test_hardlink.c
|
||||
tests/test_iconv.c
|
||||
tests/test_log.c
|
||||
tests/test_metadata.c
|
||||
tests/test_motd.c
|
||||
tests/test_multiprocessing.c
|
||||
tests/test_property.c
|
||||
tests/test_protocol.c
|
||||
tests/test_protocol_error.c
|
||||
tests/test_queue.c
|
||||
tests/test_receiver_timeout.c
|
||||
tests/test_robustness.c
|
||||
tests/test_scanner.c
|
||||
tests/test_server.c
|
||||
tests/test_server_cli.c
|
||||
tests/test_shared_utils.c
|
||||
tests/test_stop.c
|
||||
tests/test_stress.c
|
||||
tests/test_transport_ssh.c
|
||||
tests/test_transport_tcp.c
|
||||
tests/test_transport_tls.c
|
||||
tests/test_xattr.c
|
||||
)
|
||||
# Common test libraries
|
||||
set(TEST_LIBS Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
set(TEST_INCLUDES tests src/shared src/server src/client)
|
||||
|
||||
add_executable(tests ${TEST_SRCS} src/client/client_cli.c)
|
||||
target_include_directories(tests PRIVATE tests)
|
||||
# Monolithic test binary (backward compatible)
|
||||
file(GLOB TEST_SRCS "tests/test_*.c" "tests/runner.c")
|
||||
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS} src/client/scanner.c src/client/client_cli.c src/client/client_validation.c src/client/usage.c)
|
||||
target_include_directories(tests PRIVATE ${TEST_INCLUDES})
|
||||
target_compile_definitions(tests PRIVATE FASTSYNC_TEST_BUILD)
|
||||
target_link_libraries(tests PRIVATE fastsync_server_core fastsync_client_core)
|
||||
target_link_libraries(tests PRIVATE ${TEST_LIBS})
|
||||
add_test(NAME unit_all COMMAND tests)
|
||||
|
||||
# --- Fuzz targets (requires clang) ---
|
||||
@@ -275,29 +94,13 @@ if(ENABLE_FUZZ)
|
||||
if(NOT CMAKE_C_COMPILER_ID MATCHES "Clang")
|
||||
message(FATAL_ERROR "ENABLE_FUZZ requires Clang (compiler is ${CMAKE_C_COMPILER_ID})")
|
||||
endif()
|
||||
set(FUZZ_SRCS
|
||||
tests/fuzz/fuzz_chunk_deserialize.c
|
||||
tests/fuzz/fuzz_compress_decompress.c
|
||||
tests/fuzz/fuzz_config_receive.c
|
||||
tests/fuzz/fuzz_delta_deserialize.c
|
||||
tests/fuzz/fuzz_delta_signature_deserialize.c
|
||||
tests/fuzz/fuzz_glob_match.c
|
||||
tests/fuzz/fuzz_identity_parse.c
|
||||
tests/fuzz/fuzz_manifest.c
|
||||
tests/fuzz/fuzz_metadata_from_buf.c
|
||||
tests/fuzz/fuzz_protocol_framing.c
|
||||
tests/fuzz/fuzz_xattr_block.c
|
||||
)
|
||||
# Compile the sources under test directly so libFuzzer's coverage
|
||||
# instrumentation sees them (static libraries would be uninstrumented).
|
||||
set(FUZZ_CORE_SRCS ${SHARED_SRCS} src/server/receiver.c src/server/receiver_pipeline.c)
|
||||
file(GLOB FUZZ_SRCS "tests/fuzz/*.c")
|
||||
foreach(FUZZ_SRC ${FUZZ_SRCS})
|
||||
get_filename_component(FUZZ_NAME ${FUZZ_SRC} NAME_WE)
|
||||
add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${FUZZ_CORE_SRCS})
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE tests src/shared src/server)
|
||||
add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE ${TEST_INCLUDES})
|
||||
target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer)
|
||||
target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE ${TEST_LIBS})
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
+2
-16
@@ -2,22 +2,8 @@ FROM ubuntu:24.04
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \
|
||||
python3 python3-pip python3-venv openssl openssh-client \
|
||||
lcov valgrind clang libclang-rt-18-dev \
|
||||
acl attr zlib1g-dev liblz4-dev libxxhash-dev && \
|
||||
pip3 install --break-system-packages pytest pytest-xdist && \
|
||||
lcov valgrind clang libclang-rt-18-dev && \
|
||||
pip3 install --break-system-packages pytest && \
|
||||
curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \
|
||||
apt-get install -y --no-install-recommends nodejs && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# rsync is used as the reference implementation for drop-in parity tests.
|
||||
# Ubuntu 24.04 ships 3.2.7, so build the pinned 3.4.1 reference from source.
|
||||
ARG RSYNC_VERSION=3.4.1
|
||||
ARG RSYNC_SHA256=2924bcb3a1ed8b551fc101f740b9f0fe0a202b115027647cf69850d65fd88c52
|
||||
RUN curl -fsSL "https://download.samba.org/pub/rsync/src/rsync-${RSYNC_VERSION}.tar.gz" -o /tmp/rsync.tar.gz && \
|
||||
echo "${RSYNC_SHA256} /tmp/rsync.tar.gz" | sha256sum -c - && \
|
||||
tar -xzf /tmp/rsync.tar.gz -C /tmp && \
|
||||
cd "/tmp/rsync-${RSYNC_VERSION}" && \
|
||||
./configure --enable-zstd --enable-xxhash --enable-lz4 && \
|
||||
make -j"$(nproc)" && \
|
||||
make install && \
|
||||
rm -rf "/tmp/rsync-${RSYNC_VERSION}" /tmp/rsync.tar.gz
|
||||
|
||||
-262
@@ -1,262 +0,0 @@
|
||||
# FastSync — Session Handoff (2026-09-21)
|
||||
|
||||
## Current status
|
||||
- **Release `v2.28.0`** is tagged and merged to `main`: tag `v2.28.0` points at
|
||||
`ee6523a`, and the PR #304 merge commit `b4d54504` is on `main`.
|
||||
- **`dev` is at `0fbb9de`** — the merge of parity cycle 2.29 (PR #305). The old
|
||||
`558782d` (incremental-check flake fix) is an ancestor.
|
||||
- **`PROTOCOL_VERSION` = `"2.28.0"`** (`src/shared/config.h`); CMake
|
||||
`project(FastFileTransfer VERSION 2.28.0)`.
|
||||
- **Parity cycle 2.29 is merged to `dev`** (PR #305), no wire change. It closed
|
||||
the scanner-order, delete-timing, relative-basis and fuzzy-eligibility
|
||||
residuals and improved the `--info`/`--stats`/`--debug` partials. Parity
|
||||
matrix: **120 ✅ / 10 ⚠️ / 27 ❌ = 157**. Remaining ⚠️ rows: `--info`,
|
||||
`--debug`, `--msgs2stderr`, `--stats`, `--progress`, `--delete-before`, the
|
||||
three basis-dir options, and `-y`/`--fuzzy`.
|
||||
- **Audit cycle complete on branch `fix/audit-cycle`** (branched from `dev` @
|
||||
`0fbb9de`), integration PR to `dev` pending. No wire change
|
||||
(`PROTOCOL_VERSION` stays 2.28.0). It lands the receiver/client security and
|
||||
correctness fixes — `--temp-dir` symlink-escape confinement, special-bit
|
||||
masking under a super-off policy, daemon `umask(022)`, the `-z` decompression
|
||||
ceiling raised to the 256 MiB whole-file bound, `--bwlimit` pacing the
|
||||
plaintext `--sendfile` path, `--partial-dir` implying `--partial`, rejection
|
||||
of unsupported filter modifiers (`x`/`e`/`n`/`w`), client-side
|
||||
`MAX_FILTER_RULES` enforcement, unknown wire `Status` rejection, and the
|
||||
accompanying refactors/docs. The parity matrix is unchanged at
|
||||
**120 ✅ / 10 ⚠️ / 27 ❌ = 157**; this docs pass (worktree `fix/audit-docs2`)
|
||||
corrects the `RSYNC_COMPAT.md` summary tally to match the rows.
|
||||
|
||||
|
||||
## What landed this session
|
||||
1. **Wave 8 (refactors):** Config X-macro wire table; single-owner `authorized_root`;
|
||||
daemon per-module/per-host caps + cross-process auth lockout (`daemon_limits.[ch]`);
|
||||
`Data` charge returns to its owning `ProtocolSession`.
|
||||
2. **Wave 9 (protocol 2.21.0):** optional `STATUS_ERROR_DETAIL` rejection reasons;
|
||||
server-contacting `--dry-run` (`STATUS_DRY_RUN_TRANSFER`, receiver mutates nothing).
|
||||
3. **Security wave:** ran 5 parallel audits (wire parsing; daemon/transport/TLS/auth;
|
||||
receiver confinement; client/CLI/SSH; crypto/memory/limits). Fixed all HIGH and the
|
||||
confirmed MEDIUMs:
|
||||
- SSH `-o ProxyCommand=…` argument injection (RCE) — reject leading `-`, insert `--`.
|
||||
- Truncated zstd frame infinite CPU loop (remote DoS).
|
||||
- FIFO receiver opens lacked `O_NONBLOCK` (indefinite hang).
|
||||
- `--inplace` could write a FIFO/device (bypass of `--write-devices` gate).
|
||||
- `--force` not gated by server `--allow-delete`.
|
||||
- Privileged standalone server defaulted super activities on; added `--allow-super`
|
||||
(never honored with `--stdio`).
|
||||
- `--dry-run` content/hash oracle on `read only`/basis files removed.
|
||||
- Empty `hosts allow`/`deny`/`auth users` now rejected.
|
||||
- TLS: AEAD-only 1.2 + server preference, TOCTOU-safe key load, IP-SAN verify,
|
||||
CN-truncation guard. Glob backtracking bounded; line reads bounded; ACL xattrs
|
||||
gated on `--acls`; decompression/chunk memory charged; pre-auth `basis_count`
|
||||
NULL-deref fixed.
|
||||
4. **Tooling:** benchmark accuracy (data mix, verification, percentiles, `tc`,
|
||||
`build-bench/`, `--warm` mode); `shell.nix` full toolchain and no build-on-entry;
|
||||
docs state push-only / remote-source unsupported.
|
||||
5. **Preserve-attribute split (protocol 2.22.0)** landed on `feat/preserve-attr-split`: per-attribute `-p/-t/-o/-g` + `--no-*` negations, `-a` = `-rlptgoD`, and the 2.21.0 → 2.22.0 wire bump.
|
||||
6. **Rsync-parity wave (protocol 2.23.0)** on `feat/rsync-parity`: rsync short options/clustering/attached values (`-r`/`-b`/`-L`/`-B`, `-av`, `-aAX`, `-B1000`, `-essh`, `-MOPT`), `-c` checksum quick-check, `--checksum-choice`/`--compress-choice` validation and seed randomization, rsync timeout/max-alloc defaults, temp-dir confinement + `EXDEV` fallback, ownership/mapping parity (numeric-ids modifier, map ranges/`*`/empty-FROM, `--chown`+map conflicts, fake-super resolved-owner record), verbatim symlink storage with rsync `--safe-links`/`--munge-links`, socket recreation under `--specials`, `--chmod` 3.4.1 semantics, and delete scoping + `--max-delete` partial/exit-25. Wire: appended delete-manifest synchronized-directory section and `STATUS_DELETE_LIMIT`.
|
||||
7. **Parity-completion wave (protocol 2.24.0 → 2.26.0)** on `feat/parity-completion`: per-directory delete plans (`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`; receiver `STATUS_STATS` counters feeding `--stats`/`--progress` and `--out-format %b/%c/%C`, plus `-n --delete` lines; `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/`none` checksums with `auto` negotiation (default `xxh128`/`zstd`); general `-R`/`--no-implied-dirs`/`-d`; the full filter grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` + modifiers) and corrected `-F`/`-FF`; receiver-side `--chown`/map TO-name resolution; absolute basis dirs + `--link-dest` relink; receiver-side `--ignore-existing` short-circuit; `--preallocate` over `--sparse` via `fallocate(2)`; `--iconv=.`/`-`/`--no-iconv`; lone `-h` help; aliases `--ignore-non-existing`/`--protect-args`/`--msgs2stderr`; and the full `--info`/`--debug` vocabulary. `RSYNC_COMPAT.md` reclassifies the matrix to 106 ✅ / 27 ⚠️ / 23 ❌; the later rsync-parity-stats pass (`fix/parity-stats`) moves it to 107 ✅ / 25 ⚠️ / 24 ❌ (see item 8).
|
||||
8. **rsync-parity-stats pass** on `fix/parity-stats` (no wire change, `PROTOCOL_VERSION` stays `2.26.0`): `--delete-delay` now reports only entries it actually removes, while the `--max-delete` budget is charged at plan/snapshot time (`planned`, via `defer_add`) to bound the deferred list (a refilled deferred directory that survives `ENOTEMPTY` is not reported but still consumes budget); `--stats` gained the `(reg/dir/link/special)` `Number of files` breakdown and now counts only regular files actually stored for `Number of regular files transferred`/transferred size/literal data (up-to-date re-runs report 0); `Total file size` includes symlink target lengths; `--progress` prints the leading `./` root line and counts it in `to-chk` so a single-file transfer matches rsync; and `%C` uses the selected transfer checksum with `checksum_digest_file` supporting md4/sha1/none, byte-identical to rsync for every algorithm. `--out-format` reclassified ❌ (`%b`/delta-`%c` are protocol-specific). Differential + regression tests added; full suite + ASan + clang-format + cppcheck clean.
|
||||
9. **Option-parity wave (protocol 2.26.0 → 2.27.0, on `fix/parity-options`):**
|
||||
`--bwlimit` now ports rsync 3.4.1's units/quantization and paces like its
|
||||
leaky bucket; `--ignore-errors` reproduces rsync's default (an I/O error
|
||||
skips deletion unless the flag is set; the readable tree still transfers and
|
||||
the run exits 23) across every delete timing; the `--info` categories with a
|
||||
FastSync event (`name`/`flist`/`del`/`remove`/`nonreg`/`progress`) emit
|
||||
rsync's line format, with real-run `deleting`/`*deleting` lines carried over
|
||||
the new trailing config bool `report_deletes` (golden wire updated by
|
||||
`tests/test_config.c`). Two residuals were reclassified **divergent**: `-M`
|
||||
over daemon/TCP (no argv channel in FastSync's binary config handshake;
|
||||
rsync-daemon differential pins the rsync behavior) and receiver-side
|
||||
`protect`/`risk` re-derivation for destination-only entries (would need a
|
||||
receiver filter engine; differential pins the divergence — **reversed by
|
||||
track 4a below**, which adds that engine). The options pass
|
||||
stands at **110 ✅ / 21 ⚠️ / 26 ❌**. New `tests/integration/test_option_parity.py`
|
||||
holds the rsync differentials (bwlimit parse+rate, info lines, real-setpriv
|
||||
`--ignore-errors`, rsync-daemon `-M`, filter-protect pin).
|
||||
|
||||
10. **rsync-parity-fs pass** on `fix/parity-fs` (no wire change of its own; integrated
|
||||
on top of the 2.27.0 options wave): recursive transfers now recreate empty source directories (and
|
||||
`-m/--prune-empty-dirs` still suppresses them), a directory entry replaces a
|
||||
blocking destination regular file, and `-R --no-implied-dirs --files-from`
|
||||
places a listed file under a missing implied parent with default attributes
|
||||
instead of refusing (real rsync 3.4.1 parity, differential-tested). `--iconv`
|
||||
now reproduces rsync's push direction (destination charset = the spec's REMOTE
|
||||
half; a server `--iconv` overrides), and `-T/--temp-dir` relative semantics are
|
||||
confirmed identical while the absolute-path confinement is a deliberate
|
||||
divergence. The basis-dir options, `--delay-updates` and `--dry-run` were
|
||||
reclassified to ❌ after a differential test reproduced each exact residual
|
||||
(basis content verification, fixed staging-name collision, and dry-run
|
||||
would-delete over-report). `--fuzzy` was also reclassified to ❌ (deterministic
|
||||
heuristic with a 10× size window, not rsync's matcher), but its residual is the
|
||||
candidate-selection heuristic itself: the final tree is byte-exact by design, so
|
||||
it is pinned by the `TestFuzzy` threshold suite rather than a byte-level rsync
|
||||
differential. (Track 5b later found the name heuristic is rsync's own and moved
|
||||
the row ❌ → ⚠️, leaving only the narrower delta size window; see entry 15.) The parity-review pass then moved `--delete-delay` to ⚠️ (the
|
||||
plan-time `--max-delete` charge and non-recursive deferred removal differ from
|
||||
rsync when a snapshotted entry fails removal). Differential-gate allowlist
|
||||
entries `min_size`/`empty_dirs_recursive`/`dirs_plain` were removed. The
|
||||
integrated stats+options+fs branch stands at **111 ✅ / 13 ⚠️ / 33 ❌ = 157**;
|
||||
full suite + ASan + clang-format + cppcheck clean.
|
||||
|
||||
11. **No-wire parity track 1** on `feat/parity-2.28` (no protocol change):
|
||||
`-n --delete` now sends the same filter-excluded + size-pruned protected
|
||||
prefixes and synchronized-directory scope as a real run (dry-run would-delete
|
||||
matches rsync for source-derived protections; the destination-only exclude
|
||||
residual was later closed by track 4a, readdir ordering remains);
|
||||
`--delete-delay` now charges
|
||||
`--max-delete` on actual removals and re-scans a queued directory at commit
|
||||
to remove content created after the plan, with an independent deferred-list
|
||||
cap (only partial-delete ordering remains); and `--info=name2` emits `NAME is
|
||||
uptodate` plus the leading `./` root name line for `--info=name` (only the
|
||||
root-line trigger condition and receiver-side `skip` wording remain). Matrix
|
||||
now **111 ✅ / 14 ⚠️ / 32 ❌ = 157**; differential + unit tests added in
|
||||
`test_features.py`, `test_option_parity.py`, the unit test
|
||||
`tests/test_delete_plan.c`,
|
||||
`test_delete_delay_budget_parity.py`, `test_delete_timing_parity.py`.
|
||||
12. **No-wire parity track 2b** on `feat/parity-2.28` (no protocol change):
|
||||
`--progress`/`-P`/`--info=progress` (when not `--quiet`) now run an opt-in
|
||||
paths-only metadata pre-count (no file reads/hashing) that supplies rsync's
|
||||
full file-list total for the `to-chk` denominator and the directory names,
|
||||
and emits per-directory/symlink/special name lines, in both the sequential
|
||||
and `--threads` paths. `--delete-during`/`--delete-delay` reuse their
|
||||
keep-set pre-scan instead of a second walk; non-progress runs are
|
||||
unaffected. Differential tests (`progress`/`progress_threads` over a new
|
||||
`multidir` corpus) match rsync's name set and `to-chk` denominator on a
|
||||
fresh transfer, and the single-file byte-identical test still passes;
|
||||
emission order (rsync's sorted depth-first vs FastSync's readdir/BFS stream)
|
||||
plus re-run over-naming (unconditional `./`, ancestor dirs named with a
|
||||
transferred child, and no quick-check for symlinks/empty dirs) remain the
|
||||
caveats, so the row stays ⚠️ and the matrix is unchanged at
|
||||
**111 ✅ / 14 ⚠️ / 32 ❌ = 157**.
|
||||
|
||||
13. **Wire parity track 4a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0`): the receiver now has a delete-time filter engine. The sender
|
||||
compiles its root-level selection rules exactly as the scanner does
|
||||
(`filter_base_build`) and streams them as one bounded, self-describing
|
||||
config-frame block (action, sides, anchored, dir-only, negate, owner,
|
||||
pattern; bounded rule count and pattern bytes, unknown action/sides is a
|
||||
protocol error). The receiver reconstructs `protect_rules` and applies them
|
||||
first-match-wins to each extraneous destination path in every delete timing
|
||||
(the whole-tree commit walker, the `--delete-during`/`--delete-delay`
|
||||
per-directory plans, and the `-n` would-delete enumeration), so a
|
||||
`P *.log` rule protects a destination-only `extra.log` like rsync (with
|
||||
`risk` cancelling); the sender-derived protected-prefix behavior is
|
||||
preserved when no rules are sent and `--delete-excluded` semantics are
|
||||
unchanged. Per-directory merge (`:`/`.`) receiver re-derivation remains the
|
||||
residual. `TestFilterProtect` (real + dry-run) plus differential cases
|
||||
`filter_protect`, `filter_protect_during`, `filter_protect_delay` added and
|
||||
the `--filter=RULE` row moves ❌ → ✅: matrix now
|
||||
**115 ✅ / 11 ⚠️ / 31 ❌ = 157**; unit tests, the three named integration
|
||||
files, clang-format and cppcheck clean.
|
||||
|
||||
14. **Wire parity track 5a** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0` by project decision): the three basis-dir options now default to
|
||||
rsync's metadata quick-check (equal size + equal mtime, or size alone under
|
||||
`--size-only`; `-I` disables matching) instead of FastSync's historical
|
||||
xxHash64 content equality, so a same-size/different-content basis is trusted
|
||||
exactly as rsync trusts it. A new FastSync-only, long-only `--verify-basis`
|
||||
flag restores the strict whole-file content equality; its bool is appended to
|
||||
the basis block of the config frame (golden wire frame 882 → 886 bytes).
|
||||
`--verify-basis` streams the confined basis descriptor to hash it, and a
|
||||
basis hit is no longer capped at the 256 MiB whole-file payload bound:
|
||||
`--copy-dest` streams the basis through a bounded buffer and `--link-dest`'s
|
||||
copy fallback streams from the basis, so an over-limit hit materializes (a
|
||||
basis MISS still falls back to the normal transfer and keeps its own bound).
|
||||
A `--copy-dest` hit re-applies the SOURCE attributes (the sender transmits
|
||||
the source metadata with the basis check frame), matching rsync's
|
||||
"copy then fix attributes"; a `--link-dest` success keeps the shared inode's
|
||||
attributes (writing through it would mutate the basis). Differential cases
|
||||
`copy_dest` and `verify_basis` added; `test_basis_dir_size_only_content_residual`
|
||||
converted to a passing parity assertion; `TestBasisDestDirs` updated for the
|
||||
new default + `--verify-basis`; unit tests cover the quick-check/verify
|
||||
decision and the same-size/different-content handshake. The
|
||||
`--compare-dest`/`--copy-dest`/`--link-dest` rows move ❌ → ⚠️ (relative-DIR
|
||||
resolution base and over-limit MISS refusal): matrix now
|
||||
**116 ✅ / 13 ⚠️ / 28 ❌ = 157**.
|
||||
|
||||
15. **No-wire parity track 5b** on `feat/parity-2.28` (`PROTOCOL_VERSION` stays
|
||||
`2.28.0` by project decision): `-y`/`--fuzzy` reclassified ❌ → ⚠️. A probe
|
||||
against real rsync 3.4.1 (pinned `-B8192`, repeated-content 64 KiB corpus)
|
||||
showed the name heuristic is already rsync's (`util1.c fuzzy_distance` /
|
||||
`find_filename_suffix` + the exact size+mtime pass) and the output is always
|
||||
byte-exact; the only residual is candidate ELIGIBILITY, because FastSync's
|
||||
`delta_should_attempt` gate caps the size ratio at 10× and requires both
|
||||
files ≥ 16 KiB while rsync will reuse a basis from 0.25× to 10000× and below
|
||||
16 KiB. The choice is observable only as `--stats` bandwidth counters. Added
|
||||
differential case `fuzzy_basis` (same-suffix sibling, one name edit,
|
||||
identical content, block size pinned) asserting tree **and** normalized
|
||||
`--stats` parity where the choices coincide, plus `TestFuzzy` pinning the
|
||||
window boundary on both sides (>10× and <16 KiB siblings declined by
|
||||
FastSync while rsync uses them, both trees byte-identical). Matrix now
|
||||
**116 ✅ / 14 ⚠️ / 27 ❌ = 157**.
|
||||
|
||||
16. **Lockstep delete-default track 6** on `feat/parity-2.28` (`PROTOCOL_VERSION`
|
||||
stays `2.28.0`): plain `--delete` now defaults to rsync's delete-during
|
||||
(`--del`) timing, normalized on the client onto the existing `delete_during`
|
||||
wire bool. The old late whole-tree commit is opt-in via `--delete-after` or
|
||||
the FastSync-only long `--delete-commit` (identical `delete_after` timing).
|
||||
`-d/--dirs` still falls back to the end commit, `--delay-updates` still
|
||||
deletes before publication, and `--files-from`/`-R` scope is unchanged. The
|
||||
`STATUS_DELETE_PLAN` frame gained a one-int `apply` flag so the per-run
|
||||
config block (including `--delete-missing-args` exact paths) is always
|
||||
transmitted, on a config-only carrier when the scope allows no directory
|
||||
plan — fixing a latent bug with a file-only `--files-from` list. Differential
|
||||
cases `delete`/`delete_commit`/`filter_protect_after` plus the extended
|
||||
`test_delete_timing_parity.py` (plain `--delete` mid-abort removes reached
|
||||
extras, `--delete-commit` defers) pass; full `-m "not setpriv"` suite,
|
||||
clang-format and cppcheck clean. Matrix unchanged at
|
||||
**116 ✅ / 14 ⚠️ / 27 ❌ = 157** (the `--delete`/`--delete-during` rows stay
|
||||
⚠️ for the abort boundary; `--delete-after` stays ✅).
|
||||
|
||||
17. **Audit cycle** on `fix/audit-cycle` (from `dev` @ `0fbb9de`;
|
||||
`PROTOCOL_VERSION` stays `2.28.0`): a security/correctness pass over the
|
||||
parity-2.29 baseline. It raises the decompression ceiling to the 256 MiB
|
||||
protocol whole-file bound (`-z` on 100–256 MiB files now works), paces the
|
||||
plaintext-TCP `--sendfile` path with `--bwlimit`, confines the `--temp-dir`
|
||||
scratch dir by the fd's real path (symlink escape refused), masks
|
||||
client-controlled setuid/setgid/sticky bits when super activities are not
|
||||
permitted, sets the daemon umask to `022`, makes `--partial-dir` imply
|
||||
`--partial`, rejects the unsupported filter modifiers (`x`/`e`/`n`/`w`),
|
||||
enforces `MAX_FILTER_RULES` client-side, rejects unknown wire `Status`
|
||||
values, and hardens credentials/signal handling (with the accompanying
|
||||
refactors and docs). No row changes classification, so the matrix stays
|
||||
**120 ✅ / 10 ⚠️ / 27 ❌ = 157**. This docs pass is on `fix/audit-docs2`.
|
||||
|
||||
## Next steps
|
||||
1. **Open and merge the audit-cycle PR** (`fix/audit-cycle`, including this
|
||||
`fix/audit-docs2` docs pass) into `dev` once reviewed. `dev` is the default
|
||||
branch; all PRs target `dev`, never `main` directly.
|
||||
2. **Remaining deferred items:**
|
||||
- **Large structural refactors:** delete-engine consolidation
|
||||
(`delete_extras_fd`/`manifest_delete_extras`/the delete-plan path),
|
||||
god-function splits, and translation-unit splits.
|
||||
- **`--progress`/`--info` receiver→sender event channel:** the root `./`
|
||||
line, ancestor-directory suppression, receiver-side `skip`/`backup` echo,
|
||||
and symlink/empty-dir quick-check feedback.
|
||||
- **`--delete-before` phase-0 keep-set** (rsync fixes the file list before
|
||||
the data pass; FastSync keeps its pre-scan snapshot race).
|
||||
- ~~**>256 MiB single-file streaming** (B4, the general whole-file limit).~~
|
||||
Closed by #318: the whole-file payload, the basis read/verify and the fuzzy
|
||||
basis are streamed through bounded buffers (lz4/append remain buffered).
|
||||
- **Wire native-size framing:** lengths are native `size_t` and the protocol
|
||||
assumes homogeneous word size/endianness — document or move to fixed-width
|
||||
framing.
|
||||
- **SCRAM-like daemon auth channel binding:** no TLS channel binding today
|
||||
(and it is not RFC 5802).
|
||||
- Still-open security nits: the pre-auth config/daemon-auth handshake has no
|
||||
aggregate wall-clock deadline (per-message timeout only — slowloris holds
|
||||
connection slots); the per-source registry fails open when the shared table
|
||||
is full (per-module/global caps and host ACLs still apply).
|
||||
3. **Out of scope / intentional:** pull (remote source) mode is **not** planned —
|
||||
FastSync is push-only; see `RSYNC_COMPAT.md#direction`.
|
||||
|
||||
## Key facts / commands
|
||||
- CI image: `gitea.tap-tap.win/taptap/fastsync-ci:v11` (alias `fastsync-ci:local`).
|
||||
- Build/test: `cmake -B build -S . -DSTRICT_WARNINGS=ON && cmake --build build -j$(nproc) && ./build/tests`
|
||||
then `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`.
|
||||
- Dev shell: `nix-shell` (provides clang-format, cppcheck, pytest-xdist, openssh,
|
||||
rsync, iproute2, valgrind, lcov; does not build on entry).
|
||||
- Gitea API token: supplied out-of-band via the `TOKEN` environment variable; it is
|
||||
intentionally **not** recorded in this file.
|
||||
- CI polling: `GET /api/v1/repos/TapTap/FastSync/actions/runs?limit=N`, match `head_sha`,
|
||||
then `/actions/runs/<id>/jobs`.
|
||||
@@ -6,10 +6,6 @@ source/destination model and rsync-style options while adding optional
|
||||
multithreading, streaming zstd compression, chunking, zero-copy TCP transfers,
|
||||
and native TCP/TLS transports.
|
||||
|
||||
The release version is FastSync's client/server protocol version (printed by
|
||||
`./build/client --version`); client and server must match. See
|
||||
[CHANGELOG.md](CHANGELOG.md) for the history.
|
||||
|
||||
The compatibility target is straightforward:
|
||||
|
||||
- Existing rsync commands should keep the same meaning.
|
||||
@@ -28,7 +24,7 @@ FastSync uses a producer-consumer transfer pipeline and can combine several
|
||||
optimizations for large or high-latency transfers:
|
||||
|
||||
- Multithreaded scanning, loading, and sending.
|
||||
- Streaming compression (zstd by default, plus lz4/zlib/zlibx) with levels 1 through 22.
|
||||
- Streaming zstd compression with levels 1 through 22.
|
||||
- Configurable file chunking and compact chunk serialization.
|
||||
- `sendfile()` zero-copy transfers over TCP.
|
||||
- Batched incremental checks to reduce round trips.
|
||||
@@ -51,230 +47,80 @@ replacement for every rsync feature or protocol mode.
|
||||
- Rsync-style source and destination arguments.
|
||||
- SSH transport using `user@host:destination` paths below the remote authorized root.
|
||||
- TCP client/server transfers.
|
||||
- Dry runs (server-contacting since protocol 2.21.0 for server-routed targets),
|
||||
excludes, includes, size filters, backups, statistics, and bandwidth
|
||||
limiting.
|
||||
- Incremental size/mtime checks and optional content checks (`xxh128` by
|
||||
default, selectable with `--checksum-choice`).
|
||||
- Dry runs, excludes, includes, size filters, backups, statistics, and
|
||||
bandwidth limiting.
|
||||
- Incremental size/mtime checks and optional xxHash64 content checks.
|
||||
- FastSync-native delta transfer for changed files.
|
||||
- Optional mode and timestamp preservation.
|
||||
- Delete manifests with server-side delete authorization.
|
||||
- Temporary-file writes with atomic rename by default.
|
||||
- Path traversal checks and destination-root confinement.
|
||||
|
||||
### Boundaries and documented divergences
|
||||
|
||||
The items below summarize FastSync's rsync compatibility status — recently
|
||||
closed gaps and the remaining known divergences. Each row of the detailed
|
||||
matrix is classified as parity, caveat, or divergent in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
### Not yet equivalent to rsync
|
||||
|
||||
- The FastSync wire protocol is not the rsync wire protocol.
|
||||
- SSH mode requires `fastsync-server` on the remote host.
|
||||
- Archive mode covers rsync's `-rlptgoD` behavior — links, permissions, times,
|
||||
owner, group, devices, and special files — and does not imply compression or
|
||||
multithreading (see [Client](#client)). Ownership application is still
|
||||
privilege-gated: a receiver that cannot `chown` logs a warning and skips it.
|
||||
Under `-p` the source mode is copied exactly, including group/other-write
|
||||
bits; setuid/setgid/sticky bits are copied only when super-user activities are
|
||||
permitted, and are masked under `SUPER_MODE_OFF`/`--no-super` (see
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)).
|
||||
- Symlink transfer stores targets **verbatim** (`-l`/`--links`), including
|
||||
absolute and `..`-bearing targets, matching rsync. The receiver does not
|
||||
enforce a containment predicate by default; `--safe-links` drops unsafe
|
||||
targets on the sender, and `--munge-links` rewrites them with rsync's
|
||||
`/rsyncd-munged/` marker. `--trust-sender` does not affect symlink targets.
|
||||
A destination later consumed by a link-following tool can therefore follow a
|
||||
link outside the receive root — use `--safe-links` for untrusted sources.
|
||||
- Hard links (`-H`/`--hard-links`), extended attributes (`-X`/`--xattrs`), and
|
||||
POSIX ACLs (`-A`/`--acls`) are preserved; owner/group is applied through
|
||||
`-o`/`-g` (or an `-a`/`--archive` transfer), through the opt-in identity flags
|
||||
(`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`), and only when
|
||||
the receiver has permission. See
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md) for the exact semantics and documented
|
||||
divergences.
|
||||
- Device and special-file preservation is implemented with documented
|
||||
divergences: recreated device nodes require `CAP_MKNOD` on the receiver (a
|
||||
non-root receiver skips the entry), while FIFOs **and unix sockets** are
|
||||
recreated (`--specials`).
|
||||
- Sparse-file hole preservation (`-S`, `--sparse`) is implemented receiver-side:
|
||||
long all-zero runs are written as holes (no wire change; the full file image
|
||||
is already in memory).
|
||||
- `--partial`, `--partial-dir`, `-P`, `--append`, and `--append-verify` keep
|
||||
the write atomic (temp + rename). With `--partial`, a failed/interrupted write
|
||||
now retains the already-written temp at the destination path (best-effort) so
|
||||
a later `--append`/`--append-verify` run can resume it.
|
||||
- `-d`/`--dirs` and its aliases `--old-dirs`/`--old-d` transfer the named
|
||||
directory entries without recursing into their contents.
|
||||
- Short-option names are now rsync-parity (Phase 7 Wave A): FastSync's former
|
||||
collisions were renamed (`-j`/`--threads`, `--preserve`, `--sendfile`,
|
||||
`--chunk-serialization`, `--timeout`, `--ssh-port`), so `-m`, `-M`, `-f`,
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync.
|
||||
- Short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached values
|
||||
(`-B1000`, `-essh`, `-MOPT`, `--opt=value`) are accepted, matching rsync.
|
||||
- `-r`, `-b`, `-L`, and `-B` are parsed with the rsync short names.
|
||||
- `--stats` prints the counters FastSync can observe plus the receiver-only
|
||||
counters reported over the wire (`Matched data`, deleted files, and the
|
||||
created/literal counters); `Number of files` and `Number of created files`
|
||||
carry rsync's per-type breakdown. `--progress` prints rsync-style per-file
|
||||
blocks including the leading `./` line, and (when progress is requested) a
|
||||
paths-only pre-count supplies rsync's `to-chk` denominator.
|
||||
- Codecs match rsync 3.4.1: `zstd`/`lz4`/`zlib`/`zlibx` compression and
|
||||
`xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1`/`none` checksums. `auto` honors
|
||||
`RSYNC_COMPRESS_LIST`/`RSYNC_CHECKSUM_LIST` and otherwise follows rsync's
|
||||
compiled-in order. An omitted `--compress-level` uses the codec's rsync
|
||||
default (zstd 3, zlib/zlibx 6, lz4 ignored); `zlib`/`zlibx` share the
|
||||
literal-only zlib path (rsync's zlibx semantics), and the transfer checksum is
|
||||
not separately selectable.
|
||||
- Archive mode does not yet provide all of rsync's `-rlptgoD` behavior.
|
||||
- Symlink transfer is incomplete; link targets are not yet recreated in all
|
||||
modes.
|
||||
- Owner/group, ACL, xattr, hard-link, device, and special-file handling is
|
||||
incomplete or unavailable.
|
||||
- Sparse-file handling does not yet preserve all holes correctly.
|
||||
- `--partial`, `--partial-dir`, `--append`, and `--append-verify` are not yet
|
||||
full rsync-style resumable transfers.
|
||||
- Several rsync short options currently have FastSync-specific meanings. Do
|
||||
not assume every short option is interchangeable yet.
|
||||
|
||||
The detailed flag matrix is maintained in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It reports each row as **parity**,
|
||||
**caveat** (works with a documented divergence), or **divergent** (not
|
||||
supported), rather than treating "parsed" as parity.
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented,
|
||||
partial, alternate, and planned behavior.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Build
|
||||
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
This produces `./build/client` and `./build/server`. `compile_commands.json` is a symlink to `build/compile_commands.json` and is used by clangd/editor tooling; its target is generated by the build, so it dangles until the first build.
|
||||
|
||||
### Client
|
||||
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| Positional | `<source> <dest>` — automatic SSH detection if dest contains `:` |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime (implies the incremental checksum quick-check) |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh128` (default), `xxh3`, `xxh64`/`xxhash`, `md5`, `md4`, `sha1`, `none`, or `auto` (plus rsync's two-name `transfer,pre-transfer` form) |
|
||||
| `-z, --compress [level]` | Enable streaming compression (default `zstd`; level 1–22, default 5) |
|
||||
| `--compress-choice <alg>` | Compression algorithm: `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, or `auto` |
|
||||
| `--skip-compress <list>` | Skip compression for suffixes (`/`- or `,`-separated); defaults to rsync 3.4.1's built-in suffix list |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated (not compression/multithreading) |
|
||||
| `-j, --threads[=N]` | Multithreading mode; `N` (1–256) sets the parallel scanner worker count, bare `-j`/`--threads` uses the default |
|
||||
| `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) |
|
||||
| `-r, --recursive` | Recurse into directories (FastSync is always recursive; accepted for rsync compatibility) |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents; aliases `--old-dirs`/`--old-d` |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root |
|
||||
| `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) |
|
||||
| `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) |
|
||||
| `--sendfile` | Sendfile zero-copy. Incompatible with compression / chunk serialization. TCP only. Long form only. |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk) |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`) |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch) |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing; `--no-whole-file` clears it |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`) |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime |
|
||||
| `--preserve` | Preserve mode and mtime (`-p` + `-t`; add `-o`/`-g` for owner/group or `-U`/`--atimes` for atime; `-N`/`--crtimes` captures birth time but cannot apply it) |
|
||||
| `-U, --atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N, --crtimes` | Capture birth time; cannot be applied (documented divergence) |
|
||||
| `-p, --perms` | Preserve permission bits. The source mode is copied exactly, including group/other-write bits; setuid/setgid/sticky are copied only when super-user activities are permitted (`SUPER_MODE_OFF`/`--no-super` masks them) |
|
||||
| `-t, --times` | Preserve modification times |
|
||||
| `-o, --owner` | Preserve the source owner (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `-g, --group` | Preserve the source group (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group`, `--no-preserve` | Negate the per-attribute flags (short `--no-p`/`--no-t`/`--no-o`/`--no-g`; `--no-preserve` clears all four) |
|
||||
| `-E, --executability` | Preserve executable permission bits |
|
||||
| `-X, --xattrs` | Preserve user `user.*` extended attributes |
|
||||
| `-A, --acls` | Preserve POSIX ACLs |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax) |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership |
|
||||
| `--numeric-ids` | Apply source numeric uid/gid directly instead of mapping by name |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP] (requires a privileged receiver) |
|
||||
| `--fake-super` | Record the resolved owner plus full mode/rdev in rsync's reserved `user.rsync.%stat` xattr (rsync 3.4.1 grammar) and replay the permission bits; never performs a real chown |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes) |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`) |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`) |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer |
|
||||
| `-c [level]` | Compression with optional level (1–22, default 5) |
|
||||
| `-z [level]` | Alias for `-c` |
|
||||
| `-a, --archive` | Archive mode: enables `-c -m -M` (no `-s`) |
|
||||
| `-m` | Multithreading mode |
|
||||
| `-s` | Chunk serialization (batch all files per chunk) |
|
||||
| `-f, --sendfile` | Sendfile zero-copy. Incompatible with `-c` / `-s`. TCP only. |
|
||||
| `-M, --preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) |
|
||||
| `-n, --dry-run` | Scan and print what would be transferred |
|
||||
| `-p <port>` | SSH port (default: 22) |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `--progress` | Show real-time transfer speed |
|
||||
| `--delete` | Delete files on receiver not present in source |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--include-from <file>` | Read include patterns from a file |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `-x, --one-file-system` | Do not cross filesystem boundaries; the mount-point directory entry is emitted (empty at the destination) without descending |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G; `0` = no local limit, matching rsync) |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--ignore-existing` | Skip files that already exist on the receiver; like rsync it does not apply to directories or symlinks. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`; a basis of any size is supported, streamed in bounded chunks — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination (the copy is streamed, so a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win; a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `--verify-basis` | FastSync-only: require a basis hit (`--compare-dest`/`--copy-dest`/`--link-dest`) to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync) |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-during, matching rsync, so destination space is freed progressively). Scoped to the synchronized directories, so `--files-from` subsets are safe |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) |
|
||||
| `--delete-after` | Explicit delete-after timing (implies `--delete`) |
|
||||
| `--delete-commit` | FastSync-only: keep the pre-2.28 atomic timing — delete only after the whole transfer succeeded (identical timing to `--delete-after`) |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication; the fixed `.fastsync-stage` staging name diverges from rsync — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)) |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install; confined to the receive root (a relative path resolves below it; an absolute path is accepted only when it canonicalizes inside it), with an `EXDEV` non-atomic copy fallback |
|
||||
| `-n, --dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created) |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing), including the receiver-only counters reported over the wire; `Number of files` and `Number of created files` carry rsync's per-type breakdown (deleted files are reported as a single total) |
|
||||
| `-i, --itemize-changes` | Print an rsync-style per-file change line |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %c %C %i %M %%`) |
|
||||
| `--list-only` | List source files instead of transferring |
|
||||
| `--fsync` | Fsync every written file before publication |
|
||||
| `-h, --human-readable` | Format transfer byte/rate counts with rsync's decimal (base-1000) units |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `-s`. |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds (default: 30) |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 10) |
|
||||
| `--backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing) |
|
||||
| `--max-depth <n>` | Maximum directory depth to recurse (0 = unlimited, default: 0) |
|
||||
| `--log-file <path>` | Write log messages to file instead of stderr |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree (FastSync-native format, not rsync-interoperable) |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server); FastSync-native format, not rsync-interoperable |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server); FastSync-native format, not rsync-interoperable |
|
||||
| `--source-dir <path>` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) |
|
||||
| `--dest-dir <path>` | Server destination directory (overrides `FASTSYNC_DEST_DIR`) |
|
||||
| `--save-to-disk` | Write received files to disk |
|
||||
| `--server-host <ip>` | Server IP address (default: `127.0.0.1`) |
|
||||
| `--server-port <n>` | Server port (default: `8080`) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-e, --rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments, e.g. `-e "ssh -p 2222"`) |
|
||||
| `-M, --remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable) |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address |
|
||||
| `-4, --ipv4` | Force IPv4 for destination resolution |
|
||||
| `-6, --ipv6` | Force IPv6 for destination resolution |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect (`TCP_NODELAY`, `SO_KEEPALIVE`, `SO_RCVBUF`, `SO_SNDBUF`, `SO_REUSEADDR`) |
|
||||
| `--bwlimit <RATE>` | Bandwidth limit, using rsync's exact `parse_size_arg` grammar: a bare value is KiB/s; `K`/`M`/`G`/`T`/`P` are binary suffixes; `KB`/`MB` are decimal and `KiB`/`MiB` binary; decimals are accepted and quantized to whole KiB; `0` (or empty) means no limit. Also paces `--sendfile` transfers |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline. Default `0` = disabled (matching rsync); `0` disables it. `--no-timeout` is the negation. The value is not sent on the wire; the server side keeps its own safe floor. |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 60, matching rsync); `0` disables it (`--no-contimeout` is the negation) |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes (a positive integer); whatever was already transferred is kept |
|
||||
| `--stop-at=TIME` | Stop at an absolute time. Accepts rsync's `parse_time` forms (`Y-M-DTh:m`, `Y/M/DTh:m`, `Y-M-D`, `M-D`, `D`, `h:m`, `:m`, `T h:m`; omitted fields resolve to the next matching point in the local timezone), plus `now+N[smhd]` and FastSync's `HH:MM`/`HH:MM:SS` clock-time spelling. An early stop skips the late `--delete` keep-set |
|
||||
| `-b, --backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--tls` | Enable TLS encryption |
|
||||
| `--cert <path>` | TLS certificate file (PEM) |
|
||||
| `--key <path>` | TLS private key file (PEM) |
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
|
||||
The exhaustive rsync flag matrix is in [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
**Per-message vs. connection timeouts.** `--timeout` bounds each individual protocol
|
||||
send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It
|
||||
does not, by itself, stop a peer that keeps sending well-formed frames forever. The
|
||||
receiver therefore also enforces two wall-clock (`CLOCK_MONOTONIC`) bounds on a
|
||||
connection: a **1 hour** idle limit and a **24 hour** overall session cap. Only
|
||||
frames that move real work (not `STATUS_KEEPALIVE`/`STATUS_ABORT` and not an
|
||||
empty `STATUS_CHECK_BATCH`/`STATUS_DIR_TIMES`) refresh the idle timestamp, so a
|
||||
peer cannot hold a connection slot by emitting cheap empty frames; a peer that
|
||||
fabricates minimal non-empty frames can still occupy a slot until the 24 hour
|
||||
cap, since no bound can require actual payload without risking a legitimate
|
||||
long operation. Both are deliberately generous so a legitimate long-running
|
||||
transfer is never aborted.
|
||||
| `--client-cn <name>` | Required TLS client certificate common name |
|
||||
|
||||
### Server
|
||||
|
||||
@@ -288,8 +134,7 @@ transfer is never aborted.
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--destination-root <path>` | Authorized destination root (default: `.`) |
|
||||
| `--allow-delete` | Permit manifest deletion |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. **Rejected with `--stdio`** (the SSH remote argv is client-composed, so a client could otherwise pass it and defeat the secure default; operators exposing `fastsync-server --stdio` over SSH must use a forced command if the default must hold). No effect when not root. |
|
||||
| `--allow-unauthenticated` | Permit plaintext TCP clients. For an `auth users` module this opts in **loopback plaintext only**; remote auth still requires verified TLS, so the flag never permits remote plaintext auth. |
|
||||
| `--allow-unauthenticated` | Permit plaintext TCP clients |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `--help` | Show help |
|
||||
|
||||
@@ -300,70 +145,60 @@ transfer is never aborted.
|
||||
| `FASTSYNC_SOURCE_DIR` | — | Source directory fallback |
|
||||
| `FASTSYNC_DEST_DIR` | — | Destination directory fallback |
|
||||
| `FASTSYNC_SAVE_TO_DISK` | `false` | Disk persistence fallback |
|
||||
| `FASTSYNC_MAX_WHOLE_FILE_SIZE` | `268435456` | Receiver-only test hook: a byte count that lowers the whole-file streaming bound. Payloads above it are streamed through a bounded buffer. Values are clamped to the 256 MiB protocol ceiling, so it can only lower, never raise, the bound. |
|
||||
| `FASTSYNC_SSH_PORT` | `22` | Default SSH port |
|
||||
| `FASTSYNC_SERVER_HOST` | `127.0.0.1` | Default server host |
|
||||
| `FASTSYNC_SERVER_PORT` | `8080` | Default server port |
|
||||
| `FASTSYNC_TLS_CERT` | — | Default TLS certificate path |
|
||||
| `FASTSYNC_TLS_KEY` | — | Default TLS private key path |
|
||||
| `FASTSYNC_TLS_CA` | — | Default TLS CA certificate path |
|
||||
|
||||
## Implementation Details
|
||||
|
||||
### Data Structures
|
||||
|
||||
1. **Chunk** — collection of files (~10 MB total by default).
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer.
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec` (plus
|
||||
atime/crtime fields). `uid`/`gid` are applied only through the opt-in
|
||||
identity path; atime is preserved with `-U`/`--atimes`; crtime is captured
|
||||
but cannot be set on the destination.
|
||||
4. **Config** — runtime parameters. Most cross the wire (TLS settings
|
||||
excluded); `backup` and `backup_dir` are in the serialized wire table, while
|
||||
`timeout`, `contimeout`, `quiet`, `stats`, `max_depth`, and `log_file` are
|
||||
client-only.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables.
|
||||
6. **DirectoryScanner** — recursive traversal that buffers and sorts each
|
||||
directory (non-directories ascending, then directories ascending) and walks
|
||||
depth-first in rsync flist order, with exclude and include pattern support and
|
||||
max-depth enforcement.
|
||||
1. **Chunk** — collection of files (~10 MB total by default)
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`;
|
||||
uid / gid are advisory wire fields and are never applied by the receiver;
|
||||
atime is unsupported
|
||||
4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`, `queue_size`.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement
|
||||
|
||||
### Key Algorithms
|
||||
|
||||
1. **File scanning** — sorted depth-first traversal in rsync flist order (each
|
||||
directory's non-directories ascending, then its directories ascending);
|
||||
entries matched against exclude and include patterns, with max-depth
|
||||
enforced. The `--threads` parallel scanner remains unordered.
|
||||
2. **Chunking** — files accumulated until the `chunk_size` threshold (default
|
||||
10 MiB) is reached, then flushed.
|
||||
3. **Compression** — streaming zstd via `ZSTD_compressStream2()` /
|
||||
`ZSTD_decompressStream()`, with lz4 and zlib/zlibx codecs also supported
|
||||
(selectable with `--compress-choice`).
|
||||
4. **Network protocol** — status-code-driven exchange with metadata packing,
|
||||
keep-alive, and abort support.
|
||||
5. **Incremental check** — the client sends `STATUS_CHECK` + path + size +
|
||||
mtime and, with `--checksum`, a whole-file content checksum (`xxh128` by
|
||||
default; selectable via `--checksum-choice`/`--cc`, seeded by
|
||||
`--checksum-seed`); the server compares against the destination. Can be
|
||||
batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with sleep throttling on
|
||||
64 KiB write chunks.
|
||||
7. **Metadata restoration** — mode via `chmod()`/`fchmod()`, times via
|
||||
`utimensat()`/`futimens()`, and ownership only with an identity flag via
|
||||
fd-relative `fchown()`/`fchownat()`.
|
||||
8. **`--delete`** — the sender tracks all sent paths; the receiver walks the
|
||||
destination tree and removes unlisted files and directories.
|
||||
9. **SSH transport** — `socketpair()` + `fork()` + `execvp("ssh", ...)` with
|
||||
`ControlMaster` and port support.
|
||||
10. **TLS transport** — OpenSSL `SSL_CTX` with TLS 1.2 minimum, mutual CA
|
||||
verification, and transparent `SSL_read()`/`SSL_write()` via
|
||||
`io_set_ssl()`.
|
||||
11. **Path traversal protection** — `has_path_traversal()` rejects any file
|
||||
path containing `..` components, preventing directory escape attacks.
|
||||
12. **Connection limiting** — the server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100).
|
||||
13. **Keep-alive** — idle connections receive periodic `STATUS_KEEPALIVE` to
|
||||
detect half-open TCP connections.
|
||||
14. **Abort handling** — `SIGINT` sets an abort flag; the next protocol
|
||||
operation sends `STATUS_ABORT` for clean server cleanup.
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically
|
||||
renamed via `rename()`, preventing partial files.
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir`
|
||||
(or the same directory with a `~` suffix), preserving the original.
|
||||
1. **File scanning** — BFS directory traversal;
|
||||
entries matched against exclude and include patterns,
|
||||
max - depth enforced 2. * *Chunking ** — files accumulated until `chunk_size` threshold,
|
||||
then flushed 3. *
|
||||
*Compression ** — streaming zstd
|
||||
via `ZSTD_compressStream2` / `ZSTD_decompressStream` 4. *
|
||||
*Network protocol ** — status -
|
||||
code - driven exchange with metadata packing,
|
||||
keep - alive,
|
||||
and abort support 5. * *Incremental check ** — client sends `STATUS_CHECK` + path + size +
|
||||
mtime and,
|
||||
with `--checksum`, XXH64 content checksum; server compares against destination. Can be batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with `nanosleep` throttling on 64 KB write chunks
|
||||
7. **Metadata restoration** — `chmod()`, `chown()`, `utimensat()` on the receiving side
|
||||
8. **`--delete`** — sender tracks all sent paths;
|
||||
receiver walks destination tree and removes unlisted files / directories 9. *
|
||||
*SSH transport *
|
||||
* — `socketpair()` + `fork()` + `execvp("ssh",
|
||||
...)` with `ControlMaster` and port support
|
||||
10. *
|
||||
*TLS transport ** — OpenSSL `SSL_CTX` with TLS
|
||||
1.2 minimum,
|
||||
mutual CA verification,
|
||||
transparent `SSL_read`/`SSL_write` via `io_set_ssl()` 11. *
|
||||
*Path traversal protection ** — `has_path_traversal()` rejects any file path
|
||||
containing `..` components,
|
||||
preventing directory escape attacks 12. *
|
||||
*Connection limiting ** — server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100)13. *
|
||||
*Keep
|
||||
- alive ** — idle connections receive periodic `STATUS_KEEPALIVE` to detect half
|
||||
- open TCP connections 14. * *Abort handling ** — `SIGINT` sets an abort flag; the next protocol operation sends `STATUS_ABORT` for clean server cleanup
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically renamed via `rename()`, preventing partial files
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir` (or same directory with `~` suffix) preserving the original
|
||||
|
||||
## Security Features
|
||||
|
||||
@@ -387,8 +222,6 @@ Received files are written to a temporary path (suffixed with `.tmp`) and then a
|
||||
- C11 compiler
|
||||
- CMake >= 3.22
|
||||
- zstd library
|
||||
- zlib library
|
||||
- lz4 library
|
||||
- OpenSSL (development headers and libraries)
|
||||
- pthreads
|
||||
- SSH client (for SSH transport mode only)
|
||||
@@ -397,12 +230,12 @@ Received files are written to a temporary path (suffixed with `.tmp`) and then a
|
||||
|
||||
**Ubuntu/Debian:**
|
||||
```bash
|
||||
sudo apt install cmake build-essential libzstd-dev zlib1g-dev liblz4-dev libssl-dev openssh-client
|
||||
sudo apt install cmake build-essential libzstd-dev libssl-dev openssh-client
|
||||
```
|
||||
|
||||
**Nix:**
|
||||
```bash
|
||||
nix-shell # provides zstd, zlib, lz4, openssl, cmake, gcc
|
||||
nix-shell # provides zstd, openssl, cmake, gcc
|
||||
```
|
||||
|
||||
## Building
|
||||
@@ -422,34 +255,22 @@ cmake --build build -j$(nproc)
|
||||
|
||||
### SSH transfer
|
||||
|
||||
The remote host must have `fastsync-server` available in `PATH` (install or
|
||||
copy the built `./build/server` there as `fastsync-server`), or use
|
||||
The remote host must have `fastsync-server` available in `PATH`, or use
|
||||
`--fastsync-server-path`. SSH starts `fastsync-server --stdio` in its remote
|
||||
working directory, so use a destination below that directory unless the
|
||||
remote server is otherwise configured with a matching authorized root.
|
||||
|
||||
The remote `--stdio` server argv is composed by the client, so it must never
|
||||
be trusted to opt a root receiver into super-user activities: `--allow-super`
|
||||
is rejected with `--stdio` and super stays off on that path. Operators
|
||||
exposing `fastsync-server --stdio` over SSH must use a forced command (e.g. an
|
||||
`authorized_keys` `command=` entry) if the default must hold.
|
||||
|
||||
```bash
|
||||
ssh user@host 'mkdir -p destination'
|
||||
./build/client /path/to/source user@host:destination
|
||||
```
|
||||
|
||||
FastSync is **push-only**: the source (first argument) is always a local
|
||||
directory and only the destination may be remote. A remote source such as
|
||||
`client user@host:src ./local` (a "pull") is intentionally not supported; see
|
||||
[RSYNC_COMPAT.md](RSYNC_COMPAT.md#direction).
|
||||
|
||||
### TCP transfer
|
||||
|
||||
Start the FastSync server:
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to -p 8080 --allow-unauthenticated
|
||||
./build/server --destination-root /path/to -p 8080
|
||||
```
|
||||
|
||||
Then run the client:
|
||||
@@ -464,13 +285,8 @@ Plain TCP requires the explicit `--allow-unauthenticated` server option. Use TLS
|
||||
authenticated network connections.
|
||||
|
||||
### TLS transfer
|
||||
|
||||
Server TLS requires `--cert`, `--key`, `--ca`, and `--client-cn`; the client
|
||||
requires `--cert`, `--key`, and `--ca`.
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem \
|
||||
--ca ca.pem --client-cn client -p 8443
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem -p 8443
|
||||
./build/client --tls --cert client.pem --key client-key.pem --ca ca.pem \
|
||||
--server-host example.com --server-port 8443 \
|
||||
--source-dir /path/to/source --dest-dir /path/to/destination \
|
||||
@@ -506,7 +322,7 @@ FastSync-native are optional performance or transport extensions.
|
||||
./build/client --incremental --checksum /source/ user@host:destination/
|
||||
|
||||
#Preserve supported mode and timestamp metadata
|
||||
./build/client --preserve /source/ user@host:destination/
|
||||
./build/client -M /source/ user@host:destination/
|
||||
|
||||
#Keep backups of overwritten destination files
|
||||
./build/client --backup --backup-dir backups \
|
||||
@@ -520,41 +336,28 @@ features without changing the meaning of ordinary compatibility options.
|
||||
|
||||
| Option | Purpose |
|
||||
|---|---|
|
||||
| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming compression (default `zstd`), levels 1-22. |
|
||||
| `--compress-level <n>` | Set the compression level (1-22). Omitted, each codec uses its rsync default: zstd 3, zlib/zlibx 6, lz4 ignored. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, and `auto`; `zlib`/`zlibx` share the same literal-only zlib path. |
|
||||
| `--zl <n>` | Alias for `--compress-level`. |
|
||||
| `--skip-compress <list>` | Skip compression for `/`- or `,`-separated suffixes; defaults to rsync 3.4.1's built-in list. Incompatible with `--chunk-serialization`. |
|
||||
| `--compress-threads <n>` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. |
|
||||
| `-m` | Enable the multithreaded scanner/loader/sender pipeline. |
|
||||
| `-c [level]`, `-z [level]` | Enable streaming zstd compression, levels 1-22. |
|
||||
| `--compress-level <n>` | Set the zstd compression level. |
|
||||
| `--chunk-size <bytes>` | Set the transfer chunk size. |
|
||||
| `--chunk-serialization` | Enable FastSync chunk serialization (long form only; `-s` is rsync's `--secluded-args`). |
|
||||
| `--sendfile` | Use TCP `sendfile()` zero-copy transfer. Incompatible with compression and chunk serialization. Long form only. |
|
||||
| `-s` | Enable FastSync chunk serialization. |
|
||||
| `-f`, `--sendfile` | Use TCP `sendfile()` zero-copy transfer. Incompatible with compression and chunk serialization. |
|
||||
| `--delta` | Use FastSync-native block delta transfer. Requires `--incremental`. |
|
||||
| `--delta-block <bytes>` | Set the FastSync delta block size (`--block-size` is an alias). |
|
||||
| `--delta-block <bytes>` | Set the FastSync delta block size. |
|
||||
| `--delta-max <bytes>` | Limit files eligible for FastSync delta transfer. |
|
||||
| `--server-host <host>` | Select the TCP server host. |
|
||||
| `--server-port <port>` | Select the TCP server port (`--port <port>` and `--port=<port>` are rsync-friendly aliases). |
|
||||
| `--server-port <port>` | Select the TCP server port. |
|
||||
| `--tls` | Enable TLS for TCP transport. |
|
||||
| `--bwlimit <RATE>` | Apply token-bucket bandwidth limiting with rsync's exact `parse_size_arg` grammar (bare = KiB/s, `K`/`M`/`G`/`T`/`P` binary, `KB`/`MB` decimal, `KiB`/`MiB` binary, decimals quantized to whole KiB, `0`/empty = no limit; also paces `--sendfile` transfers). |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; `Number of files`/`Number of created files` carry rsync's per-type breakdown (deleted files are a single total). |
|
||||
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout. Default `0` = disabled (matching rsync); `0` disables it. |
|
||||
| `--contimeout <seconds>` | Connection timeout (default 60, matching rsync); `0` disables it. |
|
||||
| `--bwlimit <KB/s>` | Apply token-bucket bandwidth limiting. |
|
||||
| `--progress` | Show transfer progress and throughput. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `--timeout <seconds>` | Set I/O timeout. |
|
||||
| `--contimeout <seconds>` | Set connection timeout. |
|
||||
|
||||
Short-option conflicts with rsync have been resolved for the CLI namespace
|
||||
(Phase 7): `-c` is now rsync's `--checksum`, `-m` is `--prune-empty-dirs`, `-M`
|
||||
is `--remote-option`, `-f` is `--filter`, `-s` is `--secluded-args`, `-p` is
|
||||
`--perms`, and `-T` is `--temp-dir`. FastSync's own flags were renamed to
|
||||
long-form-only or new shorts: multithreading is `-j`/`--threads`, metadata
|
||||
is `--preserve`, sendfile is `--sendfile`, chunk serialization is
|
||||
`--chunk-serialization`, timeout is `--timeout`, and SSH port is `--ssh-port`.
|
||||
`-a`/`--archive` is now rsync archive `-rlptgoD` (owner/group implied, but the
|
||||
receiver still needs privilege to apply them).
|
||||
|
||||
`--secluded-args` (and its short form `-s`) is accepted as a compatibility
|
||||
no-op. It does not change FastSync's transport or protocol behavior, because
|
||||
remote SSH argv is already built injection-safe.
|
||||
Current short-option conflicts are tracked as compatibility work. In
|
||||
particular, FastSync currently uses `-p` for SSH port, `-s` for chunk
|
||||
serialization, and `-S` for sparse handling. These meanings must be reconciled
|
||||
before FastSync can claim full rsync CLI compatibility.
|
||||
|
||||
## Client Options
|
||||
|
||||
@@ -562,122 +365,47 @@ remote SSH argv is already built injection-safe.
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated. |
|
||||
| `-n`, `--dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer. |
|
||||
| `--incremental` | Skip files matching destination size and mtime. Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `-c, --checksum` | Verify content by checksum (implies the incremental quick-check). Algorithm selectable with `--checksum-choice`. |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh64`/`xxhash` (default), `xxh3`, `xxh128`, `md5`, or `auto`. |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync. |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime. |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match. |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver. |
|
||||
| `--ignore-existing` | Skip files that already exist on the receiver; like rsync it does not apply to directories or symlinks. |
|
||||
| `-@, --modify-window <sec>` | Modification-time tolerance (seconds) for the incremental/basis quick-check; `0` requires an exact mtime match. |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing (`--no-whole-file` clears it). |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`). |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents (aliases `--old-dirs`/`--old-d`). |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root. |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root). |
|
||||
| `-0, --from0` | Treat entries in `--files-from` files as NUL-delimited instead of newline-delimited. |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer (the fixed `.fastsync-stage` staging name diverges from rsync; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`; a basis of any size is supported, streamed in bounded chunks — see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination (the copy is streamed, so a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win; a basis of any size works; see [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)). |
|
||||
| `--verify-basis` | FastSync-only: require a basis hit to match the source by whole-file digest instead of trusting the size+mtime quick-check (default matches rsync). |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk). |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`). |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch). |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-during (matching rsync's `--del`): extras are removed per directory as the transfer proceeds, so destination space is freed progressively. Scoped to the synchronized directories, so `--files-from` subsets are safe. |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). |
|
||||
| `--delete-during`, `--del` | Delete each directory's extras as that directory is processed (implies `--delete`). Since protocol 2.24.0 the sender streams a per-directory `STATUS_DELETE_PLAN` frame as it reaches each source directory; this is also the default timing of a plain `--delete`. |
|
||||
| `--delete-delay` | Record extras per directory during the scan but remove them only after a successful transfer (implies `--delete`). Uses the same per-directory `STATUS_DELETE_PLAN` frames as `--delete-during`, applied late. |
|
||||
| `--delete-commit` | FastSync-only: atomic delete-after timing (only after the whole transfer succeeded). |
|
||||
| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected). |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync. |
|
||||
| `--force` | Allow an incoming file/symlink to replace a destination directory (also during `--delay-updates` publication). |
|
||||
| `-a`, `--archive` | Enable current archive preset. Full rsync archive semantics are planned. |
|
||||
| `-n`, `--dry-run` | Scan and report without writing files. |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. |
|
||||
| `--exclude <pattern>` | Exclude matching paths. Repeatable. |
|
||||
| `--include <pattern>` | Include matching paths. Repeatable. |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file. |
|
||||
| `--include-from <file>` | Read include patterns from a file. |
|
||||
| `-f, --filter=RULE` | Add an rsync-style filter rule (`+`/`-`, `include`/`exclude`, `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R`, `clear`/`!`, and modifiers; repeatable). |
|
||||
| `--max-size <bytes>` | Skip files larger than the limit. |
|
||||
| `--min-size <bytes>` | Skip files smaller than the limit. |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units; default 1G; `0` = no local limit). |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth; zero means unlimited. |
|
||||
| `-b, --backup` | Back up overwritten files. |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install (confined to the receive root: relative resolves below it, absolute must canonicalize inside it; `EXDEV` falls back to a non-atomic copy). |
|
||||
| `--backup-dir <dir>` | Store backups under a separate directory (requires `--backup`). |
|
||||
| `--suffix <suffix>` | Set the backup filename suffix (default: `~`). |
|
||||
| `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir <dir>`, completed files are written under the partial directory and installed atomically. |
|
||||
| `--partial-dir <dir>` | Set a relative partial-transfer directory below the server destination root. Implies `--partial`. Rejected together with `--inplace` (`--inplace cannot be used with --partial-dir`, matching rsync), because the inplace path bypasses partial/temp staging. |
|
||||
| `--inplace` | Write directly to the destination instead of using a temporary file. Cannot be combined with `--partial-dir`. |
|
||||
| `--fsync` | Fsync every written file before publication. |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree (FastSync-native format, not rsync-interoperable). |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server); FastSync-native format, not rsync-interoperable. |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server); FastSync-native format, not rsync-interoperable. |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes; whatever was already transferred is kept. |
|
||||
| `--stop-at=TIME` | Stop at an absolute time. Accepts rsync's `parse_time` forms (`Y-M-DTh:m`, `Y/M/DTh:m`, `Y-M-D`, `M-D`, `D`, `h:m`, `:m`, `T h:m`; omitted fields resolve to the next matching point in the local timezone), plus `now+N[smhd]` and FastSync's `HH:MM`/`HH:MM:SS` clock-time spelling. An early stop skips the late `--delete` keep-set. |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth;
|
||||
zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.|
|
||||
| `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` |
|
||||
Back up overwritten files.| | `--backup - dir<dir>` | Store backups under a separate directory.|
|
||||
| `--suffix<suffix>` | Set the backup filename suffix.| | `--partial` |
|
||||
Select partial - transfer handling.With `--partial - dir`,
|
||||
completed files are written there;
|
||||
resumable transfers are not implemented.| | `--partial - dir<dir>` |
|
||||
Set a relative partial - transfer directory below the server destination root;
|
||||
use with `--partial`. |
|
||||
| `--inplace` | Write directly to the destination instead of using a temporary file. |
|
||||
|
||||
### Metadata and links
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--preserve` | Preserve mode and mtime (long form only; equivalent to `-p` + `-t`). Add `-o`/`-g` for owner/group, `-U`/`--atimes` for atime, or an identity flag (`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`) for mapped ownership. |
|
||||
| `-U`, `--atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N`, `--crtimes` | Capture birth time and transmit it; it cannot be applied because no portable filesystem call can set a birth time (documented divergence). |
|
||||
| `-p`, `--perms` | Preserve permission bits. One of the four per-attribute preserve flags (with `-t`/`-o`/`-g`); under `-p` the source mode is copied exactly (group/other-write included; setuid/setgid/sticky included only when super-user activities are permitted, masked under `SUPER_MODE_OFF`/`--no-super`), matching rsync otherwise. |
|
||||
| `-t`, `--times` | Preserve modification times. Independent of the other attributes; `-O`/`--omit-dir-times` suppresses directories only. |
|
||||
| `-O`, `--omit-dir-times` | Do not apply modification times to directories. |
|
||||
| `-J`, `--omit-link-times` | Do not apply times to symlinks. |
|
||||
| `--open-noatime` | Open source files with `O_NOATIME` so reading for a transfer does not update their access time (client-only). |
|
||||
| `-o`, `--owner` | Preserve the source owner (uid). Mapped by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); application is privilege-gated. |
|
||||
| `-g`, `--group` | Preserve the source group (gid). Same name-mapping/numeric-fallback and privilege gating as `-o`. |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group` | Negate each per-attribute flag (also `--no-p`/`--no-t`/`--no-o`/`--no-g`); `--no-preserve` clears all four. |
|
||||
| `-E`, `--executability` | Preserve executable permission bits. |
|
||||
| `-X`, `--xattrs` | Preserve user `user.*` extended attributes. |
|
||||
| `-A`, `--acls` | Preserve POSIX ACLs. |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax, including `D`/`F`/`X` selectors and `s`/`t`); does not imply `-p`. |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files (`USER:GROUP`, `USER`, or `:GROUP`); conflicts with `--usermap`/`--groupmap` on the same side. |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership (`FROM:TO` rules; names, ids, `LOW-HIGH` ranges, `*`, empty-`FROM`). |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership (same syntax as `--usermap`). |
|
||||
| `--numeric-ids` | Mapping modifier: apply the source numeric uid/gid directly instead of mapping by name (combine with `-o`/`-g`, `-a`, or a map). |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP]; requires a privileged receiver. |
|
||||
| `--fake-super` | Record the resolved owner plus full mode/rdev in rsync's reserved `user.rsync.%stat` xattr (rsync 3.4.1 grammar) and replay the permission bits; never performs a real chown. |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes). |
|
||||
| `--no-super` | Forbid those super-user activities even when the receiver is root. |
|
||||
| `-l`, `--links` | Copy symlinks as symlinks; the target is stored verbatim (absolute and `..`-bearing targets included), matching rsync. |
|
||||
| `-L`, `--copy-links` | Copy symlink referents (a broken referent makes the run exit 23, matching rsync). |
|
||||
| `--safe-links` | Skip symlinks whose target points outside the transfer tree (applied on the sender). |
|
||||
| `-M`, `--preserve` | Preserve supported file metadata, currently mode and modification time. |
|
||||
| `-l`, `--links` | Request symlink preservation;
|
||||
link-target transfer remains incomplete. |
|
||||
| `--copy-links` | Copy symlink referents. |
|
||||
| `--safe-links` | Skip symlinks that point outside the transfer tree. |
|
||||
| `--copy-unsafe-links` | Copy unsafe symlink referents. |
|
||||
| `--munge-links` | Rewrite stored symlink targets with rsync's `/rsyncd-munged/` marker. |
|
||||
| `-k`, `--copy-dirlinks` | Treat a symlink to a directory as a real directory on the sender. |
|
||||
| `-K`, `--keep-dirlinks` | Follow an existing destination symlink-to-directory (confined to the receive root). |
|
||||
| `-H`, `--hard-links` | Preserve hard-link relationships across the transfer. |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`). |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`). |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets. |
|
||||
| `--copy-devices` | Copy a source device's content as an ordinary regular file on the destination (rsync's non-privileged safe mode) instead of recreating the device node. |
|
||||
| `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). |
|
||||
| `-S`, `--sparse` | Request sparse-file handling; full hole preservation is planned. |
|
||||
|
||||
### Output and logging
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `-q`, `--quiet` | Suppress non-error output. |
|
||||
| `--progress` | Show rsync-style per-file progress blocks; the root `./` line is printed whenever progress is active (rsync prints it only when the transfer root is created). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; `Number of files`/`Number of created files` carry rsync's per-type breakdown (deleted files are a single total). |
|
||||
| `-i, --itemize-changes` | Print an rsync-style per-file change line. |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %c %C %i %M %%`). |
|
||||
| `--list-only` | List source files instead of transferring. |
|
||||
| `--outbuf=MODE` | stdout/stderr buffering: `N` (none/unbuffered), `L` (line-buffered), or `B` (block-buffered, default). |
|
||||
| `--progress` | Show live transfer progress. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `--log-file <path>` | Write log output to a file. |
|
||||
| `--log-file-format=FORMAT` | Per-file log-line format (requires `--log-file`). |
|
||||
| `--stderr=MODE` | Route logging: `errors` (default), `all`, or `client` (forward the client's diagnostics to the server's stderr over the client-message channel). |
|
||||
| `--msgs2stderr` | Route all messages to stderr (deprecated spelling of `--stderr=all`). |
|
||||
| `--no-msgs2stderr` | Forward the client's diagnostics to the server (deprecated spelling of `--stderr=client`). |
|
||||
| `-V`, `--version` | Print the FastSync protocol version. |
|
||||
| `--help` | Print command usage. |
|
||||
|
||||
@@ -685,130 +413,34 @@ remote SSH argv is already built injection-safe.
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--ssh-port <port>` | SSH port for the SSH transport (default: 22). Note the short `-p` is now rsync's `--perms`. |
|
||||
| `-e`, `--rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments). |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode (client-only; never crosses the wire). |
|
||||
| `--rsync-path <path>` | Alias for `--fastsync-server-path`. |
|
||||
| `-M`, `--remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable; rejected for daemon/TCP destinations). |
|
||||
| `--trust-sender` | Receiver-local: trust the remote sender's file list and skip path re-validation (does not affect symlink targets). **On the client this flag alone is inert** — it is never sent on the wire; the server must be started with its own `--trust-sender`, or the client must forward it with `-M--trust-sender` (SSH only). |
|
||||
| `--timeout <sec>` | Socket + per-message I/O timeout; default `0` = disabled. |
|
||||
| `--contimeout <sec>` | Connection timeout; default 60; `0` disables. |
|
||||
| `-p <port>` | SSH port in the current CLI. This conflicts with rsync's `-p` permissions option and is planned for correction. |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode. |
|
||||
| `--source-dir <path>` | Set the source directory explicitly. |
|
||||
| `--dest-dir <path>` | Set the destination directory explicitly. |
|
||||
| `--save-to-disk` | Enable server-side disk persistence. |
|
||||
| `--server-host <host>` | TCP server address. |
|
||||
| `--server-port <port>` | TCP server port. `--port <port>` / `--port=<port>` is an alias. |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address. |
|
||||
| `-4`, `--ipv4` | Force IPv4 for destination resolution. |
|
||||
| `-6`, `--ipv6` | Force IPv6 for destination resolution. |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect. |
|
||||
| `--blocking-io` | SSH transport only: leave the socket without read/write timeouts so it blocks naturally (no effect on TCP). |
|
||||
| `--protocol=NUM` | Force the wire protocol version; must equal the current `PROTOCOL_VERSION` (FastSync cannot speak older/virtual wire formats). |
|
||||
| `--old-args` | Accepted for rsync CLI compatibility; no effect (the remote server path is always safely quoted). |
|
||||
| `--iconv=LOCAL[,REMOTE]` | Convert file-name charsets at the wire boundary (`LOCAL` is our names' charset, `REMOTE` the peer's, defaulting to `LOCAL`). |
|
||||
| `--no-iconv` | Disable `--iconv` charset conversion (same as `--iconv=-`). |
|
||||
| `--tls` | Enable TLS. Requires `--cert`, `--key`, and `--ca`. |
|
||||
| `--server-port <port>` | TCP server port. |
|
||||
| `--tls` | Enable TLS. Requires `--cert` and `--key`. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification (always required with `--tls`). |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
|
||||
## Server Options
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--stdio` | Serve one SSH connection over standard input/output. |
|
||||
| `--daemon` | Run as a persistent daemon listener using a module config file; the daemon default port is 873 (unlike `-p`, which defaults to 8080). |
|
||||
| `--config=FILE` | Daemon config file (default: `~/.config/fastsync/fastsyncd.conf`, else `/etc/fastsyncd.conf`). Requires `--daemon`. |
|
||||
| `--dparam=KEY=VALUE` | Override one global config key on the command line. Requires `--daemon`. |
|
||||
| `--no-detach` | Stay in the foreground (default detaches to the background when running `--daemon`). |
|
||||
| `-p, --port <port>` | TCP listen port (default: 8080, range: 1–65535). |
|
||||
| `-p <port>` | TCP listen port. |
|
||||
| `--tls` | Enable TLS. |
|
||||
| `--cert <path>` | TLS certificate file (PEM). |
|
||||
| `--key <path>` | TLS private key file (PEM). |
|
||||
| `--ca <path>` | CA file for peer verification (PEM). |
|
||||
| `--client-cn <name>` | TLS client certificate CN; mandatory with `--tls` (the server verifies the client CN). |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root; defaults to the current directory. |
|
||||
| `--address <addr>` | Bind the listening socket to this address. |
|
||||
| `-4`, `--ipv4` | Bind an IPv4 socket (default). |
|
||||
| `-6`, `--ipv6` | Bind an IPv6 socket. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. This also gates `--force` (which can recursively replace/remove a destination directory tree). |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. Rejected with `--stdio` (the SSH remote argv is client-composed; use a forced command if the default must hold). No effect when not root. Daemon modules opt in per module with `client owner = yes`. |
|
||||
| `--trust-sender` | Trust the remote sender's file list: skip the receiver's up-front path-traversal re-validation (fewer checks, faster, potentially unsafe; off by default). It does not affect symlink targets, which are stored verbatim either way. A client `--trust-sender` is never sent over the wire — the server must set this flag itself, or the client must forward it via `-M--trust-sender`. |
|
||||
| `--no-super` | Operator veto: never attempt super-user activities (ownership, device nodes) even as root, and refuse any client `--copy-as`/`--super` request. |
|
||||
| `--allow-unauthenticated` | Permit plaintext/anonymous network clients; an auth-required module still accepts only opted-in loopback plaintext. |
|
||||
| `--iconv=LOCAL[,REMOTE]` | Declare this server's LOCAL charset for file-name conversion. |
|
||||
| `--password-file=FILE` | Credential store for modules that declare `auth users`. Requires `--daemon`. FastSync-native SCRAM/PBKDF2 format, not rsync-interoperable. |
|
||||
| `--early-input=FILE` | Second credential store layered over `--password-file`. Requires `--daemon`. FastSync-native format, not rsync-interoperable. |
|
||||
| `--hash-credentials <file>` | Read `<file>`'s `user:password` lines and print PBKDF2 credential-store lines to stdout, then exit. Cannot be combined with `--daemon` or `--stdio`. FastSync-native, not rsync-interoperable. |
|
||||
| `--iterations N` | PBKDF2 iteration count for `--hash-credentials` (default 600000, range 100000–10000000). Requires `--hash-credentials`. FastSync-native, not rsync-interoperable. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root;
|
||||
defaults to the current directory. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. |
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--help` | Print server usage. |
|
||||
|
||||
### Daemon configuration
|
||||
|
||||
`fastsync-server --daemon --config FILE` reads a line-based module config (an
|
||||
implicit global section, then `[module]` sections). Besides `port`, `motd file`,
|
||||
and `address`, the global section accepts:
|
||||
|
||||
- `max connections = N` — global cap on concurrent connections, default 100. The
|
||||
listener enforces it; `0`, negative, and non-numeric values are parse errors.
|
||||
- `max connections per host = N` — cap on concurrent connections from a single
|
||||
source IP, default 0 (unlimited). Enforced across all forked connection
|
||||
children through a shared registry.
|
||||
- `auth failure delay = MS` — milliseconds to sleep after a failed
|
||||
authentication, default 500. `0` disables it and the value is capped at 5000,
|
||||
so online password guessing is rate-limited per connection. Successful auths
|
||||
are never delayed.
|
||||
- `auth lockout threshold = N` — number of failed authentications from one source
|
||||
IP before that source is locked out, default 10; `0` disables the lockout. The
|
||||
failure counter is shared across every connection child, so the lockout holds
|
||||
even when the next attempt is handled by a different forked child.
|
||||
- `auth lockout duration = SECONDS` — how long a locked-out source is refused
|
||||
(default 300). A locked-out client is refused before any SCRAM challenge is
|
||||
sent; a successful authentication clears the counter.
|
||||
- `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access
|
||||
patterns.
|
||||
|
||||
A `[module]` requires `path`, and may also set `read only`, `write only`,
|
||||
`client owner`, `auth users`, `max connections` (0 = unlimited; enforced per
|
||||
module across all connection children), and its own `hosts allow`/`hosts deny`.
|
||||
|
||||
Like rsync, a module is **read-only by default**: a bare `[module]` with only a
|
||||
`path` refuses a write transfer. Opt a module into writability explicitly with
|
||||
`read only = no` or `write only = yes`; a global `read only` value in the
|
||||
section before the first `[module]` sets the default for later modules, and a
|
||||
module's own `read only`/`write only = yes` always wins over it. An
|
||||
rsync-style `write only = yes` is mapped to writability because FastSync is
|
||||
push-only (a module can never be read from the network).
|
||||
|
||||
The per-host cap and the shared auth lockout identify a source by its numeric
|
||||
peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local
|
||||
client shares that one address, so counting or locking them out would let one
|
||||
local process deny service to all the others. The per-module and global
|
||||
`max connections` caps still apply to loopback. Because the key is the peer IP,
|
||||
`max connections per host` and `auth lockout` also cannot distinguish clients
|
||||
behind the same NAT, proxy, or reverse-proxy address — they share one budget and
|
||||
one lockout counter, so an over-aggressive lockout can affect unrelated users
|
||||
behind that address. Prefer TLS client certificates (`--client-cn`) plus
|
||||
`hosts allow`/`hosts deny` for per-client policy when clients share an address,
|
||||
and size `auth lockout threshold` accordingly.
|
||||
|
||||
The shared per-source table has a bounded lifetime: an entry with no live
|
||||
connection is reclaimed once its lockout has expired, or after it has been idle
|
||||
(300 s). If every entry is still live or locked, a new source is admitted without
|
||||
per-host accounting (fail open) and a rate-limited warning is logged; the
|
||||
per-module cap and host ACLs still apply. The occupancy counters are re-derived
|
||||
from the shared slot table after every child exit, so a child killed mid-transfer
|
||||
(or mid-registration) cannot leak a slot or an occupancy count.
|
||||
|
||||
Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR
|
||||
(`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs
|
||||
are rejected at parse time rather than silently never matching. A matching
|
||||
`hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of
|
||||
them is rejected; deny takes precedence over allow. The global list is checked
|
||||
before the module list, before authentication, and the connecting peer address
|
||||
(IPv4 or IPv6) appears in the connection and authentication audit log lines.
|
||||
|
||||
## Architecture
|
||||
|
||||
### Client
|
||||
@@ -833,61 +465,16 @@ before the module list, before authentication, and the connecting peer address
|
||||
|
||||
## Protocol and Security
|
||||
|
||||
FastSync protocol version `2.30.0` is shared by the client and server. The
|
||||
FastSync protocol version `2.3.0` is shared by the client and server. The
|
||||
current protocol is sender-driven and includes configuration negotiation,
|
||||
including the maximum allocation limit, incremental checks, checksums,
|
||||
manifests, keep-alives, abort handling, per-file remove-source results, and
|
||||
FastSync-native delta messages.
|
||||
Client and server versions must currently match exactly.
|
||||
incremental checks, checksums, manifests, keep-alives, abort handling, and
|
||||
FastSync-native delta messages. Client and server versions must currently
|
||||
match exactly.
|
||||
|
||||
Daemon modules that declare `auth users` authenticate with a SCRAM-SHA-256-style
|
||||
challenge/response against a salted PBKDF2 verifier store: no password and no
|
||||
replayable bearer credential crosses the wire or is stored on the daemon. All
|
||||
store entries share one iteration count, and an unknown user is answered with a
|
||||
deterministic per-username dummy challenge, so probing the daemon cannot
|
||||
enumerate users. Store lines are generated with
|
||||
`fastsync-server --hash-credentials <plaintext-file>` (see `RSYNC_COMPAT.md`);
|
||||
redirect that output to an owner-only (mode 0600) file, and note that legacy
|
||||
`user:SHA256HEX` stores are rejected. FastSync also maintains an owner-only
|
||||
(mode 0600) `<store>.dummykey` sidecar next to the store: it holds the store-wide
|
||||
dummy key, is auto-created on first load, and must be preserved across daemon
|
||||
restarts so the dummy challenge for an unknown user stays stable (the key is
|
||||
never regenerated while the sidecar exists). The sidecar is secret material and
|
||||
must be protected like the credential store: keep it owner-only (mode 0600) and
|
||||
include it with the store in backups and credential rotation. If the sidecar
|
||||
cannot be created (a process-substitution/FIFO store path such as `/dev/fd/N`, a
|
||||
read-only filesystem, a missing directory, or a create, write, fsync, link, or
|
||||
fchmod failure), the daemon logs a warning and uses a transient key, so the
|
||||
cross-restart guarantee does not hold for those deployments. One residual is
|
||||
accepted: the store
|
||||
iteration count is observable pre-auth by design, since the miss path must match
|
||||
a hit.
|
||||
|
||||
An `auth users` module accepts credentials only when one of two conditions
|
||||
holds: (a) the connection is an encrypted, verified TLS connection whose client
|
||||
certificate matches the server's `--client-cn`, or (b) the connection is
|
||||
plaintext from a loopback peer **and** the operator explicitly passed
|
||||
`--allow-unauthenticated`. A remote plaintext peer is refused before any
|
||||
challenge is sent, and `--allow-unauthenticated` never permits remote plaintext
|
||||
auth: remote peers still require verified TLS regardless of the flag. Clients
|
||||
sending daemon credentials with `--password-file` to a non-loopback daemon must
|
||||
therefore use `--tls`; the client rejects a non-local plaintext credential
|
||||
destination before any network I/O. Daemon modules are a `--daemon`-only
|
||||
feature: the SSH `--stdio` path never loads a daemon config and is not an auth
|
||||
transport for them.
|
||||
|
||||
Because the loopback allowance trusts whichever peer the kernel reports as
|
||||
`127.0.0.1`, it assumes nothing relays remote connections to the daemon. A local
|
||||
TCP forwarder or a TLS-terminating proxy in front of an auth-module listener
|
||||
makes remote clients appear as loopback and bypasses the mutual-TLS identity
|
||||
check, so do not front an auth-module listener with such a relay. `--tls` always
|
||||
mandates `--client-cn`, so a TLS connection to an auth-required module always
|
||||
has its client CN verified (`--client-cn` matches the certificate's CN only, not
|
||||
a subjectAltName, which is acceptable for a private CA).
|
||||
|
||||
TLS provides encrypted TCP transport. Both the client and the server require
|
||||
`--ca` together with `--tls`, so peer certificates are always verified
|
||||
(`SSL_VERIFY_PEER`, depth 4). The default TCP transport is not encrypted.
|
||||
TLS provides encrypted TCP transport. Supplying `--ca` enables certificate
|
||||
verification; without it, traffic is encrypted but peer identity is not
|
||||
verified. Use certificate verification for deployments where authentication
|
||||
matters. The default TCP transport is not encrypted.
|
||||
|
||||
The receiver protects its destination root with path validation, `openat()`
|
||||
directory traversal, `O_NOFOLLOW`, temporary files, and atomic renames. Delete
|
||||
@@ -898,29 +485,18 @@ operations require the server's explicit `--allow-delete` policy.
|
||||
The project will reach the drop-in replacement goal in stages:
|
||||
|
||||
1. Correct rsync option meanings, including short options, combined options,
|
||||
and `--option=value` syntax — **done** in the rsync-parity wave: `-r`/`-b`/
|
||||
`-L`/`-B`, short-option clustering (`-av`, `-aAX`, `-rlpt`), and attached
|
||||
values (`-B1000`, `-essh`, `-MOPT`) all parse.
|
||||
and `--option=value` syntax.
|
||||
2. Add differential tests that compare FastSync and rsync contents, metadata,
|
||||
links, deletes, filters, dry runs, and exit codes — **done** for the
|
||||
completion wave's scope; the tests live in `tests/integration/` and skip
|
||||
cleanly when rsync is unavailable.
|
||||
3. `-a` implements full rsync `-rlptgoD`; under `-p` the source mode is copied
|
||||
exactly, including group/other-write bits, with setuid/setgid/sticky copied
|
||||
only when super-user activities are permitted (masked under
|
||||
`SUPER_MODE_OFF`/`--no-super`). Ownership application stays privilege-gated,
|
||||
as in rsync.
|
||||
4. Symlink (verbatim storage), sparse-file, metadata, delete-policy (including
|
||||
`--max-delete` partial + exit 25, per-directory `--delete-during`/
|
||||
`--delete-delay`), codecs, and resumable-write semantics are implemented;
|
||||
remaining work is the documented edge cases, which the **Parity Completion
|
||||
Wave** section of `RSYNC_COMPAT.md` enumerates honestly.
|
||||
links, deletes, filters, dry runs, and exit codes.
|
||||
3. Make `-a` implement the expected recursive, links, permissions, times,
|
||||
owner/group, and supported special-file behavior.
|
||||
4. Complete symlink, sparse-file, metadata, delete-policy, and resumable-write
|
||||
semantics.
|
||||
5. Add rsync remote-shell and daemon protocol interoperability.
|
||||
6. Keep FastSync performance options as negotiated, optional extensions.
|
||||
|
||||
The exhaustive implementation matrix and compatibility notes are in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md); each row is classified as parity, caveat,
|
||||
or divergent.
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
## Testing
|
||||
|
||||
@@ -933,7 +509,7 @@ Run the unit test binary:
|
||||
Run the Python integration suite:
|
||||
|
||||
```bash
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
python3 -m pytest tests/
|
||||
```
|
||||
|
||||
For stricter local validation:
|
||||
@@ -957,13 +533,10 @@ rsync protocol or filesystem-semantic compatibility.
|
||||
|
||||
## Performance Guidance
|
||||
|
||||
- Use `-j`/`--threads` for workloads with many files or enough CPU parallelism
|
||||
(`-m` is `--prune-empty-dirs`).
|
||||
- Use `-z` when network bandwidth is more constrained than CPU (`-c` is
|
||||
`--checksum`, not a bandwidth option).
|
||||
- Use `-m` for workloads with many files or enough CPU parallelism.
|
||||
- Use `-c` or `-z` when network bandwidth is more constrained than CPU.
|
||||
- Tune `--chunk-size` for file sizes, memory limits, and network latency.
|
||||
- Use `--sendfile` for large uncompressed TCP transfers where zero-copy I/O
|
||||
helps (`-f` is `--filter`).
|
||||
- Use `-f` for large uncompressed TCP transfers where zero-copy I/O helps.
|
||||
- Use `--incremental` to avoid retransmitting unchanged files.
|
||||
- Use `--delta` for changed files when both endpoints are FastSync peers.
|
||||
- Use `--bwlimit` when sharing a link with other traffic.
|
||||
|
||||
+157
-1239
File diff suppressed because one or more lines are too long
+66
-411
@@ -3,24 +3,19 @@
|
||||
|
||||
Compares FastSync configs against rsync (no compression) and rsync+zstd.
|
||||
Data is ~75% random/incompressible and ~25% structured/compressible by default,
|
||||
controllable via --random-ratio. Transfers are verified by default (source and
|
||||
destination must match) so a fast-but-broken copy is never counted.
|
||||
controllable via --random-ratio.
|
||||
|
||||
Usage:
|
||||
python3 benchmark/bench.py
|
||||
python3 benchmark/bench.py --runs 5 --profiles lan wan
|
||||
python3 benchmark/bench.py --random-ratio 0.5 --size-mb 50
|
||||
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
|
||||
python3 benchmark/bench.py --warm --runs 3
|
||||
python3 benchmark/bench.py --output json
|
||||
"""
|
||||
import argparse
|
||||
import filecmp
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import random
|
||||
import shlex
|
||||
import shutil
|
||||
import socket
|
||||
import statistics
|
||||
@@ -30,11 +25,8 @@ import tempfile
|
||||
import time
|
||||
|
||||
PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
|
||||
DEFAULT_BUILD_DIR = "build-bench"
|
||||
# Populated by configure_build_dirs(); default to the dedicated bench dir so
|
||||
# importing this module never depends on the user's existing build/ tree.
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, DEFAULT_BUILD_DIR)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, "build")
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server")]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
BENCH_DIR = os.path.join(PROJECT_ROOT, "bench_data")
|
||||
|
||||
@@ -52,11 +44,10 @@ NETWORK_PROFILES = {
|
||||
|
||||
FASTSYNC_CONFIGS = [
|
||||
{"name": "fastsync", "flags": [], "tool": "fastsync"},
|
||||
{"name": "fastsync -z", "flags": ["-z"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j", "flags": ["-j"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j -z", "flags": ["-j", "-z"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j -z --chunk-serialization", "flags": ["-j", "-z", "--chunk-serialization"], "tool": "fastsync"},
|
||||
{"name": "fastsync --sendfile", "flags": ["--sendfile"], "tool": "fastsync"},
|
||||
{"name": "fastsync -c", "flags": ["-c"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m", "flags": ["-m"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m -c", "flags": ["-m", "-c"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m -c -s", "flags": ["-m", "-c", "-s"], "tool": "fastsync"},
|
||||
]
|
||||
|
||||
RSYNC_CONFIGS = [
|
||||
@@ -65,7 +56,6 @@ RSYNC_CONFIGS = [
|
||||
{"name": "rsync -z --zstd", "flags": ["-z", "--zc", "zstd"],"tool": "rsync"},
|
||||
]
|
||||
|
||||
|
||||
class RsyncDaemon:
|
||||
"""Manages an rsync daemon for network-fair benchmarking."""
|
||||
|
||||
@@ -127,12 +117,6 @@ STRUCTURED_FILES = {
|
||||
"nested/another.txt": b"another nested file\n" * 50,
|
||||
}
|
||||
|
||||
# Repeated text used to synthesize genuinely compressible filler of any size.
|
||||
COMPRESSIBLE_TEXT = (
|
||||
b"FastSync benchmark payload: the quick brown fox jumps over the lazy dog. "
|
||||
b"0123456789 ABCDEFGHIJKLMNOPQRSTUVWXYZ abcdefghijklmnopqrstuvwxyz\n"
|
||||
)
|
||||
|
||||
|
||||
class Progress:
|
||||
"""Simple progress bar with ETA."""
|
||||
@@ -167,127 +151,35 @@ class Progress:
|
||||
sys.stderr.flush()
|
||||
|
||||
|
||||
def write_compressible(path, nbytes):
|
||||
"""Write exactly nbytes of highly compressible, repeated text content."""
|
||||
if nbytes <= 0:
|
||||
return
|
||||
block = COMPRESSIBLE_TEXT * (max(1, 8192 // len(COMPRESSIBLE_TEXT)) + 1)
|
||||
remaining = nbytes
|
||||
with open(path, "wb") as f:
|
||||
while remaining > 0:
|
||||
piece = block if remaining >= len(block) else block[:remaining]
|
||||
f.write(piece)
|
||||
remaining -= len(piece)
|
||||
|
||||
|
||||
def generate_bench_data(source_dir, size_mb=25, random_ratio=0.75):
|
||||
"""Generate test data honouring the requested random/compressible split.
|
||||
|
||||
Exactly ``random_ratio * target`` bytes are incompressible random data and
|
||||
the remainder is genuinely compressible structured/repeated content. The
|
||||
measured byte counts are returned so callers can report the real mix.
|
||||
"""
|
||||
"""Generate test data. ~random_ratio is incompressible, rest is structured."""
|
||||
if os.path.exists(source_dir):
|
||||
shutil.rmtree(source_dir)
|
||||
os.makedirs(source_dir)
|
||||
|
||||
target = size_mb * 1024 * 1024
|
||||
random_budget = int(target * random_ratio)
|
||||
compressible_budget = target - random_budget
|
||||
compressible_written = 0
|
||||
random_written = 0
|
||||
files = 0
|
||||
structured_budget = int(target * (1 - random_ratio))
|
||||
written = 0
|
||||
|
||||
# A handful of fixed, human-meaningful files (directories, small files, a
|
||||
# binary blob) as long as they fit inside the compressible budget.
|
||||
for rel_path, content in STRUCTURED_FILES.items():
|
||||
if compressible_written + len(content) > compressible_budget:
|
||||
if written >= structured_budget:
|
||||
break
|
||||
full_path = os.path.join(source_dir, rel_path)
|
||||
os.makedirs(os.path.dirname(full_path), exist_ok=True)
|
||||
with open(full_path, "wb") as f:
|
||||
f.write(content)
|
||||
compressible_written += len(content)
|
||||
files += 1
|
||||
written += len(content)
|
||||
|
||||
# Fill the rest of the compressible share with generated repeated content.
|
||||
if compressible_written < compressible_budget:
|
||||
os.makedirs(os.path.join(source_dir, "compressible"), exist_ok=True)
|
||||
i = 0
|
||||
while compressible_written < compressible_budget:
|
||||
chunk = min(1024 * 1024, compressible_budget - compressible_written)
|
||||
write_compressible(os.path.join(source_dir, "compressible", f"text_{i}.dat"), chunk)
|
||||
compressible_written += chunk
|
||||
files += 1
|
||||
i += 1
|
||||
|
||||
# Incompressible share.
|
||||
if random_written < random_budget:
|
||||
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
|
||||
i = 0
|
||||
while random_written < random_budget:
|
||||
chunk = min(5 * 1024 * 1024, random_budget - random_written)
|
||||
with open(os.path.join(source_dir, "bulk", f"file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk))
|
||||
random_written += chunk
|
||||
files += 1
|
||||
while written < target:
|
||||
chunk_size = min(5 * 1024 * 1024, target - written)
|
||||
with open(os.path.join(source_dir, f"bulk/file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk_size))
|
||||
written += chunk_size
|
||||
i += 1
|
||||
|
||||
return {
|
||||
"total_bytes": compressible_written + random_written,
|
||||
"compressible_bytes": compressible_written,
|
||||
"random_bytes": random_written,
|
||||
"files": files,
|
||||
}
|
||||
|
||||
|
||||
def list_relative_files(root):
|
||||
"""Return the set of file paths (relative to root) under a directory."""
|
||||
found = set()
|
||||
for dirpath, _dirnames, filenames in os.walk(root):
|
||||
for name in filenames:
|
||||
full = os.path.join(dirpath, name)
|
||||
found.add(os.path.relpath(full, root))
|
||||
return found
|
||||
|
||||
|
||||
def verify_transfer(source_dir, dest_dir):
|
||||
"""Recursively check dest matches source (paths, sizes, content).
|
||||
|
||||
Returns (ok, detail). Content is compared byte-for-byte, never hashed, so
|
||||
collisions are impossible. This is intentionally not part of the timing.
|
||||
"""
|
||||
if not os.path.isdir(dest_dir):
|
||||
return False, "destination directory missing"
|
||||
src_files = list_relative_files(source_dir)
|
||||
dst_files = list_relative_files(dest_dir)
|
||||
if src_files != dst_files:
|
||||
missing = src_files - dst_files
|
||||
extra = dst_files - src_files
|
||||
return False, f"path set mismatch (missing {len(missing)}, extra {len(extra)})"
|
||||
for rel in sorted(src_files):
|
||||
src = os.path.join(source_dir, rel)
|
||||
dst = os.path.join(dest_dir, rel)
|
||||
if os.path.getsize(src) != os.path.getsize(dst):
|
||||
return False, f"size mismatch: {rel}"
|
||||
if not filecmp.cmp(src, dst, shallow=False):
|
||||
return False, f"content mismatch: {rel}"
|
||||
return True, ""
|
||||
|
||||
|
||||
def percentile(values, pct):
|
||||
"""Linear-interpolation percentile (matches numpy's default method)."""
|
||||
if not values:
|
||||
return None
|
||||
ordered = sorted(values)
|
||||
if len(ordered) == 1:
|
||||
return ordered[0]
|
||||
rank = (len(ordered) - 1) * (pct / 100.0)
|
||||
low = math.floor(rank)
|
||||
high = math.ceil(rank)
|
||||
if low == high:
|
||||
return ordered[int(rank)]
|
||||
return ordered[low] + (ordered[high] - ordered[low]) * (rank - low)
|
||||
return written
|
||||
|
||||
|
||||
def find_free_port():
|
||||
@@ -315,45 +207,18 @@ def wait_proc(proc, timeout=5):
|
||||
proc.wait()
|
||||
|
||||
|
||||
def _tc_base_cmd():
|
||||
"""Return the command prefix for tc, honouring root vs sudo."""
|
||||
tc = shutil.which("tc")
|
||||
if not tc:
|
||||
raise RuntimeError(
|
||||
"tc (iproute2) not found in PATH; install iproute2 to use network profiles")
|
||||
if os.geteuid() == 0:
|
||||
return [tc]
|
||||
sudo = shutil.which("sudo")
|
||||
if sudo:
|
||||
return [sudo, tc]
|
||||
raise RuntimeError(
|
||||
"applying network limits requires root or sudo; "
|
||||
"re-run as root or install sudo")
|
||||
|
||||
|
||||
def _run_tc(args, check=True):
|
||||
return subprocess.run(_tc_base_cmd() + args, check=check, capture_output=True)
|
||||
|
||||
|
||||
def netem_apply(delay=None, jitter=None, throughput=None, loss=None):
|
||||
"""Apply tc/netem rules to loopback. Pass None to skip a parameter."""
|
||||
netem_reset()
|
||||
params = []
|
||||
cmd = ["sudo", "tc", "qdisc", "add", "dev", "lo", "root", "netem"]
|
||||
if throughput:
|
||||
params += ["rate", throughput]
|
||||
cmd += ["rate", throughput]
|
||||
if delay:
|
||||
params += ["delay", delay, jitter or "0ms"]
|
||||
cmd += ["delay", delay, jitter or "0ms"]
|
||||
if loss:
|
||||
params += ["loss", loss]
|
||||
if not params:
|
||||
return
|
||||
try:
|
||||
_run_tc(["qdisc", "add", "dev", "lo", "root", "netem"] + params)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
detail = exc.stderr.decode(errors="replace").strip() if exc.stderr else str(exc)
|
||||
raise RuntimeError(f"failed to apply network profile via tc/netem: {detail}") from exc
|
||||
except RuntimeError:
|
||||
raise
|
||||
cmd += ["loss", loss]
|
||||
if len(cmd) > 6:
|
||||
subprocess.run(cmd, check=True, capture_output=True)
|
||||
|
||||
|
||||
def netem_apply_profile(profile_name):
|
||||
@@ -370,11 +235,7 @@ def netem_apply_profile(profile_name):
|
||||
|
||||
|
||||
def netem_reset():
|
||||
"""Best-effort removal of any loopback qdisc. Always safe to call."""
|
||||
try:
|
||||
_run_tc(["qdisc", "del", "dev", "lo", "root"], check=False)
|
||||
except Exception:
|
||||
pass
|
||||
subprocess.run("sudo tc qdisc del dev lo root".split(), capture_output=True)
|
||||
|
||||
|
||||
def run_fastsync(source_dir, dest_dir, flags, port):
|
||||
@@ -391,10 +252,8 @@ def run_fastsync(source_dir, dest_dir, flags, port):
|
||||
duration = time.monotonic() - start
|
||||
if result.returncode == 0:
|
||||
return duration
|
||||
sys.stderr.write(f" fastsync failed (exit {result.returncode}): "
|
||||
f"{result.stderr.strip()[:500]}\n")
|
||||
except subprocess.TimeoutExpired:
|
||||
sys.stderr.write(" fastsync timed out after 120s\n")
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
@@ -410,10 +269,8 @@ def run_rsync(source_dir, dest_dir, flags, rsync_daemon=None):
|
||||
duration = time.monotonic() - start
|
||||
if result.returncode == 0:
|
||||
return duration
|
||||
sys.stderr.write(f" rsync failed (exit {result.returncode}): "
|
||||
f"{result.stderr.strip()[:500]}\n")
|
||||
except subprocess.TimeoutExpired:
|
||||
sys.stderr.write(" rsync timed out after 120s\n")
|
||||
pass
|
||||
return None
|
||||
|
||||
|
||||
@@ -425,81 +282,7 @@ def run_transfer(config, source_dir, dest_dir, port=None, rsync_daemon=None):
|
||||
return run_fastsync(source_dir, dest_dir, config["flags"], port)
|
||||
|
||||
|
||||
def apply_incremental_changes(source_dir, target_bytes):
|
||||
"""Add and modify a few files so a warm transfer has real work to do.
|
||||
|
||||
Returns a mutation record (changed byte count plus enough data to revert
|
||||
and re-apply it) so every warm run can start from a pristine source.
|
||||
"""
|
||||
modified_n = 3
|
||||
added_n = 2
|
||||
per_file = max(4096, target_bytes // (modified_n + added_n))
|
||||
modified = {}
|
||||
added = {}
|
||||
changed = 0
|
||||
|
||||
existing = sorted(list_relative_files(source_dir))
|
||||
if existing:
|
||||
step = max(1, len(existing) // modified_n)
|
||||
for rel in existing[::step][:modified_n]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
original_size = os.path.getsize(path)
|
||||
with open(path, "ab") as f:
|
||||
f.write(random.randbytes(per_file))
|
||||
modified[rel] = (original_size, per_file)
|
||||
changed += per_file
|
||||
|
||||
for i in range(added_n):
|
||||
os.makedirs(os.path.join(source_dir, "incremental"), exist_ok=True)
|
||||
rel = os.path.join("incremental", f"new_{i}.dat")
|
||||
write_compressible(os.path.join(source_dir, rel), per_file)
|
||||
added[rel] = per_file
|
||||
changed += per_file
|
||||
|
||||
return {"changed": changed, "modified": modified, "added": added}
|
||||
|
||||
|
||||
def revert_incremental_changes(source_dir, mutation):
|
||||
"""Undo apply_incremental_changes so the source is pristine again."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (original_size, _appended) in mutation["modified"].items():
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
with open(path, "r+b") as f:
|
||||
f.truncate(original_size)
|
||||
for rel in mutation["added"]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
os.remove(path)
|
||||
|
||||
|
||||
def reapply_incremental_changes(source_dir, mutation):
|
||||
"""Re-apply a mutation after an untimed pristine seed transfer."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (_original_size, appended) in mutation["modified"].items():
|
||||
with open(os.path.join(source_dir, rel), "ab") as f:
|
||||
f.write(random.randbytes(appended))
|
||||
for rel, size in mutation["added"].items():
|
||||
write_compressible(os.path.join(source_dir, rel), size)
|
||||
|
||||
|
||||
def expected_received_root(dest_dir, source_dir, tool):
|
||||
"""Where a tool places transferred files inside dest_dir.
|
||||
|
||||
FastSync mirrors the absolute source path under dest_dir (see the
|
||||
integration suite's get_dest_received_dir); rsync copies the source tree
|
||||
contents directly into dest_dir.
|
||||
"""
|
||||
if tool == "rsync":
|
||||
return dest_dir
|
||||
return os.path.join(dest_dir, os.path.abspath(source_dir).lstrip(os.sep))
|
||||
|
||||
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
measure_bytes, verify=True, warm=False, mutation=None,
|
||||
progress=None):
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=None):
|
||||
"""Run benchmark for all configs, returns list of results."""
|
||||
is_limited = profile_name != "unlimited"
|
||||
has_rsync = any(c["tool"] == "rsync" for c in configs)
|
||||
@@ -515,10 +298,7 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
results = []
|
||||
for config in configs:
|
||||
times = []
|
||||
invalid = 0
|
||||
for run_idx in range(runs):
|
||||
if warm:
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
if os.path.exists(dest_dir):
|
||||
shutil.rmtree(dest_dir)
|
||||
os.makedirs(dest_dir, exist_ok=True)
|
||||
@@ -526,33 +306,15 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
port = find_free_port()
|
||||
server = None
|
||||
try:
|
||||
if config["tool"] == "fastsync" or warm:
|
||||
if config["tool"] == "fastsync":
|
||||
server = subprocess.Popen(
|
||||
SERVER_CMD + ["-p", str(port)],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
)
|
||||
wait_for_port(port)
|
||||
|
||||
if warm:
|
||||
seed = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if seed is None:
|
||||
invalid += 1
|
||||
sys.stderr.write(" warm-mode seeding failed; run not counted\n")
|
||||
continue
|
||||
reapply_incremental_changes(source_dir, mutation)
|
||||
|
||||
t = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if t is None:
|
||||
invalid += 1
|
||||
elif verify:
|
||||
root = expected_received_root(dest_dir, source_dir, config["tool"])
|
||||
ok, detail = verify_transfer(source_dir, root)
|
||||
if ok:
|
||||
times.append(t)
|
||||
else:
|
||||
invalid += 1
|
||||
sys.stderr.write(f" verification FAILED ({detail}); run not counted\n")
|
||||
else:
|
||||
if t is not None:
|
||||
times.append(t)
|
||||
finally:
|
||||
if server:
|
||||
@@ -565,22 +327,15 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
"config": config["name"],
|
||||
"tool": config["tool"],
|
||||
"profile": profile_name,
|
||||
"warm": warm,
|
||||
"runs": len(times),
|
||||
"invalid": invalid,
|
||||
"times": [round(t, 4) for t in times],
|
||||
}
|
||||
if times:
|
||||
p50 = percentile(times, 50)
|
||||
p95 = percentile(times, 95)
|
||||
entry["p50"] = round(p50, 4)
|
||||
entry["p95"] = round(p95, 4)
|
||||
entry["p50"] = round(statistics.median(times), 4)
|
||||
entry["p95"] = round(sorted(times)[int(len(times) * 0.95)], 4) if len(times) > 1 else entry["p50"]
|
||||
entry["min"] = round(min(times), 4)
|
||||
entry["max"] = round(max(times), 4)
|
||||
entry["stdev"] = round(statistics.stdev(times), 4) if len(times) > 1 else 0.0
|
||||
if measure_bytes:
|
||||
entry["throughput_mbps"] = round(
|
||||
(measure_bytes / (1024 * 1024)) / p50, 3)
|
||||
results.append(entry)
|
||||
return results
|
||||
finally:
|
||||
@@ -590,59 +345,44 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
netem_reset()
|
||||
|
||||
|
||||
def print_table(results, measure_bytes, stats, warm):
|
||||
def print_table(results, total_bytes, random_ratio):
|
||||
"""Print results as a human-readable table grouped by profile."""
|
||||
profiles = {}
|
||||
for r in results:
|
||||
profiles.setdefault(r["profile"], []).append(r)
|
||||
|
||||
total = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total * 100 if total else 0
|
||||
rand_pct = stats["random_bytes"] / total * 100 if total else 0
|
||||
|
||||
for profile, entries in profiles.items():
|
||||
params = NETWORK_PROFILES.get(profile, {})
|
||||
print(f"\n{'=' * 95}")
|
||||
print(f"\n{'=' * 85}")
|
||||
print(f" Profile: {profile.upper()}")
|
||||
if params.get("rate"):
|
||||
print(f" Network: {params['rate']}, {params['delay']} +/- {params['jitter']}, loss {params['loss']}")
|
||||
else:
|
||||
print(f" Network: unlimited")
|
||||
print(f" Data: {total / (1024*1024):.1f} MB "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)")
|
||||
if warm:
|
||||
print(f" Mode: warm (incremental) — measured {measure_bytes / (1024*1024):.2f} MB "
|
||||
f"changed after an untimed full seed")
|
||||
else:
|
||||
print(" Mode: cold (full copy)")
|
||||
print(f"{'=' * 95}")
|
||||
print(f" Data: {total_bytes / (1024*1024):.1f} MB ({random_ratio*100:.0f}% random, {(1-random_ratio)*100:.0f}% compressible)")
|
||||
print(f"{'=' * 85}")
|
||||
|
||||
fs_entries = [e for e in entries if e.get("tool") == "fastsync"]
|
||||
rsync_entries = [e for e in entries if e.get("tool") == "rsync"]
|
||||
|
||||
header = (f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} "
|
||||
f"{'stdev':>8} {'MB/s':>9} {'runs':>5} {'bad':>4}")
|
||||
rule = (f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} "
|
||||
f"{'-' * 8} {'-' * 9} {'-' * 5} {'-' * 4}")
|
||||
|
||||
if fs_entries:
|
||||
print(f"\n FastSync:")
|
||||
print(header)
|
||||
print(rule)
|
||||
print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
for e in sorted(fs_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if rsync_entries:
|
||||
print(f"\n rsync:")
|
||||
print(header)
|
||||
print(rule)
|
||||
print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
for e in sorted(rsync_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if params.get("rate_bps") and fs_entries and rsync_entries:
|
||||
fs_best = min((e["p50"] for e in fs_entries if "p50" in e), default=None)
|
||||
rsync_best = min((e["p50"] for e in rsync_entries if "p50" in e), default=None)
|
||||
theoretical = measure_bytes / params["rate_bps"]
|
||||
theoretical = total_bytes / params["rate_bps"]
|
||||
if fs_best and rsync_best:
|
||||
print(f"\n Theoretical max (line rate): {theoretical:.4f}s")
|
||||
print(f" FastSync best: {fs_best:.4f}s ({theoretical/fs_best:.2f}x vs line rate)")
|
||||
@@ -652,43 +392,10 @@ def print_table(results, measure_bytes, stats, warm):
|
||||
|
||||
def _print_entry(e):
|
||||
if "p50" in e:
|
||||
tp = f"{e['throughput_mbps']:.2f}" if "throughput_mbps" in e else "N/A"
|
||||
print(f" {e['config']:<38} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} "
|
||||
f"{tp:>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
print(f" {e['config']:<25} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} {e['runs']:>5}")
|
||||
else:
|
||||
print(f" {e['config']:<38} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} "
|
||||
f"{'N/A':>8} {'N/A':>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
|
||||
|
||||
def configure_build_dirs(build_dir):
|
||||
"""Install the selected build directory and derived binary paths."""
|
||||
global BUILD_DIR, SERVER_CMD, CLIENT_CMD
|
||||
if not os.path.isabs(build_dir):
|
||||
build_dir = os.path.join(PROJECT_ROOT, build_dir)
|
||||
BUILD_DIR = os.path.abspath(build_dir)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
|
||||
|
||||
def build_project():
|
||||
"""Configure (Release) and build into the dedicated bench build dir."""
|
||||
if shutil.which("cmake") is None:
|
||||
sys.stderr.write("cmake not found in PATH; cannot build\n")
|
||||
sys.exit(1)
|
||||
os.makedirs(BUILD_DIR, exist_ok=True)
|
||||
configure = ["cmake", "-B", BUILD_DIR, "-S", PROJECT_ROOT,
|
||||
"-DCMAKE_BUILD_TYPE=Release"]
|
||||
result = subprocess.run(configure, capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("CMake configure failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
jobs = str(os.cpu_count() or 1)
|
||||
result = subprocess.run(["cmake", "--build", BUILD_DIR, "-j", jobs],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("Build failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
print(f" {e['config']:<25} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {e['runs']:>5}")
|
||||
|
||||
|
||||
def main():
|
||||
@@ -705,20 +412,12 @@ Custom network limits (--delay/--jitter/--throughput) override profiles.
|
||||
|
||||
Data mix:
|
||||
Default is ~75%% random/incompressible + ~25%% structured/compressible,
|
||||
reflecting typical real-world file sets. The actual mix is measured and
|
||||
reported. Transfers are verified (destination must match source) unless
|
||||
--no-verify is given.
|
||||
|
||||
Warm mode:
|
||||
--warm seeds the destination with an untimed full copy of a pristine base,
|
||||
then measures only the incremental transfer after modifying a few files.
|
||||
reflecting typical real-world file sets.
|
||||
|
||||
Examples:
|
||||
%(prog)s --profiles wan --runs 5
|
||||
%(prog)s --throughput 50mbit --delay 30ms --jitter 5ms
|
||||
%(prog)s --random-ratio 0.5 --size-mb 100
|
||||
%(prog)s --warm --runs 3 --no-rsync
|
||||
%(prog)s --dry-run --size-mb 4 --random-ratio 0.25
|
||||
""")
|
||||
parser.add_argument("--runs", type=int, default=3,
|
||||
help="Number of runs per config (default: 3)")
|
||||
@@ -726,8 +425,7 @@ Examples:
|
||||
choices=list(NETWORK_PROFILES.keys()),
|
||||
help="Predefined network profiles (default: unlimited)")
|
||||
parser.add_argument("--configs", nargs="+", default=None,
|
||||
help="Custom FastSync config flags (shell-quoted, e.g. "
|
||||
"\"-j -z --chunk-serialization\")")
|
||||
help="Custom FastSync config flags")
|
||||
parser.add_argument("--size-mb", type=int, default=25,
|
||||
help="Test data size in MB (default: 25)")
|
||||
parser.add_argument("--random-ratio", type=float, default=0.75,
|
||||
@@ -742,14 +440,6 @@ Examples:
|
||||
help="Custom packet loss (e.g. 1%%)")
|
||||
parser.add_argument("--no-rsync", action="store_true",
|
||||
help="Skip rsync comparison")
|
||||
parser.add_argument("--no-verify", action="store_true",
|
||||
help="Skip source/destination verification after each run")
|
||||
parser.add_argument("--warm", action="store_true",
|
||||
help="Incremental mode: seed dest first, measure only changes")
|
||||
parser.add_argument("--build-dir", default=DEFAULT_BUILD_DIR,
|
||||
help=f"Build directory (default: {DEFAULT_BUILD_DIR})")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="Only generate data and report its composition, then exit")
|
||||
parser.add_argument("--progress", action="store_true",
|
||||
help="Show progress bar with ETA")
|
||||
parser.add_argument("--output", choices=["table", "json"], default="table",
|
||||
@@ -758,48 +448,12 @@ Examples:
|
||||
help="Don't clean up test data")
|
||||
args = parser.parse_args()
|
||||
|
||||
if not 0.0 <= args.random_ratio <= 1.0:
|
||||
parser.error("--random-ratio must be between 0.0 and 1.0")
|
||||
if args.size_mb <= 0:
|
||||
parser.error("--size-mb must be positive")
|
||||
|
||||
configure_build_dirs(args.build_dir)
|
||||
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
stats = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
total_bytes = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
rand_pct = stats["random_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB in {stats['files']} files "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)",
|
||||
file=sys.stderr)
|
||||
|
||||
if args.dry_run:
|
||||
print(f"size_mb={args.size_mb} random_ratio={args.random_ratio:.4f} "
|
||||
f"total_bytes={stats['total_bytes']} "
|
||||
f"compressible_bytes={stats['compressible_bytes']} "
|
||||
f"random_bytes={stats['random_bytes']} files={stats['files']}")
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
return
|
||||
|
||||
# Warm mode: keep a pristine base copy, then mutate the live source.
|
||||
base_dir = None
|
||||
measure_bytes = total_bytes
|
||||
mutation = None
|
||||
if args.warm:
|
||||
change_target = max(64 * 1024, min(int(total_bytes * 0.01), 4 * 1024 * 1024))
|
||||
mutation = apply_incremental_changes(source_dir, change_target)
|
||||
measure_bytes = mutation["changed"]
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
print(f"Warm mode: each run seeds a full copy, then measures "
|
||||
f"{measure_bytes / (1024*1024):.3f} MB of add/change deltas", file=sys.stderr)
|
||||
|
||||
# Build (Release: benchmarking a debug build is meaningless)
|
||||
print(f"Building (Release) into {BUILD_DIR}...", file=sys.stderr)
|
||||
build_project()
|
||||
# Build
|
||||
print("Building...")
|
||||
if os.system(f"cmake -B {BUILD_DIR} -S {PROJECT_ROOT} > /dev/null 2>&1") != 0:
|
||||
print("CMake configure failed"); sys.exit(1)
|
||||
if os.system(f"cmake --build {BUILD_DIR} -j$(nproc) > /dev/null 2>&1") != 0:
|
||||
print("Build failed"); sys.exit(1)
|
||||
|
||||
# Determine active profile for display
|
||||
has_custom_net = args.delay or args.jitter or args.throughput or args.loss
|
||||
@@ -819,10 +473,18 @@ Examples:
|
||||
else:
|
||||
profiles_to_run = args.profiles or ["unlimited"]
|
||||
|
||||
# Build config list (shlex so quoted/space-separated flags survive)
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
total_bytes = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
compressible_pct = (1 - args.random_ratio) * 100
|
||||
random_pct = args.random_ratio * 100
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB "
|
||||
f"({random_pct:.0f}% random, {compressible_pct:.0f}% compressible)")
|
||||
|
||||
# Build config list
|
||||
if args.configs:
|
||||
fastsync_configs = [{"name": c, "flags": shlex.split(c), "tool": "fastsync"}
|
||||
for c in args.configs]
|
||||
fastsync_configs = [{"name": c, "flags": c.split(), "tool": "fastsync"} for c in args.configs]
|
||||
else:
|
||||
fastsync_configs = list(FASTSYNC_CONFIGS)
|
||||
|
||||
@@ -834,21 +496,14 @@ Examples:
|
||||
total_runs = len(configs) * args.runs * len(profiles_to_run)
|
||||
progress = Progress(total_runs, "Benchmarking") if args.progress else None
|
||||
if progress:
|
||||
print(f"Running {total_runs} transfers...", file=sys.stderr)
|
||||
print(f"Running {total_runs} transfers...")
|
||||
|
||||
all_results = []
|
||||
try:
|
||||
for profile in profiles_to_run:
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile,
|
||||
measure_bytes, verify=not args.no_verify,
|
||||
warm=args.warm, mutation=mutation,
|
||||
progress=progress)
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile, progress)
|
||||
all_results.extend(results)
|
||||
except RuntimeError as exc:
|
||||
sys.stderr.write(f"error: {exc}\n")
|
||||
sys.exit(1)
|
||||
finally:
|
||||
netem_reset()
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
|
||||
@@ -856,7 +511,7 @@ Examples:
|
||||
if args.output == "json":
|
||||
print(json.dumps(all_results, indent=2))
|
||||
else:
|
||||
print_table(all_results, measure_bytes, stats, args.warm)
|
||||
print_table(all_results, total_bytes, args.random_ratio)
|
||||
print()
|
||||
|
||||
|
||||
|
||||
-13
@@ -1,13 +0,0 @@
|
||||
[pytest]
|
||||
; Fast integration subset run on every pull request (see .gitea/workflows/ci.yaml).
|
||||
markers =
|
||||
ci: fast, representative integration tests run on the PR CI gate
|
||||
setpriv: privilege-dependent tests (drop to an unprivileged user); excluded
|
||||
from CI because their result depends on the runner/container uid and the
|
||||
host mount permissions, but run locally as root
|
||||
daemon_detach: real double-fork backgrounding path (--daemon without
|
||||
--no-detach); slower/fragile, so it runs in the full suite but not the
|
||||
fast PR gate
|
||||
parity: differential rsync-parity case (full set; runs on push to
|
||||
dev/main)
|
||||
parity_ci: fast differential rsync-parity subset (runs on the PR gate)
|
||||
@@ -1 +0,0 @@
|
||||
nested
|
||||
@@ -1 +0,0 @@
|
||||
nested
|
||||
@@ -3,59 +3,26 @@
|
||||
}:
|
||||
|
||||
pkgs.mkShell {
|
||||
# Development shell for FastSync. Provides the host-side toolchain needed to
|
||||
# build, lint, unit-test, integration-test and benchmark the project.
|
||||
# It deliberately does NOT build on entry: run the CMake commands in README.md
|
||||
# (or use the CI Docker image for exact CI parity).
|
||||
nativeBuildInputs = with pkgs; [
|
||||
# build
|
||||
gcc
|
||||
cmake
|
||||
gnumake
|
||||
pkg-config
|
||||
# lint / static analysis (matches CI)
|
||||
clang-tools # clang-format
|
||||
cppcheck
|
||||
# tests
|
||||
(python3.withPackages (ps: with ps; [ pytest pytest-xdist psutil ]))
|
||||
openssh # SSH transport integration tests
|
||||
# debugging
|
||||
gdb
|
||||
valgrind
|
||||
# coverage
|
||||
lcov
|
||||
# benchmark tooling
|
||||
rsync
|
||||
iproute2 # tc/netem for network shaping
|
||||
# misc
|
||||
git
|
||||
curl
|
||||
nodejs
|
||||
nixpkgs-fmt
|
||||
docker
|
||||
tea
|
||||
];
|
||||
|
||||
buildInputs = with pkgs; [
|
||||
zstd
|
||||
zlib
|
||||
lz4
|
||||
openssl
|
||||
(python3.withPackages (ps: with ps; [ pytest ]))
|
||||
];
|
||||
|
||||
# The CMake configure step fetches xxHash via FetchContent, which needs
|
||||
# network access; NIX_ENFORCE_PURITY must be off so the sandbox does not block.
|
||||
NIX_ENFORCE_PURITY = 0;
|
||||
|
||||
shellHook = ''
|
||||
export NIX_ENFORCE_PURITY=0
|
||||
# Make an existing build tree available on PATH, but never build here.
|
||||
if [ -d "$PWD/build" ]; then
|
||||
cmake -B build
|
||||
export PATH="$PWD/build:$PATH"
|
||||
fi
|
||||
echo "FastSync dev shell ready."
|
||||
echo " Build: cmake -B build -S . && cmake --build build -j\$(nproc)"
|
||||
echo " Unit: ./build/tests"
|
||||
echo " CI parity: docker run --rm --user \"\$(id -u):\$(id -g)\" -v \"\$PWD:/workspace\" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 ..."
|
||||
'';
|
||||
}
|
||||
|
||||
@@ -1,799 +0,0 @@
|
||||
#include "change_list.h"
|
||||
#include "checksum.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
typedef struct {
|
||||
char* data;
|
||||
size_t length;
|
||||
size_t capacity;
|
||||
} StrBuf;
|
||||
|
||||
static void strbuf_free(StrBuf* buf) {
|
||||
if (buf == NULL)
|
||||
return;
|
||||
free(buf->data);
|
||||
buf->data = NULL;
|
||||
buf->length = 0;
|
||||
buf->capacity = 0;
|
||||
}
|
||||
|
||||
static bool strbuf_reserve(StrBuf* buf, size_t extra) {
|
||||
if (buf->length > SIZE_MAX - extra - 1)
|
||||
return false;
|
||||
size_t need = buf->length + extra + 1;
|
||||
if (need <= buf->capacity)
|
||||
return true;
|
||||
size_t capacity = buf->capacity > 0 ? buf->capacity : 32;
|
||||
while (capacity < need) {
|
||||
if (capacity > SIZE_MAX / 2) {
|
||||
capacity = need;
|
||||
break;
|
||||
}
|
||||
capacity *= 2;
|
||||
}
|
||||
char* grown = realloc(buf->data, capacity);
|
||||
if (!grown)
|
||||
return false;
|
||||
buf->data = grown;
|
||||
buf->capacity = capacity;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool strbuf_append_char(StrBuf* buf, char c) {
|
||||
if (!strbuf_reserve(buf, 1))
|
||||
return false;
|
||||
buf->data[buf->length++] = c;
|
||||
buf->data[buf->length] = '\0';
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool strbuf_append(StrBuf* buf, const char* text) {
|
||||
if (text == NULL)
|
||||
return true;
|
||||
size_t length = strlen(text);
|
||||
if (!strbuf_reserve(buf, length))
|
||||
return false;
|
||||
memcpy(buf->data + buf->length, text, length);
|
||||
buf->length += length;
|
||||
buf->data[buf->length] = '\0';
|
||||
return true;
|
||||
}
|
||||
|
||||
bool change_list_enabled(const Config* config) {
|
||||
return config != NULL && (config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL) ||
|
||||
(config->info_level & LOG_INFO_NAME) != 0);
|
||||
}
|
||||
|
||||
/* Emitted once, lazily, ahead of the first --info=name entry: rsync prints the
|
||||
* transfer-root `./` name line when the root directory is (re)created. */
|
||||
static bool name_root_printed = false;
|
||||
|
||||
void change_reset_name_root(void) {
|
||||
name_root_printed = false;
|
||||
}
|
||||
|
||||
/* ---- Itemize code ---- */
|
||||
|
||||
/* Format the permission bits as an `ls -l` string, e.g. `-rw-r--r--`. */
|
||||
static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[0] = S_ISDIR(mode) ? 'd'
|
||||
: S_ISLNK(mode) ? 'l'
|
||||
: S_ISCHR(mode) ? 'c'
|
||||
: S_ISBLK(mode) ? 'b'
|
||||
: S_ISFIFO(mode) ? 'p'
|
||||
: S_ISSOCK(mode) ? 's'
|
||||
: '-';
|
||||
mode_t bits = mode & 07777;
|
||||
out[1] = (bits & S_IRUSR) ? 'r' : '-';
|
||||
out[2] = (bits & S_IWUSR) ? 'w' : '-';
|
||||
out[3] = (bits & S_IXUSR) ? (bits & S_ISUID ? 's' : 'x') : (bits & S_ISUID ? 'S' : '-');
|
||||
out[4] = (bits & S_IRGRP) ? 'r' : '-';
|
||||
out[5] = (bits & S_IWGRP) ? 'w' : '-';
|
||||
out[6] = (bits & S_IXGRP) ? (bits & S_ISGID ? 's' : 'x') : (bits & S_ISGID ? 'S' : '-');
|
||||
out[7] = (bits & S_IROTH) ? 'r' : '-';
|
||||
out[8] = (bits & S_IWOTH) ? 'w' : '-';
|
||||
out[9] = (bits & S_IXOTH) ? (bits & S_ISVTX ? 't' : 'x') : (bits & S_ISVTX ? 'T' : '-');
|
||||
out[10] = '\0';
|
||||
}
|
||||
|
||||
static char itemize_type_char(const ChangeEvent* event) {
|
||||
if (event->is_directory)
|
||||
return 'd';
|
||||
if (event->is_symlink)
|
||||
return 'L';
|
||||
if (event->is_special) {
|
||||
if (S_ISCHR(event->mode) || S_ISBLK(event->mode))
|
||||
return 'D';
|
||||
return 'S';
|
||||
}
|
||||
return 'f';
|
||||
}
|
||||
|
||||
static bool times_match(const Config* config, const ChangeEvent* event) {
|
||||
if (!event->dest.known || !event->dest.existed)
|
||||
return false;
|
||||
if (event->mtime_sec == event->dest.mtime_sec) {
|
||||
/* A regular file's sub-second mtime IS preserved by the receiver, so an nsec
|
||||
difference is a real change. A directory or symlink has no preserved
|
||||
sub-second mtime (rsync's quick-check compares whole seconds there), so a
|
||||
nanosecond-only difference must not render a spurious `.d..t` / `.L..t`. */
|
||||
if (event->is_directory || event->is_symlink || event->is_special)
|
||||
return true;
|
||||
return event->mtime_nsec == event->dest.mtime_nsec;
|
||||
}
|
||||
long long delta = (long long)event->mtime_sec - (long long)event->dest.mtime_sec;
|
||||
if (delta < 0)
|
||||
delta = -delta;
|
||||
return delta <= (long long)config->modify_window;
|
||||
}
|
||||
|
||||
/* Fill the 11-character itemize code (10 chars + NUL). `created` means the
|
||||
* destination entry did not exist, so every attribute marker is `+`. */
|
||||
static void itemize_code(const Config* config, const ChangeEvent* event, char code[12]) {
|
||||
bool known = event->dest.known;
|
||||
bool created = !known || !event->dest.existed;
|
||||
char update;
|
||||
if (event->is_hardlink)
|
||||
update = 'h';
|
||||
else if (created)
|
||||
update = (event->is_directory || event->is_symlink || event->is_special) ? 'c' : '>';
|
||||
else if (event->is_directory)
|
||||
/* rsync: an existing directory that only has attribute changes carries no
|
||||
transfer, so the update column is `.` rather than `>`. */
|
||||
update = '.';
|
||||
else if (event->is_symlink)
|
||||
/* rsync: an existing symlink whose target is unchanged is a `.` update
|
||||
(attributes only); a changed target is `c` (the link value changed). */
|
||||
update = event->dest.target_matches ? '.' : 'c';
|
||||
else
|
||||
update = '>';
|
||||
code[0] = update;
|
||||
code[1] = itemize_type_char(event);
|
||||
if (created) {
|
||||
for (int i = 0; i < 9; i++)
|
||||
code[2 + i] = '+';
|
||||
code[11] = '\0';
|
||||
return;
|
||||
}
|
||||
/* rsync's value/checksum column: `c` for a symlink whose target changed (the
|
||||
link value is the compared content); no destination digest is available for
|
||||
a regular file. */
|
||||
bool value_diff = event->is_symlink && !event->dest.target_matches;
|
||||
/* rsync itemizes size only for regular files: a directory's st_size and a
|
||||
symlink's target length are not compared. */
|
||||
bool size_diff = !event->is_directory && !event->is_symlink && !event->is_special &&
|
||||
event->size != event->dest.size;
|
||||
/* rsync itemizes the time column only when -t/--times is in effect. */
|
||||
bool time_diff = config->preserve_times && !times_match(config, event);
|
||||
bool perms_diff = (event->mode & 07777) != (event->dest.mode & 07777);
|
||||
bool owner_diff = event->uid != (uid_t)event->dest.uid;
|
||||
bool group_diff = event->gid != (gid_t)event->dest.gid;
|
||||
code[2] = value_diff ? 'c' : '.';
|
||||
code[3] = size_diff ? 's' : '.';
|
||||
code[4] = time_diff ? 't' : '.';
|
||||
code[5] = (config->preserve_perms && perms_diff) ? 'p' : '.';
|
||||
code[6] = (config->preserve_owner && owner_diff) ? 'o' : '.';
|
||||
code[7] = (config->preserve_group && group_diff) ? 'g' : '.';
|
||||
code[8] = '.'; /* reserved */
|
||||
code[9] = '.'; /* acl: not compared */
|
||||
code[10] = '.';
|
||||
code[11] = '\0';
|
||||
}
|
||||
|
||||
/* True when the itemized destination entry is unchanged, i.e. rsync would print
|
||||
* no line at all. Reuses itemize_code so suppression is exactly consistent
|
||||
* with what would have been rendered: the update column must be `.` and every
|
||||
* attribute column must be `.`. */
|
||||
static bool itemize_is_unchanged(const Config* config, const ChangeEvent* event) {
|
||||
if (!event->dest.known || !event->dest.existed)
|
||||
return false;
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
if (code[0] != '.')
|
||||
return false;
|
||||
for (int i = 2; i < 11; i++) {
|
||||
if (code[i] != '.')
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* rsync %n: the transfer-relative name, with a trailing slash for directories.
|
||||
* The transfer root is `.` (so `%n` renders `./`), matching rsync's root entry. */
|
||||
static bool append_name(StrBuf* buf, const ChangeEvent* event) {
|
||||
const char* name = event->name != NULL ? event->name : "";
|
||||
if (event->is_directory && name[0] == '\0')
|
||||
return strbuf_append(buf, "./");
|
||||
if (!strbuf_append(buf, name))
|
||||
return false;
|
||||
if (event->is_directory && name[strlen(name) - 1] != '/')
|
||||
return strbuf_append_char(buf, '/');
|
||||
return true;
|
||||
}
|
||||
|
||||
/* rsync %L: " -> target" for a symlink, " => target" for a hard link, else "". */
|
||||
static bool append_link_suffix(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (event->is_symlink && event->symlink_target != NULL)
|
||||
return strbuf_append(buf, " -> ") && strbuf_append(buf, event->symlink_target);
|
||||
if (event->is_hardlink && event->hardlink_target != NULL)
|
||||
return strbuf_append(buf, " => ") && strbuf_append(buf, event->hardlink_target);
|
||||
return true;
|
||||
}
|
||||
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
StrBuf line = {0};
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append_char(&line, ' ') &&
|
||||
append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* rsync's `--info=name` line for an updated entry: the transfer-relative name
|
||||
* (trailing slash for directories) plus the ` -> target` / ` => target` link
|
||||
* suffix. `--info=name` does not alter an itemize/out-format run. */
|
||||
static char* change_render_name(const ChangeEvent* event) {
|
||||
StrBuf line = {0};
|
||||
bool ok = append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* rsync's `--info=name2` line for an unchanged entry: `NAME is uptodate`. */
|
||||
static char* change_render_name_uptodate(const ChangeEvent* event) {
|
||||
char* name = change_render_name(event);
|
||||
if (name == NULL)
|
||||
return NULL;
|
||||
size_t length = strlen(name);
|
||||
char* line = malloc(length + sizeof(" is uptodate"));
|
||||
if (line == NULL) {
|
||||
free(name);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(line, name, length);
|
||||
memcpy(line + length, " is uptodate", sizeof(" is uptodate"));
|
||||
free(name);
|
||||
return line;
|
||||
}
|
||||
|
||||
/* ---- --out-format / --log-file-format ---- */
|
||||
|
||||
/* rsync 3.4.1's `%C` uses the negotiated TRANSFER checksum (the first name of a
|
||||
* two-name "transfer,pre-transfer" --checksum-choice), not the pre-transfer
|
||||
* whole-file digest FastSync compares against on the wire. The default "auto"
|
||||
* resolves to xxh128, so an explicit selection and the default both render the
|
||||
* selected algorithm's digest. */
|
||||
static ChecksumAlgo out_format_checksum_algo(const Config* config) {
|
||||
return (ChecksumAlgo)config->cli.checksum_transfer_algo;
|
||||
}
|
||||
|
||||
/* Render a digest as rsync's sum_as_hex: xxh128 prints the HIGH 64-bit half
|
||||
* before the low half, and xxh64/xxh3 print their 64-bit value big-endian; every
|
||||
* other algorithm prints its bytes in order. */
|
||||
static void digest_to_hex(ChecksumAlgo algo, const uint8_t* digest, size_t len, char* out) {
|
||||
if (algo == CHECKSUM_ALGO_XXH128 && len == 16) {
|
||||
uint64_t low = 0;
|
||||
uint64_t high = 0;
|
||||
memcpy(&low, digest, sizeof(low));
|
||||
memcpy(&high, digest + 8, sizeof(high));
|
||||
snprintf(out, len * 2 + 1, "%016llx%016llx", (unsigned long long)high, (unsigned long long)low);
|
||||
return;
|
||||
}
|
||||
if ((algo == CHECKSUM_ALGO_XXH64 || algo == CHECKSUM_ALGO_XXH3) && len == 8) {
|
||||
uint64_t value = 0;
|
||||
memcpy(&value, digest, sizeof(value));
|
||||
snprintf(out, len * 2 + 1, "%016llx", (unsigned long long)value);
|
||||
return;
|
||||
}
|
||||
static const char hex[] = "0123456789abcdef";
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
out[i * 2] = hex[(digest[i] >> 4) & 0xf];
|
||||
out[i * 2 + 1] = hex[digest[i] & 0xf];
|
||||
}
|
||||
out[len * 2] = '\0';
|
||||
}
|
||||
|
||||
static bool format_uses_checksum(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0')
|
||||
break;
|
||||
if (token == 'C')
|
||||
return true;
|
||||
p += 2;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Fill event->checksum/checksum_known for a transferred regular file. A
|
||||
* non-regular entry (or a hard-link sibling) leaves checksum_known false, which
|
||||
* renders as spaces like rsync. */
|
||||
static void fill_event_checksum(const Config* config, const File* file, ChangeEvent* event) {
|
||||
if (file == NULL || file->is_dir || file->is_symlink || file->is_special ||
|
||||
(file->link_group != 0 && !file->link_first))
|
||||
return;
|
||||
if (!format_uses_checksum(config->out_format) && !format_uses_checksum(config->log_file_format))
|
||||
return;
|
||||
if (file->path == NULL)
|
||||
return;
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
/* rsync renders `--checksum-choice=none` as a blank 2-character column. */
|
||||
if (algo == CHECKSUM_ALGO_NONE)
|
||||
return;
|
||||
uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
|
||||
size_t len = 0;
|
||||
/* rsync's %C is the transfer checksum, which is always seeded with 0 (it is
|
||||
* independent of --checksum-seed, as rsync 3.4.1 demonstrates). */
|
||||
if (!checksum_digest_file(algo, 0, file->path, digest, sizeof(digest), &len))
|
||||
return;
|
||||
digest_to_hex(algo, digest, len, event->checksum);
|
||||
event->checksum_known = true;
|
||||
}
|
||||
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event) {
|
||||
if (format == NULL || event == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'i': {
|
||||
if (event->deleted) {
|
||||
/* rsync's ITEM_DELETED itemize code: `*deleting ` (11 chars). */
|
||||
ok = strbuf_append(&line, "*deleting ");
|
||||
break;
|
||||
}
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
ok = strbuf_append(&line, code);
|
||||
break;
|
||||
}
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = append_name(&line, event);
|
||||
break;
|
||||
case 'L':
|
||||
ok = append_link_suffix(&line, event);
|
||||
break;
|
||||
case 'l': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->size);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'b': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_sent);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'c': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_read);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'C': {
|
||||
if (event->checksum_known) {
|
||||
ok = strbuf_append(&line, event->checksum);
|
||||
} else {
|
||||
/* rsync pads a non-regular / untransferred / `none` entry with spaces;
|
||||
`none` renders as a blank 2-character column. */
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
int width = algo == CHECKSUM_ALGO_NONE ? 2 : checksum_digest_len(algo) * 2;
|
||||
for (int i = 0; i < width && ok; i++)
|
||||
ok = strbuf_append_char(&line, ' ');
|
||||
}
|
||||
} break;
|
||||
case 'M': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(event->mtime_sec, true, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 't': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(time(NULL), false, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 'o':
|
||||
ok = strbuf_append(&line, "send");
|
||||
break;
|
||||
case 'p': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%ld", (long)getpid());
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'B': {
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
ok = strbuf_append(&line, permission + 1);
|
||||
} break;
|
||||
case 'U': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->uid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'G': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->gid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- --list-only ---- */
|
||||
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event) {
|
||||
(void)config;
|
||||
if (event == NULL)
|
||||
return NULL;
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
char date[32];
|
||||
if (!format_rsync_datetime(event->mtime_sec, false, date, sizeof(date)))
|
||||
snprintf(date, sizeof(date), "?");
|
||||
StrBuf line = {0};
|
||||
char size_field[40];
|
||||
char grouped[32];
|
||||
if (!format_big_num(event->size, false, grouped, sizeof(grouped))) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
int written = snprintf(size_field, sizeof(size_field), "%15s", grouped);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
const char* name = event->name != NULL && event->name[0] != '\0' ? event->name : ".";
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append(&line, size_field) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, date) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, name);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- Event emission ---- */
|
||||
|
||||
static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_output) {
|
||||
char* escaped = output_escape(line, eight_bit_output);
|
||||
if (escaped != NULL) {
|
||||
fprintf(stream, "%s\n", escaped);
|
||||
free(escaped);
|
||||
} else {
|
||||
fprintf(stream, "%s\n", line);
|
||||
}
|
||||
fflush(stream);
|
||||
}
|
||||
|
||||
void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
bool to_stdout = config->itemize_changes || config->out_format != NULL;
|
||||
bool to_log = config->log_file != NULL && config->log_file_format != NULL;
|
||||
bool progress_active = config->show_progress || (config->info_level & LOG_INFO_PROGRESS);
|
||||
if (event->decision == CHANGE_UP_TO_DATE) {
|
||||
/* --info=name2 prints `NAME is uptodate` for entries the receiver already
|
||||
had. An itemize/out-format run reports them through its own format (or
|
||||
not at all), the progress stream has no frame for them, and neither the
|
||||
itemize nor the log-file stream previously reported an up-to-date entry,
|
||||
so nothing else here changes. */
|
||||
if (!to_stdout && (config->info_level & LOG_INFO_NAME_UPTODATE) != 0 && !progress_active) {
|
||||
char* line = change_render_name_uptodate(event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (to_stdout) {
|
||||
char* line = config->out_format != NULL
|
||||
? change_render_format(config->out_format, config, event)
|
||||
: change_render_itemize(config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
} else if ((config->info_level & LOG_INFO_NAME) != 0 && !progress_active) {
|
||||
/* --info=name without -i/--out-format: print the updated entry's name. The
|
||||
--progress path owns the name line when progress output is active (it
|
||||
emits the same names before the progress frames), so do not duplicate.
|
||||
The transfer-root `./` line precedes the first such name. */
|
||||
if (!name_root_printed) {
|
||||
name_root_printed = true;
|
||||
fputs("./\n", stdout);
|
||||
}
|
||||
char* line = change_render_name(event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
if (to_log) {
|
||||
char* line = change_render_format(config->log_file_format, config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(config->log_file, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static bool format_uses_mtime(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0')
|
||||
break;
|
||||
if (token == 'M')
|
||||
return true;
|
||||
p += 2;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Relative path of an entry below the transfer root (no leading slash). Uses
|
||||
* the sender-side send_path override when present (bare-relative -R layout). */
|
||||
static char* relative_name(const Config* config, const File* file) {
|
||||
const char* full = file_wire_path(file);
|
||||
if (file->send_path != NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL || full == NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, full, root_len) == 0) {
|
||||
if (full[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (full[root_len] == '/')
|
||||
return str_dup(full + root_len + 1);
|
||||
}
|
||||
return str_dup(full);
|
||||
}
|
||||
|
||||
/* rsync %f long form: the source argument as typed (leading '/' removed,
|
||||
* trailing '/' removed, leading "./" removed) joined to the relative name. */
|
||||
static char* display_name(const Config* config, const char* name) {
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL)
|
||||
return str_dup(name != NULL ? name : "");
|
||||
const char* p = root;
|
||||
while (*p == '/')
|
||||
p++;
|
||||
if (p[0] == '.' && p[1] == '/')
|
||||
p += 2;
|
||||
size_t root_len = strlen(p);
|
||||
while (root_len > 0 && p[root_len - 1] == '/')
|
||||
root_len--;
|
||||
size_t name_len = name != NULL ? strlen(name) : 0;
|
||||
if (root_len == 0 && name_len == 0)
|
||||
return str_dup("");
|
||||
char* out = malloc(root_len + (root_len > 0 && name_len > 0 ? 1 : 0) + name_len + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t offset = 0;
|
||||
if (root_len > 0) {
|
||||
memcpy(out, p, root_len);
|
||||
offset = root_len;
|
||||
}
|
||||
if (root_len > 0 && name_len > 0)
|
||||
out[offset++] = '/';
|
||||
if (name_len > 0)
|
||||
memcpy(out + offset, name, name_len);
|
||||
out[offset + name_len] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
static void fill_event_from_file(const Config* config, const File* file, ChangeEvent* event,
|
||||
char** name_out, char** path_out) {
|
||||
char* name = relative_name(config, file);
|
||||
char* path = display_name(config, name);
|
||||
event->name = name;
|
||||
event->path = path;
|
||||
*name_out = name;
|
||||
*path_out = path;
|
||||
if (file->metadata != NULL) {
|
||||
event->mtime_sec = file->metadata->mtime_sec;
|
||||
event->mtime_nsec = file->metadata->mtime_nsec;
|
||||
event->mode = file->metadata->mode;
|
||||
event->uid = file->metadata->uid;
|
||||
event->gid = file->metadata->gid;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0) {
|
||||
event->mtime_sec = st.st_mtime;
|
||||
event->mtime_nsec = st.st_mtim.tv_nsec;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = false;
|
||||
event.is_special = false;
|
||||
event.is_hardlink = false;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
event.dest = file->dest_state;
|
||||
if (file->is_symlink) {
|
||||
event.is_symlink = true;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.size = file->symlink_target != NULL ? strlen(file->symlink_target) : 0;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->is_special) {
|
||||
event.is_special = true;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->link_group != 0 && !file->link_first) {
|
||||
event.is_hardlink = true;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.bytes_sent = 0;
|
||||
} else {
|
||||
event.bytes_sent = bytes_sent;
|
||||
/* rsync's %c is the block-checksum bytes received for the file. Even a
|
||||
* whole-file transfer (no basis; --append/--inplace included) receives
|
||||
* rsync's 16-byte sum header, so rsync reports 16; a dry run transfers
|
||||
* nothing and reports 0. FastSync's whole-file path has no sum header, so
|
||||
* report rsync's value for parity. With delta enabled the real received
|
||||
* bytes are kept, but FastSync's signature framing differs from rsync's so
|
||||
* those stay numerically divergent. */
|
||||
bool delta_active = config->use_delta && !config->whole_file;
|
||||
event.bytes_read = (!config->dry_run && !delta_active) ? 16 : bytes_read;
|
||||
}
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (file->is_symlink) {
|
||||
/* Output parity (protocol 2.30.0): an unchanged symlink is silent, like
|
||||
rsync's quick check. The itemize/log stream suppresses it only when every
|
||||
attribute matches; the name stream suppresses it whenever the link target
|
||||
is unchanged (rsync names a symlink only when it relinks or creates it). */
|
||||
bool itemize_output = config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL);
|
||||
bool suppress = itemize_output
|
||||
? itemize_is_unchanged(config, &event)
|
||||
: (event.dest.known && event.dest.existed && event.dest.target_matches);
|
||||
if (suppress)
|
||||
event.decision = CHANGE_UP_TO_DATE;
|
||||
}
|
||||
if (name != NULL && path != NULL) {
|
||||
fill_event_checksum(config, file, &event);
|
||||
change_emit(config, &event);
|
||||
}
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
if (file == NULL)
|
||||
return;
|
||||
unsigned long long payload = file->data != NULL ? file->data->size : 0;
|
||||
change_emit_file_sent_bytes(config, file, payload, 0);
|
||||
}
|
||||
|
||||
void change_emit_file_uptodate(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.decision = CHANGE_UP_TO_DATE;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = file->is_symlink;
|
||||
event.is_special = file->is_special;
|
||||
event.is_hardlink = file->link_group != 0 && !file->link_first;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
void change_emit_dir_sent(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = true;
|
||||
event.size = 0;
|
||||
event.bytes_sent = 0;
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
/* Output parity (protocol 2.30.0): suppress a directory rsync would leave
|
||||
silent. The itemize/log stream suppresses it only when every attribute
|
||||
matches (`.d.........`); the name stream suppresses any pre-existing
|
||||
directory (rsync names a directory only when it is created). */
|
||||
bool itemize_output = config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL);
|
||||
bool suppress = itemize_output ? itemize_is_unchanged(config, &event)
|
||||
: (event.dest.known && event.dest.existed);
|
||||
/* The transfer root's line is an unconditional FastSync residual (rsync keys
|
||||
it off the root's own attribute change); keep emitting it. */
|
||||
bool is_root = event.name != NULL && event.name[0] == '\0';
|
||||
if (suppress && !is_root)
|
||||
event.decision = CHANGE_UP_TO_DATE;
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
@@ -1,114 +0,0 @@
|
||||
#ifndef CHANGE_LIST_H
|
||||
#define CHANGE_LIST_H
|
||||
|
||||
#include "config.h"
|
||||
#include "checksum.h"
|
||||
#include "file_types.h"
|
||||
#include "format.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
/*
|
||||
* Shared per-file change-event / output model (rsync --itemize-changes,
|
||||
* --out-format, --log-file-format, and --list-only all render from here).
|
||||
*
|
||||
* FastSync is a push-style tool: the client sends files from the source tree
|
||||
* to a server that writes them under the destination root. Events are
|
||||
* emitted by whichever code path decides a file's fate (the single-threaded
|
||||
* send loop and the `-m` sender thread both call the same per-file sender), so
|
||||
* all change events are emitted by exactly one thread and itemize/out-format
|
||||
* lines never interleave with each other. They may still interleave with
|
||||
* legacy log messages (log.c) that share the same stdout/log-file stream.
|
||||
*/
|
||||
|
||||
typedef enum {
|
||||
CHANGE_SENT, /* file data (full or delta) was transmitted */
|
||||
CHANGE_UP_TO_DATE, /* receiver already had an identical file; skipped */
|
||||
} ChangeDecision;
|
||||
|
||||
typedef struct {
|
||||
const char* path; /* long-form display path (rsync %f) */
|
||||
const char* name; /* transfer-relative path (rsync %n), no trailing slash */
|
||||
ChangeDecision decision;
|
||||
bool is_directory;
|
||||
bool is_symlink;
|
||||
bool is_special;
|
||||
bool is_hardlink; /* a hard-link sibling (linked, no data sent) */
|
||||
bool deleted; /* a would-delete report (-n --delete); no source file */
|
||||
const char* symlink_target;
|
||||
const char* hardlink_target;
|
||||
unsigned long long size; /* source file length in bytes */
|
||||
unsigned long long bytes_sent; /* wire bytes actually transferred (rsync %b) */
|
||||
unsigned long long bytes_read; /* wire bytes read back for this file (rsync %c) */
|
||||
/* rsync %C: whole-file checksum hex for a transferred regular file. Only
|
||||
* filled when the active format uses %C (checksum_known == false otherwise,
|
||||
* which renders as spaces like rsync for non-regular entries). */
|
||||
bool checksum_known;
|
||||
char checksum[CHECKSUM_MAX_DIGEST_LEN * 2 + 1];
|
||||
time_t mtime_sec;
|
||||
long mtime_nsec;
|
||||
mode_t mode;
|
||||
uid_t uid;
|
||||
gid_t gid;
|
||||
/* Receiver-reported pre-transfer destination state (OutputDestState.known is
|
||||
* false when no report was requested/received). */
|
||||
OutputDestState dest;
|
||||
} ChangeEvent;
|
||||
|
||||
/* True when any output mode is active and per-file events matter. */
|
||||
bool change_list_enabled(const Config* config);
|
||||
|
||||
/* Render the rsync-style itemize line for a transferred item
|
||||
* (`%i %n%L`): `>f+++++++++ sub/b.txt`. Caller frees the result. */
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Expand an --out-format/--log-file-format template. Supported tokens:
|
||||
* %i itemize code %n transfer-relative name (dir: trailing /)
|
||||
* %f long display path %l file length in bytes
|
||||
* %b wire bytes transferred %c block-checksum bytes received (rsync: 16
|
||||
* for a whole-file transfer, 0 for a dry run)
|
||||
* %C whole-file checksum hex (xxh128 by default; spaces for non-regular)
|
||||
* %M mtime (YYYY/MM/DD-HH:MM:SS)
|
||||
* %t current time %o operation ("send"/"del.")
|
||||
* %p pid %B permission bits without the type char
|
||||
* %U uid %G gid
|
||||
* %L " -> target" / " => target" %% a literal percent sign
|
||||
* Unknown %X sequences are preserved verbatim. Caller frees the result. */
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Render one --list-only long-listing entry:
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 sub/b.txt`
|
||||
* (ls -l style columns; mtime in the local time zone). Caller frees it. */
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Emit an event to every active destination:
|
||||
* stdout: --itemize-changes line, or the --out-format expansion when set;
|
||||
* log file: the --log-file-format expansion (requires --log-file).
|
||||
* CHANGE_UP_TO_DATE events produce no output. */
|
||||
void change_emit(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent. `bytes_sent`
|
||||
* is the process-wide wire-byte delta for this file (rsync's %b) and `bytes_read`
|
||||
* the received bytes used for the delta handshake; pass 0 when unknown. For a
|
||||
* whole-file transfer %c is pinned to rsync's 16-byte sum header regardless. */
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent, deriving
|
||||
* the wire byte counts from the source payload length. */
|
||||
void change_emit_file_sent(const Config* config, const File* file);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
void change_emit_dir_sent(const Config* config, const File* file);
|
||||
|
||||
/* Build and emit a CHANGE_UP_TO_DATE event for a file the receiver already had.
|
||||
* With --info=name2 it renders rsync's "NAME is uptodate" line (no output
|
||||
* otherwise). */
|
||||
void change_emit_file_uptodate(const Config* config, const File* file);
|
||||
|
||||
/* Reset the lazy transfer-root `./` line emitted ahead of the first
|
||||
* --info=name entry. Call once at the start of a transfer. */
|
||||
void change_reset_name_root(void);
|
||||
|
||||
#endif
|
||||
+156
-2980
File diff suppressed because it is too large
Load Diff
@@ -1,679 +0,0 @@
|
||||
#include "client_send_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "change_list.h"
|
||||
#include "charset.h"
|
||||
#include "config.h"
|
||||
#include "data.h"
|
||||
#include "delta.h"
|
||||
#include "file.h"
|
||||
#include "format.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "scanner.h"
|
||||
#include "transport_tls.h"
|
||||
#include "utils.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
/* True when --dry-run should contact a receiver rather than running the
|
||||
* client-side local manifest. Any target a real run would reach over the wire
|
||||
* selects the server-contacting path: a remote (SSH host:path), a daemon
|
||||
* (host::module/path), an explicit --server-host, --server-port/--port, TLS, or
|
||||
* a source-bind --address. A plain local destination (none of these) keeps the
|
||||
* original client-side behavior, which never dials the default 127.0.0.1:8080. */
|
||||
bool dry_run_targets_server(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
if (config->transport == TRANSPORT_SSH)
|
||||
return true;
|
||||
if (config->module && config->module[0] != '\0')
|
||||
return true;
|
||||
if (config->cli.server_host_set || config->cli.server_port_set)
|
||||
return true;
|
||||
if (config->use_tls)
|
||||
return true;
|
||||
if (config->address != NULL)
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk) {
|
||||
if (!manifest)
|
||||
return true;
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
const char* path = file_wire_path(chunk->items[i]);
|
||||
if (*path == '/')
|
||||
path++;
|
||||
char* entry = str_dup(path);
|
||||
if (!entry) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate manifest entry");
|
||||
return false;
|
||||
}
|
||||
if (!array_list_add(manifest, entry)) {
|
||||
free(entry);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Print dry-run manifest showing files that would be transferred. Returns 0 on success. */
|
||||
int send_dry_run_manifest(const Config* config) {
|
||||
int skipped = 0;
|
||||
ArrayList* missing_dest = NULL;
|
||||
if (!client_prepare_files_from(config, &missing_dest, &skipped))
|
||||
return -1;
|
||||
PreparedScanner prepared;
|
||||
if (!prepare_scanner(config, 0, &prepared)) {
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
return -1;
|
||||
}
|
||||
DirectoryScanner* scanner =
|
||||
directory_scanner_create_with_options(config->send_directory, &prepared.options);
|
||||
if (!scanner) {
|
||||
prepared_scanner_destroy(&prepared);
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
return -1;
|
||||
}
|
||||
Chunk* chunk;
|
||||
int file_count = 0;
|
||||
unsigned long long total_bytes = 0;
|
||||
char size_buffer[32];
|
||||
if (!config->quiet)
|
||||
printf("Dry run: files to be transferred\n");
|
||||
while ((chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
if (!config->quiet) {
|
||||
char* escaped_path =
|
||||
output_escape(file_wire_path(chunk->items[i]), config->eight_bit_output);
|
||||
if (!escaped_path) {
|
||||
chunk_destroy(chunk);
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
return -1;
|
||||
}
|
||||
if (config->human_readable)
|
||||
printf(
|
||||
" %s (%s)\n", escaped_path,
|
||||
display_bytes(chunk->items[i]->data->size, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf(" %s (%zu bytes)\n", escaped_path, chunk->items[i]->data->size);
|
||||
free(escaped_path);
|
||||
}
|
||||
total_bytes += chunk->items[i]->data->size;
|
||||
file_count++;
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
/* --delete-missing-args: the missing entries' destination mirrors render as
|
||||
would-be deletions (rsync's dry-run also lists its *deleting lines). */
|
||||
if (missing_dest && !config->quiet) {
|
||||
for (int i = 0; i < missing_dest->size; i++) {
|
||||
char* escaped = output_escape((char*)missing_dest->items[i], config->eight_bit_output);
|
||||
printf(" %s (missing; would be deleted)\n", escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
}
|
||||
}
|
||||
if (missing_dest)
|
||||
array_list_delete(missing_dest);
|
||||
if (!config->quiet) {
|
||||
if (config->human_readable)
|
||||
printf("Total: %d files, %s\n", file_count,
|
||||
display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
typedef struct {
|
||||
char* name; /* transfer-relative name ("" == the source root) */
|
||||
mode_t mode;
|
||||
unsigned long long size;
|
||||
time_t mtime;
|
||||
long mtime_nsec;
|
||||
bool is_dir;
|
||||
bool is_symlink;
|
||||
char* link_target;
|
||||
} ListEntry;
|
||||
|
||||
static void list_entries_destroy(ListEntry* entries, size_t count) {
|
||||
if (entries == NULL)
|
||||
return;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
free(entries[i].name);
|
||||
free(entries[i].link_target);
|
||||
}
|
||||
free(entries);
|
||||
}
|
||||
|
||||
static int compare_list_entries(const void* left, const void* right) {
|
||||
const ListEntry* a = (const ListEntry*)left;
|
||||
const ListEntry* b = (const ListEntry*)right;
|
||||
return strcmp(a->name, b->name);
|
||||
}
|
||||
|
||||
/* Relative path of an entry below `root` ("" for the root itself). Mirrors
|
||||
* change_list's relative_name for list-only rendering. */
|
||||
static char* list_relative_name(const char* root, const char* full) {
|
||||
if (root == NULL || full == NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, full, root_len) == 0) {
|
||||
if (full[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (full[root_len] == '/')
|
||||
return str_dup(full + root_len + 1);
|
||||
}
|
||||
return str_dup(full);
|
||||
}
|
||||
|
||||
/* --list-only: print an ls-style listing of the entries that WOULD be
|
||||
* transferred and exit without contacting the server or writing anything.
|
||||
* Names are transfer-relative (rsync prints `a.txt`, `sub/b.txt`, `.`) and
|
||||
* directory entries are included. Returns 0 on success, 1 on error. */
|
||||
int send_list_only(const Config* config) {
|
||||
int skipped = 0;
|
||||
if (!files_from_list_check(config, NULL, &skipped))
|
||||
return 1;
|
||||
PreparedScanner prepared;
|
||||
if (!prepare_scanner(config, 0, &prepared))
|
||||
return 1;
|
||||
prepared.options.use_metadata = true; /* capture mode + mtime for the listing */
|
||||
prepared.options.list_dirs = true;
|
||||
DirectoryScanner* scanner =
|
||||
directory_scanner_create_with_options(config->send_directory, &prepared.options);
|
||||
if (!scanner) {
|
||||
prepared_scanner_destroy(&prepared);
|
||||
return 1;
|
||||
}
|
||||
ListEntry* entries = NULL;
|
||||
size_t count = 0;
|
||||
size_t capacity = 0;
|
||||
bool oom = false;
|
||||
|
||||
/* rsync lists the source root itself (as "."). Only when the source is a
|
||||
* directory and no --files-from subset is in effect. */
|
||||
if (config->files_from_set == NULL && config->send_directory != NULL) {
|
||||
struct stat st;
|
||||
if (stat(config->send_directory, &st) == 0 && S_ISDIR(st.st_mode)) {
|
||||
capacity = 64;
|
||||
entries = calloc(capacity, sizeof(ListEntry));
|
||||
if (entries == NULL) {
|
||||
oom = true;
|
||||
} else if ((entries[0].name = str_dup("")) == NULL) {
|
||||
/* A NULL name would be dereferenced by qsort/render: fail the listing. */
|
||||
oom = true;
|
||||
} else {
|
||||
entries[0].mode = st.st_mode;
|
||||
entries[0].mtime = st.st_mtime;
|
||||
entries[0].mtime_nsec = st.st_mtim.tv_nsec;
|
||||
entries[0].size = (unsigned long long)st.st_size;
|
||||
entries[0].is_dir = true;
|
||||
count = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Chunk* chunk;
|
||||
while (!oom && (chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* f = chunk->items[i];
|
||||
if (f == NULL)
|
||||
continue;
|
||||
if (count == capacity) {
|
||||
size_t new_capacity = capacity > 0 ? capacity * 2 : 64;
|
||||
if (new_capacity <= capacity) {
|
||||
oom = true;
|
||||
break;
|
||||
}
|
||||
ListEntry* grown = realloc(entries, new_capacity * sizeof(ListEntry));
|
||||
if (!grown) {
|
||||
oom = true;
|
||||
break;
|
||||
}
|
||||
entries = grown;
|
||||
memset(entries + capacity, 0, (new_capacity - capacity) * sizeof(ListEntry));
|
||||
capacity = new_capacity;
|
||||
}
|
||||
char* name = list_relative_name(config->send_directory, file_wire_path(f));
|
||||
if (!name) {
|
||||
oom = true;
|
||||
break;
|
||||
}
|
||||
mode_t mode = 0;
|
||||
time_t mtime = 0;
|
||||
long mtime_nsec = 0;
|
||||
if (f->metadata != NULL) {
|
||||
mode = f->metadata->mode;
|
||||
mtime = f->metadata->mtime_sec;
|
||||
mtime_nsec = f->metadata->mtime_nsec;
|
||||
} else {
|
||||
struct stat st;
|
||||
if (lstat(f->path, &st) == 0) {
|
||||
mode = st.st_mode;
|
||||
mtime = st.st_mtime;
|
||||
mtime_nsec = st.st_mtim.tv_nsec;
|
||||
}
|
||||
}
|
||||
entries[count].name = name;
|
||||
entries[count].mode = mode;
|
||||
entries[count].mtime = mtime;
|
||||
entries[count].mtime_nsec = mtime_nsec;
|
||||
if (f->is_symlink)
|
||||
entries[count].size = f->symlink_target != NULL ? strlen(f->symlink_target) : 0;
|
||||
else if (f->is_dir) {
|
||||
struct stat dir_st;
|
||||
entries[count].size = stat(f->path, &dir_st) == 0 ? (unsigned long long)dir_st.st_size : 0;
|
||||
} else
|
||||
entries[count].size = f->data != NULL ? f->data->size : 0;
|
||||
entries[count].is_dir = f->is_dir;
|
||||
entries[count].is_symlink = f->is_symlink;
|
||||
entries[count].link_target =
|
||||
f->is_symlink && f->symlink_target ? str_dup(f->symlink_target) : NULL;
|
||||
count++;
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
bool failed = oom || directory_scanner_failed(scanner) || directory_scanner_had_io_error(scanner);
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
if (failed) {
|
||||
list_entries_destroy(entries, count);
|
||||
if (oom)
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed while listing");
|
||||
return 1;
|
||||
}
|
||||
if (count > 1)
|
||||
qsort(entries, count, sizeof(ListEntry), compare_list_entries);
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.name = entries[i].name;
|
||||
event.path = entries[i].name;
|
||||
event.mode = entries[i].mode;
|
||||
event.size = entries[i].size;
|
||||
event.mtime_sec = entries[i].mtime;
|
||||
event.mtime_nsec = entries[i].mtime_nsec;
|
||||
event.is_directory = entries[i].is_dir;
|
||||
event.is_symlink = entries[i].is_symlink;
|
||||
event.symlink_target = entries[i].link_target;
|
||||
char* line = change_render_list_line(config, &event);
|
||||
if (line != NULL) {
|
||||
char* escaped = output_escape(line, config->eight_bit_output);
|
||||
printf("%s\n", escaped != NULL ? escaped : line);
|
||||
free(escaped);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
list_entries_destroy(entries, count);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Send the delete manifest to the server. Returns 0 on success, -1 on
|
||||
failure. It carries FOUR path sections (keep-set paths, protected excluded
|
||||
prefixes, --delete-missing-args exact-delete paths, and the destination-
|
||||
relative directories the sender synchronized this run) followed by the
|
||||
protocol-2.30.0 per-directory filter-rule block (`per_dir_rules`, the rules
|
||||
the scan compiled from each directory's merge files).
|
||||
When --delete-excluded is given `protected` is empty: excluded destination
|
||||
mirrors are then ordinary extras and are removed. When
|
||||
--delete-missing-args is active `missing_args` holds the destination mirrors
|
||||
of missing --files-from entries: each is an explicit receiver-side deletion
|
||||
request, independent of the extras walk. `synced_dirs` confines the extras
|
||||
walk to entries directly inside a synchronized directory. A NULL
|
||||
keep-set / protected / missing / dirs list transmits an empty section. All
|
||||
four sections are unbounded on the sender; the receiver enforces
|
||||
MAX_MANIFEST_ENTRIES per section and a single MAX_MANIFEST_BYTES budget
|
||||
shared across the sections, rejecting (with STATUS_ERROR) an over-budget
|
||||
frame. A heavily filtered source whose exclusion list is large therefore
|
||||
fails the run cleanly on the receiver rather than being truncated. */
|
||||
int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs,
|
||||
const FilterRuleList* per_dir_rules) {
|
||||
if (!send_status(fd, STATUS_MANIFEST))
|
||||
return -1;
|
||||
int keep_count = manifest ? manifest->size : 0;
|
||||
if (!send_int(fd, keep_count))
|
||||
return -1;
|
||||
for (int i = 0; i < keep_count; i++) {
|
||||
if (!send_wire_str(fd, (char*)manifest->items[i]))
|
||||
return -1;
|
||||
}
|
||||
/* The receiver has ONE protected-prefix section; filter-excluded prefixes
|
||||
(dropped under --delete-excluded) and size-pruned prefixes (always
|
||||
protected) are concatenated into it. */
|
||||
int protected_count =
|
||||
(protected_prefixes ? protected_prefixes->size : 0) + (size_skipped ? size_skipped->size : 0);
|
||||
if (!send_int(fd, protected_count))
|
||||
return -1;
|
||||
if (protected_prefixes) {
|
||||
for (int i = 0; i < protected_prefixes->size; i++) {
|
||||
if (!send_wire_str(fd, (char*)protected_prefixes->items[i]))
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
if (size_skipped) {
|
||||
for (int i = 0; i < size_skipped->size; i++) {
|
||||
if (!send_wire_str(fd, (char*)size_skipped->items[i]))
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
int missing_count = missing_args ? missing_args->size : 0;
|
||||
if (!send_int(fd, missing_count))
|
||||
return -1;
|
||||
for (int i = 0; i < missing_count; i++) {
|
||||
if (!send_wire_str(fd, (char*)missing_args->items[i]))
|
||||
return -1;
|
||||
}
|
||||
int dirs_count = synced_dirs ? synced_dirs->size : 0;
|
||||
if (!send_int(fd, dirs_count))
|
||||
return -1;
|
||||
for (int i = 0; i < dirs_count; i++) {
|
||||
if (!send_wire_str(fd, (char*)synced_dirs->items[i]))
|
||||
return -1;
|
||||
}
|
||||
/* Protocol 2.30.0: the receiver-side per-directory filter rules discovered by
|
||||
the sender's scan, so the whole-tree commit walker can shield a
|
||||
destination-only entry that matches only a per-directory merge rule. */
|
||||
if (!delete_filter_dir_rules_send(fd, per_dir_rules))
|
||||
return -1;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Transmit the keep-set manifest and wait for the receiver's verdict. Used by
|
||||
--delete-before/--delete-during, where the extras are removed on the receiver
|
||||
BEFORE the first byte of file data is sent: the receiver acknowledges with
|
||||
STATUS_OK once the bounded delete committed, or STATUS_ERROR if it could not
|
||||
(in which case the sender aborts without streaming any data). The ACK may
|
||||
take much longer than an ordinary per-message round trip because the receiver
|
||||
performs the whole bounded deletion walk (up to MAX_SERVER_DELETE_COUNT
|
||||
unlinks) before replying, so the wait uses a generous explicit deadline
|
||||
instead of the default 60 s receive window. */
|
||||
#define DELETE_ACK_TIMEOUT_SEC 3600
|
||||
/* While waiting for the (potentially slow) receiver-side deletion, send a
|
||||
* STATUS_KEEPALIVE at most this often so the connection is demonstrably alive
|
||||
* and neither side's per-message timeout trips. */
|
||||
#define DELETE_ACK_KEEPALIVE_SEC 10
|
||||
|
||||
bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args,
|
||||
ArrayList* synced_dirs, const FilterRuleList* per_dir_rules) {
|
||||
if (!client || !manifest)
|
||||
return false;
|
||||
if (send_delete_manifest(client->file_descriptor, manifest, protected_prefixes, size_skipped,
|
||||
missing_args, synced_dirs, per_dir_rules) != 0)
|
||||
return false;
|
||||
Status ack;
|
||||
/* The wait is long (up to an hour) and runs inline on this thread: a helper
|
||||
* thread would race the non-thread-safe protocol send path, so keepalives are
|
||||
* emitted from this wait loop itself. A Ctrl-C/SIGTERM abort flag also ends
|
||||
* the wait; the caller then best-effort sends STATUS_ABORT. */
|
||||
if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC,
|
||||
DELETE_ACK_KEEPALIVE_SEC, client_abort_pending)) {
|
||||
/* A Ctrl-C/SIGTERM abort ends the wait above; tell the receiver before the
|
||||
caller tears the connection down (best-effort). */
|
||||
if (client_abort_pending()) {
|
||||
log_info_message(LOG_INFO_MISC,
|
||||
"Abort requested while awaiting delete ack; sending STATUS_ABORT");
|
||||
send_status(client->file_descriptor, STATUS_ABORT);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
if (ack != STATUS_OK) {
|
||||
log_server_rejection("Server failed to delete files before the transfer");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Server-contacting --dry-run. Connects to the configured remote/daemon and
|
||||
* runs the normal per-file incremental decision WITHOUT transmitting any file
|
||||
* data: the receiver (which also sees dry_run=true on the wire) answers
|
||||
* STATUS_OK for an up-to-date file and STATUS_DRY_RUN_TRANSFER for a file it
|
||||
* would otherwise write, mutating nothing on either side. The would-transfer
|
||||
* set and the same trailer as the local dry-run are printed. A
|
||||
* --compare-dest exact basis hit with no destination copy is reported as a
|
||||
* skip by the receiver.
|
||||
*
|
||||
* Only regular files take the receiver-consulted check; directory / symlink /
|
||||
* special / hard-link-sibling entries have no per-file content check, so they
|
||||
* are reported conservatively as would-transfer and their frames are never
|
||||
* sent (which is what keeps the receiver mutation-free). --delete* is
|
||||
* deliberately NOT transmitted in dry-run, so no deletion can occur; the
|
||||
* would-delete manifest report is a documented follow-up.
|
||||
*
|
||||
* Returns 0 on success, 1 on error. */
|
||||
int send_dry_run_remote(Config* config) {
|
||||
int from_skipped = 0;
|
||||
ArrayList* missing_args = NULL;
|
||||
if (!client_prepare_files_from(config, &missing_args, &from_skipped))
|
||||
return 1;
|
||||
if (missing_args)
|
||||
array_list_delete(missing_args);
|
||||
/* A live session may follow, so arm graceful abort handling. */
|
||||
client_set_abort_armed(true);
|
||||
ProtocolSession session;
|
||||
Client* client = client_connect_and_bind_session(config, &session);
|
||||
if (!client) {
|
||||
client_set_abort_armed(false);
|
||||
return 1;
|
||||
}
|
||||
|
||||
int ret = 1;
|
||||
bool partial = false;
|
||||
time_t dry_start = time(NULL);
|
||||
ReceiverStats dry_stats;
|
||||
memset(&dry_stats, 0, sizeof(dry_stats));
|
||||
PreparedScanner prepared;
|
||||
memset(&prepared, 0, sizeof(prepared));
|
||||
DirectoryScanner* scanner = NULL;
|
||||
ArrayList* dry_manifest = NULL;
|
||||
ArrayList* dry_dirs = NULL;
|
||||
ArrayList* dry_excluded = NULL;
|
||||
ArrayList* dry_size_skipped = NULL;
|
||||
FilterRuleList* dry_per_dir = NULL;
|
||||
if (!config_send(client->file_descriptor, config))
|
||||
goto dry_fail;
|
||||
receive_daemon_motd(client, config);
|
||||
if (!prepare_scanner(config, 0, &prepared))
|
||||
goto dry_fail;
|
||||
/* -n --delete: build the same keep-set manifest, protected prefixes, and
|
||||
synchronized-directory scope a real run would send, so the receiver's
|
||||
read-only extras walk enumerates exactly the deletions a real run makes. */
|
||||
if (config->use_delete) {
|
||||
dry_manifest = array_list_create(free);
|
||||
dry_dirs = array_list_create(free);
|
||||
dry_size_skipped = array_list_create(free);
|
||||
dry_per_dir = filter_rule_list_create();
|
||||
if (!dry_manifest || !dry_dirs || !dry_size_skipped || !dry_per_dir)
|
||||
goto dry_fail;
|
||||
prepared.options.per_dir_rules = dry_per_dir;
|
||||
if (!config->delete_excluded) {
|
||||
dry_excluded = array_list_create(free);
|
||||
if (!dry_excluded)
|
||||
goto dry_fail;
|
||||
prepared.options.excluded_paths = dry_excluded;
|
||||
}
|
||||
prepared.options.size_skipped_paths = dry_size_skipped;
|
||||
/* A --files-from subset confines the extras walk to the directories the
|
||||
scan synchronized; a full recursive transfer marks the root itself. */
|
||||
if (config->files_from_set == NULL) {
|
||||
char* root_marker = delete_scope_root_marker(config);
|
||||
if (!root_marker || !array_list_add(dry_dirs, root_marker)) {
|
||||
free(root_marker);
|
||||
goto dry_fail;
|
||||
}
|
||||
} else {
|
||||
prepared.options.synced_dirs = dry_dirs;
|
||||
}
|
||||
}
|
||||
scanner = directory_scanner_create_with_options(config->send_directory, &prepared.options);
|
||||
if (!scanner)
|
||||
goto dry_fail;
|
||||
|
||||
int file_count = 0;
|
||||
unsigned long long total_bytes = 0;
|
||||
char size_buffer[32];
|
||||
if (!config->quiet)
|
||||
printf("Dry run: files to be transferred\n");
|
||||
Chunk* chunk;
|
||||
while ((chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
if (dry_manifest && !add_chunk_to_manifest(dry_manifest, chunk)) {
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* f = chunk->items[i];
|
||||
if (!f)
|
||||
continue;
|
||||
unsigned long long fsize = f->data ? f->data->size : 0;
|
||||
bool would;
|
||||
if (f->is_dir || f->is_symlink || f->is_special ||
|
||||
(f->link_group != 0 && !f->link_first && f->hardlink_target != NULL)) {
|
||||
/* No receiver-side content check exists for these frame types; a real
|
||||
run would (re)create them, so report would-transfer and send no
|
||||
frame (the receiver must stay mutation-free). */
|
||||
would = true;
|
||||
} else if (fsize > MAX_RECEIVE_WHOLE_FILE_SIZE && !config->use_incremental &&
|
||||
!config_has_basis(config)) {
|
||||
/* A non-incremental run streams a >whole-file-limit source without the
|
||||
STATUS_CHECK handshake, so no read-only receiver decision is possible
|
||||
(and none is needed: a real run would transfer it). */
|
||||
would = true;
|
||||
} else {
|
||||
DeltaSignature* sig = NULL;
|
||||
unsigned long long resume_offset = 0;
|
||||
int rc = incremental_check(client, f, config, &sig, &resume_offset);
|
||||
delta_signature_destroy(sig);
|
||||
if (rc < 0) {
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
if (rc == 1)
|
||||
continue; /* up to date; nothing to report */
|
||||
if (rc != 4) {
|
||||
log_message(LOG_LEVEL_ERROR, "Unexpected receiver reply during dry-run");
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
would = true;
|
||||
}
|
||||
if (would) {
|
||||
if (!config->quiet) {
|
||||
char* escaped_path = output_escape(file_wire_path(f), config->eight_bit_output);
|
||||
if (!escaped_path) {
|
||||
chunk_destroy(chunk);
|
||||
goto dry_fail;
|
||||
}
|
||||
if (config->human_readable)
|
||||
printf(" %s (%s)\n", escaped_path,
|
||||
display_bytes(fsize, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf(" %s (%llu bytes)\n", escaped_path, fsize);
|
||||
free(escaped_path);
|
||||
}
|
||||
total_bytes += fsize;
|
||||
file_count++;
|
||||
}
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
bool io_error = directory_scanner_had_io_error(scanner);
|
||||
if (directory_scanner_failed(scanner))
|
||||
goto dry_fail;
|
||||
if (io_error)
|
||||
log_message(LOG_LEVEL_WARNING, "source scan hit an unreadable directory");
|
||||
/* Send the keep-set manifest (no data frames) so the receiver can enumerate
|
||||
the destination extras; an early-timing delete ACKs before it will accept
|
||||
the terminal FINISHED. */
|
||||
bool early_delete = config->use_delete && config_delete_timing_early(config);
|
||||
if (dry_manifest) {
|
||||
if (send_delete_manifest(client->file_descriptor, dry_manifest, dry_excluded, dry_size_skipped,
|
||||
NULL, dry_dirs, dry_per_dir) != 0)
|
||||
goto dry_fail;
|
||||
if (early_delete) {
|
||||
Status ack;
|
||||
if (!receive_status_keepalive(client->file_descriptor, &ack, DELETE_ACK_TIMEOUT_SEC,
|
||||
DELETE_ACK_KEEPALIVE_SEC, client_abort_pending) ||
|
||||
ack != STATUS_OK)
|
||||
goto dry_fail;
|
||||
}
|
||||
}
|
||||
/* Terminate the stream so the receiver emits its success frame; no data frame
|
||||
is ever sent in dry-run. */
|
||||
if (!send_status(client->file_descriptor, STATUS_FINISHED))
|
||||
goto dry_fail;
|
||||
Status status;
|
||||
if (!receive_status(client->file_descriptor, &status))
|
||||
goto dry_fail;
|
||||
if (status == STATUS_STATS) {
|
||||
ArrayList* would_delete = array_list_create(free);
|
||||
if (!would_delete)
|
||||
goto dry_fail;
|
||||
if (!receive_stats_record(client->file_descriptor, &dry_stats, would_delete)) {
|
||||
array_list_delete(would_delete);
|
||||
goto dry_fail;
|
||||
}
|
||||
print_delete_reports(config, would_delete);
|
||||
array_list_delete(would_delete);
|
||||
if (!receive_status(client->file_descriptor, &status))
|
||||
goto dry_fail;
|
||||
}
|
||||
/* A per-entry receiver failure is rsync's PARTIAL transfer (exit 23), not a
|
||||
hard failure: a dry run transfers nothing, but keep the verdict consistent
|
||||
with the normal path instead of treating it as a protocol error. */
|
||||
if (status == STATUS_PARTIAL) {
|
||||
partial = true;
|
||||
} else if (status != STATUS_OK) {
|
||||
goto dry_fail;
|
||||
}
|
||||
if (!config->quiet) {
|
||||
if (config->human_readable)
|
||||
printf("Total: %d files, %s\n", file_count,
|
||||
display_bytes(total_bytes, true, size_buffer, sizeof(size_buffer)));
|
||||
else
|
||||
printf("Total: %d files, %.1f MB\n", file_count, (double)total_bytes / (double)BYTES_PER_MIB);
|
||||
}
|
||||
{
|
||||
TransferStats dry_transfer;
|
||||
memset(&dry_transfer, 0, sizeof(dry_transfer));
|
||||
dry_transfer.flist_reg = (unsigned long long)file_count;
|
||||
dry_transfer.total_file_size = total_bytes;
|
||||
dry_transfer.transferred_regular = (unsigned long long)file_count;
|
||||
dry_transfer.transferred_file_size = total_bytes;
|
||||
dry_transfer.literal_data = total_bytes;
|
||||
report_transfer_stats(config, &dry_transfer, dry_start, &dry_stats);
|
||||
}
|
||||
ret = io_error ? 1 : (partial ? 23 : 0);
|
||||
|
||||
dry_fail:
|
||||
if (dry_manifest)
|
||||
array_list_delete(dry_manifest);
|
||||
if (dry_dirs)
|
||||
array_list_delete(dry_dirs);
|
||||
if (dry_excluded)
|
||||
array_list_delete(dry_excluded);
|
||||
if (dry_size_skipped)
|
||||
array_list_delete(dry_size_skipped);
|
||||
if (dry_per_dir)
|
||||
filter_rule_list_free(dry_per_dir);
|
||||
if (scanner)
|
||||
directory_scanner_destroy(scanner);
|
||||
prepared_scanner_destroy(&prepared);
|
||||
disconnect_transfer_client(client);
|
||||
protocol_session_unbind();
|
||||
client_set_abort_armed(false);
|
||||
return ret;
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,470 +0,0 @@
|
||||
#include "client_send_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "charset.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "file_list.h"
|
||||
#include "filter.h"
|
||||
#include "hardlink.h"
|
||||
#include "log.h"
|
||||
#include "scanner.h"
|
||||
#include "utils.h"
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Build the scanner options for one scan. Returns false and logs on failure. */
|
||||
bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out) {
|
||||
if (!out)
|
||||
return false;
|
||||
out->base_filters = NULL;
|
||||
out->hardlinks = NULL;
|
||||
out->relative_prefix = NULL;
|
||||
memset(&out->options, 0, sizeof(out->options));
|
||||
|
||||
int rule_count = config->filters ? config->filters->size : 0;
|
||||
const char** texts = NULL;
|
||||
if (rule_count > 0) {
|
||||
texts = malloc((size_t)rule_count * sizeof(char*));
|
||||
if (!texts) {
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed for filter rules");
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < rule_count; i++)
|
||||
texts[i] = (const char*)config->filters->items[i];
|
||||
}
|
||||
if (rule_count > 0 || config->cvs_exclude) {
|
||||
char err[160];
|
||||
out->base_filters = filter_base_build(texts, rule_count, config->cvs_exclude,
|
||||
config->delete_excluded, err, sizeof(err));
|
||||
free(texts);
|
||||
if (!out->base_filters) {
|
||||
log_message(LOG_LEVEL_ERROR, "invalid filter rule: %s", err);
|
||||
return false;
|
||||
}
|
||||
} else {
|
||||
free(texts);
|
||||
}
|
||||
|
||||
ScannerOptions* options = &out->options;
|
||||
options->use_metadata = config->use_metadata;
|
||||
options->preserve_atimes = config->preserve_atimes;
|
||||
options->preserve_crtimes = config->preserve_crtimes;
|
||||
options->preserve_xattrs = config->preserve_xattrs;
|
||||
options->preserve_acls = config->preserve_acls;
|
||||
options->chunk_size = config->chunk_size;
|
||||
/* --exclude/--include are compiled, in command-line order, into the SAME
|
||||
* ordered filter rule list as --filter/-f (see config_add_selection_rule), so
|
||||
* the legacy per-kind arrays are deliberately NOT passed to the scanner:
|
||||
* doing so would re-apply them with the old "excludes first, then includes as
|
||||
* a mandatory whitelist" precedence and defeat rsync's first-match-wins
|
||||
* ordering. The arrays remain populated purely for the Config API surface. */
|
||||
options->exclude_patterns = NULL;
|
||||
options->exclude_count = 0;
|
||||
options->include_patterns = NULL;
|
||||
options->include_count = 0;
|
||||
options->max_size = config->max_size;
|
||||
options->min_size = config->min_size;
|
||||
options->max_depth = config->max_depth;
|
||||
options->num_threads = num_threads;
|
||||
options->follow_symlinks = config->follow_symlinks;
|
||||
options->copy_links = config->copy_links;
|
||||
options->safe_links = config->safe_links;
|
||||
options->copy_unsafe_links = config->copy_unsafe_links;
|
||||
options->copy_dirlinks = config->copy_dirlinks;
|
||||
options->munge_links = config->munge_links;
|
||||
options->checksum = config->checksum;
|
||||
options->one_file_system = config->one_file_system;
|
||||
options->preserve_devices = config->preserve_devices;
|
||||
options->preserve_specials = config->preserve_specials;
|
||||
options->copy_devices = config->copy_devices;
|
||||
options->file_list = (const FileListSet*)config->files_from_set;
|
||||
options->base_filters = out->base_filters;
|
||||
options->per_dir_filters = config->per_dir_filter;
|
||||
options->delete_excluded = config->delete_excluded;
|
||||
options->exclude_per_dir_filter_files = config->per_dir_filter_count >= 2;
|
||||
options->dirs = config->dirs;
|
||||
options->relative = config->relative;
|
||||
/* A real recursive transfer recreates empty source directories (rsync
|
||||
parity); low-level scanner users leave this off. */
|
||||
options->emit_empty_dirs = true;
|
||||
/* --no-implied-dirs only has meaning with -R (rsync): without it the option
|
||||
is a documented no-op, so the scanner must not suppress directory
|
||||
metadata. */
|
||||
options->no_implied_dirs = config->no_implied_dirs && config->relative;
|
||||
/* -R/--relative outside --files-from reconstructs every destination path from
|
||||
* the source spec (rsync's '/./' cut point). With --files-from the listed
|
||||
* entry already supplies the bare relative path, so no prefix is built. */
|
||||
if (config->relative && config->files_from_set == NULL && config->send_directory) {
|
||||
out->relative_prefix = scanner_relative_prefix(config->send_directory);
|
||||
if (!out->relative_prefix) {
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed building --relative path prefix");
|
||||
filter_rule_list_free(out->base_filters);
|
||||
out->base_filters = NULL;
|
||||
return false;
|
||||
}
|
||||
options->relative_prefix = out->relative_prefix;
|
||||
}
|
||||
options->prune_empty_dirs = config->prune_empty_dirs;
|
||||
options->ignore_io_errors = config->ignore_errors;
|
||||
options->ignore_missing_args = config->ignore_missing_args || config->delete_missing_args;
|
||||
options->note_nonreg = (config->info_level & LOG_INFO_NONREG) != 0 && !config->quiet;
|
||||
options->note_mount = (config->info_level & LOG_INFO_MOUNT) != 0 && !config->quiet;
|
||||
options->send_directory = config->send_directory;
|
||||
options->eight_bit_output = config->eight_bit_output;
|
||||
options->excluded_paths = NULL;
|
||||
options->excluded_mutex = NULL;
|
||||
options->size_skipped_paths = NULL;
|
||||
options->synced_dirs = NULL;
|
||||
options->hardlinks = NULL;
|
||||
/* Set by the real send paths; NULL for the metadata-only scans (progress
|
||||
pre-count, batch) that must not perturb the sender's --stats counter. */
|
||||
options->dir_count = NULL;
|
||||
/* P7 Wave D: capture source directory metadata when a directory attribute is
|
||||
requested (-p for modes, -t for times unless -O omits them). Whether they
|
||||
are APPLIED is decided receiver-side. */
|
||||
options->capture_dir_times = dir_metadata_should_capture(config);
|
||||
options->dir_entries = NULL;
|
||||
options->dir_entries_mutex = NULL;
|
||||
if (config->preserve_hard_links) {
|
||||
out->hardlinks = hardlink_table_create();
|
||||
if (!out->hardlinks) {
|
||||
filter_rule_list_free(out->base_filters);
|
||||
out->base_filters = NULL;
|
||||
return false;
|
||||
}
|
||||
options->hardlinks = out->hardlinks;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void prepared_scanner_destroy(PreparedScanner* prepared) {
|
||||
if (!prepared)
|
||||
return;
|
||||
filter_rule_list_free(prepared->base_filters);
|
||||
prepared->base_filters = NULL;
|
||||
hardlink_table_destroy(prepared->hardlinks);
|
||||
prepared->hardlinks = NULL;
|
||||
free(prepared->relative_prefix);
|
||||
prepared->relative_prefix = NULL;
|
||||
}
|
||||
|
||||
/* -R/--relative implied directories: rsync transmits the metadata of the
|
||||
* parent directories implied by the source path (every prefix component above
|
||||
* the source root) so the receiver applies their attributes to the created
|
||||
* parents. FastSync's scan only covers the source root and below, so append
|
||||
* one metadata-only directory entry per implied ancestor. --no-implied-dirs
|
||||
* suppresses this exactly like rsync. A missing ancestor is never fatal. */
|
||||
bool append_implied_dir_times(const Config* config, ArrayList* dir_entries) {
|
||||
if (!dir_entries || !config->relative || config->files_from_set != NULL ||
|
||||
config->no_implied_dirs || !config->send_directory)
|
||||
return true;
|
||||
char* prefix = scanner_relative_prefix(config->send_directory);
|
||||
if (!prefix)
|
||||
return true;
|
||||
int ncomp = 0;
|
||||
for (const char* s = prefix; *s;) {
|
||||
while (*s == '/')
|
||||
s++;
|
||||
if (!*s)
|
||||
break;
|
||||
while (*s && *s != '/')
|
||||
s++;
|
||||
ncomp++;
|
||||
}
|
||||
if (ncomp <= 1) {
|
||||
free(prefix);
|
||||
return true;
|
||||
}
|
||||
char* fs = str_dup(config->send_directory);
|
||||
if (!fs) {
|
||||
free(prefix);
|
||||
return true;
|
||||
}
|
||||
size_t flen = strlen(fs);
|
||||
while (flen > 1 && fs[flen - 1] == '/')
|
||||
fs[--flen] = '\0';
|
||||
bool ok = true;
|
||||
/* Walk the source path upwards one component at a time (fs is truncated in
|
||||
place, so each step targets the next implied ancestor). */
|
||||
for (int depth = ncomp - 2; depth >= 0 && ok; depth--) {
|
||||
char* slash = strrchr(fs, '/');
|
||||
if (!slash || slash == fs)
|
||||
break;
|
||||
*slash = '\0';
|
||||
char* p = prefix;
|
||||
int c = 0;
|
||||
while (c <= depth) {
|
||||
while (*p == '/')
|
||||
p++;
|
||||
while (*p && *p != '/')
|
||||
p++;
|
||||
c++;
|
||||
}
|
||||
char saved = *p;
|
||||
*p = '\0';
|
||||
struct stat st;
|
||||
if (stat(fs, &st) == 0 && S_ISDIR(st.st_mode)) {
|
||||
File* file = file_create(fs);
|
||||
if (!file) {
|
||||
ok = false;
|
||||
} else {
|
||||
file->is_dir = true;
|
||||
file->metadata =
|
||||
file_metadata_create(fs, &st, config->preserve_atimes, config->preserve_crtimes);
|
||||
file->send_path = str_dup(prefix);
|
||||
if (!file->metadata || !file->send_path || !array_list_add(dir_entries, file)) {
|
||||
file_destroy(file);
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
}
|
||||
*p = saved;
|
||||
}
|
||||
free(fs);
|
||||
free(prefix);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* The delete-walk root scope for a full (non---files-from) transfer: rsync
|
||||
* confines --delete to the directories it actually transferred. A plain
|
||||
* recursive run mirrors the source under the receive root, so "." (the whole
|
||||
* tree) is correct; an -R run transfers only the reconstructed prefix subtree,
|
||||
* so the walk is scoped to that prefix instead. Returns a malloc'd wire path
|
||||
* (or "."), or NULL on allocation failure. */
|
||||
char* delete_scope_root_marker(const Config* config) {
|
||||
if (config->relative && config->files_from_set == NULL && config->send_directory) {
|
||||
char* prefix = scanner_relative_prefix(config->send_directory);
|
||||
if (!prefix)
|
||||
return NULL;
|
||||
if (prefix[0] != '\0')
|
||||
return prefix;
|
||||
free(prefix);
|
||||
}
|
||||
return str_dup(".");
|
||||
}
|
||||
|
||||
/* The -R destination prefix that confines a per-directory delete walk, or NULL
|
||||
* when the whole receive root is in scope. The marker was installed into
|
||||
* `synced_dirs` by delete_scope_root_marker(); for a plain recursive transfer
|
||||
* it is "." (whole root) and for --files-from the list is not a single prefix. */
|
||||
const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs) {
|
||||
if (!config || config->files_from_set != NULL || !config->relative || !config->send_directory)
|
||||
return NULL;
|
||||
if (!synced_dirs || synced_dirs->size != 1)
|
||||
return NULL;
|
||||
const char* marker = (const char*)synced_dirs->items[0];
|
||||
if (marker[0] == '\0' || strcmp(marker, ".") == 0)
|
||||
return NULL;
|
||||
return marker;
|
||||
}
|
||||
|
||||
/* The destination-relative mirror path for a missing --files-from entry: where
|
||||
a PRESENT entry with the same name would have been written. With -R that is
|
||||
the entry's bare relative path (the bare wire path the receiver uses);
|
||||
otherwise it is the full source mirror below the destination root
|
||||
(`send_directory` joined to the entry, leading '/' stripped), exactly the
|
||||
path the manifest records for a present sibling. Returns an owned string, or
|
||||
NULL on allocation failure. */
|
||||
static char* files_from_missing_dest_path(const Config* config, const char* entry) {
|
||||
if (config->relative)
|
||||
return str_dup(entry);
|
||||
char* joined = path_cat(config->send_directory, entry);
|
||||
if (!joined)
|
||||
return NULL;
|
||||
const char* rel = *joined == '/' ? joined + 1 : joined;
|
||||
char* dup = str_dup(rel);
|
||||
free(joined);
|
||||
return dup;
|
||||
}
|
||||
|
||||
/* --files-from semantics: every listed entry must resolve under the source
|
||||
* root, otherwise rsync reports a hard error instead of silently transferring
|
||||
* nothing. An entry of "." (the whole tree) and listed-but-empty directories
|
||||
* are valid. An empty list is valid too: rsync transfers nothing and exits 0.
|
||||
* With --ignore-missing-args
|
||||
* (implied by --delete-missing-args) a listed-but-missing entry is instead
|
||||
* skipped: nothing is transferred for it, it never enters the keep-set and the
|
||||
* run succeeds for the rest (an all-missing non-empty list succeeds
|
||||
* transferring nothing, matching rsync). With --delete-missing-args
|
||||
* `missing_dest` (when non-NULL) collects the entry's destination-relative
|
||||
* mirror for the receiver's exact-deletion request. Runs before any
|
||||
* transfer so the failure/skip is surfaced uniformly in the single-threaded,
|
||||
* -m, dry-run and --list-only paths. */
|
||||
bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out) {
|
||||
*skipped_out = 0;
|
||||
const FileListSet* set = (const FileListSet*)config->files_from_set;
|
||||
if (!set)
|
||||
return true;
|
||||
if (!config->send_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "--files-from requires a source directory");
|
||||
return false;
|
||||
}
|
||||
if (set->count == 0) {
|
||||
/* rsync treats an empty --files-from list as "nothing to transfer" and
|
||||
exits 0 (the source directory is still a valid source arg), so this is
|
||||
not an error. Nothing passes the (empty) allow-set, so no file is sent
|
||||
and no keep-set entry is produced. */
|
||||
return true;
|
||||
}
|
||||
bool ignore = config->ignore_missing_args || config->delete_missing_args;
|
||||
for (int i = 0; i < set->count; i++) {
|
||||
const char* entry = set->entries[i];
|
||||
if (entry[0] == '\0')
|
||||
continue; /* "." == list the whole tree */
|
||||
char* full = path_cat(config->send_directory, entry);
|
||||
if (!full) {
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from");
|
||||
return false;
|
||||
}
|
||||
struct stat st;
|
||||
if (lstat(full, &st) != 0) {
|
||||
free(full);
|
||||
if (ignore) {
|
||||
(*skipped_out)++;
|
||||
char* escaped_entry = output_escape(entry, log_get_8_bit_output());
|
||||
log_info_message(LOG_INFO_MISC, "skipping missing --files-from entry '%s'",
|
||||
escaped_entry ? escaped_entry : "<allocation failed>");
|
||||
free(escaped_entry);
|
||||
if (config->delete_missing_args && missing_dest) {
|
||||
char* mirror = files_from_missing_dest_path(config, entry);
|
||||
if (!mirror || !array_list_add(missing_dest, mirror)) {
|
||||
free(mirror);
|
||||
log_message(LOG_LEVEL_ERROR, "memory allocation failed while validating --files-from");
|
||||
return false;
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
char* escaped_entry = output_escape(entry, log_get_8_bit_output());
|
||||
char* escaped_src = output_escape(config->send_directory, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_ERROR, "--files-from entry '%s' not found in source '%s'",
|
||||
escaped_entry ? escaped_entry : "<allocation failed>",
|
||||
escaped_src ? escaped_src : "<allocation failed>");
|
||||
free(escaped_entry);
|
||||
free(escaped_src);
|
||||
return false;
|
||||
}
|
||||
free(full);
|
||||
}
|
||||
if (*skipped_out > 0) {
|
||||
if (config->delete_missing_args) {
|
||||
/* --list-only never deletes and a --dry-run only shows intent, so the
|
||||
summary must not claim a real deletion happened in those modes. */
|
||||
if (config->list_only)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--delete-missing-args: %d missing --files-from entr%s skipped (--list-only "
|
||||
"never deletes)",
|
||||
*skipped_out, *skipped_out == 1 ? "y" : "ies");
|
||||
else if (config->dry_run)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--delete-missing-args: %d missing --files-from entr%s would be deleted from "
|
||||
"the destination (dry run)",
|
||||
*skipped_out, *skipped_out == 1 ? "y" : "ies");
|
||||
else
|
||||
log_message(
|
||||
LOG_LEVEL_WARNING,
|
||||
"--delete-missing-args: %d missing --files-from entr%s will be deleted from the "
|
||||
"destination",
|
||||
*skipped_out, *skipped_out == 1 ? "y" : "ies");
|
||||
} else if (config->ignore_missing_args)
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"--ignore-missing-args: ignored %d missing --files-from entr%s", *skipped_out,
|
||||
*skipped_out == 1 ? "y" : "ies");
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Walk the whole source tree once collecting only destination-relative wire
|
||||
paths, loading and sending nothing. --delete-before/--delete-during need the
|
||||
complete keep-set manifest before the first data byte, so it is built by a
|
||||
dedicated pre-scan pass and transmitted early; the data pass then re-scans
|
||||
with a fresh scanner. --delete-before additionally replays this very scan as
|
||||
its data pass (rsync's single file list), so `chunks_out` (optional) retains
|
||||
the scanned Chunk objects for the caller to send instead of destroying them;
|
||||
the caller owns the list and must give it a chunk_destroy destructor. A
|
||||
source I/O error is fatal unless the options carry --ignore-errors, in which
|
||||
case the scan continues past the unreadable directory and *io_error_out
|
||||
reports it (the caller still performs the deletion but reports the run as
|
||||
errored). */
|
||||
bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest,
|
||||
DeletePlanSender* plans, bool* io_error_out,
|
||||
unsigned long long* non_dir_count_out, ArrayList* chunks_out,
|
||||
bool emit_nonreg) {
|
||||
if (io_error_out)
|
||||
*io_error_out = false;
|
||||
if (non_dir_count_out)
|
||||
*non_dir_count_out = 0;
|
||||
ScannerOptions local = *options;
|
||||
/* The pre-scan is normally a paths-only pass with no client output: it must
|
||||
not emit --info=nonreg lines because the data pass re-scans and emits them
|
||||
once. When the caller replays this scan as the data pass (--delete-before)
|
||||
there is no later scan, so it opts in and the lines are emitted here. */
|
||||
local.note_nonreg = emit_nonreg && options->note_nonreg;
|
||||
DirectoryScanner* scanner = directory_scanner_create_with_options(config->send_directory, &local);
|
||||
if (!scanner)
|
||||
return false;
|
||||
bool ok = true;
|
||||
Chunk* chunk;
|
||||
while ((chunk = directory_scanner_next(scanner)) != NULL) {
|
||||
if (non_dir_count_out) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
const File* f = chunk->items[i];
|
||||
if (f && !f->is_dir)
|
||||
(*non_dir_count_out)++;
|
||||
}
|
||||
}
|
||||
if (manifest && !add_chunk_to_manifest(manifest, chunk)) {
|
||||
ok = false;
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
if (plans) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* f = chunk->items[i];
|
||||
if (!f)
|
||||
continue;
|
||||
const char* path = file_wire_path(f);
|
||||
if (!delete_plan_sender_add(plans, path, f->is_dir)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!ok) {
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (chunks_out) {
|
||||
/* Retain the chunk for the caller's data pass; ownership moves with it. */
|
||||
if (!array_list_add(chunks_out, chunk)) {
|
||||
ok = false;
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
}
|
||||
if (ok) {
|
||||
/* Keep every traversed source directory, including empty ones, so a plan
|
||||
no longer removes the destination directory itself. Their own plans are
|
||||
emitted after the data stream (no file frame triggers them). */
|
||||
if (plans && options->plan_dirs) {
|
||||
for (int i = 0; i < options->plan_dirs->size; i++) {
|
||||
if (!delete_plan_sender_add(plans, (const char*)options->plan_dirs->items[i], true)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (ok && directory_scanner_failed(scanner))
|
||||
ok = false;
|
||||
if (io_error_out)
|
||||
*io_error_out = directory_scanner_had_io_error(scanner);
|
||||
directory_scanner_destroy(scanner);
|
||||
return ok;
|
||||
}
|
||||
+307
-1897
File diff suppressed because it is too large
Load Diff
@@ -4,33 +4,10 @@
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "transport_tcp.h"
|
||||
#include <signal.h>
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Set ONLY by the client's SIGINT/SIGTERM handler (async-signal-safe: the
|
||||
* handler stores 1 and does nothing else). The send loops poll it via
|
||||
* client_abort_pending() and, when set, best-effort send STATUS_ABORT so the
|
||||
* receiver can clean up before the client exits. */
|
||||
extern volatile sig_atomic_t client_abort_requested;
|
||||
bool client_abort_pending(void);
|
||||
/* Arm/disarm abort handling around the network phase. While disarmed, a
|
||||
* SIGINT/SIGTERM takes the default action (immediate termination) so local-only
|
||||
* modes are not left unresponsive. Defined in client_cli.c. */
|
||||
void client_set_abort_armed(bool armed);
|
||||
|
||||
/* Both sender entry points BORROW `config` for the duration of the call; they
|
||||
* never free it, and the caller retains ownership (freeing it with
|
||||
* config_delete() once the call returns). */
|
||||
int send_chunk(Client* client, Chunk* chunk, Config* config);
|
||||
int send_files(Config* config);
|
||||
int send_files_multithreaded(Config* config);
|
||||
/* rsync's --ignore-errors deletion gate: with no I/O error during the scan the
|
||||
* deletion phase always proceeds; with one it is suppressed unless
|
||||
* `--ignore-errors` was given. Exposed so the decision can be unit-tested
|
||||
* without a privileged (mode-000) source directory. See client_send.c. */
|
||||
bool ignore_errors_allows_delete(const Config* config, bool had_io_error);
|
||||
|
||||
/* Phase 6 residual-batch (client-only). See client_send.c. */
|
||||
int write_batch_from_source(const Config* config, const char* batch_path);
|
||||
int apply_batch_to_dest(const Config* config, const char* batch_path, const char* dest_root);
|
||||
/* Takes ownership only when *config is set to NULL on return. */
|
||||
int send_files_multithreaded(Config** config);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,126 +0,0 @@
|
||||
#ifndef CLIENT_SEND_INTERNAL_H
|
||||
#define CLIENT_SEND_INTERNAL_H
|
||||
|
||||
/* Declarations shared between the client_send.c transfer orchestration and the
|
||||
* reporting (client_report.c), scanner-preparation (client_scan.c) and
|
||||
* manifest/list/dry-run (client_manifest.c) translation units that were split
|
||||
* out of it. Nothing here is part of the public client_send.h facade. */
|
||||
|
||||
#include "array_list.h"
|
||||
#include "client_send.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "delta.h"
|
||||
#include "format.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "scanner.h"
|
||||
#include <stdatomic.h>
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <time.h>
|
||||
|
||||
/* One mebibyte in bytes; the unit used by the --stats/--progress lines.
|
||||
Always cast to double when dividing so the output stays fractional. */
|
||||
#define BYTES_PER_MIB (1024ULL * 1024ULL)
|
||||
|
||||
/* Compiled scanner inputs that are shared read-only across scanner instances
|
||||
* and, in -m mode, across worker threads. `base_filters` owns the compiled
|
||||
* command-line + -C rules; the FileListSet allow-set lives in the Config.
|
||||
* `hardlinks` owns the --hard-links/-H link-group detection table (NULL when
|
||||
* off) and is shared (mutex-guarded) across every scanner/worker of one scan. */
|
||||
typedef struct {
|
||||
ScannerOptions options;
|
||||
FilterRuleList* base_filters; /* owned; may be NULL */
|
||||
HardLinkTable* hardlinks; /* owned; may be NULL */
|
||||
char* relative_prefix; /* owned -R prefix; may be NULL */
|
||||
} PreparedScanner;
|
||||
|
||||
/* client_scan.c */
|
||||
bool prepare_scanner(const Config* config, int num_threads, PreparedScanner* out);
|
||||
void prepared_scanner_destroy(PreparedScanner* prepared);
|
||||
bool append_implied_dir_times(const Config* config, ArrayList* dir_entries);
|
||||
char* delete_scope_root_marker(const Config* config);
|
||||
const char* delete_plan_walk_root(const Config* config, const ArrayList* synced_dirs);
|
||||
bool files_from_list_check(const Config* config, ArrayList* missing_dest, int* skipped_out);
|
||||
bool scan_paths_only(const Config* config, const ScannerOptions* options, ArrayList* manifest,
|
||||
DeletePlanSender* plans, bool* io_error_out,
|
||||
unsigned long long* non_dir_count_out, ArrayList* chunks_out,
|
||||
bool emit_nonreg);
|
||||
|
||||
/* client_report.c */
|
||||
void log_server_rejection(const char* context);
|
||||
const char* display_bytes(unsigned long long bytes, bool human_readable, char* buffer,
|
||||
size_t buffer_size);
|
||||
unsigned long long dir_count_for_stats(const Config* config, const ArrayList* dir_entries,
|
||||
atomic_ullong* counter);
|
||||
void report_transfer_stats(const Config* config, const TransferStats* stats, time_t start,
|
||||
const ReceiverStats* recv);
|
||||
void transfer_stats_note_entry(TransferStats* stats, const File* file);
|
||||
void transfer_stats_note_transferred(TransferStats* stats, const File* file);
|
||||
bool info_flag_enabled(const Config* config, LogInfoFlag flag);
|
||||
void print_delete_reports(const Config* config, const ArrayList* paths);
|
||||
const char* delete_display_path(const Config* config, const char* path);
|
||||
bool progress_requested(const Config* config);
|
||||
void client_progress_cleanup(void);
|
||||
void client_progress_begin(const Config* config);
|
||||
void client_progress_file(const Config* config, const File* file);
|
||||
void client_progress_name(const Config* config, const File* file);
|
||||
/* Emit a transferred entry's ancestor directories (as -i/--out-format change
|
||||
* lines or --progress name lines) before the entry's own line. */
|
||||
void client_change_emit_ancestors(const Config* config, const File* file);
|
||||
/* Output parity (protocol 2.30.0): probe each not-yet-known ancestor directory's
|
||||
* pre-transfer destination state before the entry that first triggers it is
|
||||
* sent. Returns false on a protocol/transport error. */
|
||||
bool client_change_probe_ancestors(const Config* config, const File* file, int fd);
|
||||
/* Mark a transferred directory entry as already reported, and flush the
|
||||
* itemize lines for changed directories that had no transferred child. */
|
||||
void client_change_mark_dir(const Config* config, const File* file);
|
||||
void client_change_emit_pending_dirs(const Config* config, int fd);
|
||||
void client_progress_uptodate(const Config* config, const File* file);
|
||||
void client_progress_prepare(const Config* config, const ArrayList* plan_dirs,
|
||||
unsigned long long plan_non_dir_count);
|
||||
bool receive_stats_record(int fd, ReceiverStats* stats, ArrayList* would_delete);
|
||||
/* --stderr=client diagnostic channel (client_report.c): install the queueing
|
||||
* log sink for a transfer, mark the session live, flush queued diagnostics over
|
||||
* the wire at a frame boundary, and tear the sink down. */
|
||||
void client_messages_install(void);
|
||||
void client_messages_activate(bool active);
|
||||
void client_flush_client_messages(int fd);
|
||||
void client_messages_end(void);
|
||||
|
||||
/* client_send.c */
|
||||
void receive_daemon_motd(Client* client, const Config* config);
|
||||
Client* connect_transfer_client(const Config* config);
|
||||
/* Connect the configured transport and install `session` on it: init with the
|
||||
* socket fd pair, apply the I/O timeout and (when negotiated) the TLS object,
|
||||
* then bind the session to this thread. Returns the connected client, or NULL
|
||||
* after logging the connect failure. The caller owns the client and must keep
|
||||
* `session` alive until it calls protocol_session_unbind(). */
|
||||
Client* client_connect_and_bind_session(const Config* config, ProtocolSession* session);
|
||||
/* Shared --files-from/--delete-missing-args preamble for the send entry points:
|
||||
* when --delete-missing-args is set, allocate the list that
|
||||
* files_from_list_check fills with the destination mirrors of missing entries;
|
||||
* then validate the --files-from list. On success returns true and stores the
|
||||
* (possibly NULL) owned list in *missing_args_out plus the skipped count; on
|
||||
* failure returns false after freeing the list. */
|
||||
bool client_prepare_files_from(const Config* config, ArrayList** missing_args_out,
|
||||
int* skipped_out);
|
||||
void disconnect_transfer_client(Client* client);
|
||||
int incremental_check(Client* client, File* file, const Config* config, DeltaSignature** out_sig,
|
||||
unsigned long long* resume_offset);
|
||||
|
||||
/* client_manifest.c */
|
||||
bool dry_run_targets_server(const Config* config);
|
||||
bool add_chunk_to_manifest(ArrayList* manifest, const Chunk* chunk);
|
||||
int send_dry_run_manifest(const Config* config);
|
||||
int send_list_only(const Config* config);
|
||||
int send_dry_run_remote(Config* config);
|
||||
int send_delete_manifest(int fd, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args, ArrayList* synced_dirs,
|
||||
const FilterRuleList* per_dir_rules);
|
||||
bool send_delete_manifest_early(Client* client, ArrayList* manifest, ArrayList* protected_prefixes,
|
||||
ArrayList* size_skipped, ArrayList* missing_args,
|
||||
ArrayList* synced_dirs, const FilterRuleList* per_dir_rules);
|
||||
|
||||
#endif
|
||||
+18
-100
@@ -1,89 +1,44 @@
|
||||
#include "client_validation.h"
|
||||
#include "log.h"
|
||||
#include "usage.h"
|
||||
#include "utils.h"
|
||||
#include <string.h>
|
||||
#include <stdio.h>
|
||||
|
||||
/* Validate config after parsing. Returns true if valid. */
|
||||
bool validate_config(const Config* config) {
|
||||
/* Phase 6 residual-batch modes relax the normal source+destination pair: the
|
||||
batch driver is local and needs only what it consumes. --only-write-batch
|
||||
emits a batch from the source (no destination, no server);
|
||||
--read-batch applies a batch to the destination (no source, no server);
|
||||
--write-batch runs the live transfer AND emits a batch, so it keeps the
|
||||
full pair. */
|
||||
bool write_batch = config->write_batch != NULL;
|
||||
bool only_write_batch = config->only_write_batch != NULL;
|
||||
bool read_batch = config->read_batch != NULL;
|
||||
if ((write_batch && only_write_batch) || (write_batch && read_batch) ||
|
||||
(only_write_batch && read_batch)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--write-batch, --only-write-batch, and --read-batch are mutually exclusive");
|
||||
return false;
|
||||
}
|
||||
/* A dry-run of a local batch apply is not meaningful: --read-batch bypasses
|
||||
the client-side scan/server decision entirely, so dry-run would have no
|
||||
wire state to report (and must not be used as a mutation escape hatch).
|
||||
--only-write-batch likewise never contacts a receiver. --write-batch DOES
|
||||
run a live transfer but additionally mutates the filesystem by emitting the
|
||||
batch file, so a dry-run must not write it either. Reject all three up
|
||||
front instead of silently ignoring --dry-run. */
|
||||
if (config->dry_run && (read_batch || only_write_batch || write_batch)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--dry-run cannot be combined with --read-batch, --only-write-batch, or "
|
||||
"--write-batch; a dry-run must not mutate anything, including batch files");
|
||||
return false;
|
||||
}
|
||||
if (read_batch) {
|
||||
if (!config->receive_root_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "--read-batch requires a destination directory");
|
||||
print_usage();
|
||||
return false;
|
||||
}
|
||||
} else if (only_write_batch) {
|
||||
if (!config->send_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "--only-write-batch requires a source directory");
|
||||
print_usage();
|
||||
return false;
|
||||
}
|
||||
} else if (!config->send_directory || !config->receive_root_directory) {
|
||||
if (!config->send_directory || !config->receive_root_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "source and destination directories are required");
|
||||
print_usage();
|
||||
return false;
|
||||
}
|
||||
if (config->compression_threads > 0 && !config->use_compression) {
|
||||
log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-z/--compress)");
|
||||
if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s "
|
||||
"(chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->transport == TRANSPORT_SSH && config->use_sendfile) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport");
|
||||
return false;
|
||||
}
|
||||
/* -M/--remote-option appends an option to the REMOTE server's argv, which
|
||||
* only exists on the SSH (user@host:path) transport. A daemon
|
||||
* (host::module/path) or local TCP destination has no remote command line,
|
||||
* so the option would be silently ignored; reject it by name instead. */
|
||||
if (config->remote_option_count > 0 && config->transport != TRANSPORT_SSH) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"-M/--remote-option is only valid with the SSH transport (user@host:path); it "
|
||||
"cannot be used with a daemon (host::module/path) or local TCP destination");
|
||||
if (config->use_incremental && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR, "--incremental is not supported with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
/* -4 and -6 are mutually exclusive: a socket address family cannot be both. */
|
||||
if (config->ipv4 && config->ipv6) {
|
||||
log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive");
|
||||
if (config->use_delta && !config->use_incremental) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta requires --incremental");
|
||||
return false;
|
||||
}
|
||||
/* rsync 3.4.1 rejects --inplace together with --partial-dir (exit 1): the
|
||||
inplace write path bypasses partial staging, so a partial-dir name would be
|
||||
silently ignored. Match rsync's message and refuse before any I/O. */
|
||||
if (config->inplace && config->partial_dir) {
|
||||
log_message(LOG_LEVEL_ERROR, "--inplace cannot be used with --partial-dir");
|
||||
if (config->use_delta && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->log_file_format && !config->log_file) {
|
||||
log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file");
|
||||
if (config->use_delta && config->use_sendfile) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -f (sendfile)");
|
||||
return false;
|
||||
}
|
||||
if (config->append || config->append_verify) {
|
||||
fprintf(
|
||||
stderr,
|
||||
"Error: --append and --append-verify are not supported yet; refusing to ignore option\n");
|
||||
return false;
|
||||
}
|
||||
if (config->use_tls) {
|
||||
@@ -92,42 +47,5 @@ bool validate_config(const Config* config) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
/* Daemon credentials (A7, protocol 2.19.0): a --password-file would send the
|
||||
username in the clear and derive a SCRAM proof a network sniffer could
|
||||
attack offline, so it is only allowed over TLS (which itself mandates a
|
||||
verified --cert/--key/--ca set above) or to a loopback destination. A
|
||||
remote plaintext daemon is refused here, before any network I/O. */
|
||||
if (config->password_file && !config->use_tls && !utils_host_is_loopback(config->server_host)) {
|
||||
log_message(LOG_LEVEL_ERROR, "sending daemon credentials to a non-local server requires --tls");
|
||||
return false;
|
||||
}
|
||||
/* Every cross-field invariant the receiver enforces lives in one shared
|
||||
predicate so the client and the server can never disagree. The client
|
||||
reports the specific reason here, before any network I/O. */
|
||||
const char* invariants_error = config_invariants_error(config);
|
||||
if (invariants_error) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s", invariants_error);
|
||||
return false;
|
||||
}
|
||||
/* The receiver rejects a protect-rule block with more than MAX_FILTER_RULES
|
||||
entries as an opaque protocol error; reject an over-limit --filter set here,
|
||||
before any network I/O, with an actionable message. send_protect_entries()
|
||||
re-checks the final built count because cvs-exclude / merge rules can
|
||||
expand it beyond config->filters->size. */
|
||||
if (config->filters && config->filters->size > MAX_FILTER_RULES) {
|
||||
log_message(LOG_LEVEL_ERROR, "too many filter rules: %d (maximum %d)", config->filters->size,
|
||||
MAX_FILTER_RULES);
|
||||
return false;
|
||||
}
|
||||
/* --protocol: FastSync has exactly one wire format, so the forced version
|
||||
must equal the current PROTOCOL_VERSION exactly. Rejected here, before any
|
||||
network I/O, rather than letting the server hit its own mismatch check. */
|
||||
if (strcmp(config->version, PROTOCOL_VERSION) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--protocol must be %s (FastSync supports only its current wire "
|
||||
"protocol version and cannot speak an older or virtual one)",
|
||||
PROTOCOL_VERSION);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
+574
-1045
File diff suppressed because it is too large
Load Diff
+16
-248
@@ -2,38 +2,14 @@
|
||||
#define SCANNER_H
|
||||
|
||||
#include "chunk.h"
|
||||
#include "file_list.h"
|
||||
#include "filter.h"
|
||||
#include "hardlink.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "stop_condition.h"
|
||||
#include <dirent.h>
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdatomic.h>
|
||||
#include <sys/types.h>
|
||||
#include <threads.h>
|
||||
|
||||
/* Upper bound on the configurable parallel scanner worker count (--threads=N):
|
||||
* keeps one transfer from spawning an unbounded pool on a very large machine. */
|
||||
#define MAX_SCANNER_THREADS 256
|
||||
|
||||
/* Depth of the scanner's work queues: the sequential scanner's pending-directory
|
||||
* stack and the parallel scanner's result queue. Bounds memory for a very wide
|
||||
* or very deep tree while leaving ample headroom for normal scans. */
|
||||
#define SCANNER_RESULT_QUEUE_CAP 100
|
||||
#include <stdatomic.h>
|
||||
|
||||
typedef struct {
|
||||
bool use_metadata;
|
||||
/* Phase 4 metadata capture: -U/--atimes and -N/--crtimes tell the scanner to
|
||||
* capture the source access / birth time into each entry's FileMetadata. */
|
||||
bool preserve_atimes;
|
||||
bool preserve_crtimes;
|
||||
/* Phase 4 xattrs: when preserve_xattrs || preserve_acls is set the scanner
|
||||
* captures each regular file's whitelisted xattr set onto the File. */
|
||||
bool preserve_xattrs;
|
||||
bool preserve_acls;
|
||||
unsigned long long chunk_size;
|
||||
char** exclude_patterns;
|
||||
int exclude_count;
|
||||
@@ -47,209 +23,29 @@ typedef struct {
|
||||
bool copy_links;
|
||||
bool safe_links;
|
||||
bool copy_unsafe_links;
|
||||
/* Phase 4 symlink-trust sender options: -k/--copy-dirlinks (dereference a
|
||||
* symlink to a directory as a directory, keeping symlinks-to-files as
|
||||
* symlinks) and --munge-links (rewrite each transmitted symlink target with a
|
||||
* marker; escaping targets are never transmitted). Both are client/sender
|
||||
* side only and never serialized to the wire (keep_dirlinks is the
|
||||
* receiver-side counterpart). */
|
||||
bool copy_dirlinks;
|
||||
bool munge_links;
|
||||
bool checksum;
|
||||
int one_file_system;
|
||||
/* Phase 4 special/devices: whether device nodes (--devices) and special files
|
||||
* (--specials) are preserved via recreation, and whether --copy-devices
|
||||
* copies a device's content as an ordinary regular file. */
|
||||
bool preserve_devices;
|
||||
bool preserve_specials;
|
||||
bool copy_devices;
|
||||
/* Phase 2 (files-from / filter layer). All pointers are shared read-only
|
||||
* across scanner instances and worker threads; ownership stays with the
|
||||
* caller (client_send). */
|
||||
const FileListSet* file_list; /* --files-from allow-set, or NULL */
|
||||
const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */
|
||||
bool per_dir_filters; /* -F: read .rsync-filter per directory */
|
||||
/* --delete-excluded: per-directory plain rules become sender-only, so they no
|
||||
longer protect the receiver from deletion. */
|
||||
bool delete_excluded;
|
||||
/* -FF: also exclude the per-directory filter files themselves from the
|
||||
transfer (single -F transfers them). */
|
||||
bool exclude_per_dir_filter_files;
|
||||
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
|
||||
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
|
||||
/* -R/--relative outside --files-from: the destination-relative path prefix
|
||||
* reconstructed from the source spec (rsync's '/./' cut point), or NULL when
|
||||
* -R is off or --files-from is in use (the bare-relative path then comes from
|
||||
* the listed entry). Borrowed read-only; owned by client_send. */
|
||||
const char* relative_prefix;
|
||||
/* --list-only: emit an is_dir File for every traversed directory (the listing
|
||||
* includes directory entries, matching rsync). Client-only; never set on a
|
||||
* real transfer, which relies on implicit parent creation. */
|
||||
bool list_dirs;
|
||||
/* --prune-empty-dirs (long only): in --dirs mode an empty source directory's
|
||||
explicit entry is omitted from the transfer file list (so nothing is
|
||||
created at the destination and it can be pruned by --delete); explicitly
|
||||
--files-from-listed directories always pass through. Recursive transfers
|
||||
never emit empty directories, so the flag has no additional effect there. */
|
||||
bool prune_empty_dirs;
|
||||
/* Delete-excluded protection sink (optional): when non-NULL the scanner
|
||||
* appends the destination-relative path of every entry it prunes because a
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules and the legacy
|
||||
* --exclude/--include layer). The sender turns this list into the manifest's
|
||||
* protected prefixes so `--delete` leaves the destination mirror of excluded
|
||||
* source paths alone (rsync's default), and drops it when --delete-excluded
|
||||
* opts back into deleting them. NOT recorded for --files-from subset pruning
|
||||
* (whose delete semantics derive from the synchronized-directory set) or for
|
||||
* -R/--files-from relative wire paths. When `excluded_mutex` is non-NULL it
|
||||
* is taken around every append (the parallel scanner shares one list across
|
||||
* its worker threads). */
|
||||
ArrayList* excluded_paths;
|
||||
mtx_t* excluded_mutex;
|
||||
/* Size-prune protection sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every entry it skipped because of
|
||||
* --max-size/--min-size. rsync never deletes a size-skipped source mirror,
|
||||
* even under --delete-excluded, so the sender always transmits this list as
|
||||
* protected prefixes (unlike excluded_paths, which --delete-excluded drops).
|
||||
* Guarded by `excluded_mutex` like excluded_paths. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Synchronized-directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it is about to traverse
|
||||
* that lies inside a --files-from listed directory (or of every traversed
|
||||
* directory when there is no list). The sender sends this set with the delete
|
||||
* manifest so the receiver confines its extras walk to synchronized
|
||||
* directories, exactly like rsync; the receive root is the "." sentinel.
|
||||
* Guarded by `excluded_mutex`. */
|
||||
ArrayList* synced_dirs;
|
||||
/* Per-directory filter-rule sink (optional): when non-NULL the scanner appends
|
||||
* a deep copy of every rule it reads from a per-directory merge file, each
|
||||
* carrying its owner directory and no-inherit flag (see filter.h). The delete
|
||||
* carriers transmit them so the receiver re-derives the per-directory
|
||||
* protect/risk set for destination-only entries. Guarded by `excluded_mutex`
|
||||
* like the other sinks. */
|
||||
FilterRuleList* per_dir_rules;
|
||||
/* Delete-plan directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it traverses (except the
|
||||
* receive root). The per-directory --delete-during/--delete-delay plan
|
||||
* builder uses this to keep an empty in-scope source directory (rsync keeps
|
||||
* it) and to emit its plan after the data stream, when no file frame would
|
||||
* otherwise trigger it. Guarded by `excluded_mutex`. */
|
||||
ArrayList* plan_dirs;
|
||||
/* --ignore-errors: an unreadable subdirectory no longer aborts the scan (it
|
||||
* is always skipped so the rest of the tree transfers); this flag is kept so
|
||||
* the client can distinguish the option state when deciding deletion policy.
|
||||
* Client-only. */
|
||||
bool ignore_io_errors;
|
||||
/* --info=nonreg: print rsync's `skipping non-regular file "NAME"` line for a
|
||||
* non-regular entry that is not being preserved. Client-only. */
|
||||
bool note_nonreg;
|
||||
/* --info=mount: print rsync's `[sender] skipping mount-point dir NAME` when
|
||||
* -xx drops a mount-point directory. Client-only. */
|
||||
bool note_mount;
|
||||
/* --stats directory accounting for a `-r` run (no -t/-p): a shared counter of
|
||||
* traversed directories that are NOT otherwise represented by an inline
|
||||
* directory entry (rsync still counts every directory in `Number of files`).
|
||||
* Incremented when a directory is opened and decremented when an empty
|
||||
* directory is emitted inline (so it is counted exactly once). Atomic
|
||||
* because the parallel scanner's workers share it; NULL disables the
|
||||
* accounting. Client-only. */
|
||||
atomic_ullong* dir_count;
|
||||
/* Source root and 8-bit-output policy used to render a `--info=nonreg` name
|
||||
* relative to the transfer root. Borrowed read-only. */
|
||||
const char* send_directory;
|
||||
bool eight_bit_output;
|
||||
/* --ignore-missing-args (implied by --delete-missing-args): an explicitly
|
||||
* --files-from-listed entry that does not exist under the source is skipped
|
||||
* instead of failing (the --dirs generator is the only scanner path that
|
||||
* observes a listed-but-missing entry). */
|
||||
bool ignore_missing_args;
|
||||
/* --hard-links (-H): shared, mutable (mutex-guarded) link-group detection
|
||||
* table, NULL when -H is off. Owned by the caller (client_send), shared
|
||||
* read-only here; the parallel scanner passes it unchanged to every worker so
|
||||
* one table detects every group across all subdirectories. */
|
||||
HardLinkTable* hardlinks;
|
||||
/* Phase 6: optional sender stop deadline. When non-NULL the scanner checks
|
||||
* it at natural loop boundaries and stops emitting chunks once reached
|
||||
* (without marking the scan as failed), so a busy scan itself stops early.
|
||||
* Client-only, never serialized to the wire. */
|
||||
const StopCondition* stop_condition;
|
||||
/* P7 Wave D (protocol 2.17.0): directory-time capture sink. When
|
||||
* `capture_dir_times` is true the recursive scan appends one is_dir File
|
||||
* (with metadata, no payload) per source directory it traverses to
|
||||
* `dir_entries`, so the sender can transmit trailing STATUS_DIR_TIMES
|
||||
* frame(s) and the receiver can apply directory mtimes AFTER all children
|
||||
* are written. `dir_entries_mutex` (optional) guards the list
|
||||
* for the parallel scanner's shared worker threads; the caller owns both.
|
||||
* The --dirs generator does not use this (its directory entries carry their
|
||||
* metadata inline through STATUS_MKDIR). */
|
||||
bool capture_dir_times;
|
||||
ArrayList* dir_entries;
|
||||
mtx_t* dir_entries_mutex;
|
||||
/* Recreate empty source directories on a recursive transfer: emit a
|
||||
* payload-less directory entry for every traversed directory that produced
|
||||
* no transferred/descended child. Off by default so low-level scanner users
|
||||
* (unit helpers, --list-only) see only the historical file list; the real
|
||||
* sender sets it in prepare_scanner. */
|
||||
bool emit_empty_dirs;
|
||||
/* --no-implied-dirs with -R + --files-from: a directory that is only an
|
||||
* implied parent of a listed entry (not itself listed, nor below a listed
|
||||
* directory) must not carry source metadata; it is created with default
|
||||
* attributes at the destination, matching rsync. */
|
||||
bool no_implied_dirs;
|
||||
} ScannerOptions;
|
||||
|
||||
/* Internal per-scanner filter state. FilterNode chains represent the ordered
|
||||
* per-directory .rsync-filter rules that apply below a directory. */
|
||||
typedef struct FilterNode FilterNode;
|
||||
|
||||
typedef struct {
|
||||
/* Scan inputs, copied once at create time. Everything that is also a
|
||||
ScannerOptions field lives here (with the normalized chunk_size); only
|
||||
scanner-owned bookkeeping stays as direct members below. */
|
||||
ScannerOptions options;
|
||||
Queue* directories;
|
||||
DIR* current_dir;
|
||||
char* current_path;
|
||||
bool use_metadata;
|
||||
unsigned long long chunk_size;
|
||||
char** exclude_patterns;
|
||||
int exclude_count;
|
||||
char** include_patterns;
|
||||
int include_count;
|
||||
unsigned long long max_size;
|
||||
unsigned long long min_size;
|
||||
int max_depth;
|
||||
int current_depth;
|
||||
dev_t root_dev;
|
||||
bool follow_symlinks;
|
||||
bool copy_links;
|
||||
bool safe_links;
|
||||
bool copy_unsafe_links;
|
||||
bool checksum;
|
||||
bool failed;
|
||||
/* rsync-order traversal: each opened directory's entries are inspected once
|
||||
and buffered (an internal SortedEntry[] owned here) sorted as rsync's flist
|
||||
orders them -- non-directories ascending, then directories ascending. The
|
||||
entries are walked in order and child directories are collected in
|
||||
`pending_dirs` (an ArrayList of DirEntry*, owned here) and pushed onto the
|
||||
LIFO `directories` stack in reverse at directory exhaustion, so the emitted
|
||||
stream is depth-first like rsync. `sorted_*` are reset per directory. */
|
||||
void* sorted_entries;
|
||||
size_t sorted_count;
|
||||
size_t sorted_index;
|
||||
void* pending_dirs;
|
||||
/* Recursive scan: whether the open directory yielded any transferred or
|
||||
descended entry. When it did not, closing it emits a directory entry so
|
||||
the empty source directory is recreated at the destination (rsync
|
||||
parity). */
|
||||
bool current_dir_produced;
|
||||
/* Phase 2 (files-from / filter layer). */
|
||||
char* root_path; /* transfer root (fs path) for rel computation */
|
||||
char* current_rel; /* rel path of the open directory ("" == root) */
|
||||
bool at_seed_dir; /* next open is the seed directory */
|
||||
FilterNode* seed_node; /* inherited context of the seed dir, or NULL */
|
||||
FilterNode* current_node; /* filter context of the open directory */
|
||||
ArrayList* filter_nodes; /* owned FilterNode arena (may be NULL) */
|
||||
/* --dirs / -R state for the directory-entry generator (options.dirs replaces
|
||||
the recursive scan). */
|
||||
bool relative_mode; /* file_list && relative: send bare relative wire paths */
|
||||
bool dirs_root_emitted;
|
||||
int list_index;
|
||||
ArrayList* dirs_batch; /* owned when non-NULL */
|
||||
unsigned long long dirs_batch_size;
|
||||
/* A directory could not be opened (I/O error, e.g. EACCES). With
|
||||
--ignore-errors the scan continues past it and the caller decides what to
|
||||
do; `failed` is reserved for fatal errors that always abort the scan. */
|
||||
bool io_error;
|
||||
/* The transfer ROOT could not be opened. It is always fatal, even under
|
||||
--ignore-errors, but the client still maps it to rsync's partial-transfer
|
||||
exit (23) rather than a generic failure. */
|
||||
bool root_io_error;
|
||||
} DirectoryScanner;
|
||||
|
||||
typedef struct {
|
||||
@@ -263,14 +59,9 @@ typedef struct {
|
||||
thrd_t* threads;
|
||||
bool done;
|
||||
bool failed;
|
||||
/* A worker skipped an unreadable directory under --ignore-errors (non-fatal). */
|
||||
bool io_error;
|
||||
atomic_bool cancelled;
|
||||
int completed;
|
||||
Chunk* initial_chunk;
|
||||
ProtocolSession* allocation_session;
|
||||
FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
|
||||
const ScannerOptions* options; /* borrowed scan options (--info=nonreg output) */
|
||||
} ParallelScanner;
|
||||
|
||||
DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata,
|
||||
@@ -286,33 +77,10 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner);
|
||||
bool directory_scanner_failed(const DirectoryScanner* scanner);
|
||||
void directory_scanner_destroy(DirectoryScanner* scanner);
|
||||
|
||||
/* --one-file-system (-x) decision: a directory entry may be descended into
|
||||
* only when the option is disabled or the entry lives on the same device as
|
||||
* the transfer root. Exposed so tests can exercise the rule directly. */
|
||||
bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device);
|
||||
|
||||
/* Relative path of an on-disk path below `root` ("" == the root itself, NULL
|
||||
* when `fs_path` is not under `root`). Handles trailing slashes and a root of
|
||||
* "/". Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_path_relative(const char* root, const char* fs_path);
|
||||
|
||||
/* -R/--relative destination-relative prefix reconstructed from a source spec:
|
||||
* the path after rsync's first '.' path component (the '/./' cut point), with
|
||||
* leading/trailing slashes removed, or the whole spec (normalized) when there
|
||||
* is no cut. Returns "" for the receive root, or NULL when `spec` is NULL or
|
||||
* allocation fails. Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_relative_prefix(const char* spec);
|
||||
|
||||
ParallelScanner* parallel_scanner_create_with_options(const char* root_directory,
|
||||
const ScannerOptions* options,
|
||||
ProtocolSession* allocation_session);
|
||||
const ScannerOptions* options);
|
||||
Chunk* parallel_scanner_next(ParallelScanner* scanner);
|
||||
bool parallel_scanner_failed(const ParallelScanner* scanner);
|
||||
bool parallel_scanner_had_io_error(const ParallelScanner* scanner);
|
||||
void parallel_scanner_destroy(ParallelScanner* scanner);
|
||||
|
||||
/* True when a directory could not be opened during the scan (an I/O error,
|
||||
recorded even when --ignore-errors keeps the scan going past it). */
|
||||
bool directory_scanner_had_io_error(const DirectoryScanner* scanner);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,793 +0,0 @@
|
||||
#include "log.h"
|
||||
#include "scanner.h"
|
||||
#include "scanner_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "file.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/sysmacros.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <limits.h>
|
||||
|
||||
#include "xattr.h"
|
||||
|
||||
/* A chain node: `own` holds the .rsync-filter rules of one directory, `parent`
|
||||
* the context that directory inherited (nearest ancestor with a filter file).
|
||||
* The chain for a directory's contents runs from that directory's own node up
|
||||
* to the root; the command-line base rules are evaluated after the whole
|
||||
* chain. */
|
||||
struct FilterNode {
|
||||
FilterNode* parent;
|
||||
FilterRuleList* own;
|
||||
};
|
||||
|
||||
void filter_node_destroy(void* item) {
|
||||
if (item) {
|
||||
FilterNode* node = (FilterNode*)item;
|
||||
if (node->own)
|
||||
filter_rule_list_free(node->own);
|
||||
free(node);
|
||||
}
|
||||
}
|
||||
|
||||
FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own) {
|
||||
FilterNode* node = malloc(sizeof(FilterNode));
|
||||
if (!node)
|
||||
return NULL;
|
||||
node->parent = parent;
|
||||
node->own = own;
|
||||
return node;
|
||||
}
|
||||
|
||||
/* Evaluate a rule chain for one entry. rsync precedence, highest first: the
|
||||
* innermost (current) directory's .rsync-filter rules, then each ancestor's,
|
||||
* then the root's, and finally the command-line base rules (--filter/-C). The
|
||||
* sender-side verdict decides whether the entry is hidden from the transfer;
|
||||
* the receiver-side verdict decides whether its destination mirror is protected
|
||||
* from --delete. Each side takes the FIRST matching rule independently. */
|
||||
typedef struct {
|
||||
bool hide; /* sender-side exclude matched */
|
||||
bool protect; /* receiver-side exclude matched */
|
||||
} FilterOutcome;
|
||||
|
||||
static void chain_rules_outcome(const FilterRuleList* base, const FilterNode* node, const char* rel,
|
||||
const char* leaf, bool is_dir, FilterOutcome* out) {
|
||||
memset(out, 0, sizeof(*out));
|
||||
bool sender_decided = false;
|
||||
bool receiver_decided = false;
|
||||
const FilterNode* n = node;
|
||||
while (!sender_decided || !receiver_decided) {
|
||||
const FilterRuleList* list = n ? n->own : base;
|
||||
if (list) {
|
||||
if (!sender_decided) {
|
||||
FilterAction action = filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_SENDER);
|
||||
if (action != FILTER_ACTION_NONE) {
|
||||
out->hide = action == FILTER_ACTION_EXCLUDE;
|
||||
sender_decided = true;
|
||||
}
|
||||
}
|
||||
if (!receiver_decided) {
|
||||
FilterAction action =
|
||||
filter_rules_apply_side(list, rel, leaf, is_dir, FILTER_SIDE_RECEIVER);
|
||||
if (action != FILTER_ACTION_NONE) {
|
||||
out->protect = action == FILTER_ACTION_PROTECT;
|
||||
receiver_decided = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!n)
|
||||
break;
|
||||
n = n->parent;
|
||||
}
|
||||
}
|
||||
|
||||
static bool entry_allowed(const FilterRuleList* base, const FilterNode* node, const char* rel,
|
||||
const char* leaf, bool is_dir, bool exclude_filter_files,
|
||||
bool* protect_out) {
|
||||
/* -FF: per-directory .rsync-filter files are never transferred (single -F
|
||||
transfers them, matching rsync). */
|
||||
if (exclude_filter_files && !is_dir && strcmp(leaf, ".rsync-filter") == 0) {
|
||||
if (protect_out)
|
||||
*protect_out = false;
|
||||
return false;
|
||||
}
|
||||
FilterOutcome outcome;
|
||||
chain_rules_outcome(base, node, rel, leaf, is_dir, &outcome);
|
||||
if (protect_out)
|
||||
*protect_out = outcome.protect;
|
||||
return !outcome.hide;
|
||||
}
|
||||
|
||||
void dir_entry_destroy(void* item) {
|
||||
if (item) {
|
||||
DirEntry* de = (DirEntry*)item;
|
||||
free(de->path);
|
||||
free(de);
|
||||
}
|
||||
}
|
||||
|
||||
DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context) {
|
||||
DirEntry* de = malloc(sizeof(DirEntry));
|
||||
if (!de)
|
||||
return NULL;
|
||||
de->path = str_dup(path);
|
||||
if (!de->path) {
|
||||
free(de);
|
||||
return NULL;
|
||||
}
|
||||
de->depth = depth;
|
||||
de->context = context;
|
||||
return de;
|
||||
}
|
||||
|
||||
/* Apply rsync's symlink-resolution precedence to one S_ISLNK entry:
|
||||
* --copy-links dereferences every symlink;
|
||||
* --copy-unsafe-links dereferences only targets unsafe_symlink() flags;
|
||||
* -k/--copy-dirlinks dereferences only a symlink whose referent is a dir;
|
||||
* --safe-links (receiver-side in rsync; modelled here) ignores an unsafe
|
||||
* target that would otherwise be carried; with --munge-links
|
||||
* every stored target becomes absolute, so --safe-links then
|
||||
* ignores every symlink, exactly as rsync documents;
|
||||
* -l/--links carries the link.
|
||||
* `link_rel` is the symlink's transfer-relative path (incl. name) and is used
|
||||
* only for the lexical unsafe test. `target` receives the raw link value. */
|
||||
LinkAction scanner_link_action(const ScannerOptions* options, const char* path,
|
||||
const char* link_rel, char* target, size_t target_size) {
|
||||
if (!options->follow_symlinks && !options->copy_links && !options->safe_links &&
|
||||
!options->copy_unsafe_links && !options->copy_dirlinks)
|
||||
return LINK_ACTION_SKIP;
|
||||
ssize_t length = readlink(path, target, target_size - 1);
|
||||
if (length < 0)
|
||||
return LINK_ACTION_SKIP;
|
||||
target[length] = '\0';
|
||||
|
||||
bool unsafe = file_symlink_unsafe(target, link_rel);
|
||||
if (options->copy_links || (options->copy_unsafe_links && unsafe))
|
||||
return LINK_ACTION_DEREF;
|
||||
if (options->copy_dirlinks) {
|
||||
struct stat ref;
|
||||
if (stat(path, &ref) == 0 && S_ISDIR(ref.st_mode))
|
||||
return LINK_ACTION_DEREF;
|
||||
}
|
||||
if (options->safe_links && (unsafe || options->munge_links))
|
||||
return LINK_ACTION_SKIP_PROTECTED;
|
||||
if (!options->follow_symlinks || target[0] == '\0')
|
||||
return LINK_ACTION_SKIP;
|
||||
return LINK_ACTION_CARRY;
|
||||
}
|
||||
|
||||
/* --one-file-system (-x) decision. Only directories can carry a different
|
||||
* device than their parent (mount points), so this is checked when a child
|
||||
* directory is about to be descended into. */
|
||||
bool scanner_same_filesystem(int one_file_system, dev_t root_device, dev_t entry_device) {
|
||||
return one_file_system <= 0 || entry_device == root_device;
|
||||
}
|
||||
|
||||
/* Build a payload-less directory File carrying the captured metadata (when
|
||||
* requested). Used by -x mount-point emission and --list-only directory
|
||||
* entries. Returns NULL on allocation failure. */
|
||||
File* scanner_build_dir_file(const char* path, const struct stat* stats,
|
||||
const ScannerOptions* options) {
|
||||
File* dir = file_create(path);
|
||||
if (dir == NULL)
|
||||
return NULL;
|
||||
dir->is_dir = true;
|
||||
if (options->use_metadata) {
|
||||
dir->metadata =
|
||||
file_metadata_create(dir->path, stats, options->preserve_atimes, options->preserve_crtimes);
|
||||
if (!dir->metadata) {
|
||||
file_destroy(dir);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
return dir;
|
||||
}
|
||||
|
||||
/* Relative path of an on-disk path below `root`. The transfer root may be
|
||||
* given with a trailing slash; the returned rel path never has one and is ""
|
||||
* for the root itself. A root of "/" is handled (its children start at "/").
|
||||
* Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_path_relative(const char* root, const char* fs_path) {
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, fs_path, root_len) != 0)
|
||||
return NULL;
|
||||
if (root_len == 1 && root[0] == '/') {
|
||||
if (fs_path[1] == '\0')
|
||||
return str_dup("");
|
||||
return str_dup(fs_path + 1);
|
||||
}
|
||||
if (fs_path[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (fs_path[root_len] != '/')
|
||||
return NULL;
|
||||
return str_dup(fs_path + root_len + 1);
|
||||
}
|
||||
|
||||
/* -R/--relative destination-relative prefix reconstructed from a source spec:
|
||||
* everything after the first '.' path component (rsync's '/./' cut point),
|
||||
* with leading/trailing slashes removed; or the whole spec (normalized) when
|
||||
* there is no cut. Returns "" for the receive root. Exposed for tests. */
|
||||
char* scanner_relative_prefix(const char* spec) {
|
||||
if (!spec || spec[0] == '\0')
|
||||
return NULL;
|
||||
const char* after = spec;
|
||||
if (spec[0] == '.' && spec[1] == '/') {
|
||||
after = spec + 2;
|
||||
} else {
|
||||
const char* cut = strstr(spec, "/./");
|
||||
if (cut)
|
||||
after = cut + 3;
|
||||
}
|
||||
size_t cap = strlen(spec) + 1;
|
||||
char* out = malloc(cap);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t len = 0;
|
||||
for (const char* s = after; *s;) {
|
||||
while (*s == '/')
|
||||
s++;
|
||||
const char* comp = s;
|
||||
while (*s && *s != '/')
|
||||
s++;
|
||||
size_t clen = (size_t)(s - comp);
|
||||
if (clen == 0 || (clen == 1 && comp[0] == '.'))
|
||||
continue;
|
||||
if (len)
|
||||
out[len++] = '/';
|
||||
memcpy(out + len, comp, clen);
|
||||
len += clen;
|
||||
}
|
||||
out[len] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Relative path of a child entry below the current directory. */
|
||||
char* child_rel_path(const char* parent_rel, const char* name) {
|
||||
if (!parent_rel || parent_rel[0] == '\0')
|
||||
return str_dup(name);
|
||||
return path_cat(parent_rel, name);
|
||||
}
|
||||
|
||||
/* Destination-relative wire path for an entry under an -R prefix. */
|
||||
char* scanner_prefix_send_path(const char* prefix, const char* rel) {
|
||||
if (prefix[0] == '\0')
|
||||
return str_dup(rel);
|
||||
if (rel[0] == '\0')
|
||||
return str_dup(prefix);
|
||||
return path_cat(prefix, rel);
|
||||
}
|
||||
|
||||
/* Apply the --files-from allow-set and the filter layer to one entry. On
|
||||
* return `*protect_out` is true when a receiver-side rule protects the entry's
|
||||
* destination mirror from deletion. */
|
||||
bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base,
|
||||
const FilterNode* node, const char* rel, const char* leaf, bool is_dir,
|
||||
bool per_dir_filters, bool exclude_filter_files, bool* protect_out) {
|
||||
if (protect_out)
|
||||
*protect_out = false;
|
||||
if (file_list && !file_list_affects(file_list, rel))
|
||||
return false;
|
||||
if (base || per_dir_filters)
|
||||
return entry_allowed(base, node, rel, leaf, is_dir, exclude_filter_files, protect_out);
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Best-effort capture of the file's whitelisted xattrs (-X/-A). A failure to
|
||||
* read xattrs is non-fatal: the file is transferred without them. A symlink
|
||||
* entry reads the LINK's own xattrs (never the referent's) with the no-follow
|
||||
* variant; on Linux the VFS refuses xattrs on symlinks, so that yields NULL. */
|
||||
void scanner_capture_xattrs_opts(const ScannerOptions* options, File* file) {
|
||||
if (!options || !file || !(options->preserve_xattrs || options->preserve_acls))
|
||||
return;
|
||||
file->xattrs = file->is_symlink ? xattr_capture_path_nofollow(file->path, options->preserve_acls)
|
||||
: xattr_capture_path(file->path, options->preserve_acls);
|
||||
}
|
||||
|
||||
void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file) {
|
||||
if (!scanner)
|
||||
return;
|
||||
scanner_capture_xattrs_opts(&scanner->options, file);
|
||||
}
|
||||
|
||||
/* Apply --hard-links (-H) detection to one regular File. On a sibling (a
|
||||
* later member of an already-seen source inode) the File keeps the group id
|
||||
* and the first member's wire path but carries NO data payload (size 0); the
|
||||
* first member is left untouched (data present, link_first). Returns false on
|
||||
* allocation failure (the caller marks the scan failed); the File stays usable
|
||||
* either way. */
|
||||
bool scanner_assign_hardlink(HardLinkTable* table, File* file, const struct stat* stats) {
|
||||
if (!table || !file || !stats)
|
||||
return true;
|
||||
int gid;
|
||||
bool is_first;
|
||||
char* first_path = NULL;
|
||||
if (!hardlink_table_assign(table, file_wire_path(file), stats->st_dev, stats->st_ino, &gid,
|
||||
&is_first, &first_path))
|
||||
return false;
|
||||
file->link_group = gid;
|
||||
file->link_first = is_first;
|
||||
if (!is_first) {
|
||||
file->hardlink_target = first_path;
|
||||
file->data->size = 0;
|
||||
} else {
|
||||
free(first_path);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Phase 4 special/devices decision for one non-regular entry, matching rsync:
|
||||
- a char/block device is RECREATED as a node under -D/--devices, unless
|
||||
--copy-devices asks for its content to be copied into a regular file;
|
||||
- a FIFO/socket is RECREATED under --specials;
|
||||
- when the matching flag is absent the entry is SKIPPED ("skipping
|
||||
non-regular file"), exactly like rsync's default, instead of being
|
||||
silently copied as a zero-length regular file;
|
||||
- anything else (regular/directory) is left to the normal data path. */
|
||||
ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials,
|
||||
bool copy_devices, File* file, const struct stat* stats) {
|
||||
if (!file || !stats)
|
||||
return SCANNER_SPECIAL_REGULAR;
|
||||
bool is_device = S_ISCHR(stats->st_mode) || S_ISBLK(stats->st_mode);
|
||||
bool is_fifo = S_ISFIFO(stats->st_mode);
|
||||
bool is_socket = S_ISSOCK(stats->st_mode);
|
||||
if (!is_device && !is_fifo && !is_socket)
|
||||
return SCANNER_SPECIAL_REGULAR;
|
||||
if (is_device && copy_devices)
|
||||
return SCANNER_SPECIAL_REGULAR; /* copy device content as a regular file */
|
||||
bool preserve = is_device ? preserve_devices : preserve_specials;
|
||||
if (!preserve)
|
||||
return SCANNER_SPECIAL_SKIP;
|
||||
file->is_special = true;
|
||||
file->data->size = 0;
|
||||
file->data->data = NULL;
|
||||
if (is_device) {
|
||||
file->rdev_major = (int32_t)major(stats->st_rdev);
|
||||
file->rdev_minor = (int32_t)minor(stats->st_rdev);
|
||||
}
|
||||
return SCANNER_SPECIAL_RECREATE;
|
||||
}
|
||||
|
||||
/* Append `rel` to the caller's exclusion sink, taking `mtx` when shared across
|
||||
parallel worker threads. Returns false on allocation failure (list left
|
||||
unchanged). */
|
||||
bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel) {
|
||||
if (!list)
|
||||
return true;
|
||||
char* dup = str_dup(rel);
|
||||
if (!dup)
|
||||
return false;
|
||||
if (mtx)
|
||||
mtx_lock(mtx);
|
||||
bool ok = array_list_add(list, dup);
|
||||
if (mtx)
|
||||
mtx_unlock(mtx);
|
||||
if (!ok)
|
||||
free(dup);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Record one pruned filesystem path in a delete-protection sink. The stored
|
||||
form is the entry's wire/destination-relative path (a single leading '/'
|
||||
removed, exactly how manifest keep entries are stored), so the receiver's
|
||||
walker prefixes match the destination layout. An allocation failure is a
|
||||
fatal scan error. */
|
||||
static void scanner_record_protected(DirectoryScanner* scanner, const char* fs_path,
|
||||
ArrayList* sink) {
|
||||
if (!sink || !fs_path)
|
||||
return;
|
||||
const char* rel = *fs_path == '/' ? fs_path + 1 : fs_path;
|
||||
if (!excluded_sink_append(sink, scanner->options.excluded_mutex, rel))
|
||||
scanner->failed = true;
|
||||
}
|
||||
|
||||
/* rsync's `--info=nonreg` line for a non-regular entry that is not being
|
||||
* preserved: `skipping non-regular file "NAME"`. The name is the path relative
|
||||
* to the transfer root, so it matches rsync's displayed name. */
|
||||
void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path) {
|
||||
if (!options || !options->note_nonreg || !fs_path)
|
||||
return;
|
||||
const char* rel = utils_strip_transfer_root(fs_path, options->send_directory);
|
||||
char* escaped = output_escape(rel, options->eight_bit_output);
|
||||
printf("skipping non-regular file \"%s\"\n", escaped ? escaped : rel);
|
||||
free(escaped);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
/* Construct one non-directory File from an inspected entry. Shared by the
|
||||
* sequential and parallel scanners so entry construction has a single
|
||||
* implementation: data size (or carried symlink), -R wire path, special/devices
|
||||
* classification, hardlink group, metadata and xattr capture all happen here in
|
||||
* the same order for both. See the declaration for the ownership contract. */
|
||||
ScannerBuildStatus scanner_build_file_entry(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, File** out_file, bool* failed) {
|
||||
*out_file = NULL;
|
||||
if (failed)
|
||||
*failed = false;
|
||||
File* file = file_create(inspected->path);
|
||||
if (!file) {
|
||||
/* The File never existed, so drop the not-yet-transferred symlink target
|
||||
here; the caller's entry teardown would otherwise double-free it. */
|
||||
free(inspected->link_target);
|
||||
inspected->link_target = NULL;
|
||||
return SCANNER_BUILD_FAIL_CONTINUE;
|
||||
}
|
||||
if (inspected->is_symlink) {
|
||||
file->is_symlink = true;
|
||||
file->symlink_target = inspected->link_target;
|
||||
inspected->link_target = NULL;
|
||||
} else {
|
||||
file->data->size = inspected->stats.st_size;
|
||||
}
|
||||
/* -R + --files-from uses the bare transfer-relative path; -R without
|
||||
--files-from prefixes it. Plain scans keep the source path. */
|
||||
bool relative_mode = options->relative && options->file_list != NULL;
|
||||
if (relative_mode) {
|
||||
file->send_path = str_dup(rel);
|
||||
} else if (options->relative_prefix) {
|
||||
file->send_path = scanner_prefix_send_path(options->relative_prefix, rel);
|
||||
}
|
||||
if ((relative_mode || options->relative_prefix) && !file->send_path) {
|
||||
file_destroy(file);
|
||||
return SCANNER_BUILD_FAIL_BREAK;
|
||||
}
|
||||
/* --devices/--specials: a device/FIFO/socket entry marked for preservation
|
||||
becomes a node to recreate (is_special, no data, rdev captured); an
|
||||
unrequested non-regular entry is skipped (rsync default). */
|
||||
ScannerSpecial special =
|
||||
scanner_prepare_special(options->preserve_devices, options->preserve_specials,
|
||||
options->copy_devices, file, &inspected->stats);
|
||||
if (special == SCANNER_SPECIAL_SKIP) {
|
||||
scanner_note_nonreg(options, file->path);
|
||||
file_destroy(file);
|
||||
return SCANNER_BUILD_SKIP;
|
||||
}
|
||||
if (options->hardlinks && S_ISREG(inspected->stats.st_mode) &&
|
||||
!scanner_assign_hardlink(options->hardlinks, file, &inspected->stats)) {
|
||||
/* Allocation failure is non-fatal to this entry (it is still emitted) but
|
||||
marks the scan failed, matching the historical inlined behaviour. */
|
||||
if (failed)
|
||||
*failed = true;
|
||||
}
|
||||
if (options->use_metadata) {
|
||||
file->metadata = file_metadata_create(file->path, &inspected->stats, options->preserve_atimes,
|
||||
options->preserve_crtimes);
|
||||
if (!file->metadata) {
|
||||
file_destroy(file);
|
||||
return SCANNER_BUILD_FAIL_BREAK;
|
||||
}
|
||||
}
|
||||
/* A hardlink sibling carries no data, so it carries no xattrs. */
|
||||
if (!(file->link_group != 0 && !file->link_first))
|
||||
scanner_capture_xattrs_opts(options, file);
|
||||
*out_file = file;
|
||||
return SCANNER_BUILD_OK;
|
||||
}
|
||||
|
||||
/* rsync 3.4.1's `--info=mount` line, emitted when `-xx` drops a mount-point
|
||||
* directory: `[sender] skipping mount-point dir NAME` (the client is the
|
||||
* sender). Plain `-x` keeps the empty directory and prints nothing, matching
|
||||
* rsync. */
|
||||
void scanner_note_mount(const ScannerOptions* options, const char* fs_path) {
|
||||
if (!options || !options->note_mount || !fs_path)
|
||||
return;
|
||||
const char* rel = utils_strip_transfer_root(fs_path, options->send_directory);
|
||||
char* escaped = output_escape(rel, options->eight_bit_output);
|
||||
printf("[sender] skipping mount-point dir %s\n", escaped ? escaped : rel);
|
||||
free(escaped);
|
||||
fflush(stdout);
|
||||
}
|
||||
|
||||
/* --debug=filter: a selection/filter decision dropped an entry. */
|
||||
void scanner_note_filter(const ScannerOptions* options, const char* name) {
|
||||
if (!options || !log_debug_enabled(LOG_DEBUG_FILTER) || !name)
|
||||
return;
|
||||
log_debug_message(LOG_DEBUG_FILTER, "filter: excluded %s", name);
|
||||
}
|
||||
|
||||
/* Account for a directory that will not be represented by an inline directory
|
||||
* entry. Paired with scanner_dir_count_uncount for empty directories that are
|
||||
* emitted inline, so every traversed directory is counted exactly once. */
|
||||
void scanner_dir_count_count(const ScannerOptions* options) {
|
||||
if (options && options->dir_count)
|
||||
atomic_fetch_add(options->dir_count, 1);
|
||||
}
|
||||
|
||||
void scanner_dir_count_uncount(const ScannerOptions* options) {
|
||||
if (options && options->dir_count)
|
||||
atomic_fetch_sub(options->dir_count, 1);
|
||||
}
|
||||
|
||||
/* A user-selection exclusion (--filter/-C/per-dir or --exclude/--include). */
|
||||
void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path) {
|
||||
scanner_record_protected(scanner, fs_path, scanner->options.excluded_paths);
|
||||
}
|
||||
|
||||
/* A --max-size/--min-size prune (always protected, even under --delete-excluded). */
|
||||
void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path) {
|
||||
scanner_record_protected(scanner, fs_path, scanner->options.size_skipped_paths);
|
||||
}
|
||||
|
||||
/* The destination-relative coordinate the receiver's delete walkers match
|
||||
against for an entry at `fs_path` (with `rel` its path relative to the
|
||||
transfer root, "" for the root): `relative_prefix + rel` under -R+--relative,
|
||||
the bare relative path under -R+--files-from, else the source path with a
|
||||
leading '/' removed, with "." for the receive root. Shared by the
|
||||
synchronized-directory sink and the mirrored per-directory rule owners so
|
||||
both live in the same coordinate system. Returns an owned string, or NULL on
|
||||
allocation failure. */
|
||||
char* scanner_dest_rel_path(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode) {
|
||||
char* prefixed = NULL;
|
||||
const char* dest;
|
||||
if (relative_mode) {
|
||||
dest = rel;
|
||||
} else if (options->relative_prefix) {
|
||||
prefixed = scanner_prefix_send_path(options->relative_prefix, rel);
|
||||
if (!prefixed)
|
||||
return NULL;
|
||||
dest = prefixed;
|
||||
} else {
|
||||
dest = fs_path;
|
||||
}
|
||||
if (dest[0] == '/')
|
||||
dest++;
|
||||
if (dest[0] == '\0')
|
||||
dest = ".";
|
||||
char* out = str_dup(dest);
|
||||
free(prefixed);
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Record a directory the scan synchronized. `fs_path` is its absolute path and
|
||||
`rel` its path relative to the transfer root ("" for the root); the stored
|
||||
form matches the wire layout (see scanner_dest_rel_path). Returns false on
|
||||
allocation failure. */
|
||||
bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode) {
|
||||
if (!options->synced_dirs && !options->plan_dirs)
|
||||
return true;
|
||||
if (!file_list_dir_in_scope(options->file_list, rel))
|
||||
return true;
|
||||
char* dest = scanner_dest_rel_path(options, fs_path, rel, relative_mode);
|
||||
if (!dest)
|
||||
return false;
|
||||
bool ok = true;
|
||||
if (options->synced_dirs)
|
||||
ok = excluded_sink_append(options->synced_dirs, options->excluded_mutex, dest);
|
||||
/* The delete-plan keep set needs an entry for every traversed source
|
||||
directory, including empty ones, so its destination mirror is kept rather
|
||||
than deleted as an extra; the receive root (".") is implicit. */
|
||||
if (ok && options->plan_dirs && strcmp(dest, ".") != 0)
|
||||
ok = excluded_sink_append(options->plan_dirs, options->excluded_mutex, dest);
|
||||
free(dest);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Read every per-directory filter file that applies to `dir_path` (its
|
||||
* .rsync-filter when -F is active, plus each registered "dir-merge NAME") into a
|
||||
* fresh list. Returns NULL on allocation/parse failure (message in `err`);
|
||||
* returns an empty list (and *any_exists=false) when no file exists. */
|
||||
FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path,
|
||||
const char* rel, bool relative_mode, bool* any_exists, char* err,
|
||||
size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
const FilterRuleList* base = options->base_filters;
|
||||
bool have_names = options->per_dir_filters || (base && base->dir_merge_count > 0);
|
||||
if (any_exists)
|
||||
*any_exists = false;
|
||||
if (!have_names)
|
||||
return NULL;
|
||||
FilterRuleList* own = filter_rule_list_create();
|
||||
if (!own) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
FilterParseOptions opts = {.delete_excluded = options->delete_excluded, .cvs_exclude = false};
|
||||
bool exists = false;
|
||||
if (options->per_dir_filters) {
|
||||
if (!filter_file_append(own, dir_path, ".rsync-filter", rel, &opts, &exists, err, err_size))
|
||||
goto fail;
|
||||
if (exists && any_exists)
|
||||
*any_exists = true;
|
||||
}
|
||||
if (base) {
|
||||
for (int i = 0; i < base->dir_merge_count; i++) {
|
||||
if (!filter_dir_merge_append(own, dir_path, &base->dir_merges[i], rel, &opts, &exists, err,
|
||||
err_size))
|
||||
goto fail;
|
||||
if (exists && any_exists)
|
||||
*any_exists = true;
|
||||
}
|
||||
}
|
||||
/* Mirror the directory's rules into the delete-carrier sink so the receiver
|
||||
* can reconstruct its per-directory protect/risk set. The mirrored rules
|
||||
* carry the destination-relative owner coordinate (not the transfer-root-
|
||||
* relative one the sender's own evaluation uses) so the receiver's delete
|
||||
* walkers, which match against receive-root-relative paths, find them. */
|
||||
if (options->per_dir_rules && own->count > 0) {
|
||||
char* owner = scanner_dest_rel_path(options, dir_path, rel, relative_mode);
|
||||
if (!owner)
|
||||
goto fail;
|
||||
mtx_t* mtx = options->excluded_mutex;
|
||||
if (mtx)
|
||||
mtx_lock(mtx);
|
||||
for (int i = 0; i < own->count; i++) {
|
||||
FilterRule* copy = filter_rule_clone(own->items[i]);
|
||||
if (!copy || !filter_rule_set_owner(copy, owner) ||
|
||||
!filter_rule_list_add(options->per_dir_rules, copy)) {
|
||||
filter_rule_free(copy);
|
||||
if (mtx)
|
||||
mtx_unlock(mtx);
|
||||
free(owner);
|
||||
goto fail;
|
||||
}
|
||||
}
|
||||
if (mtx)
|
||||
mtx_unlock(mtx);
|
||||
free(owner);
|
||||
}
|
||||
return own;
|
||||
fail:
|
||||
filter_rule_list_free(own);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Merge the open directory's own per-directory filter files (the default
|
||||
* .rsync-filter when -F is active, plus every "dir-merge NAME" registered on the
|
||||
* base rule list) into the inherited context, returning the context used for
|
||||
* this directory's entries. On a parse error the scanner is marked failed.
|
||||
* Returns 0 on success, -1 on failure. */
|
||||
int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited) {
|
||||
char err[256];
|
||||
bool any_exists = false;
|
||||
FilterRuleList* own = read_dir_filters(&scanner->options, scanner->current_path,
|
||||
scanner->current_rel ? scanner->current_rel : "",
|
||||
scanner->relative_mode, &any_exists, err, sizeof(err));
|
||||
if (!own) {
|
||||
/* read_dir_filters() leaves `err` set on a parse/allocation failure even
|
||||
when an earlier merge file in the same directory existed (any_exists true);
|
||||
key off the error text rather than any_exists so an invalid per-directory
|
||||
filter file can never be silently ignored. */
|
||||
if (err[0] == '\0') {
|
||||
scanner->current_node = (FilterNode*)inherited;
|
||||
return 0;
|
||||
}
|
||||
char* escaped_path = output_escape(scanner->current_path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", err);
|
||||
free(escaped_path);
|
||||
scanner->failed = true;
|
||||
return -1;
|
||||
}
|
||||
if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) {
|
||||
FilterNode* node = filter_node_alloc((FilterNode*)inherited, own);
|
||||
if (!node || !array_list_add(scanner->filter_nodes, node)) {
|
||||
filter_node_destroy(node);
|
||||
scanner->failed = true;
|
||||
return -1;
|
||||
}
|
||||
scanner->current_node = node;
|
||||
} else {
|
||||
filter_rule_list_free(own);
|
||||
scanner->current_node = (FilterNode*)inherited;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Inspect symlinks, resolve the entry type, and apply file filters once for both scanners.
|
||||
* `link_rel` is the entry's path relative to the transfer root (including its
|
||||
* name), used for the lexical rsync unsafe-symlink test. */
|
||||
int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir,
|
||||
const char* link_rel, const char* name, ScannerEntry* entry) {
|
||||
entry->excluded = false;
|
||||
entry->size_excluded = false;
|
||||
entry->referent_error = false;
|
||||
entry->is_symlink = false;
|
||||
entry->link_target = NULL;
|
||||
entry->path = path_cat(containing_dir, name);
|
||||
if (!entry->path)
|
||||
return -1;
|
||||
|
||||
struct stat link_stats;
|
||||
if (lstat(entry->path, &link_stats) != 0) {
|
||||
free(entry->path);
|
||||
return 0;
|
||||
}
|
||||
if (!S_ISLNK(link_stats.st_mode))
|
||||
goto regular;
|
||||
|
||||
char link_target[4096];
|
||||
switch (scanner_link_action(options, entry->path, link_rel, link_target, sizeof(link_target))) {
|
||||
case LINK_ACTION_SKIP:
|
||||
goto skip;
|
||||
case LINK_ACTION_SKIP_PROTECTED:
|
||||
/* --safe-links ignored the link, but rsync still counts it as present in
|
||||
the transfer, so its destination mirror survives --delete. Record it as
|
||||
an excluded path (the same delete-protection channel as a filter prune). */
|
||||
entry->excluded = true;
|
||||
goto skip;
|
||||
case LINK_ACTION_DEREF:
|
||||
if (stat(entry->path, &entry->stats) != 0) {
|
||||
/* rsync reports "symlink has no referent" and continues with a partial
|
||||
transfer (exit 23); record the error so the run exits 23 too. */
|
||||
char* escaped = output_escape(entry->path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "symlink has no referent: %s",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
entry->referent_error = true;
|
||||
goto skip;
|
||||
}
|
||||
entry->is_directory = S_ISDIR(entry->stats.st_mode);
|
||||
if (entry->is_directory)
|
||||
return 1;
|
||||
goto apply_filters;
|
||||
case LINK_ACTION_CARRY:
|
||||
break;
|
||||
}
|
||||
|
||||
/* Carry the link as a symlink. --munge-links is applied by the RECEIVER (it
|
||||
prefixes every stored target with /rsyncd-munged/); when the SOURCE already
|
||||
holds a munged value the sender strips it so the receiver re-munges a clean
|
||||
target, round-tripping a munged tree exactly like rsync. */
|
||||
entry->is_symlink = true;
|
||||
entry->stats = link_stats;
|
||||
entry->is_directory = false;
|
||||
entry->link_target = str_dup(link_target);
|
||||
if (!entry->link_target)
|
||||
goto skip;
|
||||
if (options->munge_links)
|
||||
file_symlink_unmunge(entry->link_target);
|
||||
goto apply_filters;
|
||||
|
||||
regular:
|
||||
/* Not a symlink: the lstat() above already described this entry, and lstat
|
||||
and stat are identical for every non-symlink, so reuse that result instead
|
||||
of issuing a redundant stat() on the scanner hot path. stat() is still
|
||||
used on the dereference paths above/below for actual symlinks (copy-links,
|
||||
safe/copy-unsafe links, and -k symlinks-to-directories). */
|
||||
entry->stats = link_stats;
|
||||
entry->is_directory = S_ISDIR(link_stats.st_mode);
|
||||
if (entry->is_directory)
|
||||
return 1;
|
||||
|
||||
apply_filters:
|
||||
for (int i = 0; i < options->exclude_count; i++)
|
||||
if (glob_match(options->exclude_patterns[i], name)) {
|
||||
entry->excluded = true;
|
||||
goto skip;
|
||||
}
|
||||
if (options->include_count > 0) {
|
||||
bool included = false;
|
||||
for (int i = 0; i < options->include_count; i++)
|
||||
if (glob_match(options->include_patterns[i], name))
|
||||
included = true;
|
||||
if (!included) {
|
||||
entry->excluded = true;
|
||||
goto skip;
|
||||
}
|
||||
}
|
||||
if ((options->max_size > 0 && (unsigned long long)entry->stats.st_size > options->max_size) ||
|
||||
(options->min_size > 0 && (unsigned long long)entry->stats.st_size < options->min_size)) {
|
||||
entry->excluded = true;
|
||||
entry->size_excluded = true;
|
||||
goto skip;
|
||||
}
|
||||
return 1;
|
||||
|
||||
skip:
|
||||
free(entry->path);
|
||||
entry->path = NULL;
|
||||
free(entry->link_target);
|
||||
entry->link_target = NULL;
|
||||
return 0;
|
||||
}
|
||||
@@ -1,132 +0,0 @@
|
||||
#ifndef SCANNER_INTERNAL_H
|
||||
#define SCANNER_INTERNAL_H
|
||||
|
||||
/* Internal declarations shared between the scanner translation units
|
||||
* (scanner_filter.c, scanner.c, scanner_parallel.c). Nothing here is part of
|
||||
* the public scanner façade (scanner.h); every symbol stays internal to the
|
||||
* client module. */
|
||||
|
||||
#include "array_list.h"
|
||||
#include "file.h"
|
||||
#include "scanner.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
typedef struct {
|
||||
char* path;
|
||||
int depth;
|
||||
FilterNode* context; /* inherited per-directory filter context */
|
||||
} DirEntry;
|
||||
|
||||
/* How rsync's readlink_stat()/generator resolves one source symlink. */
|
||||
typedef enum {
|
||||
LINK_ACTION_SKIP, /* not transferred (no link option) */
|
||||
LINK_ACTION_SKIP_PROTECTED, /* ignored as unsafe by --safe-links; rsync keeps
|
||||
it in the transfer, so its destination mirror
|
||||
must be protected from --delete */
|
||||
LINK_ACTION_DEREF, /* follow the referent (--copy-links, an unsafe
|
||||
target under --copy-unsafe-links, or -k dir) */
|
||||
LINK_ACTION_CARRY, /* transmit the link itself (-l) */
|
||||
} LinkAction;
|
||||
|
||||
typedef struct {
|
||||
char* path;
|
||||
struct stat stats;
|
||||
bool is_directory;
|
||||
/* True when the entry should be carried through as a SYMLINK (is_symlink)
|
||||
rather than a dereferenced file/directory. When true, `link_target` holds
|
||||
the owned target string to transmit (sender-munged under --munge-links);
|
||||
ownership transfers to the File built from this entry. */
|
||||
bool is_symlink;
|
||||
char* link_target;
|
||||
/* True when the entry was pruned by a user selection rule (--filter/-C/per-dir
|
||||
rules or the --exclude/--include layer) rather than skipped for another
|
||||
reason (unreadable, symlink policy, not applicable). */
|
||||
bool excluded;
|
||||
/* True when the entry was skipped specifically by --max-size/--min-size.
|
||||
Size pruning protects the destination mirror even under --delete-excluded,
|
||||
so it is recorded into a separate sink from `excluded`. */
|
||||
bool size_excluded;
|
||||
/* True when a symlink selected for dereferencing (-L/--copy-links or an
|
||||
unsafe target under --copy-unsafe-links) had no usable referent (a broken
|
||||
link or a stat() failure). rsync still reports this as a partial transfer
|
||||
(exit 23) even though the entry is skipped, so the scanner records it as a
|
||||
non-fatal I/O error. */
|
||||
bool referent_error;
|
||||
} ScannerEntry;
|
||||
|
||||
typedef enum {
|
||||
SCANNER_SPECIAL_REGULAR, /* ordinary file: transfer content */
|
||||
SCANNER_SPECIAL_RECREATE, /* is_special node to recreate on the receiver */
|
||||
SCANNER_SPECIAL_SKIP, /* non-regular entry not requested: skip */
|
||||
} ScannerSpecial;
|
||||
|
||||
/* Result of scanner_build_file_entry(). The two failure variants preserve the
|
||||
* sequential scanner's historical distinction between a failure before the
|
||||
* File existed (which kept walking the directory) and one afterwards (which cut
|
||||
* the chunk short); both mark the scan failed. */
|
||||
typedef enum {
|
||||
SCANNER_BUILD_OK, /* File built; caller owns it */
|
||||
SCANNER_BUILD_SKIP, /* non-regular entry not preserved; no File */
|
||||
SCANNER_BUILD_FAIL_CONTINUE, /* failed before the File existed */
|
||||
SCANNER_BUILD_FAIL_BREAK, /* failed after the File existed */
|
||||
} ScannerBuildStatus;
|
||||
|
||||
/* scanner_filter.c */
|
||||
void filter_node_destroy(void* item);
|
||||
FilterNode* filter_node_alloc(FilterNode* parent, FilterRuleList* own);
|
||||
void dir_entry_destroy(void* item);
|
||||
DirEntry* dir_entry_create(const char* path, int depth, FilterNode* context);
|
||||
LinkAction scanner_link_action(const ScannerOptions* options, const char* path,
|
||||
const char* link_rel, char* target, size_t target_size);
|
||||
File* scanner_build_dir_file(const char* path, const struct stat* stats,
|
||||
const ScannerOptions* options);
|
||||
char* child_rel_path(const char* parent_rel, const char* name);
|
||||
char* scanner_prefix_send_path(const char* prefix, const char* rel);
|
||||
char* scanner_dest_rel_path(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode);
|
||||
bool entry_passes_selection(const FileListSet* file_list, const FilterRuleList* base,
|
||||
const FilterNode* node, const char* rel, const char* leaf, bool is_dir,
|
||||
bool per_dir_filters, bool exclude_filter_files, bool* protect_out);
|
||||
void scanner_capture_xattrs(const DirectoryScanner* scanner, File* file);
|
||||
void scanner_capture_xattrs_opts(const ScannerOptions* options, File* file);
|
||||
bool scanner_assign_hardlink(HardLinkTable* table, File* file, const struct stat* stats);
|
||||
ScannerSpecial scanner_prepare_special(bool preserve_devices, bool preserve_specials,
|
||||
bool copy_devices, File* file, const struct stat* stats);
|
||||
/* Build one non-directory transfer File from an inspected entry. `rel` is the
|
||||
* entry's transfer-root-relative path (used for the -R wire path); `inspected`
|
||||
* supplies the on-disk path, stats and (for a carried symlink) the target whose
|
||||
* ownership transfers to the File. Populates data size, send_path, special-node
|
||||
* state, hardlink group, metadata and xattrs. On SCANNER_BUILD_OK the caller
|
||||
* owns *out_file; on SCANNER_BUILD_SKIP it is NULL and the entry is dropped; on
|
||||
* either failure it is NULL and the caller must mark the scan failed. `*failed`
|
||||
* additionally reports a non-fatal hardlink-table allocation failure, in which
|
||||
* case a usable File is still returned. */
|
||||
ScannerBuildStatus scanner_build_file_entry(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, File** out_file, bool* failed);
|
||||
bool excluded_sink_append(ArrayList* list, mtx_t* mtx, const char* rel);
|
||||
void scanner_note_nonreg(const ScannerOptions* options, const char* fs_path);
|
||||
void scanner_note_mount(const ScannerOptions* options, const char* fs_path);
|
||||
void scanner_note_filter(const ScannerOptions* options, const char* name);
|
||||
void scanner_dir_count_count(const ScannerOptions* options);
|
||||
void scanner_dir_count_uncount(const ScannerOptions* options);
|
||||
void scanner_record_excluded(DirectoryScanner* scanner, const char* fs_path);
|
||||
void scanner_record_size_skipped(DirectoryScanner* scanner, const char* fs_path);
|
||||
bool scanner_record_synced_dir(const ScannerOptions* options, const char* fs_path, const char* rel,
|
||||
bool relative_mode);
|
||||
FilterRuleList* read_dir_filters(const ScannerOptions* options, const char* dir_path,
|
||||
const char* rel, bool relative_mode, bool* any_exists, char* err,
|
||||
size_t err_size);
|
||||
int open_directory_filter_context(DirectoryScanner* scanner, const FilterNode* inherited);
|
||||
int scanner_inspect_entry(const ScannerOptions* options, const char* containing_dir,
|
||||
const char* link_rel, const char* name, ScannerEntry* entry);
|
||||
|
||||
/* scanner.c */
|
||||
bool scanner_capture_dir_time(ArrayList* dir_entries, mtx_t* mutex, const char* root_path,
|
||||
const char* fs_path, bool relative_mode, const char* relative_prefix,
|
||||
bool preserve_atimes, bool preserve_crtimes, bool preserve_xattrs,
|
||||
bool preserve_acls, bool no_implied_dirs,
|
||||
const FileListSet* file_list);
|
||||
|
||||
#endif
|
||||
@@ -1,681 +0,0 @@
|
||||
#include "log.h"
|
||||
#include "scanner.h"
|
||||
#include "scanner_internal.h"
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "file.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/sysmacros.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <limits.h>
|
||||
|
||||
#include "xattr.h"
|
||||
|
||||
typedef struct {
|
||||
ParallelScanner* ps;
|
||||
char** dirs;
|
||||
int dir_count;
|
||||
char* root_dir; /* the transfer root, for relative-path computation */
|
||||
ScannerOptions options;
|
||||
ProtocolSession* allocation_session;
|
||||
} ParallelWorkerArg;
|
||||
|
||||
static int parallel_worker_thread(void* arg) {
|
||||
ParallelWorkerArg* wa = (ParallelWorkerArg*)arg;
|
||||
ProtocolSession* allocation_session = wa->allocation_session;
|
||||
if (allocation_session)
|
||||
protocol_session_bind(allocation_session);
|
||||
for (int i = 0; i < wa->dir_count; i++) {
|
||||
DirectoryScanner* ds = directory_scanner_create_with_options(wa->dirs[i], &wa->options);
|
||||
if (!ds) {
|
||||
mtx_lock(&wa->ps->result_mutex);
|
||||
wa->ps->failed = true;
|
||||
atomic_store(&wa->ps->cancelled, true);
|
||||
cnd_broadcast(&wa->ps->result_not_empty);
|
||||
cnd_broadcast(&wa->ps->result_not_full);
|
||||
mtx_unlock(&wa->ps->result_mutex);
|
||||
for (int j = i; j < wa->dir_count; j++)
|
||||
free(wa->dirs[j]);
|
||||
break;
|
||||
}
|
||||
/* Root .rsync-filter rules (parsed by the parallel scanner) apply to the
|
||||
* contents of every assigned subdirectory. Relative paths (used by the
|
||||
* allow-set and per-directory rules) are computed against the transfer
|
||||
* root, not the subdirectory the worker is seeded with. Exclusion
|
||||
* recording shares one caller-owned list across the workers. */
|
||||
free(ds->root_path);
|
||||
ds->root_path = str_dup(wa->root_dir);
|
||||
ds->seed_node = wa->ps->root_filter_node;
|
||||
ds->options.excluded_mutex = &wa->ps->result_mutex;
|
||||
Chunk* chunk;
|
||||
while ((chunk = directory_scanner_next(ds)) != NULL) {
|
||||
if (!queue_enqueue_multithreaded_cancel(wa->ps->result_queue, chunk, &wa->ps->result_mutex,
|
||||
&wa->ps->result_not_empty, &wa->ps->result_not_full,
|
||||
&wa->ps->cancelled)) {
|
||||
chunk_destroy(chunk);
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (directory_scanner_failed(ds)) {
|
||||
mtx_lock(&wa->ps->result_mutex);
|
||||
wa->ps->failed = true;
|
||||
atomic_store(&wa->ps->cancelled, true);
|
||||
cnd_broadcast(&wa->ps->result_not_empty);
|
||||
cnd_broadcast(&wa->ps->result_not_full);
|
||||
mtx_unlock(&wa->ps->result_mutex);
|
||||
} else if (directory_scanner_had_io_error(ds)) {
|
||||
/* --ignore-errors path: an unreadable directory was skipped, not fatal. */
|
||||
mtx_lock(&wa->ps->result_mutex);
|
||||
wa->ps->io_error = true;
|
||||
mtx_unlock(&wa->ps->result_mutex);
|
||||
}
|
||||
directory_scanner_destroy(ds);
|
||||
free(wa->dirs[i]);
|
||||
}
|
||||
ParallelScanner* ps = wa->ps;
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
mtx_lock(&ps->result_mutex);
|
||||
ps->completed++;
|
||||
if (ps->completed >= ps->expected_threads) {
|
||||
ps->done = true;
|
||||
cnd_signal(&ps->result_not_empty);
|
||||
}
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
if (allocation_session)
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
|
||||
static void parallel_scanner_creation_failed(ParallelScanner* ps) {
|
||||
mtx_lock(&ps->result_mutex);
|
||||
ps->failed = true;
|
||||
atomic_store(&ps->cancelled, true);
|
||||
ps->expected_threads = ps->created_threads;
|
||||
if (ps->completed >= ps->expected_threads)
|
||||
ps->done = true;
|
||||
cnd_broadcast(&ps->result_not_empty);
|
||||
cnd_broadcast(&ps->result_not_full);
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
}
|
||||
|
||||
/* Initialize result queue and synchronization primitives. Returns true on success. */
|
||||
static bool parallel_scanner_init(ParallelScanner* ps) {
|
||||
ps->result_queue = queue_create(SCANNER_RESULT_QUEUE_CAP, chunk_destroy);
|
||||
if (!ps->result_queue)
|
||||
return false;
|
||||
atomic_init(&ps->cancelled, false);
|
||||
int init = 0;
|
||||
bool ok = true;
|
||||
if (mtx_init(&ps->result_mutex, mtx_plain) != thrd_success)
|
||||
ok = false;
|
||||
if (ok) {
|
||||
init++;
|
||||
if (cnd_init(&ps->result_not_empty) != thrd_success)
|
||||
ok = false;
|
||||
}
|
||||
if (ok) {
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
if (cnd_init(&ps->result_not_full) != thrd_success)
|
||||
ok = false;
|
||||
}
|
||||
if (!ok) {
|
||||
if (init >= 3)
|
||||
cnd_destroy(&ps->result_not_full);
|
||||
if (init >= 2)
|
||||
cnd_destroy(&ps->result_not_empty);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&ps->result_mutex);
|
||||
queue_destroy(ps->result_queue);
|
||||
ps->result_queue = NULL;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Split files into chunks of roughly chunk_size bytes. Returns the first chunk (also stored
|
||||
* chunks beyond the first are enqueued on `queue`). Nulls out consumed entries in `files`.
|
||||
* Sets *failed on allocation/enqueue errors. */
|
||||
static Chunk* batch_files(ArrayList* files, unsigned long long chunk_size, Queue* queue,
|
||||
bool* failed) {
|
||||
Chunk* first = NULL;
|
||||
if (files->size <= 0)
|
||||
return NULL;
|
||||
ArrayList* batch = array_list_create(NULL);
|
||||
if (!batch) {
|
||||
*failed = true;
|
||||
return NULL;
|
||||
}
|
||||
unsigned long long batch_size = 0;
|
||||
for (int i = 0; i < files->size; i++) {
|
||||
File* f = (File*)files->items[i];
|
||||
if (!array_list_add(batch, f)) {
|
||||
*failed = true;
|
||||
break;
|
||||
}
|
||||
batch_size += f->data->size;
|
||||
if (batch_size >= chunk_size || i == files->size - 1) {
|
||||
void** items = array_list_to_array(batch);
|
||||
if (!items) {
|
||||
*failed = true;
|
||||
array_list_delete(batch);
|
||||
batch = NULL;
|
||||
break;
|
||||
}
|
||||
Chunk* c = chunk_create((File**)items, batch->size);
|
||||
free(items);
|
||||
if (!c) {
|
||||
*failed = true;
|
||||
array_list_delete(batch);
|
||||
batch = NULL;
|
||||
break;
|
||||
}
|
||||
int batch_start = i - batch->size + 1;
|
||||
for (int j = batch_start; j <= i; j++)
|
||||
files->items[j] = NULL;
|
||||
batch->item_destroyer = NULL;
|
||||
array_list_delete(batch);
|
||||
batch = NULL;
|
||||
if (!first) {
|
||||
first = c;
|
||||
} else {
|
||||
if (!queue_enqueue(queue, c)) {
|
||||
chunk_destroy(c);
|
||||
*failed = true;
|
||||
}
|
||||
}
|
||||
if (i < files->size - 1) {
|
||||
batch = array_list_create(NULL);
|
||||
if (!batch) {
|
||||
*failed = true;
|
||||
break;
|
||||
}
|
||||
batch_size = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (batch) {
|
||||
batch->item_destroyer = NULL;
|
||||
array_list_delete(batch);
|
||||
}
|
||||
return first;
|
||||
}
|
||||
|
||||
/* Record the delete-protection mirror of a root entry that
|
||||
* scanner_inspect_entry() skipped (inspection == 0): a dereferenced symlink
|
||||
* with no referent is a partial-transfer I/O error and a user-selection or size
|
||||
* prune protects the entry's destination mirror. */
|
||||
static void scan_root_record_skipped(const ScannerOptions* options, const char* root_directory,
|
||||
const char* name, const ScannerEntry* inspected,
|
||||
ParallelScanner* ps) {
|
||||
if (inspected->referent_error)
|
||||
ps->io_error = true;
|
||||
ArrayList* sink = NULL;
|
||||
if (inspected->excluded)
|
||||
sink = inspected->size_excluded ? options->size_skipped_paths : options->excluded_paths;
|
||||
if (!sink)
|
||||
return;
|
||||
/* A root-level prune protects the destination mirror of the entry's wire
|
||||
path: under -R + --files-from that is the bare relative name, otherwise it
|
||||
is the full source path with a leading '/' removed (matching the
|
||||
send_path/file_wire_path the scanner hands the sender). */
|
||||
if (options->relative && options->file_list != NULL) {
|
||||
if (!excluded_sink_append(sink, options->excluded_mutex, name))
|
||||
ps->failed = true;
|
||||
} else if (options->relative_prefix) {
|
||||
char* wrel = scanner_prefix_send_path(options->relative_prefix, name);
|
||||
if (!wrel) {
|
||||
ps->failed = true;
|
||||
} else {
|
||||
if (!excluded_sink_append(sink, options->excluded_mutex, wrel))
|
||||
ps->failed = true;
|
||||
free(wrel);
|
||||
}
|
||||
} else {
|
||||
char* abs_path = path_cat(root_directory, name);
|
||||
if (!abs_path) {
|
||||
ps->failed = true;
|
||||
} else {
|
||||
const char* rel = *abs_path == '/' ? abs_path + 1 : abs_path;
|
||||
if (!excluded_sink_append(sink, options->excluded_mutex, rel))
|
||||
ps->failed = true;
|
||||
free(abs_path);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Record the delete-protection mirror of a root entry dropped by the
|
||||
* --files-from allow-set or a filter rule. Returns false only when the -R
|
||||
* prefix could not be built (the caller must abandon the entry immediately);
|
||||
* other allocation failures mark the scan failed but let the caller continue to
|
||||
* the filter-notice step, matching the historical inlined flow. */
|
||||
static bool scan_root_record_protection(const ScannerOptions* options, const char* rel,
|
||||
const char* name, const char* cur_path, bool protect,
|
||||
bool passes, bool use_rel, ParallelScanner* ps) {
|
||||
if (passes && !protect)
|
||||
return true;
|
||||
/* --files-from subset pruning is not a filter exclusion; -R bare-wire-path
|
||||
exclusions are never recorded (see ScannerOptions.excluded_paths). */
|
||||
bool files_from_prune = options->file_list && !file_list_affects(options->file_list, rel);
|
||||
if ((!files_from_prune && !use_rel) || protect) {
|
||||
const char* rel_path;
|
||||
char* prefixed = NULL;
|
||||
if (use_rel) {
|
||||
/* -R + --files-from: the destination/wire path is the bare relative
|
||||
name, not the source path. */
|
||||
rel_path = rel;
|
||||
} else if (options->relative_prefix) {
|
||||
prefixed = scanner_prefix_send_path(options->relative_prefix, name);
|
||||
if (!prefixed)
|
||||
return false;
|
||||
rel_path = prefixed;
|
||||
} else {
|
||||
rel_path = *cur_path == '/' ? cur_path + 1 : cur_path;
|
||||
}
|
||||
if (options->excluded_paths &&
|
||||
!excluded_sink_append(options->excluded_paths, options->excluded_mutex, rel_path))
|
||||
ps->failed = true;
|
||||
free(prefixed);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Root-level directory node: apply -x/--one-file-system and either emit the
|
||||
* mount-point directory (plain -x) or queue the directory for a worker. */
|
||||
static void scan_root_dir(const ScannerOptions* options, const char* cur_path, const char* rel,
|
||||
const struct stat* st, ArrayList* root_files, ArrayList* subdirs,
|
||||
dev_t root_dev, ParallelScanner* ps) {
|
||||
if (!scanner_same_filesystem(options->one_file_system, root_dev, st->st_dev)) {
|
||||
if (options->one_file_system > 1) {
|
||||
/* -xx: drop the mount-point directory entirely (rsync) and print the
|
||||
--info=mount line when enabled. */
|
||||
scanner_note_mount(options, cur_path);
|
||||
return;
|
||||
}
|
||||
/* -x/--one-file-system: emit the mount-point directory entry (empty) but do
|
||||
not descend into it (see the sequential scanner for the same rule). */
|
||||
File* mount = scanner_build_dir_file(cur_path, st, options);
|
||||
if (!mount) {
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
if (options->relative_prefix) {
|
||||
mount->send_path = scanner_prefix_send_path(options->relative_prefix, rel);
|
||||
if (!mount->send_path) {
|
||||
file_destroy(mount);
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (!array_list_add(root_files, mount)) {
|
||||
file_destroy(mount);
|
||||
ps->failed = true;
|
||||
}
|
||||
return;
|
||||
}
|
||||
char* dir = str_dup(cur_path);
|
||||
if (!dir || !array_list_add(subdirs, dir)) {
|
||||
free(dir);
|
||||
ps->failed = true;
|
||||
}
|
||||
}
|
||||
|
||||
/* Build a non-directory root entry through the shared construction path and add
|
||||
* it to `root_files`. A non-regular entry the options do not preserve is
|
||||
* dropped by the builder (which prints rsync's nonreg line); an allocation
|
||||
* failure marks the scan failed. */
|
||||
static void scan_root_add_non_dir(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, ArrayList* root_files, ParallelScanner* ps) {
|
||||
File* file = NULL;
|
||||
bool failed = false;
|
||||
ScannerBuildStatus status = scanner_build_file_entry(options, inspected, rel, &file, &failed);
|
||||
if (failed || status == SCANNER_BUILD_FAIL_CONTINUE || status == SCANNER_BUILD_FAIL_BREAK)
|
||||
ps->failed = true;
|
||||
if (status != SCANNER_BUILD_OK)
|
||||
return;
|
||||
if (!array_list_add(root_files, file)) {
|
||||
file_destroy(file);
|
||||
ps->failed = true;
|
||||
}
|
||||
}
|
||||
|
||||
/* Regular file or carried symlink at the transfer root. */
|
||||
static void scan_root_file(const ScannerOptions* options, ScannerEntry* inspected, const char* rel,
|
||||
ArrayList* root_files, ParallelScanner* ps) {
|
||||
scan_root_add_non_dir(options, inspected, rel, root_files, ps);
|
||||
}
|
||||
|
||||
/* Device/FIFO/socket at the transfer root: recreated under --devices/--specials,
|
||||
* otherwise dropped by the shared builder. */
|
||||
static void scan_root_special(const ScannerOptions* options, ScannerEntry* inspected,
|
||||
const char* rel, ArrayList* root_files, ParallelScanner* ps) {
|
||||
scan_root_add_non_dir(options, inspected, rel, root_files, ps);
|
||||
}
|
||||
|
||||
/* Scan one root-directory entry into either the subdirs or files list. */
|
||||
static void scan_root_entry(const ScannerOptions* options, const FilterNode* root_node,
|
||||
const char* root_directory, const struct dirent* entry,
|
||||
ArrayList* root_files, ArrayList* subdirs, dev_t root_dev,
|
||||
ParallelScanner* ps) {
|
||||
ScannerEntry inspected;
|
||||
int inspection =
|
||||
scanner_inspect_entry(options, root_directory, entry->d_name, entry->d_name, &inspected);
|
||||
if (inspection < 0) {
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
if (inspection == 0) {
|
||||
scan_root_record_skipped(options, root_directory, entry->d_name, &inspected, ps);
|
||||
return;
|
||||
}
|
||||
char* cur_path = inspected.path;
|
||||
char* rel = str_dup(entry->d_name);
|
||||
if (!rel) {
|
||||
ps->failed = true;
|
||||
goto done;
|
||||
}
|
||||
bool is_dir = inspected.is_directory;
|
||||
bool protect = false;
|
||||
bool passes = entry_passes_selection(options->file_list, options->base_filters, root_node, rel,
|
||||
entry->d_name, is_dir, options->per_dir_filters,
|
||||
options->exclude_per_dir_filter_files, &protect);
|
||||
/* -R + --files-from: root-level files keep their bare relative send path. */
|
||||
bool use_rel = options->relative && options->file_list != NULL;
|
||||
if (!passes || protect) {
|
||||
if (!scan_root_record_protection(options, rel, entry->d_name, cur_path, protect, passes,
|
||||
use_rel, ps)) {
|
||||
ps->failed = true;
|
||||
goto done;
|
||||
}
|
||||
if (!passes) {
|
||||
scanner_note_filter(options, entry->d_name);
|
||||
goto done;
|
||||
}
|
||||
}
|
||||
if (is_dir) {
|
||||
scan_root_dir(options, cur_path, rel, &inspected.stats, root_files, subdirs, root_dev, ps);
|
||||
} else if (S_ISCHR(inspected.stats.st_mode) || S_ISBLK(inspected.stats.st_mode) ||
|
||||
S_ISFIFO(inspected.stats.st_mode) || S_ISSOCK(inspected.stats.st_mode)) {
|
||||
scan_root_special(options, &inspected, rel, root_files, ps);
|
||||
} else {
|
||||
scan_root_file(options, &inspected, rel, root_files, ps);
|
||||
}
|
||||
done:
|
||||
free(rel);
|
||||
free(cur_path);
|
||||
free(inspected.link_target);
|
||||
}
|
||||
|
||||
/* Scan the root directory itself, collecting root files and subdirectories.
|
||||
* Returns false if the root directory could not be opened. */
|
||||
static bool scan_root_directory(ParallelScanner* ps, const char* root_directory,
|
||||
const ScannerOptions* options, const FilterNode* root_node,
|
||||
dev_t root_dev, ArrayList* root_files, ArrayList* subdirs) {
|
||||
DIR* dir = opendir(root_directory);
|
||||
if (!dir) {
|
||||
log_perror("Could not open root directory for parallel scan");
|
||||
return false;
|
||||
}
|
||||
/* The parallel scanner opens the transfer root directly (not through
|
||||
open_next_directory), so record it as synchronized here. */
|
||||
if (!scanner_record_synced_dir(options, root_directory, "",
|
||||
options->relative && options->file_list != NULL)) {
|
||||
closedir(dir);
|
||||
ps->failed = true;
|
||||
return false;
|
||||
}
|
||||
log_debug_message(LOG_DEBUG_FLIST, "flist: scanning %s", root_directory);
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
scan_root_entry(options, root_node, root_directory, entry, root_files, subdirs, root_dev, ps);
|
||||
}
|
||||
closedir(dir);
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Spawn worker threads, one per group of subdirectories. */
|
||||
static void spawn_parallel_workers(ParallelScanner* ps, ArrayList* subdirs,
|
||||
const ScannerOptions* options, const char* root_directory,
|
||||
unsigned long long cs) {
|
||||
if (subdirs->size <= 0)
|
||||
return;
|
||||
int n = options->num_threads > 0 ? options->num_threads : 4;
|
||||
if (n > subdirs->size)
|
||||
n = subdirs->size;
|
||||
|
||||
ps->num_threads = n;
|
||||
ps->expected_threads = n;
|
||||
ps->threads = calloc(n, sizeof(thrd_t));
|
||||
if (!ps->threads) {
|
||||
ps->num_threads = 0;
|
||||
ps->expected_threads = 0;
|
||||
ps->failed = true;
|
||||
return;
|
||||
}
|
||||
int dirs_per_thread = subdirs->size / n;
|
||||
int remainder = subdirs->size % n;
|
||||
int start = 0;
|
||||
ps->num_threads = 0;
|
||||
for (int t = 0; t < n; t++) {
|
||||
int count = dirs_per_thread + (t < remainder ? 1 : 0);
|
||||
if (count == 0)
|
||||
break;
|
||||
ParallelWorkerArg* wa = calloc(1, sizeof(ParallelWorkerArg));
|
||||
if (!wa) {
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
wa->ps = ps;
|
||||
wa->dirs = calloc(count, sizeof(char*));
|
||||
wa->root_dir = str_dup(root_directory);
|
||||
if (!wa->dirs || !wa->root_dir) {
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
bool dup_ok = true;
|
||||
for (int j = 0; j < count; j++) {
|
||||
wa->dirs[j] = str_dup((char*)subdirs->items[start + j]);
|
||||
if (!wa->dirs[j])
|
||||
dup_ok = false;
|
||||
}
|
||||
if (!dup_ok) {
|
||||
for (int j = 0; j < count; j++)
|
||||
free(wa->dirs[j]);
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
wa->dir_count = count;
|
||||
wa->options = *options;
|
||||
wa->options.chunk_size = cs;
|
||||
wa->allocation_session = ps->allocation_session;
|
||||
start += count;
|
||||
if (thrd_create(&ps->threads[t], parallel_worker_thread, wa) != thrd_success) {
|
||||
for (int j = 0; j < count; j++)
|
||||
free(wa->dirs[j]);
|
||||
free(wa->root_dir);
|
||||
free(wa->dirs);
|
||||
free(wa);
|
||||
parallel_scanner_creation_failed(ps);
|
||||
break;
|
||||
}
|
||||
ps->num_threads++;
|
||||
ps->created_threads++;
|
||||
}
|
||||
}
|
||||
|
||||
ParallelScanner* parallel_scanner_create_with_options(const char* root_directory,
|
||||
const ScannerOptions* options,
|
||||
ProtocolSession* allocation_session) {
|
||||
if (!root_directory || !options)
|
||||
return NULL;
|
||||
ParallelScanner* ps = calloc(1, sizeof(ParallelScanner));
|
||||
if (!ps)
|
||||
return NULL;
|
||||
if (!parallel_scanner_init(ps)) {
|
||||
free(ps);
|
||||
return NULL;
|
||||
}
|
||||
ps->allocation_session = allocation_session;
|
||||
ps->options = options;
|
||||
|
||||
ArrayList* root_files = array_list_create(file_destroy);
|
||||
ArrayList* subdirs = array_list_create(free);
|
||||
if (!root_files || !subdirs) {
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
dev_t root_dev = 0;
|
||||
if (options->one_file_system) {
|
||||
struct stat root_stats;
|
||||
if (stat(root_directory, &root_stats) != 0) {
|
||||
log_perror("Could not stat source directory");
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
root_dev = root_stats.st_dev;
|
||||
}
|
||||
|
||||
/* Build the root directory's per-directory filter context once; workers seed
|
||||
* their scanners with it so per-dir rules behave identically to the sequential
|
||||
* scanner. */
|
||||
FilterNode* root_node = NULL;
|
||||
{
|
||||
char err[256];
|
||||
bool any_exists = false;
|
||||
FilterRuleList* own = read_dir_filters(options, root_directory, "",
|
||||
options->relative && options->file_list != NULL,
|
||||
&any_exists, err, sizeof(err));
|
||||
if (!own) {
|
||||
/* A parse/allocation failure must fail the scan even when an earlier
|
||||
merge file in the same directory existed (see the sequential scanner). */
|
||||
if (err[0] != '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "invalid per-directory filter in %s: %s", root_directory, err);
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
/* no files exist: leave root_node NULL */
|
||||
} else if (any_exists && (own->count > 0 || own->dir_merge_count > 0)) {
|
||||
root_node = filter_node_alloc(NULL, own);
|
||||
if (!root_node) {
|
||||
filter_rule_list_free(own);
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
} else {
|
||||
filter_rule_list_free(own);
|
||||
}
|
||||
}
|
||||
ps->root_filter_node = root_node;
|
||||
|
||||
if (!scan_root_directory(ps, root_directory, options, root_node, root_dev, root_files, subdirs)) {
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
/* The root itself is a traversed directory (rsync counts it in
|
||||
`Number of files`); the worker DirectoryScanners account for every
|
||||
subdirectory below it. */
|
||||
scanner_dir_count_count(options);
|
||||
/* P7 Wave D: the parallel scanner never runs a DirectoryScanner over the
|
||||
transfer root itself (it hands the root's immediate subdirectories to
|
||||
workers), so capture the root's directory time here. */
|
||||
if (options->capture_dir_times &&
|
||||
!scanner_capture_dir_time(
|
||||
options->dir_entries, options->dir_entries_mutex, root_directory, root_directory,
|
||||
options->relative && options->file_list != NULL, options->relative_prefix,
|
||||
options->preserve_atimes, options->preserve_crtimes, options->preserve_xattrs,
|
||||
options->preserve_acls, options->no_implied_dirs, options->file_list)) {
|
||||
array_list_delete(root_files);
|
||||
array_list_delete(subdirs);
|
||||
parallel_scanner_destroy(ps);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
unsigned long long cs = options->chunk_size > 0 ? options->chunk_size : DESIRED_CHUNK_SIZE;
|
||||
ps->initial_chunk = batch_files(root_files, cs, ps->result_queue, &ps->failed);
|
||||
array_list_delete(root_files);
|
||||
|
||||
spawn_parallel_workers(ps, subdirs, options, root_directory, cs);
|
||||
array_list_delete(subdirs);
|
||||
return ps;
|
||||
}
|
||||
|
||||
Chunk* parallel_scanner_next(ParallelScanner* ps) {
|
||||
if (ps->initial_chunk) {
|
||||
Chunk* c = ps->initial_chunk;
|
||||
ps->initial_chunk = NULL;
|
||||
return c;
|
||||
}
|
||||
if (ps->num_threads == 0) {
|
||||
mtx_lock(&ps->result_mutex);
|
||||
if (!queue_is_empty(ps->result_queue)) {
|
||||
Chunk* chunk = queue_dequeue(ps->result_queue);
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
return chunk;
|
||||
}
|
||||
ps->done = true;
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
return NULL;
|
||||
}
|
||||
Chunk* chunk = queue_dequeue_multithreaded(
|
||||
ps->result_queue, &ps->result_mutex, &ps->result_not_empty, &ps->result_not_full, &ps->done);
|
||||
return chunk;
|
||||
}
|
||||
|
||||
bool parallel_scanner_failed(const ParallelScanner* ps) {
|
||||
return ps == NULL || ps->failed;
|
||||
}
|
||||
|
||||
bool parallel_scanner_had_io_error(const ParallelScanner* ps) {
|
||||
return ps != NULL && ps->io_error;
|
||||
}
|
||||
|
||||
void parallel_scanner_destroy(ParallelScanner* ps) {
|
||||
if (!ps)
|
||||
return;
|
||||
mtx_lock(&ps->result_mutex);
|
||||
ps->done = true;
|
||||
atomic_store(&ps->cancelled, true);
|
||||
cnd_broadcast(&ps->result_not_empty);
|
||||
cnd_broadcast(&ps->result_not_full);
|
||||
mtx_unlock(&ps->result_mutex);
|
||||
for (int i = 0; i < ps->num_threads; i++)
|
||||
thrd_join(ps->threads[i], NULL);
|
||||
free(ps->threads);
|
||||
if (ps->root_filter_node)
|
||||
filter_node_destroy(ps->root_filter_node);
|
||||
if (ps->initial_chunk)
|
||||
chunk_destroy(ps->initial_chunk);
|
||||
queue_destroy(ps->result_queue);
|
||||
mtx_destroy(&ps->result_mutex);
|
||||
cnd_destroy(&ps->result_not_empty);
|
||||
cnd_destroy(&ps->result_not_full);
|
||||
free(ps);
|
||||
}
|
||||
+25
-336
@@ -2,7 +2,6 @@
|
||||
#include <stdio.h>
|
||||
#include <delta.h>
|
||||
#include <chunk.h>
|
||||
#include "scanner.h"
|
||||
|
||||
void print_usage(void) {
|
||||
printf("Usage:\n");
|
||||
@@ -12,375 +11,65 @@ void print_usage(void) {
|
||||
printf("Destination formats:\n");
|
||||
printf(" user@host:/path SSH transport (rsync-style)\n");
|
||||
printf(" host:/path SSH transport (current user)\n");
|
||||
printf(" host::module/path Daemon TCP transport (fastsync-server --daemon);\n");
|
||||
printf(" module names a server-side module, path is relative\n");
|
||||
printf(" within it (connect with --server-port)\n");
|
||||
printf(" /local/path TCP transport (requires server on localhost:8080)\n");
|
||||
printf("\n");
|
||||
printf("Options:\n");
|
||||
printf(" -c, --checksum Verify content by checksum instead of size+mtime\n");
|
||||
printf(" -z, --compress [level] Enable compression. The default level is\n");
|
||||
printf(" per-codec: zstd 3 (range 1-22), zlib/zlibx 6, lz4\n");
|
||||
printf(" ignores the level\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, perms, times,\n");
|
||||
printf(" owner, group, devices and specials; not\n");
|
||||
printf(" compression/multithreading\n");
|
||||
printf(" -r, --recursive Recurse into directories (FastSync is always recursive)\n");
|
||||
printf(" --inc-recursive Accepted for rsync CLI compatibility; no effect (FastSync\n");
|
||||
printf(" always performs a full scan, so the destination is identical)\n");
|
||||
printf(" --no-inc-recursive Accepted for rsync CLI compatibility; no effect\n");
|
||||
printf(" -c [level] Enable compression (level 1-22, default 5)\n");
|
||||
printf(" -z [level] Alias for -c\n");
|
||||
printf(" -a, --archive Archive mode (-c -m -M)\n");
|
||||
printf(" -n, --dry-run Show what would be transferred\n");
|
||||
printf(" --remove-source-files Remove regular source files after successful transfer\n");
|
||||
printf(" -p, --perms Preserve permission bits\n");
|
||||
printf(" -t, --times Preserve modification times\n");
|
||||
printf(" -o, --owner Preserve owner (uid)\n");
|
||||
printf(" -g, --group Preserve group (gid)\n");
|
||||
printf(" --ssh-port <port> SSH port (default: 22)\n");
|
||||
printf(" -e, --rsh <command> Remote shell to launch on the client for the SSH\n");
|
||||
printf(" transport (default: ssh). The command may include\n");
|
||||
printf(" arguments, e.g. -e \"ssh -p 2222\"\n");
|
||||
printf(" --rsync-path <path> Alias for --fastsync-server-path (path to the\n");
|
||||
printf(" fastsync server binary on the remote side)\n");
|
||||
printf(" --blocking-io SSH transport only: leave the socket without read/write\n");
|
||||
printf(" timeouts so it blocks naturally (no effect on TCP)\n");
|
||||
printf(" --outbuf=MODE stdout/stderr buffering: N (none/unbuffered),\n");
|
||||
printf(" L (line-buffered), or B (block-buffered, default)\n");
|
||||
printf(" -p <port> SSH port (default: 22)\n");
|
||||
printf(" --progress Show transfer progress\n");
|
||||
printf(" -P Partial mode with progress (retention incomplete)\n");
|
||||
printf(" -8, --8-bit-output Leave high-bit characters unescaped in output\n");
|
||||
printf(" --iconv=LOCAL[,REMOTE] Convert file-NAME charsets at the wire boundary:\n");
|
||||
printf(" LOCAL is the charset of our file names, REMOTE is the\n");
|
||||
printf(" remote side's charset (defaults to LOCAL). Names are\n");
|
||||
printf(" converted before transmission and back on receipt; a\n");
|
||||
printf(" name that cannot be represented in the target charset\n");
|
||||
printf(" fails that transfer cleanly (rsync-compatible)\n");
|
||||
printf(" --no-iconv Disable --iconv charset conversion (same as --iconv=-)\n");
|
||||
printf(" --protocol=NUM Force the wire protocol version (must equal the current\n");
|
||||
printf(" PROTOCOL_VERSION; FastSync cannot speak older/virtual\n");
|
||||
printf(" wire formats)\n");
|
||||
printf(" --write-batch=FILE Run the normal live transfer AND also emit a\n");
|
||||
printf(" self-contained batch file of the whole source tree\n");
|
||||
printf(" (implies the single-threaded transfer path)\n");
|
||||
printf(" --only-write-batch=FILE\n");
|
||||
printf(" Emit the batch file only (no destination, no server)\n");
|
||||
printf(" --read-batch=FILE Apply the batch file to the destination (no source, no\n");
|
||||
printf(" server); takes only the destination as an argument\n");
|
||||
printf(" NOTE: the FastSync batch format is NOT interoperable with rsync's batch\n");
|
||||
printf(" files (different container format); do not mix the two tools.\n");
|
||||
printf(" --delete Delete files on receiver not in source\n");
|
||||
printf(" (default timing: delete-during, like rsync --del)\n");
|
||||
printf(" --delete-before Delete extras before the transfer starts\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-during Delete a directory's extras as that directory is\n");
|
||||
printf(" processed (implies --delete)\n");
|
||||
printf(" --del Alias for --delete-during\n");
|
||||
printf(" --delete-delay Record the extras during the scan but remove them\n");
|
||||
printf(" only after a successful transfer (implies --delete)\n");
|
||||
printf(" --delete-after Delete only after the whole transfer succeeded\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-commit FastSync-only: restore the late whole-tree commit\n");
|
||||
printf(" (identical to --delete-after; implies --delete)\n");
|
||||
printf(" --delete-excluded Also delete destination files that were excluded on\n");
|
||||
printf(" the source (default protects them, matching rsync)\n");
|
||||
printf(" --max-delete=NUM Delete at most NUM destination entries per run; if the\n");
|
||||
printf(" extras exceed NUM, the rest are skipped and the run is\n");
|
||||
printf(" reported as partial (exit 25, matching rsync). Only\n");
|
||||
printf(" applies together with --delete\n");
|
||||
printf(" --ignore-errors Continue (and still delete) when a source directory is\n");
|
||||
printf(" unreadable during the scan, instead of aborting with no\n");
|
||||
printf(" deletion\n");
|
||||
printf(" --force A file may replace a destination directory by removing\n");
|
||||
printf(" that (non-empty) directory first\n");
|
||||
printf(" --ignore-missing-args A --files-from entry that does not exist under the\n");
|
||||
printf(" source is silently skipped instead of failing the run\n");
|
||||
printf(" --delete-missing-args Implies --ignore-missing-args; also deletes each missing\n");
|
||||
printf(" entry's destination mirror receiver-side. Independent of\n");
|
||||
printf(" --delete (it does not imply --delete; a non-empty directory\n");
|
||||
printf(" mirror is removed only with --force or --delete)\n");
|
||||
printf(" -m, --prune-empty-dirs Do not create empty directories (a recursive transfer\n");
|
||||
printf(" otherwise recreates them, like rsync)\n");
|
||||
printf(" Note: each timing flag implies --delete. Combining a timing flag with\n");
|
||||
printf(" --no-delete (in either order) is rejected as a config error, as is more\n");
|
||||
printf(" than one timing flag.\n");
|
||||
printf(" --ignore-existing Skip files that already exist on receiver\n");
|
||||
printf(" --delay-updates Put updated files into place only at the end of transfer\n");
|
||||
printf(" --dirs, -d, --old-dirs, --old-d Transfer the named directory entries without\n");
|
||||
printf(" recursing into their contents (-d <dir> mirrors the source\n");
|
||||
printf(" directory empty; with --files-from listed dirs are created\n");
|
||||
printf(" empty and listed files are transferred)\n");
|
||||
printf(" -R, --relative With --files-from, preserve each listed entry's relative path\n");
|
||||
printf(" below the destination root instead of mirroring the full\n");
|
||||
printf(" source path (no effect without --files-from)\n");
|
||||
printf(" --no-implied-dirs With -R, do not apply the source metadata of a listed file's\n");
|
||||
printf(" implied parent directories (they are still created with\n");
|
||||
printf(" default attributes)\n");
|
||||
printf(" --mkpath Create the destination root directory on the server when it\n");
|
||||
printf(" does not exist yet\n");
|
||||
printf(" --exclude <pattern>, --exclude=<pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern>, --include=<pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file>, --exclude-from=<file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file>, --include-from=<file> Read include patterns from file\n");
|
||||
printf(" --files-from <file> Read the source file list from FILE (paths relative to the "
|
||||
"source root)\n");
|
||||
printf(" -0, --from0 Entries in --files-from are NUL-delimited\n");
|
||||
printf(" -f, --filter=RULE rsync-style filter rule: exclude/- include/+ hide/H show/S\n");
|
||||
printf(" protect/P risk/R merge/. dir-merge/: clear/! with modifiers\n");
|
||||
printf(" (repeatable; --filter=RULE and -f RULE / -f=RULE both work)\n");
|
||||
printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n");
|
||||
printf(" -F Apply per-directory .rsync-filter files; repeated -FF also\n");
|
||||
printf(" excludes the .rsync-filter files themselves\n");
|
||||
printf(" --exclude <pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file> Read include patterns from file\n");
|
||||
printf(" --max-size <n> Skip files larger than n bytes\n");
|
||||
printf(" --min-size <n> Skip files smaller than n bytes\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G; 0 = no limit,\n");
|
||||
printf(" matching rsync)\n");
|
||||
printf(" --incremental Skip files unchanged since last transfer\n");
|
||||
printf(" --size-only Skip incremental files matching in size, ignoring mtime\n");
|
||||
printf(" -I, --ignore-times Transfer files even when size and mtime match\n");
|
||||
printf(" -@, --modify-window <sec> Modification time tolerance\n");
|
||||
printf(" -u, --update Skip files newer than the source on receiver\n");
|
||||
printf(" --existing Skip files not already present at destination\n");
|
||||
printf(" --compare-dest <dir> Treat DIR (relative to destination root) as an extra\n");
|
||||
printf(" comparison basis: unchanged files are not transferred\n");
|
||||
printf(" (requires --incremental, which is implied)\n");
|
||||
printf(" --copy-dest <dir> Like --compare-dest, but copies the unchanged file from DIR\n");
|
||||
printf(" into the destination instead of transferring its data\n");
|
||||
printf(" --link-dest <dir> Like --copy-dest, but hard-links the unchanged file from DIR\n");
|
||||
printf(" into the destination (repeatable; earlier DIRs win)\n");
|
||||
printf(" --verify-basis FastSync-only: require a basis hit's content to match the\n");
|
||||
printf(" source by whole-file digest instead of trusting rsync's\n");
|
||||
printf(" size+mtime (or --size-only) quick-check\n");
|
||||
printf(" --checksum-choice, --cc <alg> Whole-file checksum algorithm for --incremental/\n");
|
||||
printf(" --checksum compares. Accepted: xxh128 (default), xxh3, xxh64\n");
|
||||
printf(" (aka xxhash), md5, md4, sha1, or none. A two-name\n");
|
||||
printf(" 'transfer,pre-transfer' form is accepted like rsync; 'none' as\n");
|
||||
printf(" the pre-transfer algorithm is rejected with --checksum\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. A seed\n");
|
||||
printf(" of 0 (the default) is randomized per transfer, exactly like\n");
|
||||
printf(" rsync, and the chosen seed is sent to the receiver\n");
|
||||
printf(" --delta Delta transfer for changed files (requires --incremental)\n");
|
||||
printf(" -W, --whole-file Transfer changed files without delta processing\n");
|
||||
printf(" --no-whole-file rsync spelling that clears -W/--whole-file\n");
|
||||
printf(" -y, --fuzzy Use a similar-named file already in the destination\n");
|
||||
printf(" directory as the delta basis when the destination has no\n");
|
||||
printf(" usable file at the exact path (saves bandwidth; implies\n");
|
||||
printf(" --incremental and --delta; inert with --whole-file,\n");
|
||||
printf(" --no-delta, or --no-incremental)\n");
|
||||
printf(" --no-fuzzy Disable --fuzzy\n");
|
||||
printf(" -B <n>, --block-size <n>, --delta-block <n>\n");
|
||||
printf(" Delta block size in bytes (default: %u)\n", DELTA_BLOCK_SIZE_DEFAULT);
|
||||
printf(" --delta-block <n> Delta block size in bytes (default: %d)\n",
|
||||
DELTA_BLOCK_SIZE_DEFAULT);
|
||||
printf(" --delta-max <n> Max file size for delta transfer (default: %llu)\n",
|
||||
DELTA_MAX_FILE_SIZE);
|
||||
printf(" -j, --threads[=N] Enable the multithreaded scanner/loader/sender\n");
|
||||
printf(" pipeline; N (1-%d) sets the parallel scanner worker\n",
|
||||
MAX_SCANNER_THREADS);
|
||||
printf(" count (bare -j/--threads uses the default)\n");
|
||||
printf(" --chunk-serialization Enable chunk serialization (long form only)\n");
|
||||
printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n");
|
||||
printf(" SSH argv is already built injection-safe)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only;\n");
|
||||
printf(" -f is bound to --filter, not --sendfile)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm: zstd (default), lz4, zlib,\n");
|
||||
printf(" zlibx, none, or auto\n");
|
||||
printf(" --zc <alg> Alias for --compress-choice\n");
|
||||
printf(" -m Enable multithreading\n");
|
||||
printf(" -s Enable chunk serialization\n");
|
||||
printf(" -f Enable sendfile (TCP only, not with -c or -s)\n");
|
||||
printf(" -v, --verbose Enable debug logging\n");
|
||||
printf(" -q, --quiet Suppress non-error output\n");
|
||||
printf(" --debug=FLAGS Fine-grained debug logging (use --debug=help for flags)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,name,misc,skip,stats,all,none\n");
|
||||
printf(" (use --info=help for flags; none suppresses --verbose)\n");
|
||||
printf(" --preserve Preserve permissions and times (= -pt; long form only)\n");
|
||||
printf(" --no-perms Negate -p/--perms\n");
|
||||
printf(" --no-times Negate -t/--times\n");
|
||||
printf(" --no-owner Negate -o/--owner\n");
|
||||
printf(" --no-group Negate -g/--group\n");
|
||||
printf(" --no-preserve Disable metadata preservation (negates --preserve)\n");
|
||||
printf(" -E, --executability Preserve executable permission bits\n");
|
||||
printf(" -U, --atimes Preserve access times\n");
|
||||
printf(" -N, --crtimes Capture birth time; cannot be applied (documented\n");
|
||||
printf(" divergence)\n");
|
||||
printf(" -O, --omit-dir-times Do not apply modification times to directories\n");
|
||||
printf(" -J, --omit-link-times Do not apply times to symlinks\n");
|
||||
printf(" --open-noatime Open source files with O_NOATIME so reading for a\n");
|
||||
printf(" transfer does not update their access time\n");
|
||||
printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n");
|
||||
printf(" privileged security.*/trusted.* namespaces are\n");
|
||||
printf(" never captured or applied)\n");
|
||||
printf(" -A, --acls Preserve POSIX ACLs (the system.posix_acl_* xattrs;\n");
|
||||
printf(" setting an ACL the receiver is not permitted to\n");
|
||||
printf(" set is warned and skipped, never fatal)\n");
|
||||
printf(" --fake-super Store the source mode/rdev/uid/gid in rsync's\n");
|
||||
printf(" reserved user.rsync.%%stat xattr on each written\n");
|
||||
printf(" file (interoperable with rsync); it never performs a\n");
|
||||
printf(" real chown, so an unprivileged receiver records the\n");
|
||||
printf(" privileged stat for a later restore\n");
|
||||
printf(" --super Permit the receiver to attempt super-user activities\n");
|
||||
printf(" (char/block device-node creation, --write-devices)\n");
|
||||
printf(" within the confined receive root. Never elevates\n");
|
||||
printf(" privileges and never bypasses confinement; ownership\n");
|
||||
printf(" is still applied only with -o/--owner, -g/--group, or an\n");
|
||||
printf(" explicit identity flag (--chown/--usermap/--groupmap/\n");
|
||||
printf(" --copy-as); --numeric-ids only changes how ids map\n");
|
||||
printf(" --no-super Forbid those super-user activities even when the\n");
|
||||
printf(" receiver is running as root\n");
|
||||
printf(
|
||||
" --chmod <changes> Modify new/transferred permissions (rsync syntax; implies no -p)\n");
|
||||
printf(" --numeric-ids Map uid/gid by id instead of by name (a modifier, not\n");
|
||||
printf(" an ownership request: combine with -o/-g or a map)\n");
|
||||
printf(" --usermap=MAP Map usernames when applying ownership: comma-separated\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM is a name (from\n");
|
||||
printf(" the source), an id, an inclusive LOW-HIGH range, *\n");
|
||||
printf(" (any id), or empty (ids with no name). TO is an id, *\n");
|
||||
printf(" (current user), or a name resolved on the receiver.\n");
|
||||
printf(" e.g. 0-99:nobody,*:normal (cannot mix with --chown)\n");
|
||||
printf(" --groupmap=MAP Map group names when applying ownership (same syntax)\n");
|
||||
printf(" --chown=USER:GROUP Override the ownership of transferred files. Forms:\n");
|
||||
printf(" USER:GROUP, USER (owner only), :GROUP (group only); a\n");
|
||||
printf(" value of * means the current/root user as appropriate.\n");
|
||||
printf(" Names resolve on the source machine; @N for numerics.\n");
|
||||
printf(" (Implies owner/group metadata; -M now means rsync's\n");
|
||||
printf(" --remote-option.)\n");
|
||||
printf(" --copy-as=USER[:GROUP] Force every written entry (files, dirs, symlinks\n");
|
||||
printf(" and special nodes) to USER[:GROUP], resolved on the\n");
|
||||
printf(" source machine like --chown. Requires a privileged\n");
|
||||
printf(" (root) receiver and implies owner/group metadata; an\n");
|
||||
printf(" unprivileged receiver refuses the transfer. Never\n");
|
||||
printf(" switches process credentials (safe-subset; see\n");
|
||||
printf(" RSYNC_COMPAT.md). A daemon refuses it.\n");
|
||||
printf(" -M, --preserve Preserve file metadata\n");
|
||||
printf(" --chunk-size <n> Chunk size in bytes (default: %d)\n", DEFAULT_CHUNK_SIZE);
|
||||
printf(" --source-dir <path> Source directory\n");
|
||||
printf(" --dest-dir <path> Destination directory\n");
|
||||
printf(" --save-to-disk Write received files to disk\n");
|
||||
printf(" --server-host <ip> Server IP address (default: 127.0.0.1)\n");
|
||||
printf(" --server-port <n> Server port (default: 8080)\n");
|
||||
printf(" --port <n> Alias for --server-port\n");
|
||||
printf(" --password-file <f> Authenticate a host::module/path daemon destination.\n");
|
||||
printf(" FastSync-native SCRAM/PBKDF2 credential scheme (NOT\n");
|
||||
printf(" rsync's --password-file): the file's first user:password\n");
|
||||
printf(" line supplies the username and password; no password or\n");
|
||||
printf(" reusable digest is sent (keep the file mode 0600)\n");
|
||||
printf(" --no-motd Suppress display of the daemon's MOTD (the server\n");
|
||||
printf(" still sends it; the client just does not show it)\n");
|
||||
printf(" --bwlimit=RATE Limit socket I/O bandwidth (default unit KiB/s,\n");
|
||||
printf(" rsync-style: 0 = no limit; K/M/G/T/P suffixes are\n");
|
||||
printf(" binary, KB/MB decimal, KiB/MiB binary; decimals allowed)\n");
|
||||
printf(" --bwlimit <KB/s> Bandwidth limit in kilobytes per second\n");
|
||||
printf(" --tls Enable TLS encryption\n");
|
||||
printf(" --cert <path> TLS certificate file (PEM)\n");
|
||||
printf(" --key <path> TLS private key file (PEM)\n");
|
||||
printf(" --ca <path> TLS CA certificate file (PEM)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 0 = disabled, matching\n");
|
||||
printf(" rsync). 0 disables it; --no-timeout is the same\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 60, matching\n");
|
||||
printf(" rsync); 0 disables it (--no-contimeout)\n");
|
||||
printf(" --stop-after=MINS Stop the transfer after MINS minutes (a positive\n");
|
||||
printf(" integer); whatever was already transferred is kept\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time. Accepts rsync's date form\n");
|
||||
printf(" (Y-M-DTh:m, Y/M/DTh:m, abbreviable fields such as 12-31,\n");
|
||||
printf(" 14:00, :59, 1) plus FastSync's HH:MM[:SS] and now+N[smhd]\n");
|
||||
printf(" (a time already in the past stops the transfer\n");
|
||||
printf(" immediately; client-only). An early stop skips the late\n");
|
||||
printf(" --delete keep-set so it cannot delete source mirrors that\n");
|
||||
printf(" were not yet scanned\n");
|
||||
printf(" --address <ip> Bind the outgoing client socket to this source address\n");
|
||||
printf(" -4, --ipv4 Force IPv4 for destination resolution\n");
|
||||
printf(" -6, --ipv6 Force IPv6 for destination resolution\n");
|
||||
printf(" --sockopts=OPTS Comma-separated OPT=VAL socket options applied before connect:\n");
|
||||
printf(" TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR\n");
|
||||
printf(" -b, --backup Backup existing files before overwriting\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 30)\n");
|
||||
printf(" -T <sec> Alias for --timeout\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 10)\n");
|
||||
printf(" --backup Backup existing files before overwriting\n");
|
||||
printf(" --backup-dir <dir> Directory for backups (requires --backup)\n");
|
||||
printf(" --suffix <str> Backup suffix (default: ~)\n");
|
||||
printf(" --stats Print transfer statistics at end\n");
|
||||
printf(" -i, --itemize-changes Print an rsync-style per-file change line\n");
|
||||
printf(" --out-format=FORMAT Output format (%%f %%n %%l %%b %%c %%C %%i %%M %%%%)\n");
|
||||
printf(" --list-only List source files instead of transferring\n");
|
||||
printf(" --log-file-format=FORMAT Per-file log line format (needs --log-file)\n");
|
||||
printf(" -h, --human-readable Print byte sizes in human-readable form\n");
|
||||
printf(" --max-depth <n> Maximum directory depth (0=unlimited)\n");
|
||||
printf(" -x, --one-file-system Do not cross filesystem boundaries\n");
|
||||
printf(" --log-file <path>, --log-file=<path> Write log messages to file\n");
|
||||
printf(" --stderr=MODE Route logging: errors (default), all, or client\n");
|
||||
printf(" (forward the client's diagnostics to the server's\n");
|
||||
printf(" stderr)\n");
|
||||
printf(" --msgs2stderr Route all messages to stderr (deprecated spelling of\n");
|
||||
printf(" --stderr=all)\n");
|
||||
printf(" --no-msgs2stderr Forward the client's diagnostics to the server\n");
|
||||
printf(" (deprecated spelling of --stderr=client)\n");
|
||||
printf(" --log-file <path> Write log messages to file\n");
|
||||
printf(" --partial Keep partial files on interrupted transfer\n");
|
||||
printf(" --partial-dir <dir> Directory for partial files (implies --partial)\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install.\n");
|
||||
printf(" Confined to the receive root: a relative dir resolves below\n");
|
||||
printf(" it and an absolute/traversal dir is rejected. The dir must\n");
|
||||
printf(" already exist; a different filesystem falls back to a\n");
|
||||
printf(" non-atomic copy instead of aborting\n");
|
||||
printf(" --partial-dir <dir> Directory for partial files\n");
|
||||
printf(" --fastsync-server-path <path>\n");
|
||||
printf(" Path to fastsync-server on remote (default: fastsync-server)\n");
|
||||
printf(" --old-args Accepted for rsync CLI compatibility; no effect (the\n");
|
||||
printf(" remote server path is always safely quoted now)\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation. SSH\n");
|
||||
printf(" transport ONLY (user@host:path): a daemon (host::module) or\n");
|
||||
printf(" local TCP destination rejects it (no remote command line to\n");
|
||||
printf(" append to). Repeatable; each value is single-quote-escaped on\n");
|
||||
printf(" the remote command line; empty values and values with control\n");
|
||||
printf(" characters are rejected; -M OPT, -M=OPT and\n");
|
||||
printf(" --remote-option=OPT work\n");
|
||||
printf(" --trust-sender RECEIVER-LOCAL policy: trust the remote sender's file list\n");
|
||||
printf(" and skip the receiver's own up-front path-traversal/\n");
|
||||
printf(" containment re-validation of the incoming list (fewer checks,\n");
|
||||
printf(" faster, potentially unsafe). It is never sent to the peer, so\n");
|
||||
printf(" for a push it must be enabled on the receiving SERVER\n");
|
||||
printf(" (fastsync-server --trust-sender) or forwarded with\n");
|
||||
printf(" -M--trust-sender; the client flag alone has no effect\n");
|
||||
printf(" -l, --links Copy symlinks as symlinks\n");
|
||||
printf(" -L, --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks whose target points outside the tree\n");
|
||||
printf(" --copy-unsafe-links Copy unsafe symlinks (outside tree) as referent files\n");
|
||||
printf(" -k, --copy-dirlinks Transform symlinks to directories into real dirs\n");
|
||||
printf(" -K, --keep-dirlinks Keep an existing symlink-to-dir as that dir\n");
|
||||
printf(" --munge-links Munge stored symlink targets (/rsyncd-munged/) on the receiver\n");
|
||||
printf(" -H, --hard-links Preserve hard-link relationships across the transfer\n");
|
||||
printf(" --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks that point outside transfer tree\n");
|
||||
printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n");
|
||||
printf(" -S, --sparse Handle sparse files efficiently\n");
|
||||
printf(
|
||||
" -D Preserve device and special files (implies --devices --specials)\n");
|
||||
printf(
|
||||
" --devices Recreate device nodes on the destination (privileged; skipped when\n");
|
||||
printf(" the receiver lacks CAP_MKNOD)\n");
|
||||
printf(" --specials Recreate special files (FIFOs, sockets) on the destination\n");
|
||||
printf(" --copy-devices Copy a source device's content as a regular file instead\n");
|
||||
printf(" --write-devices Write received data into an existing destination device node\n");
|
||||
printf(" --inplace Update files in-place (no temp+rename)\n");
|
||||
printf(
|
||||
" --preallocate Allocate destination file space up front (fail-fast on full disk)\n");
|
||||
printf(" --append Resume a shorter destination by appending only its tail\n");
|
||||
printf(" (prefix is not verified; requires --incremental)\n");
|
||||
printf(" --append-verify Like --append, but verifies the retained prefix checksum\n");
|
||||
printf(" before appending (falls back to a full transfer on mismatch)\n");
|
||||
printf(" --fsync Fsync every written file before publication\n");
|
||||
printf(" --compress-level <n> Compression level (per-codec default: zstd 3,\n");
|
||||
printf(" zlib/zlibx 6, lz4 ignores it)\n");
|
||||
printf(" --zl <n> Alias for --compress-level\n");
|
||||
printf(" --skip-compress=LIST Skip compression for suffixes in LIST (separated by\n");
|
||||
printf(" '/' as in rsync, or ','); a leading dot is optional. The\n");
|
||||
printf(" default is rsync 3.4.1's built-in skip-compress list\n");
|
||||
printf(" --compress-threads <n> Compression worker threads (requires zstd threaded support)\n");
|
||||
printf(" --no-OPTION Disable a supported boolean option\n");
|
||||
printf(" --compress-level <n> Compression level (default: 5)\n");
|
||||
printf(" --help Show this help\n");
|
||||
printf(" -V, --version Show version\n");
|
||||
}
|
||||
|
||||
void print_debug_usage(void) {
|
||||
printf("Emitting debug flags: IO,PROTO,PACK,UTIL,FLIST,DEL,HASH,DELTASUM,\n");
|
||||
printf("RECV,FILTER,SEND,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): ACL,BACKUP,BIND,CHDIR,\n");
|
||||
printf("CONNECT,CMD,DUP,EXIT,FUZZY,GENR,HLINK,ICONV,NSTR,OWN,TIME.\n");
|
||||
printf("Flags may be comma-separated, for example: --debug=io,proto\n");
|
||||
printf("An optional level suffix is accepted (e.g. --debug=io2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
|
||||
void print_info_usage(void) {
|
||||
printf("Emitting info flags: COPY,MISC,SKIP,STATS,DEL,REMOVE,NAME,FLIST,\n");
|
||||
printf("NONREG,PROGRESS,MOUNT,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): BACKUP,SYMS,SYMSAFE.\n");
|
||||
printf("Flags may be comma-separated, for example: --info=name,stats\n");
|
||||
printf("An optional level suffix is accepted (e.g. --info=stats2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
|
||||
@@ -2,7 +2,5 @@
|
||||
#define USAGE_H
|
||||
|
||||
void print_usage(void);
|
||||
void print_debug_usage(void);
|
||||
void print_info_usage(void);
|
||||
|
||||
#endif
|
||||
|
||||
+53
-954
File diff suppressed because it is too large
Load Diff
@@ -2,126 +2,18 @@
|
||||
#define RECEIVER_H
|
||||
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include <stdbool.h>
|
||||
#include <time.h>
|
||||
|
||||
typedef bool (*ReceiverFileSink)(File* file, void* context);
|
||||
|
||||
/* Ordered per-file save outcomes for one connection. One entry is appended
|
||||
for every data-bearing file the receiver processes (in the order the files
|
||||
were sent) so the sender of a --remove-source-files transfer can be told
|
||||
which sources were actually written versus skipped on the receiver. */
|
||||
typedef struct {
|
||||
unsigned char* entries; /* FILE_SAVE_WRITTEN or FILE_SAVE_SKIPPED */
|
||||
size_t count;
|
||||
size_t capacity;
|
||||
} ReceiverOutcomes;
|
||||
|
||||
typedef bool (*ReceiverSuccessFrame)(int fd, void* context);
|
||||
|
||||
/* Records that a --max-delete commit stopped with extras left over, so the
|
||||
caller's terminal success frame can carry STATUS_DELETE_LIMIT instead of
|
||||
STATUS_OK. The commit runs on the receiver thread, so the flag is stored in
|
||||
the sink's own context rather than in a shared global. */
|
||||
typedef void (*ReceiverNoteDeleteLimit)(void* context);
|
||||
|
||||
typedef struct {
|
||||
ReceiverFileSink store_file;
|
||||
void* context;
|
||||
bool send_error;
|
||||
bool send_success;
|
||||
/* Emits the end-of-transfer success frame. When the sender requested
|
||||
--remove-source-files this includes one per-file status per processed
|
||||
data file followed by the final status; otherwise just the final status. */
|
||||
ReceiverSuccessFrame send_success_frame;
|
||||
/* Optional; may be NULL when the sink has no --max-delete handling. */
|
||||
ReceiverNoteDeleteLimit note_delete_limit;
|
||||
/* Optional end-of-transfer wire counters (protocol 2.25.0). When non-NULL
|
||||
and the wire config carries report_stats, the success frame is preceded by
|
||||
a STATUS_STATS record; `would_delete` (optional, receiver-owned strings)
|
||||
carries the -n/--dry-run --delete path list. */
|
||||
ReceiverStats* stats;
|
||||
struct ArrayList* would_delete;
|
||||
/* When --info=del requested it, receiver-owned strings for every path the
|
||||
deletion commit ACTUALLY removed, sent in the terminal STATUS_STATS frame's
|
||||
path list so the sender can print rsync's `deleting PATH` lines. */
|
||||
struct ArrayList* deleted_paths;
|
||||
} ReceiverSink;
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code);
|
||||
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes);
|
||||
|
||||
/* Delete observer context: `deleted_paths` (optional) receives owned copies of
|
||||
every truly-removed destination-relative path for --info=del; `stats`
|
||||
(optional) receives the per-type `Number of deleted files` tallies for
|
||||
--stats. Both may be NULL, in which case the observer is a no-op. */
|
||||
typedef struct {
|
||||
ReceiverStats* stats;
|
||||
struct ArrayList* deleted_paths;
|
||||
} ReceiverDeleteContext;
|
||||
|
||||
/* DeletePathObserver implementation: records each truly-removed path (when the
|
||||
context carries a path list) and tallies it by type (when it carries a stats
|
||||
record). Shared by the single-threaded receiver and the -m pipeline's
|
||||
deferred commit. */
|
||||
void receiver_record_deleted_path(void* context, const char* rel_path, DeleteEntryType type);
|
||||
|
||||
/* Send the terminal success frame. `final_status` is usually STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status);
|
||||
|
||||
/* Emit STATUS_STATS (a fixed ReceiverStats record plus, when `would_delete` is
|
||||
non-NULL, a count and that many wire strings) when the wire config requested
|
||||
report_stats. A no-op otherwise. */
|
||||
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
|
||||
const struct ArrayList* would_delete,
|
||||
const struct ArrayList* deleted_paths);
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink);
|
||||
/* receiver_process with an escape hatch for the commit-style (late) deletion:
|
||||
when `pending_manifest` is non-NULL the receiver does NOT delete at
|
||||
STATUS_FINISHED itself; instead it stores the owned keep-set manifest there
|
||||
(leaving *pending_manifest untouched on early modes/errors) so the caller can
|
||||
commit the deletion only after its disk writer has fully drained. Likewise,
|
||||
when `pending_plans` is non-NULL the --delete-delay per-directory session is
|
||||
handed to the caller instead of being committed at STATUS_FINISHED. Pass NULL
|
||||
for either to keep the default behaviour (delete before the success frame). */
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans);
|
||||
/* receiver_process_pending() with an explicit observer context for a
|
||||
per-directory delete session that is handed to the caller via
|
||||
`pending_plans`. The session outlives this call (the -m pipeline commits it
|
||||
after joining its disk writer), so its observer context must too: pass a
|
||||
long-lived object such as PipelineContextReceiver.delete_ctx. When
|
||||
`delete_ctx` is NULL an internal stack context is used, which is only safe
|
||||
when the session is committed before returning (the default behaviour). */
|
||||
int receiver_process_pending_ctx(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest,
|
||||
DeletePlanSession** pending_plans,
|
||||
ReceiverDeleteContext* delete_ctx);
|
||||
int receiver_receive_files(Config* config, int file_descriptor);
|
||||
|
||||
/* ---- Connection time bounds (anti-slowloris) ----
|
||||
* receiver_process_pending() aborts a connection that makes no forward progress
|
||||
* (only STATUS_KEEPALIVE/STATUS_ABORT frames) beyond a wall-clock idle limit,
|
||||
* and enforces a hard cap on the whole session. Both are CLOCK_MONOTONIC
|
||||
* deltas, independent of the per-message poll deadline, so a 60 s (or
|
||||
* --timeout) receive window can never reset them. Defaults are deliberately
|
||||
* generous (see MAX_SESSION_IDLE_SEC / MAX_SESSION_WALL_SEC in receiver.c). */
|
||||
|
||||
/* Test seam: override the idle/session wall-clock limits (0 = abort on the
|
||||
* next status). Always restore with receiver_reset_time_limits(). */
|
||||
void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec);
|
||||
void receiver_reset_time_limits(void);
|
||||
/* Pure predicate over explicit monotonic timestamps, exposed so the bound is
|
||||
* unit-testable without sleeping. True when either the idle or the overall
|
||||
* session limit has elapsed. */
|
||||
bool receiver_time_limit_exceeded(const struct timespec* session_start,
|
||||
const struct timespec* last_progress, const struct timespec* now);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,331 +0,0 @@
|
||||
#include "receiver_pipeline.h"
|
||||
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <threads.h>
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue,
|
||||
int file_descriptor, SSL* ssl) {
|
||||
PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver));
|
||||
if (context == NULL)
|
||||
return NULL;
|
||||
context->config = config;
|
||||
context->queue = queue;
|
||||
context->file_descriptor = file_descriptor;
|
||||
context->ssl = ssl;
|
||||
context->outcomes.entries = NULL;
|
||||
context->outcomes.count = 0;
|
||||
context->outcomes.capacity = 0;
|
||||
dir_time_list_init(&context->dir_times);
|
||||
protocol_session_init(&context->session, file_descriptor, file_descriptor);
|
||||
protocol_session_set_ssl(&context->session, ssl);
|
||||
context->receiver_done = false;
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->deferred_manifest = NULL;
|
||||
context->deferred_plans = NULL;
|
||||
context->delete_ctx.stats = NULL;
|
||||
context->delete_ctx.deleted_paths = NULL;
|
||||
context->delete_limit_reached = false;
|
||||
context->failed_entries = 0;
|
||||
memset(&context->stats, 0, sizeof(context->stats));
|
||||
context->would_delete = NULL;
|
||||
context->deleted_paths = NULL;
|
||||
atomic_init(&context->cancelled, false);
|
||||
int init = 0;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_full) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_empty) != thrd_success)
|
||||
goto fail;
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
context->would_delete = array_list_create(free);
|
||||
if (!context->would_delete)
|
||||
goto fail;
|
||||
/* The actually-removed path list is only needed to render rsync's
|
||||
`deleting PATH` lines, which the client requests via report_deletes
|
||||
(--info=del / -i / --out-format under --delete). A plain --delete run must
|
||||
not allocate it or observe every removal. */
|
||||
if (config->report_deletes) {
|
||||
context->deleted_paths = array_list_create(free);
|
||||
if (!context->deleted_paths)
|
||||
goto fail;
|
||||
}
|
||||
return context;
|
||||
|
||||
fail:
|
||||
log_perror("Error initializing synchronization objects");
|
||||
if (init >= 3)
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
if (init >= 2)
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&context->mutex);
|
||||
/* Free every list that was already created before the failing allocation:
|
||||
`context` itself is freed below, so they would otherwise leak. */
|
||||
if (context->would_delete)
|
||||
array_list_delete(context->would_delete);
|
||||
if (context->deleted_paths)
|
||||
array_list_delete(context->deleted_paths);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
|
||||
config_delete(context->config);
|
||||
if (context->deferred_manifest)
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
if (context->deferred_plans)
|
||||
delete_plan_session_destroy(context->deferred_plans);
|
||||
queue_destroy(context->queue);
|
||||
receiver_outcomes_destroy(&context->outcomes);
|
||||
dir_time_list_free(&context->dir_times);
|
||||
if (context->would_delete)
|
||||
array_list_delete(context->would_delete);
|
||||
if (context->deleted_paths)
|
||||
array_list_delete(context->deleted_paths);
|
||||
mtx_destroy(&context->mutex);
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
free(context);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes) {
|
||||
if (context == NULL)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
context->max_queue_bytes = max_bytes;
|
||||
context->queued_bytes = 0;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes) {
|
||||
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
if (released_bytes >= context->queued_bytes)
|
||||
context->queued_bytes = 0;
|
||||
else
|
||||
context->queued_bytes -= released_bytes;
|
||||
cnd_signal(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) {
|
||||
if (context == NULL || file == NULL)
|
||||
return false;
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
mtx_lock(&context->mutex);
|
||||
while (!atomic_load(&context->cancelled)) {
|
||||
bool blocked_by_count = queue_is_full(context->queue);
|
||||
bool blocked_by_budget = false;
|
||||
if (context->max_queue_bytes > 0) {
|
||||
size_t budget = context->max_queue_bytes;
|
||||
size_t used = context->queued_bytes;
|
||||
if (used >= budget) {
|
||||
blocked_by_budget = true;
|
||||
} else if (file_bytes > budget - used) {
|
||||
/* A single payload larger than the whole budget (not possible with
|
||||
the per-file receive cap) is only admitted to an empty pipeline so
|
||||
the wait can never deadlock. */
|
||||
blocked_by_budget = used != 0;
|
||||
}
|
||||
}
|
||||
if (!blocked_by_count && !blocked_by_budget)
|
||||
break;
|
||||
cnd_wait(&context->condition_not_full, &context->mutex);
|
||||
}
|
||||
if (atomic_load(&context->cancelled)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (!queue_enqueue(context->queue, file)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
context->queued_bytes += file_bytes;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool receiver_enqueue_file(File* file, void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
if (file && file->matched_bytes > 0) {
|
||||
mtx_lock(&context->mutex);
|
||||
context->stats.matched_data += file->matched_bytes;
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
return pipeline_context_receiver_enqueue_file(context, file);
|
||||
}
|
||||
|
||||
/* Early delete modes (--delete-before/--delete-during) commit the manifest
|
||||
inside receiver_process_pending on this thread; record a capped commit so
|
||||
server.c's terminal frame can report STATUS_DELETE_LIMIT. The plain bool is
|
||||
safe: receive_thread writes it before the main thread joins the thread. */
|
||||
static void receiver_pipeline_note_delete_limit(void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static void receiver_thread_fail(PipelineContextReceiver* context) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
int receive_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
int file_descriptor = context->file_descriptor;
|
||||
const Config* config = context->config;
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
ReceiverSink sink = {receiver_enqueue_file,
|
||||
context,
|
||||
false,
|
||||
false,
|
||||
NULL,
|
||||
receiver_pipeline_note_delete_limit,
|
||||
&context->stats,
|
||||
context->would_delete,
|
||||
context->deleted_paths};
|
||||
if (receiver_process_pending_ctx((Config*)config, file_descriptor, &sink,
|
||||
&context->deferred_manifest, &context->deferred_plans,
|
||||
&context->delete_ctx) != 0) {
|
||||
receiver_thread_fail(context);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
mtx_lock(&context->mutex);
|
||||
context->receiver_done = true;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
|
||||
int write_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
bool save_to_disk = context->config->save_to_disk;
|
||||
char* root_directory = str_dup(context->config->receive_root_directory);
|
||||
mtx_unlock(&context->mutex);
|
||||
if (save_to_disk && !root_directory) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
File* file =
|
||||
queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty,
|
||||
&context->condition_not_full, &context->receiver_done);
|
||||
if (file == NULL) {
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
FileSaveResult result = FILE_SAVE_SKIPPED;
|
||||
bool created = false;
|
||||
unsigned created_dirs = 0;
|
||||
/* Server-contacting --dry-run: never write. The receiver thread does not
|
||||
enqueue anything on the dry-run path, but this keeps the writer thread
|
||||
provably mutation-free if a data frame ever reached it. */
|
||||
bool dry_run = context->config->dry_run;
|
||||
if (save_to_disk && !dry_run) {
|
||||
result =
|
||||
file_save_to_disk_full_ex(root_directory, file, context->config, &created, &created_dirs);
|
||||
if (result == FILE_SAVE_WRITTEN) {
|
||||
/* Protocol 2.28.0: fold the receiver-observed literal bytes and the
|
||||
created-entry type into the shared stats block under its mutex (the
|
||||
receive thread also writes stats.matched_data). */
|
||||
mtx_lock(&context->mutex);
|
||||
receiver_stats_note_saved(&context->stats, file, created, created_dirs);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
/* --devices parity: a device node that could not be mknod'ed is counted
|
||||
per-run but does NOT abort the transfer. */
|
||||
if (result == FILE_SAVE_FAILED) {
|
||||
mtx_lock(&context->mutex);
|
||||
context->failed_entries++;
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
}
|
||||
/* P7 Wave D: a directory's times are never applied inline (a later child
|
||||
write would clobber them); accumulate the metadata here and let the
|
||||
caller apply it once every writer has drained. */
|
||||
if (!dry_run && result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
/* Record the per-file outcome so a --remove-source-files sender learns
|
||||
which sources were actually written versus skipped on the receiver.
|
||||
Explicit directory entries and recreated device/special nodes have no
|
||||
source and are never acknowledged (mirrors receiver.c). */
|
||||
if (!dry_run && context->config->remove_source_files && !file->is_dir && !file->is_special &&
|
||||
!file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
}
|
||||
}
|
||||
@@ -1,97 +0,0 @@
|
||||
#ifndef RECEIVER_PIPELINE_H
|
||||
#define RECEIVER_PIPELINE_H
|
||||
|
||||
#include <stdatomic.h>
|
||||
#include <stdbool.h>
|
||||
#include <threads.h>
|
||||
|
||||
#include "config.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "receiver.h"
|
||||
#include <openssl/ssl.h>
|
||||
|
||||
typedef struct PipelineContextReceiver {
|
||||
Queue* queue;
|
||||
Config* config;
|
||||
int file_descriptor;
|
||||
SSL* ssl;
|
||||
ProtocolSession session;
|
||||
ReceiverOutcomes outcomes;
|
||||
mtx_t mutex;
|
||||
cnd_t condition_not_full;
|
||||
cnd_t condition_not_empty;
|
||||
bool receiver_done;
|
||||
atomic_bool cancelled;
|
||||
/* Aggregate payload bytes that have been received but not yet released by
|
||||
the disk writer (queued or in the writer's hand). Guarded by `mutex`.
|
||||
When `max_queue_bytes` is non-zero the receiver blocks before enqueuing
|
||||
once this total would exceed it, so decompressed/copied file payloads
|
||||
buffered ahead of a slow disk writer respect the per-connection memory
|
||||
budget instead of growing without bound. */
|
||||
size_t queued_bytes;
|
||||
size_t max_queue_bytes;
|
||||
/* Keep-set manifest for the commit-style (late) deletion
|
||||
(--delete/--delete-after/--delete-delay). receive_thread parses the whole
|
||||
protocol stream but hands the manifest here instead of deleting while the
|
||||
disk writer may still be draining; the caller (server.c) commits the
|
||||
deletion after both threads have joined, so no extra is removed unless the
|
||||
transfer truly succeeded. NULL in the early delete modes (which delete at
|
||||
the manifest). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* Per-directory delete session for --delete-delay: receive_thread snapshots
|
||||
each plan's extras as it arrives and hands the session here instead of
|
||||
committing while the disk writer may still be draining; server.c commits it
|
||||
after both threads joined. NULL for every other timing. */
|
||||
DeletePlanSession* deferred_plans;
|
||||
/* Observer context for `deferred_plans`. It must outlive the receive thread
|
||||
(the session is committed by server.c after both threads join), so it lives
|
||||
here rather than on receiver_process_pending()'s stack; receive_thread
|
||||
installs it on the session. */
|
||||
ReceiverDeleteContext delete_ctx;
|
||||
/* Set by server.c when the deferred delete commit hit the --max-delete
|
||||
budget; the terminal success frame then carries STATUS_DELETE_LIMIT
|
||||
(rsync exit 25) while the transfer itself still succeeds. */
|
||||
bool delete_limit_reached;
|
||||
/* P7 Wave D: directory metadata collected by write_thread from received
|
||||
directory entries. Only write_thread mutates it (before it joins); the
|
||||
caller (server.c) applies it after the delete/delay-updates phase. */
|
||||
DirTimeList dir_times;
|
||||
/* End-of-transfer wire counters (protocol 2.25.0). receive_thread accumulates
|
||||
matched_data under `mutex`; server.c adds the delete-commit tallies after
|
||||
both threads join and emits the STATUS_STATS frame. */
|
||||
ReceiverStats stats;
|
||||
/* -n/--dry-run --delete would-delete path list, collected by receive_thread
|
||||
and reported in the STATUS_STATS frame. */
|
||||
struct ArrayList* would_delete;
|
||||
/* --info=del actually-removed path list, collected by the deferred delete
|
||||
commit in server.c and reported in the STATUS_STATS frame. */
|
||||
struct ArrayList* deleted_paths;
|
||||
/* Per-run count of entries that failed to materialize without aborting the
|
||||
stream (currently ONLY a --devices mknod EPERM/EACCES). write_thread
|
||||
increments it under `mutex`; server.c turns a nonzero count into a non-OK
|
||||
terminal status so the client exits non-zero. */
|
||||
size_t failed_entries;
|
||||
} PipelineContextReceiver;
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
|
||||
int file_descriptor, SSL* ssl);
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context);
|
||||
/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes);
|
||||
/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue
|
||||
is full by element count or when adding `file` would push queued_bytes over
|
||||
the configured byte limit; waits until the disk writer releases bytes.
|
||||
Takes ownership of `file` on success and destroys it on failure/cancel. */
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file);
|
||||
/* Account for `released_bytes` of payload memory that has been freed by the
|
||||
disk writer, unblocking a receiver that is waiting on the byte limit. */
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes);
|
||||
int receive_thread(void* pipeline_context);
|
||||
int write_thread(void* pipeline_context);
|
||||
|
||||
#endif
|
||||
+284
-1441
File diff suppressed because it is too large
Load Diff
@@ -1,303 +0,0 @@
|
||||
#include "server_cli.h"
|
||||
#include "charset.h"
|
||||
#include "credentials.h"
|
||||
#include "utils.h"
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/socket.h>
|
||||
|
||||
#define set_error utils_set_error
|
||||
|
||||
void server_cli_options_default(ServerCliOptions* opts) {
|
||||
if (!opts)
|
||||
return;
|
||||
memset(opts, 0, sizeof(*opts));
|
||||
opts->destination_root = ".";
|
||||
opts->port = 8080;
|
||||
opts->bind_family = AF_UNSPEC;
|
||||
}
|
||||
|
||||
static bool arg_is(const char* arg, const char* name) {
|
||||
return strcmp(arg, name) == 0;
|
||||
}
|
||||
|
||||
/* Match "--opt" against "--opt=value" / separate-value forms; on the "=" form
|
||||
* *value receives the inline value. Returns true when the argument is the
|
||||
* named option in either form. */
|
||||
static bool arg_has_value(const char* arg, const char* name, const char** value) {
|
||||
if (strcmp(arg, name) == 0)
|
||||
return true; /* separate form; caller takes the next argv slot */
|
||||
size_t name_len = strlen(name);
|
||||
if (strncmp(arg, name, name_len) == 0 && arg[name_len] == '=') {
|
||||
*value = arg + name_len + 1;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static int parse_port_arg(const char* value, int* port, char* err, size_t err_size) {
|
||||
char* end;
|
||||
long p = strtol(value, &end, 10);
|
||||
if (*end != '\0' || p <= 0 || p > 65535) {
|
||||
char* escaped = output_escape(value, false);
|
||||
set_error(err, err_size, "invalid port '%s' (must be 1-65535)",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
return -1;
|
||||
}
|
||||
*port = (int)p;
|
||||
return 0;
|
||||
}
|
||||
|
||||
int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err, size_t err_size) {
|
||||
if (err && err_size)
|
||||
err[0] = '\0';
|
||||
server_cli_options_default(opts);
|
||||
|
||||
for (int i = 1; i < argc; i++) {
|
||||
const char* inline_value = NULL;
|
||||
if (arg_is(argv[i], "--help")) {
|
||||
opts->show_help = true;
|
||||
return 1;
|
||||
} else if (arg_is(argv[i], "--stdio")) {
|
||||
opts->stdio_mode = true;
|
||||
} else if (arg_is(argv[i], "--daemon")) {
|
||||
opts->daemon_mode = true;
|
||||
} else if (arg_is(argv[i], "--no-detach")) {
|
||||
opts->no_detach = true;
|
||||
} else if (arg_is(argv[i], "-v") || arg_is(argv[i], "--verbose")) {
|
||||
opts->verbose = true;
|
||||
} else if (arg_is(argv[i], "--tls")) {
|
||||
opts->use_tls = true;
|
||||
} else if (arg_is(argv[i], "--cert")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --cert");
|
||||
return -1;
|
||||
}
|
||||
opts->tls_cert = argv[++i];
|
||||
} else if (arg_is(argv[i], "--key")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --key");
|
||||
return -1;
|
||||
}
|
||||
opts->tls_key = argv[++i];
|
||||
} else if (arg_is(argv[i], "--ca")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --ca");
|
||||
return -1;
|
||||
}
|
||||
opts->tls_ca = argv[++i];
|
||||
} else if (arg_is(argv[i], "--client-cn")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --client-cn");
|
||||
return -1;
|
||||
}
|
||||
opts->client_cn = argv[++i];
|
||||
} else if (arg_is(argv[i], "--destination-root")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --destination-root");
|
||||
return -1;
|
||||
}
|
||||
opts->destination_root = argv[++i];
|
||||
opts->destination_root_set = true;
|
||||
} else if (arg_has_value(argv[i], "--password-file", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --password-file");
|
||||
return -1;
|
||||
}
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->password_file = inline_value;
|
||||
} else if (arg_has_value(argv[i], "--early-input", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --early-input");
|
||||
return -1;
|
||||
}
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->early_input_file = inline_value;
|
||||
} else if (arg_has_value(argv[i], "--hash-credentials", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --hash-credentials");
|
||||
return -1;
|
||||
}
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->hash_credentials_file = inline_value;
|
||||
} else if (arg_has_value(argv[i], "--iterations", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --iterations");
|
||||
return -1;
|
||||
}
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
char* end = NULL;
|
||||
long n = strtol(inline_value, &end, 10);
|
||||
if (!end || *end != '\0' || n < (long)CREDENTIAL_MIN_ITERS ||
|
||||
n > (long)CREDENTIAL_MAX_ITERS) {
|
||||
set_error(err, err_size, "--iterations must be in [%u,%u], got '%s'", CREDENTIAL_MIN_ITERS,
|
||||
CREDENTIAL_MAX_ITERS, inline_value);
|
||||
return -1;
|
||||
}
|
||||
opts->hash_iterations = (uint32_t)n;
|
||||
opts->hash_iterations_set = true;
|
||||
} else if (arg_is(argv[i], "--address")) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --address");
|
||||
return -1;
|
||||
}
|
||||
opts->bind_address = argv[++i];
|
||||
} else if (arg_is(argv[i], "-4") || arg_is(argv[i], "--ipv4")) {
|
||||
if (opts->bind_family == AF_INET6) {
|
||||
set_error(err, err_size, "--ipv4 and --ipv6 are mutually exclusive");
|
||||
return -1;
|
||||
}
|
||||
opts->bind_family = AF_INET;
|
||||
} else if (arg_is(argv[i], "-6") || arg_is(argv[i], "--ipv6")) {
|
||||
if (opts->bind_family == AF_INET) {
|
||||
set_error(err, err_size, "--ipv4 and --ipv6 are mutually exclusive");
|
||||
return -1;
|
||||
}
|
||||
opts->bind_family = AF_INET6;
|
||||
} else if (arg_is(argv[i], "--allow-delete")) {
|
||||
opts->allow_delete = true;
|
||||
} else if (arg_is(argv[i], "--trust-sender")) {
|
||||
opts->trust_sender = true;
|
||||
} else if (arg_is(argv[i], "--no-super")) {
|
||||
opts->no_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-super")) {
|
||||
opts->allow_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-unauthenticated")) {
|
||||
opts->allow_unauthenticated = true;
|
||||
} else if (arg_has_value(argv[i], "--iconv", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --iconv");
|
||||
return -1;
|
||||
}
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->iconv_spec = inline_value;
|
||||
} else if (arg_is(argv[i], "-p") || arg_has_value(argv[i], "--port", &inline_value)) {
|
||||
if (inline_value) {
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(inline_value, &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
} else {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for %s", argv[i]);
|
||||
return -1;
|
||||
}
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
}
|
||||
} else {
|
||||
if (arg_has_value(argv[i], "--config", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --config");
|
||||
return -1;
|
||||
}
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->config_path = inline_value;
|
||||
} else if (arg_has_value(argv[i], "--dparam", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for --dparam");
|
||||
return -1;
|
||||
}
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
const char** grown =
|
||||
realloc((char**)opts->dparams, (size_t)(opts->dparam_count + 1) * sizeof(const char*));
|
||||
if (!grown) {
|
||||
set_error(err, err_size, "out of memory parsing --dparam");
|
||||
return -1;
|
||||
}
|
||||
opts->dparams = grown;
|
||||
opts->dparams[opts->dparam_count++] = inline_value;
|
||||
} else if (argv[i][0] == '-') {
|
||||
char* escaped = output_escape(argv[i], false);
|
||||
set_error(err, err_size, "unknown option: %s", escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
return -1;
|
||||
} else {
|
||||
set_error(err, err_size, "unexpected argument '%s'", argv[i]);
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Cross-mode validation. */
|
||||
if (opts->stdio_mode && opts->daemon_mode) {
|
||||
set_error(err, err_size, "--stdio and --daemon are mutually exclusive");
|
||||
return -1;
|
||||
}
|
||||
if (opts->daemon_mode && opts->destination_root_set) {
|
||||
set_error(err, err_size,
|
||||
"--destination-root cannot be combined with --daemon (module paths "
|
||||
"replace it)");
|
||||
return -1;
|
||||
}
|
||||
if (!opts->daemon_mode &&
|
||||
(opts->config_path != NULL || opts->dparam_count > 0 || opts->no_detach ||
|
||||
opts->password_file != NULL || opts->early_input_file != NULL)) {
|
||||
set_error(err, err_size,
|
||||
"--config, --dparam, --no-detach, --password-file, and --early-input require "
|
||||
"--daemon");
|
||||
return -1;
|
||||
}
|
||||
if (opts->hash_credentials_file != NULL && (opts->daemon_mode || opts->stdio_mode)) {
|
||||
set_error(err, err_size, "--hash-credentials cannot be combined with --daemon or --stdio");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->no_super) {
|
||||
set_error(err, err_size, "--allow-super and --no-super are mutually exclusive");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->daemon_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is for a locally-launched standalone TCP server; daemon modules opt "
|
||||
"in per module with 'client owner = yes'");
|
||||
return -1;
|
||||
}
|
||||
/* --stdio is the SSH transport: the remote server argv is composed by the
|
||||
* CLIENT (directly and via --remote-option), so a client could otherwise pass
|
||||
* --allow-super to a root --stdio receiver and defeat the C3 secure default.
|
||||
* Never honor it there; the super mode stays forced OFF. An operator who
|
||||
* must keep the historical permissive behavior over SSH has to launch the
|
||||
* receiver through a forced command, not via client-composed argv. */
|
||||
if (opts->allow_super && opts->stdio_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is not accepted with --stdio (the remote argv is client-composed; "
|
||||
"use a forced command if the default must hold)");
|
||||
return -1;
|
||||
}
|
||||
if (opts->hash_iterations_set && opts->hash_credentials_file == NULL) {
|
||||
set_error(err, err_size, "--iterations requires --hash-credentials");
|
||||
return -1;
|
||||
}
|
||||
/* --iconv: reject a malformed CONVERT_SPEC or an unsupported charset name at
|
||||
startup (a probe iconv_open is attempted). */
|
||||
if (opts->iconv_spec != NULL && !charset_spec_valid(opts->iconv_spec)) {
|
||||
set_error(err, err_size, "--iconv requires LOCAL[,REMOTE] charset names supported by iconv");
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
void server_cli_options_free(ServerCliOptions* opts) {
|
||||
if (!opts)
|
||||
return;
|
||||
free((char**)opts->dparams);
|
||||
opts->dparams = NULL;
|
||||
opts->dparam_count = 0;
|
||||
}
|
||||
@@ -1,79 +0,0 @@
|
||||
#ifndef SERVER_CLI_H
|
||||
#define SERVER_CLI_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* Parsed fastsync-server command line. All string members are borrowed
|
||||
* pointers into the original argv (valid for the life of the argv array the
|
||||
* caller passed to server_cli_parse); dparams points at the raw --dparam
|
||||
* argument strings. No member owns heap memory. */
|
||||
typedef struct ServerCliOptions {
|
||||
bool stdio_mode; /* --stdio */
|
||||
bool daemon_mode; /* --daemon */
|
||||
bool no_detach; /* --no-detach */
|
||||
bool verbose; /* -v / --verbose */
|
||||
bool show_help; /* --help */
|
||||
bool use_tls; /* --tls */
|
||||
const char* tls_cert; /* --cert */
|
||||
const char* tls_key; /* --key */
|
||||
const char* tls_ca; /* --ca */
|
||||
const char* client_cn; /* --client-cn */
|
||||
bool destination_root_set; /* an explicit --destination-root was given */
|
||||
const char* destination_root; /* --destination-root value ("." if unset) */
|
||||
bool port_set; /* an explicit -p was given */
|
||||
int port; /* -p value (default 8080 when unset) */
|
||||
const char* config_path; /* --config value, or NULL */
|
||||
const char* password_file; /* --password-file value, or NULL (daemon) */
|
||||
const char* early_input_file; /* --early-input value, or NULL (daemon) */
|
||||
/* --hash-credentials=FILE: read `user:password` lines from FILE and print
|
||||
* new-format credential-store lines to stdout, then exit. Standalone mode
|
||||
* (mutually exclusive with --daemon/--stdio). */
|
||||
const char* hash_credentials_file;
|
||||
bool hash_iterations_set; /* an explicit --iterations was given */
|
||||
uint32_t hash_iterations; /* --iterations value (default CREDENTIAL_DEFAULT_ITERS) */
|
||||
const char** dparams; /* raw --dparam override strings */
|
||||
int dparam_count;
|
||||
const char* bind_address; /* --address */
|
||||
int bind_family; /* AF_UNSPEC / AF_INET / AF_INET6 */
|
||||
bool allow_delete; /* --allow-delete */
|
||||
bool trust_sender; /* --trust-sender */
|
||||
bool allow_unauthenticated; /* --allow-unauthenticated */
|
||||
/* --no-super: operator veto forcing SUPER_MODE_OFF for every connection, so
|
||||
* the receiver never attempts super-user activities (ownership application,
|
||||
* device-node creation) even when running as root. Applies to --stdio and
|
||||
* --daemon alike; also makes the server refuse any client --copy-as. */
|
||||
bool no_super; /* --no-super */
|
||||
/* --allow-super: locally-launched standalone TCP listener opt-in that keeps
|
||||
* the historical permissive behavior for a PRIVILEGED (root) receiver.
|
||||
* Without it a root standalone server forces SUPER_MODE_OFF, so a client
|
||||
* --devices / --write-devices / --super / ownership request cannot make it
|
||||
* create device nodes, write raw devices, or apply client-chosen ownership.
|
||||
* It is rejected for --stdio: that path's remote argv is composed by the
|
||||
* client (directly and via --remote-option), so it must never opt a root
|
||||
* receiver back into super mode. Non-root receivers are unaffected (the
|
||||
* kernel refuses the confined attempts). The daemon path instead uses the
|
||||
* per-module `client owner = yes` opt-in. */
|
||||
bool allow_super; /* --allow-super */
|
||||
/* --iconv=CONVERT_SPEC: the server's own LOCAL charset declaration. The
|
||||
* client's full spec rides the wire config frame anyway; when the server is
|
||||
* started with its own --iconv, its LOCAL half overrides the local charset
|
||||
* the client assumed so the server converts received names to ITS charset.
|
||||
* Borrowed pointer into argv (never owns heap). */
|
||||
const char* iconv_spec; /* --iconv value, or NULL */
|
||||
} ServerCliOptions;
|
||||
|
||||
/* Parse argc/argv into *opts. Zero-initialize *opts before calling (or use
|
||||
* server_cli_options_default). Returns:
|
||||
* 1 -- --help was requested (opts->show_help set; caller prints usage).
|
||||
* 0 -- parsed successfully.
|
||||
* -1 -- invalid arguments (err is filled with the reason).
|
||||
*/
|
||||
void server_cli_options_default(ServerCliOptions* opts);
|
||||
int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err, size_t err_size);
|
||||
/* Release the only heap the parsed options own (the dparams pointer array; the
|
||||
* strings it points at are borrowed from argv and are not freed). Safe to
|
||||
* call on a zero-initialized/defaulted struct. */
|
||||
void server_cli_options_free(ServerCliOptions* opts);
|
||||
#endif
|
||||
+7
-11
@@ -1,19 +1,17 @@
|
||||
#include "log.h"
|
||||
#include "array_list.h"
|
||||
#include "protocol.h"
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
ArrayList* array_list_create(void (*item_destroyer)(void* item)) {
|
||||
ArrayList* list = (ArrayList*)protocol_alloc(sizeof(ArrayList));
|
||||
ArrayList* list = (ArrayList*)malloc(sizeof(ArrayList));
|
||||
if (list == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not allocate memory for array list struct");
|
||||
log_perror("ERROR: Could not allocate memory for array list struct");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
list->items = protocol_alloc(INITIAL_ARRAY_SIZE * sizeof(void*));
|
||||
list->items = malloc(INITIAL_ARRAY_SIZE * sizeof(void*));
|
||||
if (list->items == NULL) {
|
||||
free(list);
|
||||
return NULL;
|
||||
@@ -40,14 +38,12 @@ void array_list_delete(ArrayList* array_list) {
|
||||
static bool array_list_extend(ArrayList* array_list) {
|
||||
if (array_list == NULL)
|
||||
return false;
|
||||
if (array_list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_capacity = array_list->capacity * 2;
|
||||
if (new_capacity == 0)
|
||||
new_capacity = INITIAL_ARRAY_SIZE;
|
||||
void* new_items = protocol_realloc(array_list->items, new_capacity * sizeof(void*));
|
||||
void* new_items = realloc(array_list->items, new_capacity * sizeof(void*));
|
||||
if (new_items == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s", "ERROR: Could not reallocate memory for array list items");
|
||||
log_perror("ERROR: Could not reallocate memory for array list items");
|
||||
return false;
|
||||
}
|
||||
array_list->items = new_items;
|
||||
@@ -71,9 +67,9 @@ void** array_list_to_array(const ArrayList* array_list) {
|
||||
if (array_list == NULL) {
|
||||
return NULL;
|
||||
}
|
||||
void** array = protocol_alloc(array_list->size * sizeof(void*));
|
||||
void** array = malloc(array_list->size * sizeof(void*));
|
||||
if (array == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s", "Could not malloc space for array from array list!");
|
||||
log_perror("Could not malloc space for array from array list!");
|
||||
return NULL;
|
||||
}
|
||||
memcpy(array, array_list->items, array_list->size * sizeof(void*));
|
||||
|
||||
@@ -1,195 +0,0 @@
|
||||
#include "batch.h"
|
||||
#include "data.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include <errno.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* Serialization metadata mode for the batch stream, captured from the config at
|
||||
* batch_write_header time. The header persists it into the file so a batch is
|
||||
* self-describing about whether per-entry metadata was CAPTURED in the stream:
|
||||
* batch_read_apply re-reads it from the file (not from the reading config) to
|
||||
* decode the chunk records correctly. Which attributes are actually APPLIED,
|
||||
* however, comes from the INVOKING process's per-attribute config (the
|
||||
* FileAttrPolicy and the dir-metadata gate), so a batch written with -M is NOT
|
||||
* automatically applied identically by an invoking process with a different
|
||||
* -p/-t/-o/-g: --read-batch must be invoked with the same -p/-t/-o/-g as the
|
||||
* write side (rsync requires the same options). The batch driver is a single
|
||||
* sequential scan pass within one thread, so this module-level flag is safe. */
|
||||
static bool batch_metadata_mode = false;
|
||||
|
||||
static bool write_all_bytes(int fd, const void* data, size_t size) {
|
||||
const unsigned char* p = (const unsigned char*)data;
|
||||
size_t done = 0;
|
||||
while (done < size) {
|
||||
ssize_t n = write(fd, p + done, size - done);
|
||||
if (n < 0 && errno == EINTR)
|
||||
continue;
|
||||
if (n <= 0)
|
||||
return false;
|
||||
done += (size_t)n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool batch_write_header(int fd, const Config* config) {
|
||||
if (fd < 0)
|
||||
return false;
|
||||
batch_metadata_mode = config != NULL && config->use_metadata;
|
||||
if (!write_all_bytes(fd, BATCH_MAGIC, BATCH_MAGIC_LEN))
|
||||
return false;
|
||||
unsigned char version = BATCH_FORMAT_VERSION;
|
||||
if (!write_all_bytes(fd, &version, 1))
|
||||
return false;
|
||||
unsigned char mode = batch_metadata_mode ? 1 : 0;
|
||||
return write_all_bytes(fd, &mode, 1);
|
||||
}
|
||||
|
||||
bool batch_write_chunk(int fd, Chunk* chunk) {
|
||||
if (fd < 0 || chunk == NULL)
|
||||
return false;
|
||||
Data* serialized = chunk_serialize(chunk, batch_metadata_mode);
|
||||
if (serialized == NULL)
|
||||
return false;
|
||||
bool ok = false;
|
||||
unsigned long long length = (unsigned long long)serialized->size;
|
||||
if (length > BATCH_MAX_RECORD) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: record size %llu exceeds the %llu-byte cap", length,
|
||||
(unsigned long long)BATCH_MAX_RECORD);
|
||||
} else if (write_all_bytes(fd, &length, sizeof(length)) &&
|
||||
(length == 0 || write_all_bytes(fd, serialized->data, (size_t)length))) {
|
||||
ok = true;
|
||||
}
|
||||
data_destroy(serialized);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Read exactly `size` bytes. Returns true on success. On reaching EOF, sets
|
||||
* *clean_eof only when no bytes had been read yet (a clean boundary) and returns
|
||||
* that value, so a truncated record (EOF mid-read) yields false. */
|
||||
static bool read_exact(int fd, void* data, size_t size, bool* clean_eof) {
|
||||
unsigned char* p = (unsigned char*)data;
|
||||
size_t done = 0;
|
||||
while (done < size) {
|
||||
ssize_t n = read(fd, p + done, size - done);
|
||||
if (n < 0 && errno == EINTR)
|
||||
continue;
|
||||
if (n == 0) {
|
||||
if (clean_eof)
|
||||
*clean_eof = done == 0;
|
||||
return done == 0;
|
||||
}
|
||||
if (n < 0)
|
||||
return false;
|
||||
done += (size_t)n;
|
||||
}
|
||||
if (clean_eof)
|
||||
*clean_eof = false;
|
||||
return true;
|
||||
}
|
||||
|
||||
int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
if (fd < 0 || dest_root == NULL || dest_root[0] == '\0')
|
||||
return -1;
|
||||
|
||||
/* Directory metadata is deferred to the end of the apply (a child write would
|
||||
* otherwise clobber its parent's mtime/mode). The batch header's single
|
||||
* metadata bit only says whether metadata is present in the stream; which
|
||||
* attributes are APPLIED comes from the invoking process's config, so
|
||||
* --read-batch must be invoked with the same -p/-t/-o/-g as the write side
|
||||
* (rsync requires the same options). The identity snapshot is activated so
|
||||
* -o/-g and the explicit ownership flags can apply. */
|
||||
DirTimeList dir_times;
|
||||
dir_time_list_init(&dir_times);
|
||||
int result = -1;
|
||||
if (!identity_set_active(config)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not activate the identity policy");
|
||||
goto done;
|
||||
}
|
||||
|
||||
char magic[BATCH_MAGIC_LEN];
|
||||
bool eof = false;
|
||||
if (!read_exact(fd, magic, BATCH_MAGIC_LEN, &eof) || eof ||
|
||||
memcmp(magic, BATCH_MAGIC, BATCH_MAGIC_LEN) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad magic)");
|
||||
goto done;
|
||||
}
|
||||
unsigned char version;
|
||||
if (!read_exact(fd, &version, 1, &eof) || eof || version != BATCH_FORMAT_VERSION) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad or missing format version)");
|
||||
goto done;
|
||||
}
|
||||
unsigned char mode;
|
||||
if (!read_exact(fd, &mode, 1, &eof) || eof || (mode != 0 && mode != 1)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad metadata flag)");
|
||||
goto done;
|
||||
}
|
||||
bool use_metadata = mode == 1;
|
||||
|
||||
while (1) {
|
||||
unsigned long long length;
|
||||
if (!read_exact(fd, &length, sizeof(length), &eof)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated length prefix");
|
||||
goto done;
|
||||
}
|
||||
if (eof)
|
||||
break; /* clean end of stream */
|
||||
if (length == 0 || length > BATCH_MAX_RECORD) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected record length %llu (valid range 1..%llu)",
|
||||
length, (unsigned long long)BATCH_MAX_RECORD);
|
||||
goto done;
|
||||
}
|
||||
char* record = (char*)malloc((size_t)length);
|
||||
if (record == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not allocate a %llu-byte record", length);
|
||||
goto done;
|
||||
}
|
||||
if (!read_exact(fd, record, (size_t)length, &eof) || eof) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated chunk record");
|
||||
free(record);
|
||||
goto done;
|
||||
}
|
||||
Data* data = data_create(record, (size_t)length);
|
||||
if (data == NULL)
|
||||
goto done; /* data_create frees `record` on failure */
|
||||
Chunk* chunk = chunk_deserialize(data, use_metadata);
|
||||
data_destroy(data);
|
||||
if (chunk == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected malformed chunk record");
|
||||
goto done;
|
||||
}
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* file = chunk->items[i];
|
||||
chunk->items[i] = NULL;
|
||||
if (file == NULL)
|
||||
continue;
|
||||
FileSaveResult save = file_save_to_disk_full(dest_root, file, config);
|
||||
/* Accumulate directory metadata (when it applies) before the File is
|
||||
* destroyed; applied once the whole stream has been consumed. */
|
||||
if (save != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(config) &&
|
||||
!dir_time_list_add(&dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
chunk_destroy(chunk);
|
||||
goto done;
|
||||
}
|
||||
file_destroy(file);
|
||||
if (save == FILE_SAVE_ERROR) {
|
||||
chunk_destroy(chunk);
|
||||
goto done;
|
||||
}
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
dir_metadata_list_apply(&dir_times, dest_root, config);
|
||||
result = 0;
|
||||
|
||||
done:
|
||||
identity_clear_active();
|
||||
dir_time_list_free(&dir_times);
|
||||
return result;
|
||||
}
|
||||
@@ -1,28 +0,0 @@
|
||||
#ifndef BATCH_H
|
||||
#define BATCH_H
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
|
||||
/* Phase 6 residual-batch codec. A residual batch is a self-contained
|
||||
* single-file record of a whole source tree: a magic+format-version header
|
||||
* followed by length-prefixed chunk blobs (each built with chunk_serialize),
|
||||
* byte-identical by construction. The batch is a client-only driver feature:
|
||||
* it never crosses the wire, so there is no PROTOCOL_VERSION bump and no server
|
||||
* change. */
|
||||
|
||||
#define BATCH_MAGIC "FSTRESBATCH"
|
||||
#define BATCH_MAGIC_LEN 11
|
||||
#define BATCH_FORMAT_VERSION 1
|
||||
/* Max size of a single length-prefixed record (a whole serialized chunk,
|
||||
* which can span several files). A single source file near the 64 MB wire
|
||||
* limit plus per-file headers can produce a record slightly over 64 MB, so a
|
||||
* large file just under the wire cap may be refused by the batch writer; this
|
||||
* is documented upstream and the failure is clean (the partial batch is
|
||||
* unlinked), never a truncated/corrupt batch. */
|
||||
#define BATCH_MAX_RECORD (64ULL * 1024 * 1024)
|
||||
|
||||
bool batch_write_header(int fd, const Config* config);
|
||||
bool batch_write_chunk(int fd, Chunk* chunk);
|
||||
int batch_read_apply(int fd, const Config* config, const char* dest_root);
|
||||
|
||||
#endif
|
||||
@@ -1,389 +0,0 @@
|
||||
#include "charset.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <errno.h>
|
||||
#include <iconv.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
typedef struct {
|
||||
iconv_t cd;
|
||||
} CharsetConversion;
|
||||
|
||||
/* Process-wide wire conversion descriptor (one direction per process: a client
|
||||
* only sends, a server only receives). CONCURRENCY CONTRACT: iconv_t is not
|
||||
* guaranteed thread-safe, so every conversion MUST run on a single thread at a
|
||||
* time. This holds today -- on the client the conversions run on the sender
|
||||
* thread (in the -m pipeline chunk_serialize/send happen on the sender thread
|
||||
* only), on the server on the receive-loop thread; the descriptor is
|
||||
* initialized on one thread before any transfer thread spawns and torn down
|
||||
* (charset_wire_free) only after all threads have joined. Do not add a
|
||||
* concurrent conversion path (e.g. parallel chunk serialization) without
|
||||
* guarding access with a mutex. */
|
||||
static CharsetConversion* g_wire_conv;
|
||||
|
||||
/* Grow *buf to double capacity, freeing it on failure. realloc preserves the
|
||||
* already-written prefix, so the caller only tracks its write offset. */
|
||||
static bool grow_charset_buffer(char** buf, size_t* cap) {
|
||||
size_t new_cap = *cap * 2;
|
||||
if (new_cap <= *cap) {
|
||||
free(*buf);
|
||||
*buf = NULL;
|
||||
return false;
|
||||
}
|
||||
char* grown = realloc(*buf, new_cap);
|
||||
if (!grown) {
|
||||
free(*buf);
|
||||
*buf = NULL;
|
||||
return false;
|
||||
}
|
||||
*buf = grown;
|
||||
*cap = new_cap;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Throw away any pending shift state so a subsequent conversion starts clean.
|
||||
* The flush output is discarded; for the stateless single-byte/UTF charsets
|
||||
* this feature targets it is a no-op. */
|
||||
static void charset_conversion_reset(const CharsetConversion* conv) {
|
||||
char scratch[64];
|
||||
char* sp = scratch;
|
||||
size_t sl = sizeof(scratch);
|
||||
(void)iconv(conv->cd, NULL, NULL, &sp, &sl);
|
||||
}
|
||||
|
||||
int charset_spec_parse(const char* spec, char** local_out, char** remote_out) {
|
||||
if (!local_out || !remote_out)
|
||||
return -1;
|
||||
*local_out = NULL;
|
||||
*remote_out = NULL;
|
||||
if (!spec || spec[0] == '\0')
|
||||
return -1;
|
||||
char* dup = str_dup(spec);
|
||||
if (!dup)
|
||||
return -1;
|
||||
char* comma = strchr(dup, ',');
|
||||
if (comma) {
|
||||
if (comma == dup || comma[1] == '\0') {
|
||||
free(dup);
|
||||
return -1;
|
||||
}
|
||||
*comma = '\0';
|
||||
*local_out = str_dup(dup);
|
||||
*remote_out = str_dup(comma + 1);
|
||||
free(dup);
|
||||
} else {
|
||||
*local_out = str_dup(dup);
|
||||
*remote_out = str_dup(dup);
|
||||
free(dup);
|
||||
}
|
||||
if (!*local_out || !*remote_out) {
|
||||
free(*local_out);
|
||||
free(*remote_out);
|
||||
*local_out = NULL;
|
||||
*remote_out = NULL;
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
void* charset_conversion_open(const char* from_charset, const char* to_charset) {
|
||||
if (!from_charset || !to_charset)
|
||||
return NULL;
|
||||
iconv_t cd = iconv_open(to_charset, from_charset);
|
||||
if (cd == (iconv_t)-1)
|
||||
return NULL;
|
||||
CharsetConversion* conv = malloc(sizeof(CharsetConversion));
|
||||
if (!conv) {
|
||||
iconv_close(cd);
|
||||
return NULL;
|
||||
}
|
||||
conv->cd = cd;
|
||||
return conv;
|
||||
}
|
||||
|
||||
void charset_conversion_close(void* conversion) {
|
||||
if (!conversion)
|
||||
return;
|
||||
CharsetConversion* conv = (CharsetConversion*)conversion;
|
||||
iconv_close(conv->cd);
|
||||
free(conv);
|
||||
}
|
||||
|
||||
/* Probe a single conversion direction: the from/to charsets both open AND a
|
||||
* representative ASCII name converts to a byte string containing no embedded
|
||||
* NUL (so a target charset like UTF-16 that emits NUL bytes for ordinary ASCII
|
||||
* names is rejected up front -- such an output would be silently truncated by
|
||||
* the C-string wire helpers). */
|
||||
static bool direction_probe_valid(const char* from, const char* to) {
|
||||
if (!from || !to)
|
||||
return false;
|
||||
void* conv = charset_conversion_open(from, to);
|
||||
if (!conv)
|
||||
return false;
|
||||
bool ok = true;
|
||||
char input = 'a';
|
||||
char* in_ptr = &input;
|
||||
size_t in_left = 1;
|
||||
char out_buf[64];
|
||||
char* out_ptr = out_buf;
|
||||
size_t out_left = sizeof(out_buf);
|
||||
if (iconv(((CharsetConversion*)conv)->cd, &in_ptr, &in_left, &out_ptr, &out_left) == (size_t)-1)
|
||||
ok = false;
|
||||
char flush_buf[64];
|
||||
char* flush_ptr = flush_buf;
|
||||
size_t flush_left = sizeof(flush_buf);
|
||||
if (ok &&
|
||||
iconv(((CharsetConversion*)conv)->cd, NULL, NULL, &flush_ptr, &flush_left) == (size_t)-1)
|
||||
ok = false;
|
||||
size_t produced = (size_t)(out_ptr - out_buf);
|
||||
if (ok && produced > 0 && memchr(out_buf, '\0', produced) != NULL)
|
||||
ok = false;
|
||||
charset_conversion_close(conv);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool charset_pair_valid(const char* local, const char* remote) {
|
||||
/* Both ends convert in opposite directions with the same two charsets, so a
|
||||
* valid spec must open (and be NUL-free) in BOTH directions: the sender
|
||||
* opens local->remote, the receiver opens remote->local. */
|
||||
return direction_probe_valid(local, remote) && direction_probe_valid(remote, local);
|
||||
}
|
||||
|
||||
bool charset_spec_valid(const char* spec) {
|
||||
if (!spec)
|
||||
return true;
|
||||
char* local;
|
||||
char* remote;
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
bool ok = charset_pair_valid(local, remote);
|
||||
free(local);
|
||||
free(remote);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool charset_spec_valid_direction(const char* from_charset, const char* to_charset) {
|
||||
return direction_probe_valid(from_charset, to_charset);
|
||||
}
|
||||
|
||||
/* The receiver's conversion is wire charset -> destination charset. rsync's
|
||||
* CONVERT_SPEC is LOCAL,REMOTE and "stays the same whether you're pushing or
|
||||
* pulling", so for a PUSH (FastSync's only direction) the destination end's
|
||||
* charset is the spec's REMOTE half: the client converts LOCAL -> REMOTE on the
|
||||
* sender and the receiver writes the wire bytes verbatim. Only a server that
|
||||
* declares its OWN --iconv (the daemon "charset" analog) has a different local
|
||||
* charset, and then it is that spec's LOCAL half and the receiver converts
|
||||
* wire -> server-local. A dedicated pre-ack check so an impossible direction is
|
||||
* rejected before the connection instead of refusing mid-transfer. */
|
||||
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec) {
|
||||
if (!spec)
|
||||
return true;
|
||||
char* local;
|
||||
char* remote;
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
const char* wire = remote;
|
||||
const char* target_local = remote;
|
||||
char* server_local = NULL;
|
||||
char* server_remote = NULL;
|
||||
if (server_spec) {
|
||||
if (charset_spec_parse(server_spec, &server_local, &server_remote) != 0) {
|
||||
free(local);
|
||||
free(remote);
|
||||
return false;
|
||||
}
|
||||
target_local = server_local;
|
||||
}
|
||||
bool ok = charset_spec_valid_direction(wire, target_local);
|
||||
free(server_local);
|
||||
free(server_remote);
|
||||
free(local);
|
||||
free(remote);
|
||||
return ok;
|
||||
}
|
||||
|
||||
char* charset_convert(const void* conversion, const char* in, int* err_out) {
|
||||
if (!conversion || !in)
|
||||
return NULL;
|
||||
const CharsetConversion* conv = (const CharsetConversion*)conversion;
|
||||
size_t in_len = strlen(in);
|
||||
size_t cap = in_len + 16;
|
||||
char* out = malloc(cap);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t in_left = in_len;
|
||||
char* in_ptr = (char*)in;
|
||||
size_t out_used = 0;
|
||||
|
||||
while (in_left > 0) {
|
||||
char* out_ptr = out + out_used;
|
||||
size_t out_left = cap - out_used;
|
||||
if (iconv(conv->cd, &in_ptr, &in_left, &out_ptr, &out_left) == (size_t)-1) {
|
||||
if (errno != E2BIG) {
|
||||
if (err_out)
|
||||
*err_out = errno;
|
||||
charset_conversion_reset(conv);
|
||||
free(out);
|
||||
return NULL;
|
||||
}
|
||||
/* Output exhausted but input remains. E2BIG does not roll the output
|
||||
pointer back: the bytes iconv already emitted before the failure must
|
||||
be preserved, so advance out_used before growing. */
|
||||
out_used = (size_t)(out_ptr - out);
|
||||
if (!grow_charset_buffer(&out, &cap))
|
||||
return NULL;
|
||||
continue;
|
||||
}
|
||||
out_used = (size_t)(out_ptr - out);
|
||||
}
|
||||
|
||||
/* Flush any pending shift state (a no-op for the stateless single-byte and
|
||||
UTF charsets this feature targets, but keeps the descriptor clean). */
|
||||
for (;;) {
|
||||
char* out_ptr = out + out_used;
|
||||
size_t out_left = cap - out_used;
|
||||
if (iconv(conv->cd, NULL, NULL, &out_ptr, &out_left) == (size_t)-1) {
|
||||
if (errno != E2BIG) {
|
||||
if (err_out)
|
||||
*err_out = errno;
|
||||
charset_conversion_reset(conv);
|
||||
free(out);
|
||||
return NULL;
|
||||
}
|
||||
out_used = (size_t)(out_ptr - out);
|
||||
if (!grow_charset_buffer(&out, &cap))
|
||||
return NULL;
|
||||
continue;
|
||||
}
|
||||
out_used = (size_t)(out_ptr - out);
|
||||
break;
|
||||
}
|
||||
|
||||
/* A successful iconv call may legitimately consume the whole buffer (output
|
||||
exactly fills cap), leaving no room for the terminator: guarantee headroom
|
||||
before the final write. */
|
||||
if (out_used >= cap && !grow_charset_buffer(&out, &cap))
|
||||
return NULL;
|
||||
|
||||
/* Defense in depth: a target charset that emits embedded NUL bytes would
|
||||
truncate at the first NUL in the C-string wire helpers; fail cleanly
|
||||
(validation already rejects such charsets up front). */
|
||||
if (memchr(out, '\0', out_used) != NULL) {
|
||||
if (err_out)
|
||||
*err_out = EILSEQ;
|
||||
charset_conversion_reset(conv);
|
||||
free(out);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
out[out_used] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
bool charset_wire_init_sender(const char* spec) {
|
||||
charset_wire_free();
|
||||
if (!spec)
|
||||
return true;
|
||||
char* local;
|
||||
char* remote;
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
void* conv = charset_conversion_open(local, remote);
|
||||
free(local);
|
||||
free(remote);
|
||||
if (!conv)
|
||||
return false;
|
||||
g_wire_conv = (CharsetConversion*)conv;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool charset_wire_init_receiver(const char* spec, const char* server_spec) {
|
||||
charset_wire_free();
|
||||
if (!spec)
|
||||
return true;
|
||||
char* local;
|
||||
char* remote;
|
||||
if (charset_spec_parse(spec, &local, &remote) != 0)
|
||||
return false;
|
||||
/* The wire charset is the client spec's REMOTE half (rsync's LOCAL,REMOTE
|
||||
* spec stays the same push or pull, so on a push the destination end's
|
||||
* charset is REMOTE and the receiver writes the wire bytes verbatim). Only a
|
||||
* server started with its own --iconv declares a different local charset (the
|
||||
* server halves above never travel), and then it is that spec's LOCAL half. */
|
||||
const char* wire = remote;
|
||||
const char* target_local = remote;
|
||||
char* server_local = NULL;
|
||||
char* server_remote = NULL;
|
||||
if (server_spec) {
|
||||
if (charset_spec_parse(server_spec, &server_local, &server_remote) != 0) {
|
||||
free(local);
|
||||
free(remote);
|
||||
return false;
|
||||
}
|
||||
target_local = server_local;
|
||||
}
|
||||
void* conv = charset_conversion_open(wire, target_local);
|
||||
free(server_local);
|
||||
free(server_remote);
|
||||
free(local);
|
||||
free(remote);
|
||||
if (!conv)
|
||||
return false;
|
||||
g_wire_conv = (CharsetConversion*)conv;
|
||||
return true;
|
||||
}
|
||||
|
||||
void charset_wire_free(void) {
|
||||
if (g_wire_conv) {
|
||||
charset_conversion_close(g_wire_conv);
|
||||
g_wire_conv = NULL;
|
||||
}
|
||||
}
|
||||
|
||||
bool charset_wire_active(void) {
|
||||
return g_wire_conv != NULL;
|
||||
}
|
||||
|
||||
char* charset_wire_apply(const char* path) {
|
||||
if (!g_wire_conv)
|
||||
return str_dup(path);
|
||||
return charset_convert(g_wire_conv, path, NULL);
|
||||
}
|
||||
|
||||
static void charset_convert_failure_log(const char* path) {
|
||||
char* escaped = output_escape(path, false);
|
||||
log_message(LOG_LEVEL_ERROR, "--iconv: cannot convert file name '%s' to the target charset",
|
||||
escaped ? escaped : "<unprintable>");
|
||||
free(escaped);
|
||||
}
|
||||
|
||||
bool send_wire_str(int file_descriptor, const char* local_path) {
|
||||
if (!g_wire_conv)
|
||||
return send_str(file_descriptor, local_path);
|
||||
char* wire = charset_wire_apply(local_path);
|
||||
if (!wire) {
|
||||
charset_convert_failure_log(local_path);
|
||||
return false;
|
||||
}
|
||||
bool ok = send_str(file_descriptor, wire);
|
||||
free(wire);
|
||||
return ok;
|
||||
}
|
||||
|
||||
char* receive_wire_str(int file_descriptor) {
|
||||
char* raw = receive_str(file_descriptor);
|
||||
if (!raw)
|
||||
return NULL;
|
||||
if (!g_wire_conv)
|
||||
return raw;
|
||||
char* local = charset_convert(g_wire_conv, raw, NULL);
|
||||
if (!local) {
|
||||
charset_convert_failure_log(raw);
|
||||
free(raw);
|
||||
return NULL;
|
||||
}
|
||||
free(raw);
|
||||
return local;
|
||||
}
|
||||
@@ -1,86 +0,0 @@
|
||||
#ifndef CHARSET_H
|
||||
#define CHARSET_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
|
||||
/* --iconv=CONVERT_SPEC file-name charset conversion (rsync compatibility).
|
||||
*
|
||||
* CONVERT_SPEC is "LOCAL[,REMOTE]": LOCAL is the charset of our own file
|
||||
* names, REMOTE is the charset of the remote side's file names and defaults
|
||||
* to LOCAL when the comma half is omitted. The sender converts every local
|
||||
* path from LOCAL to REMOTE before it goes on the wire; the receiver converts
|
||||
* every received path back from REMOTE to LOCAL. A NULL/disabled spec means
|
||||
* identity with zero overhead (the common path never consults iconv).
|
||||
*
|
||||
* All helpers are friendly to the strict cold path: the wire conversion state
|
||||
* is process-global (one direction per process -- a client only sends, a
|
||||
* server only receives) and is initialized once, before any path is
|
||||
* serialized, so conversion compiles to a single non-NULL check when disabled.
|
||||
*/
|
||||
|
||||
/* Parse CONVERT_SPEC into malloc'd LOCAL and REMOTE charset names (caller
|
||||
* frees both). REMOTE is a separate copy of LOCAL when no comma is present.
|
||||
* Returns 0 on success, -1 on a malformed spec (empty halves / missing value /
|
||||
* allocation failure); nothing is allocated on the -1 path. Both output
|
||||
* pointers are REQUIRED (non-NULL). */
|
||||
int charset_spec_parse(const char* spec, char** local_out, char** remote_out);
|
||||
|
||||
/* True when a CONVERT_SPEC is well-formed AND its charsets are usable for this
|
||||
* feature: each pair opens in a probe iconv_open in BOTH directions (a sender
|
||||
* converts local->remote, the receiver converts remote->local) and converting
|
||||
* a representative ASCII name emits no embedded NUL byte (a UTF-16-style NUL
|
||||
* emitter would be silently truncated by the C-string wire helpers). A typo'd
|
||||
* charset name is therefore rejected at startup, not mid-run. NULL (iconv
|
||||
* disabled) is always valid. */
|
||||
bool charset_spec_valid(const char* spec);
|
||||
|
||||
/* Probe a concrete from->to conversion pair without keeping the descriptor:
|
||||
* both charsets open AND a representative ASCII name converts with no embedded
|
||||
* NUL. Used for direction-specific validation (e.g. the receiver's exact
|
||||
* wire->local direction including a server-side charset override). */
|
||||
bool charset_spec_valid_direction(const char* from_charset, const char* to_charset);
|
||||
bool charset_pair_valid(const char* local, const char* remote);
|
||||
|
||||
/* One-shot conversion of a NUL-terminated input to a malloc'd NUL-terminated
|
||||
* result, or NULL on failure. On failure *err_out (when non-NULL) receives
|
||||
* the iconv errno (EILSEQ/EINVAL = the input is not representable in the
|
||||
* target charset). The caller must free the result. */
|
||||
char* charset_convert(const void* conversion, const char* in, int* err_out);
|
||||
|
||||
/* Open a conversion descriptor for direction from_charset -> to_charset.
|
||||
* Returns NULL (errno = EINVAL) when a charset name is unsupported. Freed
|
||||
* with charset_conversion_close. */
|
||||
void* charset_conversion_open(const char* from_charset, const char* to_charset);
|
||||
void charset_conversion_close(void* conversion);
|
||||
|
||||
/* Process-wide wire conversion. charset_wire_init_sender (client side) opens
|
||||
* LOCAL->REMOTE; charset_wire_init_receiver (server side) opens
|
||||
* wire(REMOTE)->server-local. server_spec is the server's own --iconv, whose
|
||||
* LOCAL half overrides the destination charset; NULL means the destination
|
||||
* charset is the client spec's REMOTE half (rsync's push semantics: the wire
|
||||
* bytes are written verbatim). Both return false on an unsupported spec.
|
||||
* The state is freed with charset_wire_free. */
|
||||
bool charset_wire_init_sender(const char* spec);
|
||||
bool charset_wire_init_receiver(const char* spec, const char* server_spec);
|
||||
void charset_wire_free(void);
|
||||
bool charset_wire_active(void);
|
||||
|
||||
/* Pre-ack receiver-direction sanity (see charset_wire_init_receiver): true
|
||||
* when the exact wire->destination conversion the receiver will use (client
|
||||
* spec's REMOTE half into the server's own LOCAL half, or REMOTE->REMOTE when
|
||||
* the server has no --iconv) opens and produces NUL-free output. */
|
||||
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec);
|
||||
|
||||
/* Convert a path across the wire in the process direction. Returns a malloc'd
|
||||
* string, or NULL when the name cannot be represented in the target charset. */
|
||||
char* charset_wire_apply(const char* path);
|
||||
|
||||
/* Convenience wire string I/O: encode+send_str / receive_str+decode. Both
|
||||
* return false/NULL (logging a clear --iconv error) on conversion failure, so
|
||||
* an unconvertible path FAILS the transfer cleanly instead of silently sending
|
||||
* a mangled name. */
|
||||
bool send_wire_str(int file_descriptor, const char* local_path);
|
||||
char* receive_wire_str(int file_descriptor);
|
||||
|
||||
#endif
|
||||
@@ -1,427 +0,0 @@
|
||||
#include "checksum.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <openssl/evp.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* delta.c owns the single XXH_IMPLEMENTATION that provides the xxHash symbols
|
||||
* for the whole binary; this TU only needs the declarations. The streaming
|
||||
* state structs and XXH3_update are exposed only with XXH_STATIC_LINKING_ONLY. */
|
||||
#define XXH_STATIC_LINKING_ONLY
|
||||
#include <xxhash.h>
|
||||
|
||||
/* ---------------------------------------------------------------------------
|
||||
* Self-contained MD4 (RFC 1320). OpenSSL's MD4 lives in the legacy provider
|
||||
* and is not guaranteed present, so FastSync carries its own implementation to
|
||||
* keep --checksum-choice=md4 working on every build.
|
||||
* ------------------------------------------------------------------------- */
|
||||
|
||||
typedef struct {
|
||||
uint32_t state[4];
|
||||
uint64_t bit_count;
|
||||
uint8_t buffer[64];
|
||||
size_t buffer_len;
|
||||
} Md4Ctx;
|
||||
|
||||
static uint32_t md4_rotl(uint32_t x, int n) {
|
||||
return (x << n) | (x >> (32 - n));
|
||||
}
|
||||
|
||||
static void md4_transform(uint32_t state[4], const uint8_t block[64]) {
|
||||
uint32_t x[16];
|
||||
for (int i = 0; i < 16; i++)
|
||||
x[i] = (uint32_t)block[i * 4] | ((uint32_t)block[i * 4 + 1] << 8) |
|
||||
((uint32_t)block[i * 4 + 2] << 16) | ((uint32_t)block[i * 4 + 3] << 24);
|
||||
|
||||
uint32_t a = state[0], b = state[1], c = state[2], d = state[3];
|
||||
|
||||
#define F(x, y, z) (((x) & (y)) | (~(x) & (z)))
|
||||
#define G(x, y, z) (((x) & (y)) | ((x) & (z)) | ((y) & (z)))
|
||||
#define H(x, y, z) ((x) ^ (y) ^ (z))
|
||||
#define ROUND1(a, b, c, d, k, s) a = md4_rotl(a + F(b, c, d) + x[k], s)
|
||||
#define ROUND2(a, b, c, d, k, s) a = md4_rotl(a + G(b, c, d) + x[k] + 0x5a827999u, s)
|
||||
#define ROUND3(a, b, c, d, k, s) a = md4_rotl(a + H(b, c, d) + x[k] + 0x6ed9eba1u, s)
|
||||
|
||||
ROUND1(a, b, c, d, 0, 3);
|
||||
ROUND1(d, a, b, c, 1, 7);
|
||||
ROUND1(c, d, a, b, 2, 11);
|
||||
ROUND1(b, c, d, a, 3, 19);
|
||||
ROUND1(a, b, c, d, 4, 3);
|
||||
ROUND1(d, a, b, c, 5, 7);
|
||||
ROUND1(c, d, a, b, 6, 11);
|
||||
ROUND1(b, c, d, a, 7, 19);
|
||||
ROUND1(a, b, c, d, 8, 3);
|
||||
ROUND1(d, a, b, c, 9, 7);
|
||||
ROUND1(c, d, a, b, 10, 11);
|
||||
ROUND1(b, c, d, a, 11, 19);
|
||||
ROUND1(a, b, c, d, 12, 3);
|
||||
ROUND1(d, a, b, c, 13, 7);
|
||||
ROUND1(c, d, a, b, 14, 11);
|
||||
ROUND1(b, c, d, a, 15, 19);
|
||||
|
||||
ROUND2(a, b, c, d, 0, 3);
|
||||
ROUND2(d, a, b, c, 4, 5);
|
||||
ROUND2(c, d, a, b, 8, 9);
|
||||
ROUND2(b, c, d, a, 12, 13);
|
||||
ROUND2(a, b, c, d, 1, 3);
|
||||
ROUND2(d, a, b, c, 5, 5);
|
||||
ROUND2(c, d, a, b, 9, 9);
|
||||
ROUND2(b, c, d, a, 13, 13);
|
||||
ROUND2(a, b, c, d, 2, 3);
|
||||
ROUND2(d, a, b, c, 6, 5);
|
||||
ROUND2(c, d, a, b, 10, 9);
|
||||
ROUND2(b, c, d, a, 14, 13);
|
||||
ROUND2(a, b, c, d, 3, 3);
|
||||
ROUND2(d, a, b, c, 7, 5);
|
||||
ROUND2(c, d, a, b, 11, 9);
|
||||
ROUND2(b, c, d, a, 15, 13);
|
||||
|
||||
ROUND3(a, b, c, d, 0, 3);
|
||||
ROUND3(d, a, b, c, 8, 9);
|
||||
ROUND3(c, d, a, b, 4, 11);
|
||||
ROUND3(b, c, d, a, 12, 15);
|
||||
ROUND3(a, b, c, d, 2, 3);
|
||||
ROUND3(d, a, b, c, 10, 9);
|
||||
ROUND3(c, d, a, b, 6, 11);
|
||||
ROUND3(b, c, d, a, 14, 15);
|
||||
ROUND3(a, b, c, d, 1, 3);
|
||||
ROUND3(d, a, b, c, 9, 9);
|
||||
ROUND3(c, d, a, b, 5, 11);
|
||||
ROUND3(b, c, d, a, 13, 15);
|
||||
ROUND3(a, b, c, d, 3, 3);
|
||||
ROUND3(d, a, b, c, 11, 9);
|
||||
ROUND3(c, d, a, b, 7, 11);
|
||||
ROUND3(b, c, d, a, 15, 15);
|
||||
|
||||
#undef F
|
||||
#undef G
|
||||
#undef H
|
||||
#undef ROUND1
|
||||
#undef ROUND2
|
||||
#undef ROUND3
|
||||
|
||||
state[0] += a;
|
||||
state[1] += b;
|
||||
state[2] += c;
|
||||
state[3] += d;
|
||||
}
|
||||
|
||||
static void md4_init(Md4Ctx* ctx) {
|
||||
ctx->state[0] = 0x67452301u;
|
||||
ctx->state[1] = 0xefcdab89u;
|
||||
ctx->state[2] = 0x98badcfeu;
|
||||
ctx->state[3] = 0x10325476u;
|
||||
ctx->bit_count = 0;
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
|
||||
static void md4_update(Md4Ctx* ctx, const uint8_t* data, size_t len) {
|
||||
ctx->bit_count += (uint64_t)len * 8;
|
||||
while (len > 0) {
|
||||
size_t space = sizeof(ctx->buffer) - ctx->buffer_len;
|
||||
size_t take = len < space ? len : space;
|
||||
memcpy(ctx->buffer + ctx->buffer_len, data, take);
|
||||
ctx->buffer_len += take;
|
||||
data += take;
|
||||
len -= take;
|
||||
if (ctx->buffer_len == sizeof(ctx->buffer)) {
|
||||
md4_transform(ctx->state, ctx->buffer);
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void md4_final(Md4Ctx* ctx, uint8_t out[16]) {
|
||||
uint64_t bit_count = ctx->bit_count;
|
||||
uint8_t pad = 0x80;
|
||||
md4_update(ctx, &pad, 1);
|
||||
uint8_t zero = 0;
|
||||
while (ctx->buffer_len != 56)
|
||||
md4_update(ctx, &zero, 1);
|
||||
uint8_t length_le[8];
|
||||
for (int i = 0; i < 8; i++)
|
||||
length_le[i] = (uint8_t)((bit_count >> (8 * i)) & 0xff);
|
||||
md4_update(ctx, length_le, sizeof(length_le));
|
||||
for (int i = 0; i < 4; i++) {
|
||||
out[i * 4] = (uint8_t)(ctx->state[i] & 0xff);
|
||||
out[i * 4 + 1] = (uint8_t)((ctx->state[i] >> 8) & 0xff);
|
||||
out[i * 4 + 2] = (uint8_t)((ctx->state[i] >> 16) & 0xff);
|
||||
out[i * 4 + 3] = (uint8_t)((ctx->state[i] >> 24) & 0xff);
|
||||
}
|
||||
}
|
||||
|
||||
/* One-shot EVP digest (md5/sha1). Returns false when OpenSSL refuses. */
|
||||
static bool evp_digest(const EVP_MD* md, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
static const uint8_t empty = 0;
|
||||
const void* input = data ? data : ∅
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_Digest(input, size, out, &digest_len, md, NULL) != 1)
|
||||
return false;
|
||||
if (digest_len > out_capacity)
|
||||
return false;
|
||||
*out_len = digest_len;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
if (data == NULL && size != 0)
|
||||
return false;
|
||||
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64: {
|
||||
uint64_t digest = XXH64(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_XXH3: {
|
||||
uint64_t digest = XXH3_64bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_XXH128: {
|
||||
XXH128_hash_t digest = XXH3_128bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
/* md5 takes no seed; the caller's seed is deliberately ignored (documented
|
||||
* in RSYNC_COMPAT.md). */
|
||||
return evp_digest(EVP_md5(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_MD4: {
|
||||
Md4Ctx ctx;
|
||||
md4_init(&ctx);
|
||||
md4_update(&ctx, (const uint8_t*)data, size);
|
||||
md4_final(&ctx, out);
|
||||
*out_len = 16;
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
/* sha1 takes no seed; the caller's seed is deliberately ignored. */
|
||||
return evp_digest(EVP_sha1(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
/* No checksum requested: an empty digest is the successful result. */
|
||||
*out_len = 0;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!path || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
|
||||
int fd = open(path, O_RDONLY | O_CLOEXEC);
|
||||
if (fd < 0)
|
||||
return false;
|
||||
|
||||
bool ok = checksum_digest_fd(algo, seed, fd, out, out_capacity, out_len);
|
||||
close(fd);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len) {
|
||||
if (fd < 0 || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
|
||||
if (algo == CHECKSUM_ALGO_NONE) {
|
||||
/* No checksum requested: nothing to read; an empty digest succeeds. */
|
||||
*out_len = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
uint8_t buffer[64 * 1024];
|
||||
bool ok = false;
|
||||
lseek(fd, 0, SEEK_SET);
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5 || algo == CHECKSUM_ALGO_SHA1) {
|
||||
const EVP_MD* md = algo == CHECKSUM_ALGO_MD5 ? EVP_md5() : EVP_sha1();
|
||||
EVP_MD_CTX* ctx = EVP_MD_CTX_new();
|
||||
if (!ctx)
|
||||
return false;
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_DigestInit_ex(ctx, md, NULL) == 1) {
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (EVP_DigestUpdate(ctx, buffer, (size_t)got) != 1) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
if (ok && EVP_DigestFinal_ex(ctx, out, &digest_len) == 1 && digest_len <= out_capacity)
|
||||
*out_len = digest_len;
|
||||
else
|
||||
ok = false;
|
||||
}
|
||||
EVP_MD_CTX_free(ctx);
|
||||
return ok;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD4) {
|
||||
Md4Ctx ctx;
|
||||
md4_init(&ctx);
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0)
|
||||
md4_update(&ctx, buffer, (size_t)got);
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
if (ok) {
|
||||
md4_final(&ctx, out);
|
||||
*out_len = 16;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
XXH64_state_t xxh64;
|
||||
XXH3_state_t* xxh3 = NULL;
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
XXH64_reset(&xxh64, seed);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3 || algo == CHECKSUM_ALGO_XXH128) {
|
||||
xxh3 = XXH3_createState();
|
||||
if (!xxh3)
|
||||
return false;
|
||||
if (algo == CHECKSUM_ALGO_XXH3)
|
||||
XXH3_64bits_reset_withSeed(xxh3, seed);
|
||||
else
|
||||
XXH3_128bits_reset_withSeed(xxh3, seed);
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64)
|
||||
XXH64_update(&xxh64, buffer, (size_t)got);
|
||||
else if (XXH3_64bits_update(xxh3, buffer, (size_t)got) == XXH_ERROR) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
|
||||
if (ok) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
uint64_t digest = XXH64_digest(&xxh64);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3) {
|
||||
uint64_t digest = XXH3_64bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else {
|
||||
XXH128_hash_t digest = XXH3_128bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
}
|
||||
}
|
||||
if (xxh3)
|
||||
XXH3_freeState(xxh3);
|
||||
return ok;
|
||||
}
|
||||
|
||||
int checksum_algo_from_name(const char* name) {
|
||||
if (!name)
|
||||
return -1;
|
||||
if (strcasecmp(name, "xxh64") == 0 || strcasecmp(name, "xxhash") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH64;
|
||||
if (strcasecmp(name, "xxh3") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH3;
|
||||
if (strcasecmp(name, "xxh128") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH128;
|
||||
if (strcasecmp(name, "md5") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD5;
|
||||
if (strcasecmp(name, "md4") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD4;
|
||||
if (strcasecmp(name, "sha1") == 0)
|
||||
return (int)CHECKSUM_ALGO_SHA1;
|
||||
if (strcasecmp(name, "none") == 0)
|
||||
return (int)CHECKSUM_ALGO_NONE;
|
||||
return -1;
|
||||
}
|
||||
|
||||
const char* checksum_algo_name(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
return "xxh64";
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return "xxh3";
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
return "xxh128";
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
return "md5";
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return "md4";
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return "sha1";
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return "none";
|
||||
}
|
||||
return "<unknown>";
|
||||
}
|
||||
|
||||
bool checksum_algo_valid(int algo) {
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5 ||
|
||||
algo == (int)CHECKSUM_ALGO_XXH3 || algo == (int)CHECKSUM_ALGO_XXH128 ||
|
||||
algo == (int)CHECKSUM_ALGO_MD4 || algo == (int)CHECKSUM_ALGO_SHA1 ||
|
||||
algo == (int)CHECKSUM_ALGO_NONE;
|
||||
}
|
||||
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return 8;
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return 16;
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return 20;
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static ChecksumAlgo compiled_checksum_preference_first(void) {
|
||||
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
|
||||
* resolves to xxh128. */
|
||||
static const ChecksumAlgo preference[] = {
|
||||
CHECKSUM_ALGO_XXH128, CHECKSUM_ALGO_XXH3, CHECKSUM_ALGO_XXH64, CHECKSUM_ALGO_MD5,
|
||||
CHECKSUM_ALGO_MD4, CHECKSUM_ALGO_SHA1, CHECKSUM_ALGO_NONE,
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(preference) / sizeof(preference[0]); i++) {
|
||||
if (checksum_algo_valid((int)preference[i]))
|
||||
return preference[i];
|
||||
}
|
||||
return CHECKSUM_ALGO_XXH64;
|
||||
}
|
||||
|
||||
int checksum_choice_resolve(void) {
|
||||
bool specified = false;
|
||||
int env = env_choice_first("RSYNC_CHECKSUM_LIST", checksum_algo_from_name, &specified);
|
||||
if (specified)
|
||||
return env; /* -1 = the list named no supported checksum */
|
||||
return (int)compiled_checksum_preference_first();
|
||||
}
|
||||
|
||||
ChecksumAlgo checksum_negotiate_default(void) {
|
||||
int resolved = checksum_choice_resolve();
|
||||
return resolved >= 0 ? (ChecksumAlgo)resolved : compiled_checksum_preference_first();
|
||||
}
|
||||
@@ -1,89 +0,0 @@
|
||||
#ifndef CHECKSUM_H
|
||||
#define CHECKSUM_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* Whole-file content-digest algorithms selectable with --checksum-choice and
|
||||
* seeded with --checksum-seed. The ids are the values actually placed on the
|
||||
* wire (config frame), so they must be kept stable and validated on receive.
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the historical FastSync default and its numeric
|
||||
* value is preserved. The full set mirrors the algorithms rsync 3.4.1 can be
|
||||
* built with; every one of them is implemented here. */
|
||||
typedef enum {
|
||||
CHECKSUM_ALGO_XXH64 = 0,
|
||||
CHECKSUM_ALGO_MD5 = 1,
|
||||
CHECKSUM_ALGO_XXH3 = 2,
|
||||
CHECKSUM_ALGO_XXH128 = 3,
|
||||
CHECKSUM_ALGO_MD4 = 4,
|
||||
CHECKSUM_ALGO_SHA1 = 5,
|
||||
CHECKSUM_ALGO_NONE = 6
|
||||
} ChecksumAlgo;
|
||||
|
||||
/* FastSync's negotiated default (rsync 3.4.1 auto-negotiates xxh128 first).
|
||||
* The wire default for Config->checksum_algo is this value. */
|
||||
#define CHECKSUM_ALGO_DEFAULT CHECKSUM_ALGO_XXH128
|
||||
|
||||
/* sha1 digest is 20 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 20
|
||||
|
||||
/* Compute the whole-file digest of the first `size` bytes of `data`.
|
||||
*
|
||||
* - CHECKSUM_ALGO_XXH64: xxHash64(data, size, seed) (full 64-bit seed).
|
||||
* - CHECKSUM_ALGO_XXH3: XXH3_64bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_XXH128: XXH3_128bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_MD4: md4(data, size), self-contained RFC 1320 (seed ignored).
|
||||
* - CHECKSUM_ALGO_SHA1: sha1(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_NONE: no digest; *out_len is 0 and nothing is written.
|
||||
* - `size == 0` hashes the empty input (plus its seed), not a NULL input.
|
||||
*
|
||||
* Writes up to `out_capacity` bytes into `out`, storing the digest length in
|
||||
* *out_len. Returns false on NULL out* or when the digest would not fit.
|
||||
* Never writes more than CHECKSUM_MAX_DIGEST_LEN bytes. */
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Streaming whole-file digest: hash the contents of `path` without holding the
|
||||
* whole file in memory. Same digest/capacity contract as checksum_digest.
|
||||
* Returns false on open/read failure or an undersized buffer. */
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Descriptor form of the streaming digest: rewinds `fd` to the start and hashes
|
||||
* to EOF without closing it. Used by the --verify-basis path to hash an
|
||||
* already-open, root-confined basis descriptor. Same contract as
|
||||
* checksum_digest_file. */
|
||||
bool checksum_digest_fd(ChecksumAlgo algo, uint64_t seed, int fd, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len);
|
||||
|
||||
/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "xxh64"/"xxhash", "xxh3", "xxh128", "md5", "md4", "sha1", "none".
|
||||
* "auto" is not an algorithm here; the caller resolves it to the negotiated
|
||||
* default. Returns -1 for any unrecognized name. */
|
||||
int checksum_algo_from_name(const char* name);
|
||||
|
||||
/* Canonical name of an algorithm (used in CLI error messages). */
|
||||
const char* checksum_algo_name(ChecksumAlgo algo);
|
||||
|
||||
/* True when `algo` is a supported id (used by config receive validation). */
|
||||
bool checksum_algo_valid(int algo);
|
||||
|
||||
/* Digest length in bytes for an algorithm (xxh64/xxh3 = 8,
|
||||
* md5/md4/xxh128 = 16, sha1 = 20, none = 0). */
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list that is
|
||||
* supported on this build (rsync 3.4.1's `--version` order:
|
||||
* xxh128 xxh3 xxh64 md5 md4 sha1 none). Used to resolve "auto". */
|
||||
ChecksumAlgo checksum_negotiate_default(void);
|
||||
|
||||
/* Resolve "auto" the way rsync does: the first supported name in
|
||||
* RSYNC_CHECKSUM_LIST (whitespace-separated, client half ends at '&'), then the
|
||||
* compiled-in preference order when the variable is unset/blank. Returns -1
|
||||
* when the variable is set but names no supported checksum (rsync's failed
|
||||
* negotiation), otherwise a valid ChecksumAlgo id. */
|
||||
int checksum_choice_resolve(void);
|
||||
|
||||
#endif /* CHECKSUM_H */
|
||||
@@ -1,183 +0,0 @@
|
||||
#include "chmod.h"
|
||||
#include "file.h"
|
||||
#include <stddef.h>
|
||||
#include <string.h>
|
||||
|
||||
/* rsync's --chmod parser (parse_chmod + tweak_mode). A single clause is
|
||||
* applied as it is completed, so repeated clauses and repeated --chmod options
|
||||
* (joined with commas by the CLI) accumulate exactly like rsync. The D/F
|
||||
* selectors restrict a clause to directories/files; X adds execute only to
|
||||
* directories or files that were already executable. */
|
||||
|
||||
#define CHMOD_BITS 07777
|
||||
#define CHMOD_FLAG_X_KEEP (1U << 0)
|
||||
#define CHMOD_FLAG_DIRS_ONLY (1U << 1)
|
||||
#define CHMOD_FLAG_FILES_ONLY (1U << 2)
|
||||
|
||||
enum chmod_op { CHMOD_OP_ADD = 1, CHMOD_OP_SUB, CHMOD_OP_EQ, CHMOD_OP_SET };
|
||||
enum chmod_state {
|
||||
CHMOD_STATE_ERROR,
|
||||
CHMOD_STATE_1ST_HALF,
|
||||
CHMOD_STATE_2ND_HALF,
|
||||
CHMOD_STATE_OCTAL
|
||||
};
|
||||
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result) {
|
||||
if (!spec || !*spec || !result)
|
||||
return false;
|
||||
const mode_t nonperm = mode & ~(mode_t)CHMOD_BITS;
|
||||
const bool initially_executable = (mode & 0111) != 0;
|
||||
mode_t changed = mode;
|
||||
int state = CHMOD_STATE_1ST_HALF;
|
||||
unsigned where = 0;
|
||||
int what = 0, op = 0, topbits = 0, topoct = 0, flags = 0;
|
||||
const char* p = spec;
|
||||
while (state != CHMOD_STATE_ERROR) {
|
||||
if (*p == '\0' || *p == ',') {
|
||||
int bits;
|
||||
if (!op) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
if (where)
|
||||
bits = (int)(where * (unsigned)what);
|
||||
else {
|
||||
where = 0111;
|
||||
bits = (int)((where * (unsigned)what) & ~(unsigned)file_process_umask());
|
||||
}
|
||||
int mode_and, mode_or;
|
||||
switch (op) {
|
||||
case CHMOD_OP_ADD:
|
||||
mode_and = CHMOD_BITS;
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
case CHMOD_OP_SUB:
|
||||
mode_and = CHMOD_BITS - bits - topoct;
|
||||
mode_or = 0;
|
||||
break;
|
||||
case CHMOD_OP_EQ:
|
||||
mode_and = CHMOD_BITS - (int)(where * 7U) - (topoct ? topbits : 0);
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
default:
|
||||
mode_and = 0;
|
||||
mode_or = bits;
|
||||
break;
|
||||
}
|
||||
bool is_dir = S_ISDIR(nonperm);
|
||||
if (!((flags & CHMOD_FLAG_DIRS_ONLY) && !is_dir) &&
|
||||
!((flags & CHMOD_FLAG_FILES_ONLY) && is_dir)) {
|
||||
changed &= (mode_t)mode_and;
|
||||
if ((flags & CHMOD_FLAG_X_KEEP) && !initially_executable && !is_dir)
|
||||
changed |= (mode_t)(mode_or & ~0111);
|
||||
else
|
||||
changed |= (mode_t)mode_or;
|
||||
}
|
||||
if (*p == '\0')
|
||||
break;
|
||||
p++;
|
||||
state = CHMOD_STATE_1ST_HALF;
|
||||
where = 0;
|
||||
what = op = topoct = topbits = flags = 0;
|
||||
continue;
|
||||
}
|
||||
switch (state) {
|
||||
case CHMOD_STATE_1ST_HALF:
|
||||
switch (*p) {
|
||||
case 'D':
|
||||
if (flags & CHMOD_FLAG_FILES_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_DIRS_ONLY;
|
||||
break;
|
||||
case 'F':
|
||||
if (flags & CHMOD_FLAG_DIRS_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_FILES_ONLY;
|
||||
break;
|
||||
case 'u':
|
||||
where |= 0100;
|
||||
topbits |= 04000;
|
||||
break;
|
||||
case 'g':
|
||||
where |= 0010;
|
||||
topbits |= 02000;
|
||||
break;
|
||||
case 'o':
|
||||
where |= 0001;
|
||||
break;
|
||||
case 'a':
|
||||
where |= 0111;
|
||||
break;
|
||||
case '+':
|
||||
op = CHMOD_OP_ADD;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '-':
|
||||
op = CHMOD_OP_SUB;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '=':
|
||||
op = CHMOD_OP_EQ;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7' && !where) {
|
||||
op = CHMOD_OP_SET;
|
||||
state = CHMOD_STATE_OCTAL;
|
||||
where = 1;
|
||||
what = *p - '0';
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case CHMOD_STATE_2ND_HALF:
|
||||
switch (*p) {
|
||||
case 'r':
|
||||
what |= 4;
|
||||
break;
|
||||
case 'w':
|
||||
what |= 2;
|
||||
break;
|
||||
case 'X':
|
||||
flags |= CHMOD_FLAG_X_KEEP;
|
||||
/* fall through */
|
||||
case 'x':
|
||||
what |= 1;
|
||||
break;
|
||||
case 's':
|
||||
if (topbits)
|
||||
topoct |= topbits;
|
||||
else
|
||||
topoct = 04000;
|
||||
break;
|
||||
case 't':
|
||||
topoct |= 01000;
|
||||
break;
|
||||
default:
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7') {
|
||||
what = what * 8 + (*p - '0');
|
||||
if (what > CHMOD_BITS)
|
||||
state = CHMOD_STATE_ERROR;
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
p++;
|
||||
}
|
||||
if (state == CHMOD_STATE_ERROR)
|
||||
return false;
|
||||
*result = (changed & (mode_t)CHMOD_BITS) | nonperm;
|
||||
return true;
|
||||
}
|
||||
@@ -1,13 +0,0 @@
|
||||
#ifndef CHMOD_H
|
||||
#define CHMOD_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Apply rsync's --chmod syntax to a permission mode, including the D/F/X
|
||||
* selectors and the s/t special bits. `mode` should carry the file type bits
|
||||
* (S_IFDIR/S_IFREG) so D/F/X can be evaluated; the type bits are preserved in
|
||||
* `result`. A spec may contain comma-separated clauses, which accumulate. */
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result);
|
||||
|
||||
#endif
|
||||
+86
-289
@@ -1,13 +1,11 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include "array_list.h"
|
||||
#include "charset.h"
|
||||
#include "chunk.h"
|
||||
#include "compression.h"
|
||||
#include "data.h"
|
||||
@@ -17,42 +15,14 @@
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
|
||||
/* Maximum individual file data size within a chunk (64 MB). Distinct from the
|
||||
* receiver's whole-file MAX_FILE_DATA_SIZE (256 MB) in file_receive.c. */
|
||||
#define MAX_CHUNK_FILE_DATA_SIZE (64ULL * 1024 * 1024)
|
||||
/* Maximum individual file data size within a chunk (64 MB) */
|
||||
#define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024)
|
||||
#define MAX_FILES_PER_CHUNK 65536U
|
||||
|
||||
/* Reserve `charge` against `session`'s connection budget. This mirrors the
|
||||
static protocol_reserve_memory() in protocol.c: the receive-side call sites
|
||||
only have the Data.owner pointer (a ProtocolSession*), and protocol.c is out
|
||||
of scope for this fix, so the same atomic CAS accounting is reproduced here.
|
||||
The matching release always goes through data_destroy()'s Data.owner path. */
|
||||
static bool chunk_session_reserve(ProtocolSession* session, size_t charge) {
|
||||
unsigned long long allocated = atomic_load(&session->total_allocated_bytes);
|
||||
while (true) {
|
||||
if (allocated > MAX_CONNECTION_MEMORY ||
|
||||
(unsigned long long)charge > MAX_CONNECTION_MEMORY - allocated)
|
||||
return false;
|
||||
if (atomic_compare_exchange_weak(&session->total_allocated_bytes, &allocated,
|
||||
allocated + (unsigned long long)charge))
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge) {
|
||||
if (!data || charge == 0 || session == NULL)
|
||||
return true;
|
||||
if (!chunk_session_reserve(session, charge))
|
||||
return false;
|
||||
data->owner = session;
|
||||
data->protocol_charge = charge;
|
||||
return true;
|
||||
}
|
||||
|
||||
Chunk* chunk_create(File** items, int element_count) {
|
||||
if (element_count < 0 || (element_count > 0 && items == NULL))
|
||||
return NULL;
|
||||
Chunk* chunk = (Chunk*)protocol_alloc(sizeof(Chunk));
|
||||
Chunk* chunk = (Chunk*)malloc(sizeof(Chunk));
|
||||
if (chunk == NULL) {
|
||||
log_perror("ERROR: Could not allocate memory for chunk structure");
|
||||
return NULL;
|
||||
@@ -65,7 +35,7 @@ Chunk* chunk_create(File** items, int element_count) {
|
||||
free(chunk);
|
||||
return NULL;
|
||||
}
|
||||
chunk->items = (File**)protocol_alloc((size_t)element_count * sizeof(File*));
|
||||
chunk->items = (File**)malloc((size_t)element_count * sizeof(File*));
|
||||
if (chunk->items == NULL) {
|
||||
free(chunk);
|
||||
return NULL;
|
||||
@@ -93,22 +63,9 @@ void chunk_destroy(void* item) {
|
||||
free(chunk);
|
||||
}
|
||||
|
||||
/* --iconv: a chunk blob carries wire-charset path/target bytes. Encode the
|
||||
* sender-side path (a no-op copy when iconv is disabled) so the blob is in the
|
||||
* same charset as every other wire string. */
|
||||
static char* chunk_encode_wire(const char* path) {
|
||||
if (!charset_wire_active())
|
||||
return str_dup(path);
|
||||
return charset_wire_apply(path);
|
||||
}
|
||||
|
||||
static unsigned long long per_file_serialize_size(File* file, bool use_metadata) {
|
||||
unsigned long long size = sizeof(size_t);
|
||||
char* wire_path = chunk_encode_wire(file_wire_path(file));
|
||||
if (!wire_path)
|
||||
return 0;
|
||||
size_t path_len = strlen(wire_path);
|
||||
free(wire_path);
|
||||
size_t path_len = strlen(file->path);
|
||||
unsigned long long metadata_size =
|
||||
use_metadata ? sizeof(int) + (file->metadata ? FILE_METADATA_WIRE_SIZE : 0) : 0;
|
||||
if ((unsigned long long)path_len > ULLONG_MAX - size)
|
||||
@@ -117,39 +74,12 @@ static unsigned long long per_file_serialize_size(File* file, bool use_metadata)
|
||||
if (metadata_size > ULLONG_MAX - size)
|
||||
return 0;
|
||||
size += metadata_size;
|
||||
/* Entry type marker: 0 = regular file, 1 = explicit directory entry,
|
||||
2 = symlink entry (carries its target string), 3 = special/device node
|
||||
(recreated by the receiver). */
|
||||
if (sizeof(int) > ULLONG_MAX - size)
|
||||
return 0;
|
||||
size += sizeof(int);
|
||||
/* A special node also carries its rdev major/minor. */
|
||||
if (file->is_special) {
|
||||
if (2 * sizeof(int32_t) > ULLONG_MAX - size)
|
||||
return 0;
|
||||
size += 2 * sizeof(int32_t);
|
||||
}
|
||||
if (sizeof(size_t) > ULLONG_MAX - size)
|
||||
return 0;
|
||||
size += sizeof(size_t);
|
||||
if ((unsigned long long)file->data->size > ULLONG_MAX - size)
|
||||
return 0;
|
||||
size += file->data->size;
|
||||
/* Symlink entries append the target string (length-prefixed). */
|
||||
if (file->is_symlink) {
|
||||
char* wire_target = chunk_encode_wire(file->symlink_target ? file->symlink_target : "");
|
||||
if (!wire_target)
|
||||
return 0;
|
||||
size_t target_len = strlen(wire_target);
|
||||
free(wire_target);
|
||||
if (sizeof(size_t) > ULLONG_MAX - size)
|
||||
return 0;
|
||||
size += sizeof(size_t);
|
||||
if ((unsigned long long)target_len > ULLONG_MAX - size)
|
||||
return 0;
|
||||
size += target_len;
|
||||
}
|
||||
return size;
|
||||
return size + file->data->size;
|
||||
}
|
||||
|
||||
Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
|
||||
@@ -159,8 +89,7 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
if (!chunk->items[i] || !chunk->items[i]->path || !chunk->items[i]->data ||
|
||||
(chunk->items[i]->data->size > 0 && !chunk->items[i]->data->data) ||
|
||||
chunk->items[i]->path[0] == '\0' || has_path_traversal(chunk->items[i]->path) ||
|
||||
(file_wire_path(chunk->items[i]))[0] == '\0')
|
||||
chunk->items[i]->path[0] == '\0' || has_path_traversal(chunk->items[i]->path))
|
||||
return NULL;
|
||||
unsigned long long file_size = per_file_serialize_size(chunk->items[i], use_metadata);
|
||||
if (file_size == 0 || file_size > ULLONG_MAX - data_size || data_size + file_size > SIZE_MAX)
|
||||
@@ -175,30 +104,11 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
|
||||
char* data_pointer = data->data;
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* file = chunk->items[i];
|
||||
char* wire_path = chunk_encode_wire(file_wire_path(file));
|
||||
if (wire_path == NULL) {
|
||||
data_destroy(data);
|
||||
return NULL;
|
||||
}
|
||||
size_t path_len = strlen(wire_path);
|
||||
size_t path_len = strlen(file->path);
|
||||
memcpy(data_pointer, &path_len, sizeof(size_t));
|
||||
data_pointer += sizeof(size_t);
|
||||
memcpy(data_pointer, wire_path, path_len);
|
||||
memcpy(data_pointer, file->path, path_len);
|
||||
data_pointer += path_len;
|
||||
free(wire_path);
|
||||
|
||||
int entry_type = file->is_dir ? 1 : (file->is_symlink ? 2 : (file->is_special ? 3 : 0));
|
||||
memcpy(data_pointer, &entry_type, sizeof(int));
|
||||
data_pointer += sizeof(int);
|
||||
|
||||
if (file->is_special) {
|
||||
int32_t special_major = file->rdev_major;
|
||||
int32_t special_minor = file->rdev_minor;
|
||||
memcpy(data_pointer, &special_major, sizeof(special_major));
|
||||
data_pointer += sizeof(special_major);
|
||||
memcpy(data_pointer, &special_minor, sizeof(special_minor));
|
||||
data_pointer += sizeof(special_minor);
|
||||
}
|
||||
|
||||
if (use_metadata)
|
||||
metadata_to_buf(&data_pointer, file->metadata);
|
||||
@@ -206,24 +116,8 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
|
||||
size_t file_data_size = file->data->size;
|
||||
memcpy(data_pointer, &file_data_size, sizeof(size_t));
|
||||
data_pointer += sizeof(size_t);
|
||||
if (file_data_size > 0)
|
||||
memcpy(data_pointer, file->data->data, file_data_size);
|
||||
data_pointer += file_data_size;
|
||||
|
||||
if (file->is_symlink) {
|
||||
char* wire_target = chunk_encode_wire(file->symlink_target ? file->symlink_target : "");
|
||||
if (wire_target == NULL) {
|
||||
data_destroy(data);
|
||||
return NULL;
|
||||
}
|
||||
size_t target_len = strlen(wire_target);
|
||||
memcpy(data_pointer, &target_len, sizeof(size_t));
|
||||
data_pointer += sizeof(size_t);
|
||||
if (target_len > 0)
|
||||
memcpy(data_pointer, wire_target, target_len);
|
||||
data_pointer += target_len;
|
||||
free(wire_target);
|
||||
}
|
||||
}
|
||||
return data;
|
||||
}
|
||||
@@ -236,20 +130,17 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
return NULL;
|
||||
char* data_pointer = data->data;
|
||||
size_t remaining_size = data->size;
|
||||
/* The element currently being parsed is owned by `files` only after the
|
||||
* array_list_add() at the end of the iteration; until then the error
|
||||
* epilogue destroys it directly. Keeping this one pointer nulled after the
|
||||
* hand-off makes the single cleanup path correct for every failure. */
|
||||
File* file = NULL;
|
||||
|
||||
while (remaining_size > 0) {
|
||||
if ((unsigned int)files->size >= MAX_FILES_PER_CHUNK) {
|
||||
log_message(LOG_LEVEL_ERROR, "Chunk contains too many files");
|
||||
goto error;
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path length");
|
||||
goto error;
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
size_t path_len;
|
||||
@@ -259,117 +150,77 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (path_len > SIZE_MAX - 1 || remaining_size < path_len) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path");
|
||||
goto error;
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
char* path = protocol_alloc(path_len + 1);
|
||||
if (path_len == SIZE_MAX) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
char* path = malloc(path_len + 1);
|
||||
if (path == NULL) {
|
||||
log_perror("Could not allocate memory for file path");
|
||||
goto error;
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(path, data_pointer, path_len);
|
||||
path[path_len] = '\0';
|
||||
if (memchr(path, '\0', path_len) != NULL) {
|
||||
free(path);
|
||||
goto error;
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
data_pointer += path_len;
|
||||
remaining_size -= path_len;
|
||||
|
||||
/* --iconv: the blob holds the wire charset; translate it to the receiver's
|
||||
local charset before validation and creation so the destination gets the
|
||||
local name. A name that cannot be decoded fails the file cleanly. */
|
||||
if (charset_wire_active()) {
|
||||
char* local_path = charset_wire_apply(path);
|
||||
free(path);
|
||||
if (local_path == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv: received chunk file name cannot be converted to the local charset");
|
||||
goto error;
|
||||
}
|
||||
path = local_path;
|
||||
path_len = strlen(path);
|
||||
}
|
||||
|
||||
if (path_len == 0 || has_path_traversal(path)) {
|
||||
free(path);
|
||||
goto error;
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
file = file_create(path);
|
||||
File* file = file_create(path);
|
||||
free(path);
|
||||
if (file == NULL)
|
||||
goto error;
|
||||
|
||||
if (remaining_size < sizeof(int)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for entry type");
|
||||
goto error;
|
||||
}
|
||||
int entry_type;
|
||||
memcpy(&entry_type, data_pointer, sizeof(int));
|
||||
if (entry_type != 0 && entry_type != 1 && entry_type != 2 && entry_type != 3) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad entry type");
|
||||
goto error;
|
||||
}
|
||||
file->is_dir = entry_type == 1;
|
||||
file->is_symlink = entry_type == 2;
|
||||
file->is_special = entry_type == 3;
|
||||
data_pointer += sizeof(int);
|
||||
remaining_size -= sizeof(int);
|
||||
|
||||
if (file->is_special) {
|
||||
if (remaining_size < 2 * (int32_t)sizeof(int32_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for special rdev");
|
||||
goto error;
|
||||
}
|
||||
int32_t special_major, special_minor;
|
||||
memcpy(&special_major, data_pointer, sizeof(special_major));
|
||||
data_pointer += sizeof(special_major);
|
||||
memcpy(&special_minor, data_pointer, sizeof(special_minor));
|
||||
data_pointer += sizeof(special_minor);
|
||||
remaining_size -= 2 * sizeof(int32_t);
|
||||
/* Reject an out-of-range/negative rdev here as a malformed chunk (the
|
||||
same 0xffff / 0x00ffffff bounds file_special_rdev_valid uses), so a
|
||||
bogus large-but-positive rdev is refused cleanly instead of being
|
||||
deferred to the creation site where it would abort after the frame. */
|
||||
if (special_major < 0 || special_minor < 0 || special_major > 0xffff ||
|
||||
special_minor > 0x00ffffff) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: out-of-range special rdev");
|
||||
goto error;
|
||||
}
|
||||
file->rdev_major = special_major;
|
||||
file->rdev_minor = special_minor;
|
||||
if (file == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
if (use_metadata) {
|
||||
if (remaining_size < sizeof(int)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata");
|
||||
goto error;
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
/* Peek at the present flag to determine the total record size before
|
||||
decoding. metadata_from_buf() independently bounds-checks every read
|
||||
against remaining_size, so a short body can never over-read. */
|
||||
// Peek at present flag to determine total size needed before reading
|
||||
int present_flag;
|
||||
memcpy(&present_flag, data_pointer, sizeof(int));
|
||||
if ((present_flag != 0 && present_flag != 1) ||
|
||||
(present_flag == 1 && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata body");
|
||||
goto error;
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
file->metadata = metadata_from_buf((const uint8_t*)data_pointer, remaining_size);
|
||||
size_t metadata_consumed = sizeof(int);
|
||||
file->metadata = metadata_from_buf(&data_pointer);
|
||||
remaining_size -= sizeof(int);
|
||||
if (present_flag == 1) {
|
||||
if (file->metadata == NULL)
|
||||
goto error;
|
||||
metadata_consumed += FILE_METADATA_WIRE_SIZE;
|
||||
if (file->metadata == NULL) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
remaining_size -= FILE_METADATA_WIRE_SIZE;
|
||||
}
|
||||
data_pointer += metadata_consumed;
|
||||
remaining_size -= metadata_consumed;
|
||||
}
|
||||
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for data size");
|
||||
goto error;
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
size_t file_data_size;
|
||||
@@ -379,120 +230,75 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (remaining_size < file_data_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for file content");
|
||||
goto error;
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
// Reject individual file data larger than the maximum allowed size.
|
||||
if (file_data_size > MAX_CHUNK_FILE_DATA_SIZE) {
|
||||
if (file_data_size > MAX_FILE_DATA_SIZE) {
|
||||
log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size,
|
||||
(unsigned long long)MAX_CHUNK_FILE_DATA_SIZE);
|
||||
goto error;
|
||||
(unsigned long long)MAX_FILE_DATA_SIZE);
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
size_t allocation_size = file_data_size > 0 ? file_data_size : 1;
|
||||
void* file_data = protocol_alloc(allocation_size);
|
||||
void* file_data = malloc(allocation_size);
|
||||
if (file_data == NULL) {
|
||||
log_perror("Could not allocate memory for file data");
|
||||
goto error;
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(file_data, data_pointer, file_data_size);
|
||||
Data* replacement = data_create(file_data, file_data_size);
|
||||
if (replacement == NULL)
|
||||
goto error;
|
||||
/* Charge the retained per-file copy to the connection budget (when the
|
||||
inbound chunk carries an owning session) so the queued copies are not
|
||||
held outside MAX_CONNECTION_MEMORY (B6). A NULL owner (e.g. a local
|
||||
batch apply) leaves the copy uncharged. */
|
||||
if (!data_charge_session(replacement, data->owner, allocation_size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for chunk file data");
|
||||
data_destroy(replacement);
|
||||
goto error;
|
||||
if (replacement == NULL) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
data_destroy(file->data);
|
||||
file->data = replacement;
|
||||
data_pointer += file_data_size;
|
||||
remaining_size -= file_data_size;
|
||||
|
||||
if (file->is_symlink) {
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for symlink target");
|
||||
goto error;
|
||||
}
|
||||
size_t target_len;
|
||||
memcpy(&target_len, data_pointer, sizeof(size_t));
|
||||
data_pointer += sizeof(size_t);
|
||||
remaining_size -= sizeof(size_t);
|
||||
if (target_len == 0 || remaining_size < target_len) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad symlink target");
|
||||
goto error;
|
||||
}
|
||||
char* target = protocol_alloc(target_len + 1);
|
||||
if (!target) {
|
||||
log_perror("Could not allocate memory for symlink target");
|
||||
goto error;
|
||||
}
|
||||
memcpy(target, data_pointer, target_len);
|
||||
target[target_len] = '\0';
|
||||
if (memchr(target, '\0', target_len) != NULL) {
|
||||
free(target);
|
||||
goto error;
|
||||
}
|
||||
/* The symlink target also rides the wire charset; decode it to the local
|
||||
charset like the path (a target is a path). */
|
||||
if (charset_wire_active()) {
|
||||
char* local_target = charset_wire_apply(target);
|
||||
free(target);
|
||||
if (local_target == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv: received chunk symlink target cannot be converted to the local "
|
||||
"charset");
|
||||
goto error;
|
||||
}
|
||||
target = local_target;
|
||||
}
|
||||
file->symlink_target = target;
|
||||
data_pointer += target_len;
|
||||
remaining_size -= target_len;
|
||||
}
|
||||
|
||||
if (!array_list_add(files, file))
|
||||
goto error;
|
||||
file = NULL;
|
||||
}
|
||||
|
||||
File** file_array = (File**)array_list_to_array(files);
|
||||
if (files->size > 0 && file_array == NULL)
|
||||
goto error;
|
||||
Chunk* chunk = chunk_create(file_array, files->size);
|
||||
free(file_array);
|
||||
if (chunk == NULL)
|
||||
goto error;
|
||||
files->item_destroyer = NULL;
|
||||
array_list_delete(files);
|
||||
return chunk;
|
||||
|
||||
error:
|
||||
if (file)
|
||||
if (!array_list_add(files, file)) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) {
|
||||
return chunk_compress_with_threads(chunk, compression_level, use_metadata, 0);
|
||||
}
|
||||
|
||||
Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_metadata,
|
||||
int compression_threads) {
|
||||
File** file_array = (File**)array_list_to_array(files);
|
||||
if (files->size > 0 && file_array == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
Chunk* chunk = chunk_create(file_array, files->size);
|
||||
|
||||
free(file_array);
|
||||
if (chunk == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
files->item_destroyer = NULL;
|
||||
array_list_delete(files);
|
||||
|
||||
return chunk;
|
||||
}
|
||||
|
||||
Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) {
|
||||
log_message(LOG_LEVEL_DEBUG, "Starting to compress chunk");
|
||||
Data* serialized = chunk_serialize(chunk, use_metadata);
|
||||
if (serialized == NULL)
|
||||
return NULL;
|
||||
Data* compressed = data_compress_with_threads(serialized, compression_level, compression_threads);
|
||||
Data* compressed = data_compress(serialized, compression_level);
|
||||
data_destroy(serialized);
|
||||
if (compressed == NULL)
|
||||
return NULL;
|
||||
log_debug_message(LOG_DEBUG_PACK, "Chunk successfully compressed");
|
||||
log_message(LOG_LEVEL_DEBUG, "Chunk successfully compressed");
|
||||
return compressed;
|
||||
}
|
||||
|
||||
@@ -504,21 +310,12 @@ Chunk* receive_chunk_data(int fd, const Config* config) {
|
||||
}
|
||||
Data* data_to_process = chunk_data;
|
||||
if (config->use_compression) {
|
||||
/* Preserve the inbound session across decompression so the (larger)
|
||||
decompressed chunk is charged to the same connection budget; the
|
||||
compressed buffer's own charge is released by data_destroy below. */
|
||||
ProtocolSession* owner = chunk_data->owner;
|
||||
data_to_process = data_decompress_limited(chunk_data, MAX_CHUNK_SIZE);
|
||||
data_destroy(chunk_data);
|
||||
if (data_to_process == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to decompress chunk");
|
||||
return NULL;
|
||||
}
|
||||
if (!data_charge_session(data_to_process, owner, data_to_process->size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for decompressed chunk");
|
||||
data_destroy(data_to_process);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
// Reject chunks larger than the maximum allowed size to prevent OOM.
|
||||
|
||||
@@ -19,19 +19,6 @@ void chunk_destroy(void* chunk);
|
||||
Data* chunk_serialize(Chunk* chunk, bool use_metadata);
|
||||
Chunk* chunk_deserialize(Data* data, bool use_metadata);
|
||||
Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata);
|
||||
Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_metadata,
|
||||
int compression_threads);
|
||||
Chunk* receive_chunk_data(int fd, const Config* config);
|
||||
|
||||
/* Charge `charge` retained bytes of `data` against `session`'s per-connection
|
||||
* budget (MAX_CONNECTION_MEMORY), mirroring the protocol layer's accounting, and
|
||||
* record them on `data` so data_destroy() returns the charge through the
|
||||
* Data.owner path. Returns false (leaving `data` uncharged) when the ceiling
|
||||
* would be exceeded. A NULL/zero-size charge or a NULL session is a no-op
|
||||
* success. The receive-side decompression and chunk-copy paths know the owning
|
||||
* session only through the Data.owner of the buffer they are processing, so
|
||||
* this is the entry point that lets them participate in the connection budget
|
||||
* without a session handle (B6). */
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
|
||||
+75
-1006
File diff suppressed because it is too large
Load Diff
+2
-130
@@ -4,137 +4,9 @@
|
||||
#include "data.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
#define COMPRESSION_MAX_THREADS 64
|
||||
|
||||
/* Compression algorithms selectable with --compress-choice / -z. The ids are
|
||||
* the values placed on the wire (Config->compression_algo), so they must be
|
||||
* kept stable. NONE is "no compression"; ZSTD is the historical FastSync
|
||||
* default and the negotiated "auto" choice. ZLIBX is rsync's zlib-without-
|
||||
* matched-data variant: FastSync compresses only the delta/token bytes (it does
|
||||
* not put matched file data in the compression stream), so its zlib codec is
|
||||
* already the "x" form and zlib/zlibx share the same implementation, recorded
|
||||
* under distinct ids. */
|
||||
typedef enum {
|
||||
COMPRESSION_ALGO_NONE = 0,
|
||||
COMPRESSION_ALGO_ZSTD = 1,
|
||||
COMPRESSION_ALGO_LZ4 = 2,
|
||||
COMPRESSION_ALGO_ZLIB = 3,
|
||||
COMPRESSION_ALGO_ZLIBX = 4
|
||||
} CompressionAlgo;
|
||||
|
||||
/* Resolve a --compress-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "zstd", "lz4", "zlib", "zlibx", "none". "auto" is not an algorithm
|
||||
* here; the caller resolves it to the negotiated default. Returns -1 for any
|
||||
* unrecognized name. */
|
||||
int compression_algo_from_name(const char* name);
|
||||
const char* compression_algo_name(CompressionAlgo algo);
|
||||
bool compression_algo_valid(int algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list
|
||||
* (rsync 3.4.1's `--version` order: zstd lz4 zlibx zlib none). Resolves
|
||||
* "auto". */
|
||||
CompressionAlgo compression_negotiate_default(void);
|
||||
|
||||
/* Resolve "auto" the way rsync does: the first supported name in
|
||||
* RSYNC_COMPRESS_LIST (whitespace-separated, client half ends at '&'), then the
|
||||
* compiled-in preference order when the variable is unset/blank. Returns -1
|
||||
* when the variable is set but names no supported codec (rsync's failed
|
||||
* negotiation), otherwise a valid CompressionAlgo id. */
|
||||
int compression_choice_resolve(void);
|
||||
|
||||
/* rsync 3.4.1's per-codec default level, applied when the user did not pass
|
||||
* --compress-level/--zl. zstd uses ZSTD_CLEVEL_DEFAULT (3) and zlib/zlibx the
|
||||
* resolved Z_DEFAULT_COMPRESSION (6). lz4 has no tunable level in rsync
|
||||
* (always the default acceleration); FastSync returns a positive placeholder so
|
||||
* its "level > 0" compression gate stays engaged, and lz4_compress ignores the
|
||||
* value, so the output is identical to rsync's. none is 0. */
|
||||
int compression_default_level(CompressionAlgo algo);
|
||||
|
||||
/* Clamp an explicit --compress-level to the codec's accepted range the way
|
||||
* rsync's init_compression_level() does: zstd 1..22, zlib/zlibx 1..9, lz4
|
||||
* ignored (fixed positive placeholder), none 0. */
|
||||
int compression_clamp_level(CompressionAlgo algo, int level);
|
||||
|
||||
/* True when the algorithm actually compresses (i.e. is not NONE). */
|
||||
bool compression_algo_enabled(CompressionAlgo algo);
|
||||
|
||||
/* Select the process-wide codec used by the legacy wrappers below. Each
|
||||
* process serves exactly one transfer config (the server forks per connection,
|
||||
* the client configures itself before spawning transfer threads), so a
|
||||
* process-global default is sufficient and constant for the lifetime of a
|
||||
* transfer. Defaults to ZSTD when never set. Thread-safe. */
|
||||
void compression_set_algo(CompressionAlgo algo);
|
||||
CompressionAlgo compression_get_algo(void);
|
||||
|
||||
/* Codec-aware primitives. The compressed buffer is self-describing: its first
|
||||
* byte is the CompressionAlgo id, so decompression never needs the codec passed
|
||||
* separately (this keeps every existing Decompress call site source-compatible).
|
||||
* `data_compress_codec` returns NULL on invalid input or an unsupported codec. */
|
||||
Data* data_compress_codec(Data* data_to_compress, CompressionAlgo algo, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
|
||||
/* Legacy zstd-default wrappers retained for existing callers/tests. */
|
||||
Data* data_compress(Data* data_to_compress, int compression_level);
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress(Data* compressed_data);
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count);
|
||||
|
||||
/* Streaming decompression for a payload too large to hold in memory. The
|
||||
* caller consumes the frame's leading codec byte (and, for lz4/zlib/zlibx, the
|
||||
* 4-byte little-endian raw-size prefix) and then feeds the remaining frame
|
||||
* bytes in bounded chunks; decompressed output is written straight to `out_fd`
|
||||
* so neither the compressed nor the decompressed image is ever materialized.
|
||||
* Only zstd (the default), zlib/zlibx and none support streaming; lz4's block
|
||||
* format is one-shot, so its stream decompressor reports failure and the caller
|
||||
* falls back (the whole-buffer path keeps its existing bound). */
|
||||
typedef struct CompressionStreamDecompressor CompressionStreamDecompressor;
|
||||
|
||||
CompressionStreamDecompressor*
|
||||
compression_stream_decompressor_create(CompressionAlgo algo, unsigned long long expected_out);
|
||||
/* Feed one chunk. Returns false on a malformed frame, an I/O error, or when the
|
||||
* total output would exceed `expected_out` (when non-zero). *done is set once
|
||||
* the frame end has been reached. */
|
||||
bool compression_stream_decompressor_feed(CompressionStreamDecompressor* d, const void* in,
|
||||
size_t in_len, int out_fd, bool* done);
|
||||
unsigned long long compression_stream_decompressor_total(const CompressionStreamDecompressor* d);
|
||||
void compression_stream_decompressor_destroy(CompressionStreamDecompressor* d);
|
||||
|
||||
/* Peek the logical (decompressed) size from the leading bytes of a compressed
|
||||
* frame (codec byte + header), returning 0 when it cannot be determined from
|
||||
* the supplied prefix. Used to decide whether a frame must take the streaming
|
||||
* path before its body is read. */
|
||||
unsigned long long compression_peek_frame_content_size(const void* buf, size_t len);
|
||||
|
||||
/* Streaming compression (sender side). Compresses a source in bounded chunks
|
||||
* into `out_fd` as one self-describing frame (codec byte, the lz4/zlib raw-size
|
||||
* prefix, then the codec stream), so a whole file can be compressed without
|
||||
* materializing it in memory. zstd/zlib/zlibx/none are supported; lz4's block
|
||||
* format is one-shot, so its create() returns NULL and the caller keeps the
|
||||
* buffered path. `raw_size` is the known source length (used for the zlib
|
||||
* prefix and, for zstd, the frame content-size field). */
|
||||
typedef struct CompressionStreamCompressor CompressionStreamCompressor;
|
||||
|
||||
/* True when `algo` can be stream-compressed (zstd/zlib/zlibx; lz4's block format
|
||||
* is one-shot). Used by the sender to decide whether an over-threshold source
|
||||
* may stay unloaded. */
|
||||
bool compression_stream_compress_supported(CompressionAlgo algo);
|
||||
CompressionStreamCompressor* compression_stream_compressor_create(CompressionAlgo algo, int level,
|
||||
int threads);
|
||||
bool compression_stream_compressor_begin(CompressionStreamCompressor* c,
|
||||
unsigned long long raw_size, int out_fd);
|
||||
bool compression_stream_compressor_feed(CompressionStreamCompressor* c, const void* in,
|
||||
size_t in_len, int out_fd);
|
||||
bool compression_stream_compressor_finish(CompressionStreamCompressor* c, int out_fd);
|
||||
void compression_stream_compressor_destroy(CompressionStreamCompressor* c);
|
||||
|
||||
/* Release the calling thread's cached zstd contexts (compressor, decompressor
|
||||
* and scratch buffer). The cache is thread-local and is also released
|
||||
* automatically when a worker thread exits (via a C11 tss destructor) and for
|
||||
* the main thread at process exit; this explicit entry point exists so tests
|
||||
* and long-lived callers can drop the cache deterministically. Safe to call
|
||||
* when no context has been created, and idempotent. */
|
||||
void compression_free_thread_contexts(void);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
bool compression_should_skip(const char* path);
|
||||
|
||||
#endif
|
||||
|
||||
+191
-1436
File diff suppressed because it is too large
Load Diff
+73
-1191
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,203 +0,0 @@
|
||||
#ifndef CREDENTIALS_H
|
||||
#define CREDENTIALS_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
|
||||
/* Daemon password authentication (A7 remediation, protocol 2.19.0).
|
||||
*
|
||||
* FastSync authenticates a daemon connection with a SCRAM-SHA-256-style
|
||||
* challenge/response handshake. The daemon stores only a salted PBKDF2
|
||||
* verifier (never the password, and never a value that can be replayed as a
|
||||
* bearer credential): the client proves knowledge of the password against a
|
||||
* per-connection server nonce, and the server proves the same shared secret
|
||||
* back. See credentials.c for the exact derivation.
|
||||
*
|
||||
* Server credential store format (--password-file and --early-input): one line
|
||||
* per entry,
|
||||
* user:$fastsync$1$pbkdf2-sha256$<iters>$<salt_b64>$<stored_key_b64>$<server_key_b64>
|
||||
* with standard base64, a 16-byte salt and 32-byte keys, and iters in
|
||||
* [CREDENTIAL_MIN_ITERS, CREDENTIAL_MAX_ITERS]. Blank lines and lines whose
|
||||
* first non-space character is '#' or ';' are comments. The parser is STRICT:
|
||||
* a malformed line fails the whole load so a typo can never silently change who
|
||||
* may log in. A line holding the legacy (unsalted SHA-256 hex) secret is
|
||||
* hard-rejected with an actionable "legacy" error; there is no auto-upgrade.
|
||||
* Use `fastsync-server --hash-credentials` to generate new-format lines.
|
||||
*
|
||||
* Alongside the store, credentials_load maintains an exact-mode-0600
|
||||
* `<store_path>.dummykey` sidecar holding the store-wide random dummy key. It
|
||||
* is auto-created on first load and MUST be preserved across restarts: it makes
|
||||
* the dummy challenge for an unknown user stable for the life of the store, so
|
||||
* a daemon restart cannot be used as a username-enumeration oracle. A sidecar
|
||||
* that is not an exact-mode-0600 regular file of exactly 32 bytes fails the load
|
||||
* (fail closed); creation forces exact 0600 with fchmod (so a restrictive umask
|
||||
* cannot leave the sidecar unreadable), and only a create/write/fsync/link or
|
||||
* fchmod failure degrades to a transient per-run key with a warning. NOTE: the
|
||||
* sidecar requires EXACT 0600, whereas the store / password files only reject
|
||||
* group/other bits (a deliberate difference).
|
||||
*
|
||||
* Client --password-file format: the FIRST meaningful (non-comment, non-blank)
|
||||
* line is `user:password`, holding the literal password. The client keeps it
|
||||
* only for the duration of the handshake and wipes it at teardown; the file
|
||||
* should be mode 0600 and readable only by its owner. */
|
||||
|
||||
/* Longest accepted credential-file line (excluding the trailing newline). */
|
||||
#define CREDENTIAL_MAX_LINE 4096
|
||||
/* Upper bound on a username in a credential file and on the wire. Kept well
|
||||
* below MAX_STRING_SIZE so a wire username can never exhaust anything. */
|
||||
#define CREDENTIAL_MAX_USER_LEN 256
|
||||
/* Upper bound on a client-file password (before derivation). */
|
||||
#define CREDENTIAL_MAX_PASSWORD_LEN 1024
|
||||
|
||||
/* SCRAM-SHA-256 parameters. Salt and client nonce sizes are fixed by the
|
||||
* shared-auth-message framing; keys are always 32 bytes (SHA-256). */
|
||||
#define CREDENTIAL_SALT_LEN 16
|
||||
#define CREDENTIAL_NONCE_LEN 32
|
||||
#define CREDENTIAL_KEY_LEN 32
|
||||
#define CREDENTIAL_DEFAULT_ITERS 600000u
|
||||
#define CREDENTIAL_MIN_ITERS 100000u
|
||||
#define CREDENTIAL_MAX_ITERS 10000000u
|
||||
/* Buffer size for the full AuthMessage (prefix + three length-prefixed fields).
|
||||
* Worst case: 16 + 4 + 256 + 4 + 32 + 4 + 32. */
|
||||
#define CREDENTIAL_AUTH_MESSAGE_MAX \
|
||||
(16 + 4 + CREDENTIAL_MAX_USER_LEN + 4 + CREDENTIAL_NONCE_LEN + 4 + CREDENTIAL_NONCE_LEN)
|
||||
|
||||
typedef struct CredentialStore CredentialStore;
|
||||
|
||||
/* One resolved verifier. `found` is false for an unknown user or a user not on
|
||||
* a module's auth list; the remaining fields then hold a deterministic dummy
|
||||
* salt (HMAC of the store-wide dummy key over the username), the store-wide
|
||||
* uniform iteration count (default for an empty store) and fixed dummy keys, so
|
||||
* the server can run the same challenge/response math with no enumeration or
|
||||
* timing oracle. */
|
||||
typedef struct {
|
||||
uint8_t salt[CREDENTIAL_SALT_LEN];
|
||||
uint32_t iters;
|
||||
uint8_t stored_key[CREDENTIAL_KEY_LEN];
|
||||
uint8_t server_key[CREDENTIAL_KEY_LEN];
|
||||
bool found;
|
||||
} CredentialVerifier;
|
||||
|
||||
/* Load the daemon credential store.
|
||||
*
|
||||
* password_file and early_input_file are both NULL-or-path, matching the
|
||||
* server's --password-file and --early-input options. A file that cannot be
|
||||
* opened or that fails the strict grammar is a hard error (err filled, NULL
|
||||
* returned) -- the daemon fails CLOSED rather than serving an auth-required
|
||||
* module with a partial store. Both files may be NULL, which yields an empty
|
||||
* store (every auth-required module then refuses connections). Every entry in
|
||||
* the resulting store must agree on the iteration count; entries that disagree
|
||||
* (within one file or across the two layered sources) are rejected. When both
|
||||
* are given, the --early-input file is layered over --password-file: a duplicate
|
||||
* username whose verifier matches is deduplicated; one whose verifier differs
|
||||
* is an error (the two sources disagree), never a silent pick.
|
||||
*
|
||||
* The returned store is heap-owned; free it with credentials_free. */
|
||||
CredentialStore* credentials_load(const char* password_file, const char* early_input_file,
|
||||
char* err, size_t err_size);
|
||||
|
||||
/* Wipe every stored key/salt and free the store. */
|
||||
void credentials_free(CredentialStore* store);
|
||||
|
||||
/* True when `user` is a single bounded token free of whitespace/control bytes
|
||||
* (the rule applied to store users, client-file users and the module list). */
|
||||
bool credentials_username_valid(const char* user);
|
||||
|
||||
/* Standard base64. encode writes NUL-terminated output to out (size out_sz).
|
||||
* decode writes the raw bytes to out (capacity out_sz) and stores the length;
|
||||
* the input must be a well-formed padded base64 string. Both return false on
|
||||
* NULL arguments, a bad character/length, or insufficient output space. */
|
||||
bool credentials_b64_encode(const uint8_t* in, size_t n, char* out, size_t out_sz);
|
||||
bool credentials_b64_decode(const char* in, uint8_t* out, size_t out_sz, size_t* out_len);
|
||||
|
||||
/* Fill out[0..n) from the CSPRNG (RAND_bytes). Returns false on failure. */
|
||||
bool credentials_random_bytes(uint8_t* out, size_t n);
|
||||
|
||||
/* Resolve `user` against the store AND the module's auth-user list. The list
|
||||
* scan is a constant-time full-length comparison with no early break. On a
|
||||
* miss, *out is filled with a dummy verifier (a deterministic per-username salt
|
||||
* derived from the store's dummy key, the store-wide uniform iteration count,
|
||||
* fixed dummy keys, found=false). Returns false on invalid arguments or an
|
||||
* HMAC/crypto primitive failure. */
|
||||
bool credentials_get_verifier(const CredentialStore* store, const char* user,
|
||||
const char* const* module_users, int n, CredentialVerifier* out);
|
||||
|
||||
/* Derive the SCRAM keys from a plaintext password:
|
||||
* K = PBKDF2-HMAC-SHA256(password, salt, iters, 32)
|
||||
* ClientKey = HMAC-SHA256(K, "Client Key"); StoredKey = SHA256(ClientKey)
|
||||
* ServerKey = HMAC-SHA256(K, "Server Key")
|
||||
* Any of client_key/stored_key/server_key may be NULL when not needed.
|
||||
* `iters` must lie in [CREDENTIAL_MIN_ITERS, CREDENTIAL_MAX_ITERS]. */
|
||||
bool credentials_compute_keys(const char* password, const uint8_t salt[CREDENTIAL_SALT_LEN],
|
||||
uint32_t iters, uint8_t client_key[CREDENTIAL_KEY_LEN],
|
||||
uint8_t stored_key[CREDENTIAL_KEY_LEN],
|
||||
uint8_t server_key[CREDENTIAL_KEY_LEN]);
|
||||
|
||||
/* Serialize the shared AuthMessage:
|
||||
* "FastSync-Auth-v1" || be32(len(user)) || user
|
||||
* || be32(32) || server_nonce
|
||||
* || be32(32) || client_nonce
|
||||
* out must hold at least CREDENTIAL_AUTH_MESSAGE_MAX bytes. *out_len receives
|
||||
* the number of bytes written. */
|
||||
bool credentials_build_auth_message(const char* user, const uint8_t* snonce, const uint8_t* cnonce,
|
||||
uint8_t* out, size_t out_sz, size_t* out_len);
|
||||
|
||||
/* Client side: ClientProof = ClientKey XOR HMAC(StoredKey, AuthMessage), and
|
||||
* the expected ServerSignature = HMAC(ServerKey, AuthMessage). */
|
||||
bool credentials_client_proof(const uint8_t client_key[CREDENTIAL_KEY_LEN],
|
||||
const uint8_t stored_key[CREDENTIAL_KEY_LEN],
|
||||
const uint8_t server_key[CREDENTIAL_KEY_LEN], const uint8_t* auth_msg,
|
||||
size_t msg_len, uint8_t proof[CREDENTIAL_KEY_LEN],
|
||||
uint8_t server_sig[CREDENTIAL_KEY_LEN]);
|
||||
|
||||
/* Server side: recompute ClientSig' = HMAC(StoredKey, AuthMessage) and
|
||||
* ClientKey' = proof XOR ClientSig', then accept iff v->found AND
|
||||
* SHA256(ClientKey') equals StoredKey (constant-time over the 32-byte keys).
|
||||
* Always computes server_sig_out = HMAC(ServerKey, AuthMessage). Returns the
|
||||
* accept decision. */
|
||||
bool credentials_verify_response(const CredentialVerifier* v, const char* user,
|
||||
const uint8_t* snonce, const uint8_t* cnonce,
|
||||
const uint8_t proof[CREDENTIAL_KEY_LEN],
|
||||
uint8_t server_sig_out[CREDENTIAL_KEY_LEN]);
|
||||
|
||||
/* Derive a new-format store line for `user`/`password` and write it (without a
|
||||
* trailing newline) into out. A random 16-byte salt is used. On failure err is
|
||||
* filled. Used by --hash-credentials and by tests. */
|
||||
bool credentials_hash_store_line(const char* user, const char* password, uint32_t iters, char* out,
|
||||
size_t out_sz, char* err, size_t err_size);
|
||||
|
||||
/* Read `user:password` lines from `path` (the same no-group/other-bits check as
|
||||
* the other secret files) and write one new-format store line per entry to
|
||||
* `out`.
|
||||
* Blank/comment lines are skipped; a malformed line fails the whole run.
|
||||
* Returns 0 on success, -1 on error (err filled). Used by
|
||||
* `--hash-credentials`. */
|
||||
int credentials_hash_file(const char* path, uint32_t iters, FILE* out, char* err, size_t err_size);
|
||||
|
||||
/* Read the CLIENT-side secret file: the first meaningful line is
|
||||
* `user:password` (the literal password). *user_out and *password_out are
|
||||
* freshly allocated on success (password is plaintext -- the caller derives the
|
||||
* proof and then burns/frees it); both are NULL on error. Returns 0 on
|
||||
* success, -1 on failure (err filled: the path is named, never the credential
|
||||
* itself). Only the line's trailing CR/LF are stripped: the password's bytes
|
||||
* are otherwise preserved exactly, so a password with leading/trailing
|
||||
* whitespace (after the ':') is kept usable. The username is trimmed of
|
||||
* surrounding space/tabs. */
|
||||
int credentials_read_secret_file(const char* path, char** user_out, char** password_out, char* err,
|
||||
size_t err_size);
|
||||
|
||||
/* Constant-time equality over exactly len bytes. */
|
||||
bool credentials_secure_equal(const char* a, const char* b, size_t len);
|
||||
|
||||
/* Overwrite secret[0..len) with zeros (best-effort wipe). */
|
||||
void credentials_burn(char* secret, size_t len);
|
||||
|
||||
/* Number of entries currently in the store (tests/introspection). */
|
||||
int credentials_store_size(const CredentialStore* store);
|
||||
|
||||
/* Whether the store contains an entry for `user` (tests/introspection). */
|
||||
bool credentials_store_has(const CredentialStore* store, const char* user);
|
||||
|
||||
#endif
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,208 +0,0 @@
|
||||
#ifndef DAEMON_CONF_H
|
||||
#define DAEMON_CONF_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
|
||||
/* FastSync-native daemon configuration (a FastSync analog of rsyncd.conf).
|
||||
*
|
||||
* This is the config the fastsync-server --daemon listener consumes. It is
|
||||
* line-based with an implicit global section followed by zero or more
|
||||
* [module] sections. The full grammar is documented in RSYNC_COMPAT.md
|
||||
* ("Daemon Mode") and summarized below; the parser lives entirely in
|
||||
* daemon_conf.c so it can be unit tested without any socket code.
|
||||
*
|
||||
* The parser is STRICT: an unknown key, a malformed line, a value that does
|
||||
* not parse, a module without a `path`, or a line longer than
|
||||
* DAEMON_CONF_MAX_LINE all fail the whole load with a clear, line-numbered
|
||||
* error instead of being silently ignored. This keeps a typo from silently
|
||||
* changing what a module serves.
|
||||
*
|
||||
* rsync compatibility: to reduce the divergence from rsync 3.4.1's rsyncd.conf
|
||||
* grammar, the parser also ACCEPTS the common rsync GLOBAL and MODULE keys.
|
||||
* Keys with a FastSync equivalent are mapped onto it (the native spellings are
|
||||
* unchanged; `read only` defaults to yes like rsync, and `write only = yes`
|
||||
* opts a module into writability). Keys with no FastSync equivalent are
|
||||
* accepted and documented as inert (they load successfully but have no effect)
|
||||
* rather than failing the whole config; the accepted inert set is listed in
|
||||
* kRsyncInertGlobalKeys / kRsyncInertModuleKeys in daemon_conf.c and in
|
||||
* RSYNC_COMPAT.md. Every inert key whose intent is access control is loudly
|
||||
* warned about at load time (kRsyncUnenforced*SecurityKeys) so an operator
|
||||
* migrating a hardened rsyncd.conf is never misled into believing the
|
||||
* restriction is enforced. A key outside both the FastSync-native grammar and
|
||||
* the recognized rsync subset is still rejected as unknown. */
|
||||
|
||||
/* A daemon module's configured root is used exactly like the standalone
|
||||
* server's --destination-root: the daemon confines every connection that
|
||||
* selects this module to this path (file_open_secure_parent /
|
||||
* has_path_traversal / path_is_within all keep the existing confinement, just
|
||||
* per-module). There is never any client-chosen root: a module path always
|
||||
* stays confined. A daemon REFUSES every client-chosen ownership / super-user
|
||||
* request by default -- --numeric-ids, --chown, --usermap/--groupmap,
|
||||
* --fake-super, --copy-as and an explicit --super -- because there is no
|
||||
* per-module opt-in unless the operator adds one. An operator opts a single
|
||||
* module in with `client owner = yes` (DaemonModule.client_owner), which allows
|
||||
* that client to choose ownership within that module's root (the standalone/SSH
|
||||
* server honors such requests for its single operator-authorized root). The
|
||||
* operator-level --no-super veto additionally forces super-user activities off
|
||||
* for every daemon connection, even an opted-in module. See server_module_gate
|
||||
* in server.c and RSYNC_COMPAT.md.
|
||||
*
|
||||
* `auth_users` is honored by Wave B daemon authentication: a module that
|
||||
* declares auth users accepts a connection only when the presented username is
|
||||
* on this list AND verifies against the daemon's credential store
|
||||
* (--password-file / --early-input). An auth-required module with no usable
|
||||
* store refuses (fail closed) rather than falling open; see server.c. Auth is
|
||||
* never bypassed by ignoring the list. */
|
||||
typedef struct DaemonModule {
|
||||
char* name; /* module name, as the client requests it */
|
||||
char* path; /* module root (daemon-side authorized root) */
|
||||
bool read_only; /* `read only = yes/no`; defaults to the global `read only`
|
||||
default (rsync allows it in the global section), which is
|
||||
itself default YES (rsync modules are read-only unless
|
||||
`read only = no` / `write only = yes` opts in) */
|
||||
bool read_only_explicit; /* set when this module set its own `read only` or
|
||||
`write only = yes`, so a later global default (from a
|
||||
`--dparam read only=`) does not override it */
|
||||
bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in
|
||||
that lets this module's clients choose ownership
|
||||
(--numeric-ids/--chown/--usermap/--groupmap/--fake-super/
|
||||
--copy-as) and request explicit --super super-user
|
||||
activities. Without it the daemon refuses all of them. */
|
||||
char** auth_users; /* `auth users = a,b`; Wave B credential list */
|
||||
int auth_user_count;
|
||||
/* `max connections = N` (optional per-module cap). 0 means unlimited. The
|
||||
* per-connection child records the selected module in the shared registry
|
||||
* (daemon_limits.c) once the config frame names it, so the cap is enforced
|
||||
* across all forked children; the parent reclaims the slot on SIGCHLD. */
|
||||
int max_connections;
|
||||
char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
char** hosts_deny; /* `hosts deny = a,b`; host access deny patterns */
|
||||
int hosts_deny_count;
|
||||
} DaemonModule;
|
||||
|
||||
/* Global (pre-module) scalar keys. `motd file` is parsed and stored but has
|
||||
* no wire effect yet (MOTD display is Wave C). */
|
||||
typedef struct DaemonConfGlobals {
|
||||
int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */
|
||||
char* motd_file; /* `motd file`, may be NULL */
|
||||
char* address; /* `address` (optional bind address), may be NULL */
|
||||
bool read_only_default; /* global `read only` default for modules defined
|
||||
after it (rsync allows the module key in the
|
||||
global section); default YES to match rsync's
|
||||
read-only modules */
|
||||
int max_connections; /* `max connections`, default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */
|
||||
int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */
|
||||
int max_connections_per_host; /* `max connections per host`, concurrent cap per
|
||||
source IP; default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST (0 =
|
||||
unlimited) */
|
||||
int auth_lockout_threshold; /* `auth lockout threshold`, failed attempts from
|
||||
one source before lockout; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD (0
|
||||
disables) */
|
||||
int auth_lockout_duration_sec; /* `auth lockout duration`, seconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC
|
||||
(0 disables) */
|
||||
char** hosts_allow; /* `hosts allow`; global host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
char** hosts_deny; /* `hosts deny`; global host access deny patterns */
|
||||
int hosts_deny_count;
|
||||
} DaemonConfGlobals;
|
||||
|
||||
typedef struct DaemonConf {
|
||||
DaemonConfGlobals global;
|
||||
DaemonModule* modules;
|
||||
int module_count;
|
||||
} DaemonConf;
|
||||
|
||||
#define DAEMON_CONF_DEFAULT_PORT 873
|
||||
/* Default global connection cap when `max connections` is absent. Matches the
|
||||
* historical hardcoded listener value. */
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100
|
||||
/* Default `auth failure delay` in milliseconds (0 disables the throttle). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500
|
||||
/* Default `max connections per host` (0 = unlimited). */
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST 0
|
||||
/* Default cross-process auth lockout: 10 failed attempts from one source lock
|
||||
* it out for 300 s (0 disables either knob). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD 10
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC 300
|
||||
/* Upper bound on a `max connections per host` or `auth lockout threshold`
|
||||
* value, so a typo cannot size the shared registry absurdly. */
|
||||
#define DAEMON_CONF_MAX_CONCURRENCY_LIMIT 1000000
|
||||
/* Upper bound on `auth lockout duration` (7 days). */
|
||||
#define DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC 604800
|
||||
/* Largest accepted `auth failure delay`, so a typo cannot pin a connection
|
||||
* child in nanosleep for an absurd time. */
|
||||
/* Bounded well below the socket I/O timeout so a failed-auth child cannot hold
|
||||
* a connection slot for long enough to amplify connection-cap exhaustion. */
|
||||
#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000
|
||||
/* Upper bound on the number of [module] sections, so the shared registry's
|
||||
* per-module counter array stays fixed-size. The parser rejects the next
|
||||
* section past this bound. */
|
||||
#define DAEMON_CONF_MAX_MODULES 256
|
||||
/* Longest accepted config line (excluding the trailing newline). Longer lines
|
||||
* are rejected rather than buffered unboundedly. */
|
||||
#define DAEMON_CONF_MAX_LINE 4096
|
||||
/* Upper bound on a module name. Kept far below MAX_STRING_SIZE so a wire
|
||||
* module name can never exhaust anything by being long. */
|
||||
#define DAEMON_MAX_MODULE_NAME 200
|
||||
|
||||
/* Allocate an empty daemon config with defaulted globals (port 873, no
|
||||
* modules, no motd/address). Never fails for an allocation failure; callers
|
||||
* must still NULL-check. */
|
||||
DaemonConf* daemon_conf_create(void);
|
||||
|
||||
/* Parse `path` into a freshly allocated DaemonConf. Returns NULL on any error
|
||||
* and fills `err` (err_size bytes) with a clear, line-numbered message. The
|
||||
* returned object is heap-owned; free it with daemon_conf_free. */
|
||||
DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size);
|
||||
|
||||
void daemon_conf_free(DaemonConf* conf);
|
||||
|
||||
/* Case-sensitive exact module lookup by name. Returns the module or NULL.
|
||||
* Module names are matched exactly (rsync semantics). */
|
||||
const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char* name);
|
||||
|
||||
/* Module-name syntax check: non-empty, at most DAEMON_MAX_MODULE_NAME chars,
|
||||
* and only [A-Za-z0-9._-]. Used by the config parser, the client's
|
||||
* host::module/path destination parser, and (implicitly) by the daemon lookup
|
||||
* (a name that fails this can never match a parsed module). */
|
||||
bool daemon_module_name_valid(const char* name);
|
||||
|
||||
/* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and
|
||||
* apply it to the global keys only. Keys are case-insensitive and cover the
|
||||
* global keys defined by the grammar (port, motd file, address, read only,
|
||||
* max connections, max connections per host, auth failure delay,
|
||||
* auth lockout threshold, auth lockout duration, hosts allow, hosts deny) plus
|
||||
* the recognized inert rsync global keys and the compact rsync spellings
|
||||
* (`motdfile`, `pidfile`, `logfile`). Applying `read only` sets the global
|
||||
* default and re-applies it to every module that did not set its own value.
|
||||
* Returns 0 on success, -1 on error (err filled). */
|
||||
int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size);
|
||||
|
||||
/* Host access-control matching (pure; no I/O). `daemon_host_pattern_match`
|
||||
* matches one configured pattern against a numeric peer IP string. Supported
|
||||
* patterns: `*` (match anything), an IPv4/IPv6 literal, an IPv4/IPv6 CIDR
|
||||
* (`10.0.0.0/8`, `2001:db8::/32`), or a glob (`*.example.com`) evaluated with
|
||||
* the same matcher as file globs; a glob only matches a peer string of the
|
||||
* same shape, so a numeric peer never matches a hostname glob. */
|
||||
bool daemon_host_pattern_match(const char* pattern, const char* peer_ip);
|
||||
|
||||
/* rsync-like combined decision over a deny list and an allow list: a matching
|
||||
* deny rejects (deny takes precedence); otherwise, when any allow entries
|
||||
* exist, a peer that matches none is rejected; with no allow entries every
|
||||
* peer not denied is accepted. An empty/unset pair returns true. */
|
||||
bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count,
|
||||
char* const* deny, int deny_count);
|
||||
|
||||
/* True when at least one allow or deny pattern is configured (i.e. an
|
||||
* unprovable peer must fail closed rather than being treated as unrestricted). */
|
||||
bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny,
|
||||
int deny_count);
|
||||
|
||||
#endif
|
||||
@@ -1,494 +0,0 @@
|
||||
#include "daemon_limits.h"
|
||||
#include "daemon_conf.h"
|
||||
#include "log.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <netinet/in.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <time.h>
|
||||
|
||||
/* The two module-count bounds must agree: the daemon config parser never
|
||||
* produces more than DAEMON_CONF_MAX_MODULES modules, so the shared registry's
|
||||
* per-module counter array is sized from the same bound. */
|
||||
_Static_assert(DAEMON_LIMITS_MAX_MODULES == DAEMON_CONF_MAX_MODULES,
|
||||
"daemon_limits module bound must match daemon_conf");
|
||||
|
||||
/* Slot lifecycle states (stored in slot_state). */
|
||||
enum {
|
||||
SLOT_FREE = 0,
|
||||
SLOT_CLAIMED = 1,
|
||||
SLOT_REGISTERED = 2,
|
||||
};
|
||||
|
||||
/* The registry header lives at the base of the shared mapping; the pointer
|
||||
* fields point at the arrays carved out of the same mapping. Absolute pointers
|
||||
* remain valid in a forked child because fork() clones the address space and
|
||||
* mapping, so parent and child observe the same virtual addresses. */
|
||||
struct DaemonLimitRegistry {
|
||||
int max_slots;
|
||||
int module_count;
|
||||
int host_slots; /* power of two; 1 when no per-source tracking is needed */
|
||||
int per_host_cap;
|
||||
int lockout_threshold;
|
||||
int lockout_duration_sec;
|
||||
size_t map_size;
|
||||
_Atomic long long host_full_warn; /* last "table full" warning epoch */
|
||||
_Atomic int* slot_state;
|
||||
_Atomic int* slot_pid;
|
||||
_Atomic int* slot_module;
|
||||
_Atomic int* slot_host; /* per-source table bucket, or -1 */
|
||||
_Atomic int* module_active;
|
||||
_Atomic uint64_t* host_key; /* 0 == empty bucket */
|
||||
_Atomic int* host_active;
|
||||
_Atomic int* host_fail;
|
||||
_Atomic long long* host_until; /* epoch seconds the lockout expires */
|
||||
_Atomic long long* host_last_use; /* epoch seconds the bucket was last touched */
|
||||
};
|
||||
|
||||
static size_t round_up(size_t n, size_t align) {
|
||||
return (n + align - 1) & ~(align - 1);
|
||||
}
|
||||
|
||||
static size_t next_pow2(size_t n) {
|
||||
size_t p = 1;
|
||||
while (p < n)
|
||||
p <<= 1;
|
||||
return p;
|
||||
}
|
||||
|
||||
/* Parse a numeric IPv4/IPv6 peer string into family + raw bytes. */
|
||||
static bool parse_peer_ip(const char* peer_ip, int* family, unsigned char* bytes) {
|
||||
if (!peer_ip || *peer_ip == '\0')
|
||||
return false;
|
||||
struct in_addr v4;
|
||||
if (inet_pton(AF_INET, peer_ip, &v4) == 1) {
|
||||
memcpy(bytes, &v4, sizeof(v4));
|
||||
*family = AF_INET;
|
||||
return true;
|
||||
}
|
||||
struct in6_addr v6;
|
||||
if (inet_pton(AF_INET6, peer_ip, &v6) == 1) {
|
||||
memcpy(bytes, &v6, sizeof(v6));
|
||||
*family = AF_INET6;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) {
|
||||
if (ok)
|
||||
*ok = false;
|
||||
unsigned char bytes[16];
|
||||
int family = AF_UNSPEC;
|
||||
if (!parse_peer_ip(peer_ip, &family, bytes))
|
||||
return 0;
|
||||
uint64_t hash = 14695981039346656037ULL ^ (uint64_t)(uint32_t)family;
|
||||
size_t length = family == AF_INET ? 4 : 16;
|
||||
for (size_t i = 0; i < length; i++) {
|
||||
hash ^= bytes[i];
|
||||
hash *= 1099511628211ULL;
|
||||
}
|
||||
if (hash == 0)
|
||||
hash = 0x9e3779b97f4a7c15ULL;
|
||||
if (ok)
|
||||
*ok = true;
|
||||
return hash;
|
||||
}
|
||||
|
||||
/* True when the registry must maintain per-source buckets: either the per-host
|
||||
* cap is configured, or the auth lockout is (threshold AND duration > 0). A
|
||||
* lockout threshold without a duration is a no-op, so it must not size or intern
|
||||
* the table. create(), register() and the lockout paths all agree on this. */
|
||||
static bool registry_tracks_hosts(const DaemonLimitRegistry* registry) {
|
||||
return registry->per_host_cap > 0 ||
|
||||
(registry->lockout_threshold > 0 && registry->lockout_duration_sec > 0);
|
||||
}
|
||||
|
||||
/* Find the bucket holding `peer_ip`, or -1 when it has no entry. Finding a
|
||||
* bucket refreshes its last-use time so the eviction policy sees it as live. */
|
||||
static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], (long long)time(NULL),
|
||||
memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0)
|
||||
return -1; /* no tombstones: an empty bucket ends the probe chain */
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* A bucket with no live connection may be repurposed: immediately when its
|
||||
* lockout deadline has already passed (the review's "expired" case), or after an
|
||||
* idle window when it holds no pending lockout. A bucket with a future lockout
|
||||
* deadline is retained so the lockout actually lasts its configured duration. */
|
||||
static bool host_bucket_reclaimable(DaemonLimitRegistry* registry, size_t idx, long long now) {
|
||||
if (atomic_load_explicit(®istry->host_active[idx], memory_order_relaxed) != 0)
|
||||
return false;
|
||||
long long until = atomic_load_explicit(®istry->host_until[idx], memory_order_relaxed);
|
||||
if (until != 0)
|
||||
return until <= now;
|
||||
long long last_use = atomic_load_explicit(®istry->host_last_use[idx], memory_order_relaxed);
|
||||
/* A bucket whose key is published but whose last_use has not yet been stamped
|
||||
* (last_use == 0) must be treated as live: reclaiming it here would steal a
|
||||
* bucket a racing child just claimed. The claim path also stamps last_use
|
||||
* before publishing the key, so this window cannot persist. */
|
||||
return last_use != 0 && now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC;
|
||||
}
|
||||
|
||||
/* Emit at most one "per-source table full" warning per
|
||||
* DAEMON_LIMITS_HOST_FULL_WARN_SEC across all forked children. Called from a
|
||||
* normal (non-signal) child path, so logging is safe here. */
|
||||
static void host_warn_table_full(DaemonLimitRegistry* registry, long long now) {
|
||||
long long last = atomic_load_explicit(®istry->host_full_warn, memory_order_relaxed);
|
||||
if (last != 0 && now - last < DAEMON_LIMITS_HOST_FULL_WARN_SEC)
|
||||
return;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_full_warn, &last, now,
|
||||
memory_order_relaxed, memory_order_relaxed)) {
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"daemon: per-source registry is full (%d slots) and no bucket can be reclaimed; "
|
||||
"'max connections per host' and the auth lockout are temporarily not enforced for "
|
||||
"new sources (the per-module cap and host ACLs still apply)",
|
||||
registry->host_slots);
|
||||
}
|
||||
}
|
||||
|
||||
/* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two
|
||||
* forked children racing on the same source converge on one bucket.
|
||||
*
|
||||
* When the probe finds no empty bucket it reclaims, via a key CAS, the first
|
||||
* bucket that is reclaimable (expired lockout or idle, and no active
|
||||
* connection) and resets its counters. This bounds the table's lifetime so it
|
||||
* cannot fill permanently and stay fail-open. Returns -1 only when the address
|
||||
* is unparseable or the table is genuinely full of live/locked buckets
|
||||
* (callers fail open: the global/module caps and ACLs still apply). */
|
||||
static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
long long now = (long long)time(NULL);
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
/* A couple of passes bound the work: the first normally claims/seeds a bucket;
|
||||
* a lost eviction CAS retries once against the freshly observed table. */
|
||||
for (int pass = 0; pass < 2; pass++) {
|
||||
int evict = -1;
|
||||
uint64_t evict_key = 0;
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0) {
|
||||
/* Stamp last_use *before* publishing the key so a reclaimer racing the
|
||||
* claim can never observe a claimed bucket with last_use == 0 and
|
||||
* evict it. A pre-stamp is harmless if the CAS loses: the bucket is
|
||||
* either still empty (never inspected for reclaim) or has just been
|
||||
* taken by another source that wants a fresh timestamp anyway. */
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
uint64_t expected = 0;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
return (int)idx;
|
||||
}
|
||||
if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) {
|
||||
return (int)idx;
|
||||
}
|
||||
continue; /* another child won this empty bucket; keep probing */
|
||||
}
|
||||
if (evict < 0 && host_bucket_reclaimable(registry, idx, now)) {
|
||||
evict = (int)idx;
|
||||
evict_key = current;
|
||||
}
|
||||
}
|
||||
if (evict >= 0) {
|
||||
/* Refresh the timestamp before the key changes hands so the reused bucket
|
||||
* is not seen as immediately idle by a racing reclaimer. */
|
||||
atomic_store_explicit(®istry->host_last_use[evict], now, memory_order_relaxed);
|
||||
uint64_t expected = evict_key;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[evict], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
/* The bucket now belongs to the new source; clear the evicted source's
|
||||
* stale lockout/failure state. */
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_fail[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_until[evict], 0, memory_order_relaxed);
|
||||
/* Two children can race to intern the same brand-new key into different
|
||||
* eviction targets, leaving the table with duplicate buckets for `key`.
|
||||
* Re-scan for the first (canonical) bucket holding `key`; when it
|
||||
* precedes `evict`, drop our duplicate's occupancy and hand back the
|
||||
* canonical bucket so per-source counts are not orphaned on the
|
||||
* duplicate. The duplicate keeps its key, so no tombstone hole is
|
||||
* created and probe chains stay intact; it ages out normally. */
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t candidate = (start + i) & mask;
|
||||
uint64_t found =
|
||||
atomic_load_explicit(®istry->host_key[candidate], memory_order_acquire);
|
||||
if (found == key) {
|
||||
if (candidate != (size_t)evict) {
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
return (int)candidate;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (found == 0)
|
||||
break; /* the key is present at `evict`, so this cannot happen first */
|
||||
}
|
||||
return evict;
|
||||
}
|
||||
continue; /* lost the race; re-probe with fresh observations */
|
||||
}
|
||||
break; /* no free and no reclaimable bucket: genuinely full */
|
||||
}
|
||||
host_warn_table_full(registry, now);
|
||||
return -1;
|
||||
}
|
||||
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec) {
|
||||
if (max_slots < DAEMON_LIMITS_MIN_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MIN_SLOTS;
|
||||
if (max_slots > DAEMON_LIMITS_MAX_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MAX_SLOTS;
|
||||
if (module_count < 1)
|
||||
module_count = 1;
|
||||
if (module_count > DAEMON_LIMITS_MAX_MODULES)
|
||||
module_count = DAEMON_LIMITS_MAX_MODULES;
|
||||
if (per_host_cap < 0)
|
||||
per_host_cap = 0;
|
||||
if (lockout_threshold < 0)
|
||||
lockout_threshold = 0;
|
||||
if (lockout_duration_sec < 0)
|
||||
lockout_duration_sec = 0;
|
||||
|
||||
bool need_hosts = per_host_cap > 0 || (lockout_threshold > 0 && lockout_duration_sec > 0);
|
||||
int host_slots = 1;
|
||||
if (need_hosts) {
|
||||
size_t want = (size_t)max_slots * 4;
|
||||
if (want < 64)
|
||||
want = 64;
|
||||
if (want > DAEMON_LIMITS_MAX_HOST_SLOTS)
|
||||
want = DAEMON_LIMITS_MAX_HOST_SLOTS;
|
||||
host_slots = (int)next_pow2(want);
|
||||
}
|
||||
|
||||
size_t header = round_up(sizeof(DaemonLimitRegistry), 16);
|
||||
size_t slot_bytes =
|
||||
round_up((size_t)max_slots * sizeof(_Atomic int), 16) * 4; /* state,pid,module,host */
|
||||
size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16);
|
||||
size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16);
|
||||
size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2;
|
||||
size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16) * 2;
|
||||
size_t total =
|
||||
header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16;
|
||||
|
||||
void* map = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0);
|
||||
if (map == MAP_FAILED)
|
||||
return NULL;
|
||||
memset(map, 0, total);
|
||||
|
||||
DaemonLimitRegistry* registry = (DaemonLimitRegistry*)map;
|
||||
registry->max_slots = max_slots;
|
||||
registry->module_count = module_count;
|
||||
registry->host_slots = host_slots;
|
||||
registry->per_host_cap = per_host_cap;
|
||||
registry->lockout_threshold = lockout_threshold;
|
||||
registry->lockout_duration_sec = lockout_duration_sec;
|
||||
registry->map_size = total;
|
||||
|
||||
unsigned char* cursor = (unsigned char*)map + header;
|
||||
registry->slot_state = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_pid = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_module = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_host = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->module_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)module_count * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_key = (_Atomic uint64_t*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic uint64_t);
|
||||
registry->host_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
registry->host_fail = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_until = (atomic_llong*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic long long);
|
||||
registry->host_last_use = (atomic_llong*)cursor;
|
||||
|
||||
for (int i = 0; i < max_slots; i++) {
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
}
|
||||
return registry;
|
||||
}
|
||||
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
munmap(registry, registry->map_size);
|
||||
}
|
||||
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
int expected = SLOT_FREE;
|
||||
if (atomic_compare_exchange_strong(®istry->slot_state[i], &expected, SLOT_CLAIMED)) {
|
||||
atomic_store(®istry->slot_pid[i], 0);
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
}
|
||||
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_store(®istry->slot_pid[slot], (int)pid);
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel);
|
||||
atomic_store_explicit(®istry->slot_pid[slot], 0, memory_order_relaxed);
|
||||
/* The module/host occupancy arrays are derived from the slot table; do not
|
||||
* decrement here or a SIGKILL between a child's increment and its REGISTERED
|
||||
* publish would leak a count. Callers that need the derived counts call
|
||||
* daemon_limits_recompute. */
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) {
|
||||
if (!registry || pid <= 0)
|
||||
return;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load(®istry->slot_state[i]) == SLOT_FREE)
|
||||
continue;
|
||||
if (atomic_load(®istry->slot_pid[i]) == (int)pid) {
|
||||
daemon_limits_reclaim_slot(registry, i);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
/* Zero the derived arrays, then re-derive solely from the REGISTERED slots.
|
||||
* A child that was SIGKILLed after incrementing a counter but before
|
||||
* publishing REGISTERED is not counted, and its leaked increment is erased by
|
||||
* the zeroing, so the leak cannot persist. */
|
||||
for (int m = 0; m < registry->module_count; m++)
|
||||
atomic_store_explicit(®istry->module_active[m], 0, memory_order_relaxed);
|
||||
for (int h = 0; h < registry->host_slots; h++)
|
||||
atomic_store_explicit(®istry->host_active[h], 0, memory_order_relaxed);
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load_explicit(®istry->slot_state[i], memory_order_acquire) != SLOT_REGISTERED)
|
||||
continue;
|
||||
int module = atomic_load_explicit(®istry->slot_module[i], memory_order_relaxed);
|
||||
if (module >= 0 && module < registry->module_count)
|
||||
atomic_fetch_add_explicit(®istry->module_active[module], 1, memory_order_relaxed);
|
||||
int host = atomic_load_explicit(®istry->slot_host[i], memory_order_relaxed);
|
||||
if (host >= 0 && host < registry->host_slots)
|
||||
atomic_fetch_add_explicit(®istry->host_active[host], 1, memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (module_index < 0 || module_index >= registry->module_count)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (atomic_load_explicit(®istry->slot_state[slot], memory_order_acquire) != SLOT_CLAIMED)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
|
||||
int host = -1;
|
||||
if (registry_tracks_hosts(registry))
|
||||
host = host_intern(registry, peer_ip);
|
||||
|
||||
int module_count = atomic_fetch_add(®istry->module_active[module_index], 1) + 1;
|
||||
if (module_cap > 0 && module_count > module_cap) {
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_MODULE_FULL;
|
||||
}
|
||||
if (host >= 0) {
|
||||
int host_count = atomic_fetch_add(®istry->host_active[host], 1) + 1;
|
||||
if (registry->per_host_cap > 0 && host_count > registry->per_host_cap) {
|
||||
atomic_fetch_sub(®istry->host_active[host], 1);
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_HOST_FULL;
|
||||
}
|
||||
}
|
||||
atomic_store(®istry->slot_module[slot], module_index);
|
||||
atomic_store(®istry->slot_host[slot], host);
|
||||
atomic_store_explicit(®istry->slot_state[slot], SLOT_REGISTERED, memory_order_release);
|
||||
return DAEMON_LIMIT_OK;
|
||||
}
|
||||
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return false;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return false;
|
||||
long long until = atomic_load(®istry->host_until[bucket]);
|
||||
long long now = (long long)time(NULL);
|
||||
if (until > now) {
|
||||
if (seconds_remaining)
|
||||
*seconds_remaining = (int)(until - now);
|
||||
return true;
|
||||
}
|
||||
if (until != 0) {
|
||||
/* The previous lockout has expired: clear the stale counter so the source
|
||||
* gets a fresh allowance. */
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return;
|
||||
int bucket = host_intern(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
int failures = atomic_fetch_add(®istry->host_fail[bucket], 1) + 1;
|
||||
if (failures >= registry->lockout_threshold) {
|
||||
long long now = (long long)time(NULL);
|
||||
atomic_store(®istry->host_until[bucket], now + (long long)registry->lockout_duration_sec);
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry)
|
||||
return;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
@@ -1,147 +0,0 @@
|
||||
#ifndef DAEMON_LIMITS_H
|
||||
#define DAEMON_LIMITS_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* Cross-process daemon connection registry.
|
||||
*
|
||||
* The daemon listener forks ONE child per accepted connection, so any
|
||||
* per-module / per-source accounting must live in state shared across the
|
||||
* forked children. This module owns a fixed-size registry carved out of an
|
||||
* anonymous shared mapping (mmap(MAP_SHARED | MAP_ANONYMOUS)) created by the
|
||||
* accept-loop PARENT before it forks; every child inherits the mapping (and the
|
||||
* pointer to it) across fork().
|
||||
*
|
||||
* Rules:
|
||||
* - ONLY C11 atomics (atomic_*); never mtx_t/pthread locks, which can deadlock
|
||||
* in a forked child if another thread held them at fork time.
|
||||
* - No heap allocation after fork: the mapping is fixed-size and all access is
|
||||
* atomic load/store/CAS over preallocated arrays.
|
||||
*
|
||||
* Slot lifecycle (the parent reclaims even when a child is SIGKILLed):
|
||||
* FREE --(parent claim_slot)--> CLAIMED
|
||||
* CLAIMED --(child register)--> REGISTERED
|
||||
* any --(parent reclaim)--> FREE
|
||||
* The child records its module index and per-source bucket into the slot before
|
||||
* publishing REGISTERED; the parent's SIGCHLD handler matches the reaped pid to
|
||||
* the slot and, when REGISTERED, decrements the module/per-source counters.
|
||||
* A child killed before registering holds no counts, so reclaiming a CLAIMED
|
||||
* slot only frees the slot.
|
||||
*
|
||||
* Per-source identity is the normalized numeric peer IP (IPv4-mapped IPv6 is
|
||||
* already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an
|
||||
* open-addressed, linear-probing table keyed by a 64-bit hash. The same table
|
||||
* also carries the cross-process auth-failure counter and lockout deadline.
|
||||
*
|
||||
* Per-source table lifetime: a bucket's key is never cleared back to empty (that
|
||||
* would break every later probe chain that passed through it). Instead the
|
||||
* table has a bounded-lifetime eviction policy: when no empty bucket exists, the
|
||||
* first bucket that is reclaimable -- no active connection AND (its lockout
|
||||
* deadline has passed OR it has been idle for
|
||||
* DAEMON_LIMITS_HOST_EVICT_IDLE_SEC) -- is atomically repurposed for the new
|
||||
* source via a CAS of its key, and its counters are reset. The table therefore
|
||||
* cannot fill permanently, and a full table degrades to fail-open for the
|
||||
* per-source cap/lockout of new sources (the per-module cap and host ACLs still
|
||||
* apply) instead of staying fail-open forever. A rate-limited warning is logged
|
||||
* on the fail-open path. The eviction race with a concurrent
|
||||
* registration/reclaim on the same bucket is benign: it can at worst lose one
|
||||
* source's counter (fail-open), never corrupt memory or the module caps.
|
||||
*/
|
||||
|
||||
typedef struct DaemonLimitRegistry DaemonLimitRegistry;
|
||||
|
||||
/* Result of a per-connection admission check. */
|
||||
typedef enum {
|
||||
DAEMON_LIMIT_OK = 0, /* admitted; slot is now REGISTERED */
|
||||
DAEMON_LIMIT_MODULE_FULL, /* module's `max connections` cap reached */
|
||||
DAEMON_LIMIT_HOST_FULL, /* global `max connections per host` cap reached */
|
||||
DAEMON_LIMIT_UNAVAILABLE, /* registry/slot unusable (caller fails open) */
|
||||
} DaemonLimitResult;
|
||||
|
||||
/* Bounds for registry sizing. A slot is one concurrently live child. */
|
||||
#define DAEMON_LIMITS_MIN_SLOTS 16
|
||||
#define DAEMON_LIMITS_MAX_SLOTS 65536
|
||||
#define DAEMON_LIMITS_MAX_HOST_SLOTS 65536
|
||||
#define DAEMON_LIMITS_NO_SLOT (-1)
|
||||
/* Upper bound on `module_count`, matching daemon_conf.h's DAEMON_CONF_MAX_MODULES
|
||||
* (asserted in daemon_limits.c) so a caller can never size the per-module counter
|
||||
* array larger than the config parser can produce. */
|
||||
#define DAEMON_LIMITS_MAX_MODULES 256
|
||||
|
||||
/* Per-source table lifetime: a bucket with no active connection and no pending
|
||||
* lockout is reclaimable once it has been idle this long, so a flood of distinct
|
||||
* sources cannot pin the table full forever. A bucket whose lockout deadline
|
||||
* has passed is reclaimable immediately (independent of this idle window). */
|
||||
#define DAEMON_LIMITS_HOST_EVICT_IDLE_SEC 300
|
||||
/* Minimum spacing between "per-source table is full" warnings, so a table-full
|
||||
* attack cannot flood the log. */
|
||||
#define DAEMON_LIMITS_HOST_FULL_WARN_SEC 60
|
||||
|
||||
/* Create the shared registry in the calling (parent) process. `max_slots` is
|
||||
* the number of concurrently live children to track (clamped to
|
||||
* [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the
|
||||
* number of daemon modules (clamped to
|
||||
* [1, DAEMON_LIMITS_MAX_MODULES]); `per_host_cap` and the lockout pair come
|
||||
* from the daemon config (0 disables). Returns NULL on failure (e.g. mmap
|
||||
* allocation); callers must degrade gracefully (global cap + ACLs still
|
||||
* apply). */
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec);
|
||||
|
||||
/* Unmap the registry. Only the creating process may call this. */
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Parent side: reserve a slot for the next fork. Returns the slot index or
|
||||
* DAEMON_LIMITS_NO_SLOT when every slot is in use. */
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry);
|
||||
/* Parent side: record the forked child's pid in a claimed slot. */
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid);
|
||||
/* Parent side: release a slot. The slot becomes FREE; the module/per-source
|
||||
* occupancy arrays are DERIVED state and are only refreshed by
|
||||
* daemon_limits_recompute, which callers must invoke afterwards when they rely
|
||||
* on the derived counts (the SIGCHLD handler batches one recompute for the whole
|
||||
* reap). Idempotent. */
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot);
|
||||
/* Parent SIGCHLD side: release the slot owned by `pid` (no-op when not found).
|
||||
* Like reclaim_slot this does not touch the derived occupancy arrays; call
|
||||
* daemon_limits_recompute after a batch of releases. */
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid);
|
||||
|
||||
/* Parent side (async-signal-safe; atomics only, no malloc/log): rebuild
|
||||
* module_active[] / host_active[] from scratch by scanning the REGISTERED slots.
|
||||
* The slot table is the single source of truth, so this self-heals any
|
||||
* count leaked by a child that was SIGKILLed mid-registration (it zeroes the
|
||||
* arrays and re-derives them). Bounded by max_slots + host_slots. A
|
||||
* registration racing this call can be transiently undercounted until the next
|
||||
* recompute, which can only relax a cap briefly -- never corrupt memory. */
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Child side: admit the connection for `module_index` from `peer_ip`. Always
|
||||
* tracks the module/per-source occupancy (so the parent's reclaim is
|
||||
* symmetric); when `module_cap` > 0 it additionally enforces the per-module
|
||||
* cap. A NULL/empty or non-numeric `peer_ip` skips the per-source track (the
|
||||
* callers use that to exempt a trusted loopback peer from the per-host cap; the
|
||||
* per-module cap still applies). Returns DAEMON_LIMIT_OK and publishes the
|
||||
* slot, or a refusal reason. */
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap);
|
||||
|
||||
/* Child side: true when `peer_ip` is currently locked out after too many failed
|
||||
* authentications. `seconds_remaining` may be NULL. */
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining);
|
||||
/* Child side: count one failed authentication for `peer_ip`; once the threshold
|
||||
* is reached the source is locked out for the configured duration. */
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
/* Child side: clear the failure counter/lockout for a source that authenticated
|
||||
* successfully (no-op when the source has no table entry). */
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
|
||||
/* Pure helper: 64-bit FNV-1a hash of a numeric peer IP plus its family, used to
|
||||
* index the per-source table. *ok is set false (and 0 returned) for a NULL or
|
||||
* non-numeric address. Exposed for unit testing. */
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok);
|
||||
|
||||
#endif
|
||||
+4
-11
@@ -1,12 +1,11 @@
|
||||
#include "data.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include <stdlib.h>
|
||||
|
||||
Data* data_create_empty(size_t data_size) {
|
||||
/* malloc(0) is UB; allocate at least 1 byte but preserve requested size */
|
||||
size_t alloc_size = data_size > 0 ? data_size : 1;
|
||||
void* data = protocol_alloc(alloc_size);
|
||||
void* data = malloc(alloc_size);
|
||||
if (data == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Could not allocate memory for empty data");
|
||||
return NULL;
|
||||
@@ -15,7 +14,7 @@ Data* data_create_empty(size_t data_size) {
|
||||
}
|
||||
|
||||
Data* data_create_reserve(size_t size) {
|
||||
Data* d = protocol_alloc(sizeof(Data));
|
||||
Data* d = malloc(sizeof(Data));
|
||||
if (d == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data");
|
||||
return NULL;
|
||||
@@ -23,12 +22,11 @@ Data* data_create_reserve(size_t size) {
|
||||
d->data = NULL;
|
||||
d->size = size;
|
||||
d->protocol_charge = 0;
|
||||
d->owner = NULL;
|
||||
return d;
|
||||
}
|
||||
|
||||
Data* data_create(void* data, size_t data_size) {
|
||||
Data* new_data = protocol_alloc(sizeof(Data));
|
||||
Data* new_data = malloc(sizeof(Data));
|
||||
if (new_data == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data");
|
||||
free(data);
|
||||
@@ -37,19 +35,14 @@ Data* data_create(void* data, size_t data_size) {
|
||||
new_data->data = data;
|
||||
new_data->size = data_size;
|
||||
new_data->protocol_charge = 0;
|
||||
new_data->owner = NULL;
|
||||
return new_data;
|
||||
}
|
||||
|
||||
void data_destroy(Data* data) {
|
||||
if (data == NULL)
|
||||
return;
|
||||
if (data->protocol_charge != 0) {
|
||||
if (data->owner != NULL)
|
||||
protocol_release_memory_for_session(data->owner, data->protocol_charge);
|
||||
else
|
||||
if (data->protocol_charge != 0)
|
||||
protocol_release_memory(data->protocol_charge);
|
||||
}
|
||||
free(data->data);
|
||||
free(data);
|
||||
}
|
||||
|
||||
@@ -3,25 +3,11 @@
|
||||
|
||||
#include <stdlib.h>
|
||||
|
||||
/* Forward declaration for the connection budget a received Data is charged
|
||||
* against; defined in protocol.h (which includes this header). */
|
||||
typedef struct ProtocolSession ProtocolSession;
|
||||
|
||||
typedef struct {
|
||||
void* data;
|
||||
size_t size;
|
||||
/* Non-zero only for a buffer charged to the protocol connection budget. */
|
||||
size_t protocol_charge;
|
||||
/* Session whose budget `protocol_charge` was reserved from. When non-NULL,
|
||||
* the charge is returned to this session directly, regardless of which
|
||||
* session (if any) is bound to the destroying thread. owner is not
|
||||
* guaranteed to be set whenever protocol_charge is non-zero: it is NULL for
|
||||
* uncharged Data and for Data that has no recorded owner, in which case any
|
||||
* charge falls back to the session bound at destroy time.
|
||||
*
|
||||
* Lifetime contract: a Data with a non-NULL owner must not outlive that
|
||||
* ProtocolSession -- data_destroy dereferences owner to return the charge. */
|
||||
ProtocolSession* owner;
|
||||
} Data;
|
||||
|
||||
Data* data_create_empty(size_t data_size);
|
||||
@@ -29,9 +15,5 @@ Data* data_create_reserve(size_t size);
|
||||
Data* data_create(void* data, size_t data_size);
|
||||
void data_destroy(Data* data);
|
||||
void protocol_release_memory(size_t charge);
|
||||
/* Release `charge` against `session` directly instead of the thread-local bound
|
||||
* session. Used by data_destroy to honor Data.owner; `session` must outlive
|
||||
* the Data whose charge is being returned. A NULL session is a no-op. */
|
||||
void protocol_release_memory_for_session(ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,434 +0,0 @@
|
||||
#include "delay_updates.h"
|
||||
|
||||
#include "config.h"
|
||||
#include "file.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/file.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* Process-wide counter so two staging contexts created in the same process (or
|
||||
within the same clock tick) can never pick the same name. */
|
||||
static unsigned long long delay_updates_next_sequence(void) {
|
||||
static atomic_ullong sequence;
|
||||
return atomic_fetch_add_explicit(&sequence, 1, memory_order_relaxed);
|
||||
}
|
||||
|
||||
/* Build the per-run staging directory basename: the reserved prefix plus the
|
||||
pid and an entropy token. A fixed name could collide with a genuine
|
||||
destination entry; the token makes such a collision vanishingly unlikely and,
|
||||
if it ever happens, prepare() refuses to touch the existing directory. */
|
||||
static char* delay_updates_make_staging_name(void) {
|
||||
unsigned long long entropy = 0;
|
||||
int fd = open("/dev/urandom", O_RDONLY | O_CLOEXEC);
|
||||
if (fd >= 0) {
|
||||
ssize_t got = read(fd, &entropy, sizeof(entropy));
|
||||
close(fd);
|
||||
if (got != (ssize_t)sizeof(entropy))
|
||||
entropy = 0;
|
||||
}
|
||||
if (entropy == 0)
|
||||
entropy = ((unsigned long long)time(NULL) << 20) ^ ((unsigned long long)getpid() << 8) ^
|
||||
delay_updates_next_sequence();
|
||||
int length = snprintf(NULL, 0, DELAY_UPDATES_STAGING_DIR ".%ld.%llx", (long)getpid(), entropy);
|
||||
if (length < 0)
|
||||
return NULL;
|
||||
char* name = malloc((size_t)length + 1);
|
||||
if (!name)
|
||||
return NULL;
|
||||
snprintf(name, (size_t)length + 1, DELAY_UPDATES_STAGING_DIR ".%ld.%llx", (long)getpid(),
|
||||
entropy);
|
||||
return name;
|
||||
}
|
||||
|
||||
DelayUpdatesContext* delay_updates_context_create(const char* root_directory) {
|
||||
if (!root_directory)
|
||||
return NULL;
|
||||
DelayUpdatesContext* context = calloc(1, sizeof(DelayUpdatesContext));
|
||||
if (!context)
|
||||
return NULL;
|
||||
context->root_directory = str_dup(root_directory);
|
||||
if (!context->root_directory) {
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
context->staging_name = delay_updates_make_staging_name();
|
||||
if (!context->staging_name) {
|
||||
free(context->root_directory);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
context->staging_root = path_cat(root_directory, context->staging_name);
|
||||
if (!context->staging_root) {
|
||||
free(context->staging_name);
|
||||
free(context->root_directory);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
context->entries = NULL;
|
||||
context->count = 0;
|
||||
context->capacity = 0;
|
||||
context->prepared = false;
|
||||
context->lock_fd = -1;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success) {
|
||||
free(context->staging_root);
|
||||
free(context->staging_name);
|
||||
free(context->root_directory);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
return context;
|
||||
}
|
||||
|
||||
void delay_updates_context_destroy(DelayUpdatesContext* context) {
|
||||
if (!context)
|
||||
return;
|
||||
mtx_destroy(&context->mutex);
|
||||
if (context->lock_fd >= 0)
|
||||
close(context->lock_fd);
|
||||
context->lock_fd = -1;
|
||||
free(context->staging_root);
|
||||
free(context->staging_name);
|
||||
free(context->root_directory);
|
||||
for (size_t i = 0; i < context->count; i++) {
|
||||
free(context->entries[i].staged_path);
|
||||
free(context->entries[i].final_path);
|
||||
free(context->entries[i].file_path);
|
||||
}
|
||||
free(context->entries);
|
||||
free(context);
|
||||
}
|
||||
|
||||
bool delay_updates_staging_name_conflict(const char* dir) {
|
||||
if (!dir || !*dir)
|
||||
return false;
|
||||
size_t length = strlen(dir);
|
||||
while (length > 0 && dir[length - 1] == '/')
|
||||
length--;
|
||||
size_t reserved_length = strlen(DELAY_UPDATES_STAGING_DIR);
|
||||
if (length != reserved_length)
|
||||
return false;
|
||||
return strncmp(dir, DELAY_UPDATES_STAGING_DIR, length) == 0;
|
||||
}
|
||||
|
||||
/* Recursively delete every entry inside an open directory (never following
|
||||
symlinks). The directory itself is left in place. Mirrors the fd-relative
|
||||
walk used by the delete code so a symlink planted inside the staging tree
|
||||
can never redirect removal outside of it. */
|
||||
static bool delay_wipe_dir_fd(int dirfd) {
|
||||
int scanfd = dup(dirfd);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
DIR* dir = fdopendir(scanfd);
|
||||
if (!dir) {
|
||||
close(scanfd);
|
||||
return false;
|
||||
}
|
||||
bool operation_ok = true;
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT)
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (S_ISDIR(st.st_mode)) {
|
||||
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool child_removed = false;
|
||||
if (childfd >= 0) {
|
||||
child_removed = delay_wipe_dir_fd(childfd);
|
||||
close(childfd);
|
||||
} else if (errno != ENOENT) {
|
||||
operation_ok = false;
|
||||
}
|
||||
if (child_removed && unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0 && errno != ENOENT)
|
||||
operation_ok = false;
|
||||
} else {
|
||||
if (unlinkat(dirfd, entry->d_name, 0) != 0 && errno != ENOENT)
|
||||
operation_ok = false;
|
||||
}
|
||||
}
|
||||
closedir(dir);
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
bool delay_updates_prepare(DelayUpdatesContext* context) {
|
||||
if (!context)
|
||||
return false;
|
||||
if (context->prepared)
|
||||
return true;
|
||||
/* Create the per-run staging directory with O_EXCL semantics. The name is
|
||||
unique to this transfer, so if the path already exists it is NOT ours:
|
||||
either a genuine destination entry that happens to share the name or a
|
||||
leftover from another session. Refuse rather than wipe it -- the old
|
||||
fixed-name design could destroy a real destination entry. A crash
|
||||
leftover is never reused (the next run picks a fresh name). */
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(context->staging_root, &leaf, true);
|
||||
if (parent_fd < 0) {
|
||||
int saved_errno = errno;
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not create --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
int fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (fd >= 0) {
|
||||
close(fd);
|
||||
close(parent_fd);
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--delay-updates staging directory '%s' already exists and is not owned by this "
|
||||
"transfer; refusing to overwrite it",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
if (errno != ENOENT) {
|
||||
int saved_errno = errno;
|
||||
close(parent_fd);
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
if (mkdirat(parent_fd, leaf, 0700) != 0) {
|
||||
int saved_errno = errno;
|
||||
close(parent_fd);
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not create --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
if (fd < 0) {
|
||||
int saved_errno = errno;
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
/* Keep the exclusive advisory lock as defense in depth: the unique name
|
||||
already prevents two sessions from sharing a staging directory, but the
|
||||
lock also catches an improbable same-name collision that raced between the
|
||||
existence check above and the open. */
|
||||
if (flock(fd, LOCK_EX | LOCK_NB) != 0) {
|
||||
int saved_errno = errno;
|
||||
close(fd);
|
||||
char* escaped = output_escape(context->staging_root, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not lock --delay-updates staging directory '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
context->lock_fd = fd;
|
||||
context->prepared = true;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool delay_updates_record(DelayUpdatesContext* context, const char* staged_path,
|
||||
const char* final_path, const char* file_path) {
|
||||
if (!context || !staged_path || !final_path || !file_path)
|
||||
return false;
|
||||
char* staged_copy = str_dup(staged_path);
|
||||
char* final_copy = str_dup(final_path);
|
||||
char* file_copy = str_dup(file_path);
|
||||
if (!staged_copy || !final_copy || !file_copy) {
|
||||
free(staged_copy);
|
||||
free(final_copy);
|
||||
free(file_copy);
|
||||
return false;
|
||||
}
|
||||
mtx_lock(&context->mutex);
|
||||
bool ok = true;
|
||||
if (context->count == context->capacity) {
|
||||
size_t new_capacity = context->capacity == 0 ? 64 : context->capacity * 2;
|
||||
if (new_capacity < context->capacity) {
|
||||
ok = false;
|
||||
} else {
|
||||
StagedFileEntry* grown = realloc(context->entries, new_capacity * sizeof(StagedFileEntry));
|
||||
if (!grown) {
|
||||
ok = false;
|
||||
} else {
|
||||
context->entries = grown;
|
||||
context->capacity = new_capacity;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (ok) {
|
||||
context->entries[context->count].staged_path = staged_copy;
|
||||
context->entries[context->count].final_path = final_copy;
|
||||
context->entries[context->count].file_path = file_copy;
|
||||
context->count++;
|
||||
}
|
||||
mtx_unlock(&context->mutex);
|
||||
if (!ok) {
|
||||
free(staged_copy);
|
||||
free(final_copy);
|
||||
free(file_copy);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Move an existing final destination file aside before the staged replacement
|
||||
is installed. Deferred from stage time so the final destination is not
|
||||
modified until publication. Mirrors the immediate-mode backup logic. */
|
||||
static bool delay_publish_backup(const DelayUpdatesContext* context, const Config* config,
|
||||
const StagedFileEntry* entry) {
|
||||
bool backup_enabled = config && config->backup && !config->ignore_existing;
|
||||
if (!backup_enabled)
|
||||
return true;
|
||||
const char* backup_suffix = (config && config->suffix) ? config->suffix : "~";
|
||||
struct stat backup_stat;
|
||||
if (!file_stat_secure(entry->final_path, &backup_stat))
|
||||
return true; /* nothing to back up */
|
||||
|
||||
char* backup_path = NULL;
|
||||
if (config->backup_dir) {
|
||||
char* confined_backup = path_cat(context->root_directory, config->backup_dir);
|
||||
if (!confined_backup)
|
||||
return false;
|
||||
backup_path = path_cat(confined_backup, entry->file_path);
|
||||
free(confined_backup);
|
||||
} else {
|
||||
size_t path_len = strlen(entry->final_path);
|
||||
size_t suffix_len = strlen(backup_suffix);
|
||||
if (path_len > SIZE_MAX - suffix_len - 1)
|
||||
return false;
|
||||
backup_path = malloc(path_len + suffix_len + 1);
|
||||
if (backup_path) {
|
||||
memcpy(backup_path, entry->final_path, path_len);
|
||||
memcpy(backup_path + path_len, backup_suffix, suffix_len + 1);
|
||||
}
|
||||
}
|
||||
if (!backup_path)
|
||||
return false;
|
||||
char* parent_copy = str_dup(backup_path);
|
||||
if (!parent_copy || !file_ensure_directory_secure(dirname(parent_copy))) {
|
||||
free(parent_copy);
|
||||
free(backup_path);
|
||||
return false;
|
||||
}
|
||||
free(parent_copy);
|
||||
bool ok = file_rename_secure(entry->final_path, backup_path);
|
||||
free(backup_path);
|
||||
return ok;
|
||||
}
|
||||
|
||||
static bool delay_publish_entry(DelayUpdatesContext* context, const Config* config,
|
||||
const StagedFileEntry* entry) {
|
||||
if (!delay_publish_backup(context, config, entry))
|
||||
return false;
|
||||
/* An incoming regular file/symlink may replace a destination DIRECTORY that
|
||||
blocks it. rsync removes the blocker recursively when --delete or --force
|
||||
is active (its generator's "make way" deletion), and a --delay-updates run
|
||||
stages elsewhere so it only discovers the blocker here. FastSync's
|
||||
immediate-install path clears it too; without --delete/--force a non-empty
|
||||
blocker fails the run (rsync's "could not make way for new regular file").
|
||||
use_delete is gated by the server --allow-delete policy, so a client can
|
||||
never use this to bypass deletion authorization. */
|
||||
if (config && (config->force_delete || config->use_delete) &&
|
||||
file_directory_exists_secure(entry->final_path)) {
|
||||
if (!file_remove_tree_secure(entry->final_path)) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not remove destination directory blocking '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!file_rename_secure(entry->staged_path, entry->final_path)) {
|
||||
if (errno == EXDEV) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"staging directory is on a different filesystem than the destination; cannot "
|
||||
"atomically install file (EXDEV): %s",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
} else {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not install staged file '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(errno));
|
||||
free(escaped);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Remove the staging tree (contents plus the directory itself). Returns true
|
||||
when nothing is left behind (including the case where it never existed). */
|
||||
static bool delay_updates_remove_staging_tree(DelayUpdatesContext* context) {
|
||||
int fd = open(context->staging_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (fd < 0)
|
||||
return errno == ENOENT;
|
||||
bool ok = delay_wipe_dir_fd(fd);
|
||||
if (close(fd) != 0)
|
||||
ok = false;
|
||||
if (ok && rmdir(context->staging_root) != 0 && errno != ENOENT)
|
||||
ok = false;
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool delay_updates_publish(DelayUpdatesContext* context, const Config* config) {
|
||||
if (!context)
|
||||
return false;
|
||||
mtx_lock(&context->mutex);
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < context->count; i++) {
|
||||
if (!delay_publish_entry(context, config, &context->entries[i])) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
/* Renaming files out of the staging tree leaves the mirrored directories
|
||||
behind, and a mid-publish failure leaves the remaining staged files.
|
||||
Remove whatever is left so a later run starts from a clean staging area
|
||||
and no staged content can linger after a failed publish. If that cleanup
|
||||
fails, tell the operator: a stale staging directory would otherwise
|
||||
silently accumulate and make the next transfer's prepare-wipe fail. */
|
||||
if (!delay_updates_remove_staging_tree(context)) {
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"could not fully remove --delay-updates staging directory '%s' after publish; a "
|
||||
"later --delay-updates transfer to this destination will try to clear it",
|
||||
context->staging_root);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
void delay_updates_cleanup(DelayUpdatesContext* context) {
|
||||
if (!context)
|
||||
return;
|
||||
/* Only a context that gained exclusive ownership may touch the shared
|
||||
staging directory. If prepare never succeeded (e.g. lock contention with
|
||||
another live session) the directory belongs to that other session and must
|
||||
be left alone. */
|
||||
if (!context->prepared)
|
||||
return;
|
||||
delay_updates_remove_staging_tree(context);
|
||||
}
|
||||
@@ -1,74 +0,0 @@
|
||||
#ifndef DELAY_UPDATES_H
|
||||
#define DELAY_UPDATES_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <threads.h>
|
||||
|
||||
/* Forward-declared in config.h; full type needed by file_save_to_disk. */
|
||||
typedef struct Config Config;
|
||||
|
||||
/* One staged file awaiting publication. */
|
||||
typedef struct {
|
||||
char* staged_path; /* full path inside the staging tree */
|
||||
char* final_path; /* full final destination path */
|
||||
char* file_path; /* the file path as received on the wire */
|
||||
} StagedFileEntry;
|
||||
|
||||
/* Receiver-side --delay-updates staging registry. All successfully written
|
||||
files land under a private staging directory inside the receive root and are
|
||||
atomically renamed into their final destination only at the very end of the
|
||||
transfer. A single receiver pipeline (see src/server/receiver_pipeline.h)
|
||||
has exactly one writer thread, but the registry is still mutex-protected so
|
||||
the same object can be safely
|
||||
shared with the publish/cleanup phase that runs after the threads join. */
|
||||
typedef struct DelayUpdatesContext {
|
||||
char* root_directory; /* receive root the staging dir lives under */
|
||||
char* staging_name; /* per-run unique staging dir basename */
|
||||
char* staging_root; /* root_directory/<staging dir name> */
|
||||
mtx_t mutex;
|
||||
StagedFileEntry* entries;
|
||||
size_t count;
|
||||
size_t capacity;
|
||||
bool prepared; /* staging dir created, wiped, and exclusively locked */
|
||||
int lock_fd; /* advisory exclusive flock held on the staging dir, or -1 */
|
||||
} DelayUpdatesContext;
|
||||
|
||||
/* Reserved prefix for the private staging subdirectory created under the
|
||||
receive root. The actual directory name is per-run unique (the prefix plus a
|
||||
pid/entropy token) so it can never clobber a genuine destination entry that
|
||||
happens to share the name; the bare prefix is still what a --backup-dir must
|
||||
not collide with. */
|
||||
#define DELAY_UPDATES_STAGING_DIR ".fastsync-stage"
|
||||
|
||||
/* True when `dir` (ignoring a trailing "/") is the reserved staging directory
|
||||
name. Used to reject a --backup-dir that would collide with the internal
|
||||
staging area. */
|
||||
bool delay_updates_staging_name_conflict(const char* dir);
|
||||
|
||||
/* Create an empty staging context rooted below root_directory. Does not touch
|
||||
the filesystem yet. */
|
||||
DelayUpdatesContext* delay_updates_context_create(const char* root_directory);
|
||||
void delay_updates_context_destroy(DelayUpdatesContext* context);
|
||||
|
||||
/* Create the private 0700 staging directory (on first call) and wipe any
|
||||
leftovers from a previously interrupted delayed transfer. Idempotent. */
|
||||
bool delay_updates_prepare(DelayUpdatesContext* context);
|
||||
|
||||
/* Record a fully-written staged file for later publication. Copies all three
|
||||
paths. Returns false on allocation failure. */
|
||||
bool delay_updates_record(DelayUpdatesContext* context, const char* staged_path,
|
||||
const char* final_path, const char* file_path);
|
||||
|
||||
/* Atomically rename every staged file into its final destination. Deferred
|
||||
--backup handling runs immediately before each rename. On any failure the
|
||||
remaining staged files are removed (best effort); already-published files
|
||||
are not rolled back. Afterwards the staging tree is removed so a successful
|
||||
or failed publish leaves no staging leftovers. */
|
||||
bool delay_updates_publish(DelayUpdatesContext* context, const Config* config);
|
||||
|
||||
/* Best-effort removal of every staged file and the staging directory itself.
|
||||
Safe to call when nothing was staged or after a successful publish. */
|
||||
void delay_updates_cleanup(DelayUpdatesContext* context);
|
||||
|
||||
#endif
|
||||
@@ -1,777 +0,0 @@
|
||||
#include "delete.h"
|
||||
|
||||
#include "delay_updates.h"
|
||||
#include "filter.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* Build the keep-set index from the exact manifest entries only. A lookup of
|
||||
`rel` succeeds iff `rel` is a kept entry, a kept directory, or an ancestor
|
||||
directory of kept content (the old is_dir_in_manifest predicate); the sorted
|
||||
view answers "is an ancestor of kept content" without materializing any
|
||||
per-component prefix copy, so the index is O(manifest size) memory. */
|
||||
static bool build_keep_index(const ArrayList* manifest, PathIndex* index) {
|
||||
if (!manifest || manifest->size <= 0)
|
||||
return path_index_build(index, NULL, 0);
|
||||
return path_index_build(index, (const char* const*)manifest->items, (size_t)manifest->size);
|
||||
}
|
||||
|
||||
static bool keep_is_dir(const PathIndex* index, const char* rel_path) {
|
||||
return path_index_contains(index, rel_path) || path_index_has_descendant(index, rel_path);
|
||||
}
|
||||
|
||||
static bool keep_is_file(const PathIndex* index, const char* rel_path) {
|
||||
return path_index_contains(index, rel_path);
|
||||
}
|
||||
|
||||
/* rsync's receiver-side verdict for one candidate extra: the per-directory
|
||||
* chain first (deepest directory before ancestors), then the command-line base
|
||||
* rules. Either rule set may be absent. */
|
||||
FilterAction delete_protect_verdict(const DeleteProtectRules* protect, const char* rel_path,
|
||||
const char* leaf, bool is_dir) {
|
||||
if (!protect)
|
||||
return FILTER_ACTION_NONE;
|
||||
/* rsync protects its own --backup files from the delete pass: a name ending
|
||||
in the backup suffix is never an extra. Checked before the filter rules so
|
||||
an explicit exclude cannot be bypassed (the suffix is always a shield). */
|
||||
if (protect->backup_suffix && protect->backup_suffix[0] != '\0') {
|
||||
size_t name_len = strlen(leaf);
|
||||
size_t suffix_len = strlen(protect->backup_suffix);
|
||||
if (name_len > suffix_len &&
|
||||
strcmp(leaf + (name_len - suffix_len), protect->backup_suffix) == 0)
|
||||
return FILTER_ACTION_PROTECT;
|
||||
}
|
||||
FilterAction action = filter_dir_rules_apply_side(protect->dir_rules, rel_path, leaf, is_dir);
|
||||
if (action != FILTER_ACTION_NONE)
|
||||
return action;
|
||||
return filter_rules_apply_side(protect->base_rules, rel_path, leaf, is_dir, FILTER_SIDE_RECEIVER);
|
||||
}
|
||||
|
||||
const char* delete_backup_suffix(const Config* config) {
|
||||
if (!config || !config->backup || config->ignore_existing)
|
||||
return NULL;
|
||||
const char* suffix = config->suffix ? config->suffix : "~";
|
||||
if (!suffix[0] || strchr(suffix, '/'))
|
||||
return NULL;
|
||||
return suffix;
|
||||
}
|
||||
|
||||
/* Classify a removed entry from its st_mode for the per-type delete counters. */
|
||||
DeleteEntryType delete_entry_type_of_mode(mode_t mode) {
|
||||
if (S_ISDIR(mode))
|
||||
return DELETE_ENTRY_DIR;
|
||||
if (S_ISLNK(mode))
|
||||
return DELETE_ENTRY_LINK;
|
||||
if (S_ISREG(mode))
|
||||
return DELETE_ENTRY_REG;
|
||||
return DELETE_ENTRY_SPECIAL;
|
||||
}
|
||||
|
||||
/* True when child_rel is, or lies below, a protected entry. A prefix "a"
|
||||
therefore protects "a" and "a/b/c" but not "ab". Entries with top_level_only
|
||||
set only protect DIRECT children of the receive root (at_root); nested
|
||||
directories that share such a name stay ordinary destination content. */
|
||||
bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips,
|
||||
int skip_count) {
|
||||
for (int i = 0; i < skip_count; i++) {
|
||||
if (skips[i].top_level_only && !at_root)
|
||||
continue;
|
||||
size_t prefix_len = strlen(skips[i].prefix);
|
||||
if (strncmp(child_rel, skips[i].prefix, prefix_len) == 0 &&
|
||||
(child_rel[prefix_len] == '\0' || child_rel[prefix_len] == '/'))
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Per-run deletion budget and tallies. `max_delete` is the cap on the number
|
||||
of entries the walker may remove (SIZE_MAX = unlimited); once it is reached
|
||||
the remaining extras are counted in `skipped` and left in place, matching
|
||||
rsync's partial --max-delete behavior. */
|
||||
typedef struct {
|
||||
size_t max_delete;
|
||||
size_t deleted;
|
||||
size_t skipped;
|
||||
bool limit_hit;
|
||||
} DeleteBudget;
|
||||
|
||||
/* True when direct children of the directory named by `rel` may be removed.
|
||||
With no synchronization info (dirs == NULL) the whole tree is deletable; when
|
||||
a dirs index is supplied only its exact entries are (the receive root is the
|
||||
"." sentinel). */
|
||||
static bool is_synced_dir(const PathIndex* dirs, const char* rel) {
|
||||
if (!dirs)
|
||||
return true;
|
||||
return path_index_contains(dirs, rel[0] == '\0' ? "." : rel);
|
||||
}
|
||||
|
||||
/* Unsigned byte-wise string compare, matching rsync's u_strcmp (a signed
|
||||
strcmp would order bytes >= 0x80 differently). */
|
||||
static int delete_name_cmp(const char* a, const char* b) {
|
||||
const unsigned char* pa = (const unsigned char*)a;
|
||||
const unsigned char* pb = (const unsigned char*)b;
|
||||
while (*pa != '\0' && *pa == *pb) {
|
||||
pa++;
|
||||
pb++;
|
||||
}
|
||||
return (int)*pa - (int)*pb;
|
||||
}
|
||||
|
||||
bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count,
|
||||
bool* operation_ok) {
|
||||
*out = NULL;
|
||||
*count = 0;
|
||||
if (operation_ok)
|
||||
*operation_ok = true;
|
||||
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
DIR* dir = fdopendir(scanfd);
|
||||
if (!dir) {
|
||||
close(scanfd);
|
||||
return false;
|
||||
}
|
||||
DeleteDirEntry* entries = NULL;
|
||||
size_t used = 0;
|
||||
size_t capacity = 0;
|
||||
bool ok = true;
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT && operation_ok)
|
||||
*operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (used == capacity) {
|
||||
size_t next = capacity == 0 ? 16 : capacity * 2;
|
||||
DeleteDirEntry* grown = realloc(entries, next * sizeof(*grown));
|
||||
if (!grown) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
entries = grown;
|
||||
capacity = next;
|
||||
}
|
||||
entries[used].name = str_dup(entry->d_name);
|
||||
if (!entries[used].name) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
entries[used].is_dir = S_ISDIR(st.st_mode);
|
||||
entries[used].mode = st.st_mode;
|
||||
used++;
|
||||
}
|
||||
closedir(dir);
|
||||
if (!ok) {
|
||||
delete_dir_entries_free(entries, used);
|
||||
return false;
|
||||
}
|
||||
*out = entries;
|
||||
*count = used;
|
||||
return true;
|
||||
}
|
||||
|
||||
void delete_dir_entries_free(DeleteDirEntry* entries, size_t count) {
|
||||
if (!entries)
|
||||
return;
|
||||
for (size_t i = 0; i < count; i++)
|
||||
free(entries[i].name);
|
||||
free(entries);
|
||||
}
|
||||
|
||||
/* rsync's extraneous-entry order: subdirectories before files, each group in
|
||||
descending name order. */
|
||||
int delete_dir_entry_cmp_desc(const void* a, const void* b) {
|
||||
const DeleteDirEntry* ea = a;
|
||||
const DeleteDirEntry* eb = b;
|
||||
if (ea->is_dir != eb->is_dir)
|
||||
return ea->is_dir ? -1 : 1;
|
||||
return -delete_name_cmp(ea->name, eb->name);
|
||||
}
|
||||
|
||||
/* rsync's kept-subdirectory order: plain ascending name. */
|
||||
int delete_dir_entry_cmp_asc(const void* a, const void* b) {
|
||||
const DeleteDirEntry* ea = a;
|
||||
const DeleteDirEntry* eb = b;
|
||||
return delete_name_cmp(ea->name, eb->name);
|
||||
}
|
||||
|
||||
/* How the shared classification/descent walk disposes of an extra it has
|
||||
identified. LIST records the destination-relative path without touching disk
|
||||
(the -n/--dry-run would-delete enumeration); DELETE unlinks/rmdirs it, charges
|
||||
the shared --max-delete budget and notifies the observer. Both modes classify
|
||||
and traverse identically, so the dry-run enumeration and the real deletion
|
||||
cannot drift. */
|
||||
typedef enum { DELETE_WALK_MODE_DELETE, DELETE_WALK_MODE_LIST } DeleteWalkMode;
|
||||
|
||||
typedef struct {
|
||||
DeleteWalkMode mode;
|
||||
DeleteBudget* budget; /* DELETE mode */
|
||||
ArrayList* out; /* LIST mode: receives strdup'd relative paths */
|
||||
size_t* recorded; /* LIST mode */
|
||||
DeletePathObserver observer; /* DELETE mode */
|
||||
void* observer_context; /* DELETE mode */
|
||||
} DeleteWalkState;
|
||||
|
||||
/* The per-walk invariants threaded unchanged through every recursive descent:
|
||||
the keep/synchronized-dir indexes, the destination mode and the protection
|
||||
rules. Bundling them keeps the recursive helpers below to a handful of
|
||||
positional arguments. */
|
||||
typedef struct {
|
||||
const PathIndex* keep;
|
||||
const PathIndex* dirs;
|
||||
DeleteWalkState* state;
|
||||
const DeleteSkipEntry* skips;
|
||||
int skip_count;
|
||||
const DeleteProtectRules* protect;
|
||||
} DeleteWalkContext;
|
||||
|
||||
/* Duplicate `path` with rsync's trailing-slash convention, used to report a
|
||||
removed (or would-be-removed) directory. Returns NULL on allocation
|
||||
failure. */
|
||||
static char* with_trailing_slash(const char* path) {
|
||||
size_t len = strlen(path);
|
||||
char* copy = malloc(len + 2);
|
||||
if (!copy)
|
||||
return NULL;
|
||||
memcpy(copy, path, len);
|
||||
copy[len] = '/';
|
||||
copy[len + 1] = '\0';
|
||||
return copy;
|
||||
}
|
||||
|
||||
/* Forward declaration: the ordered passes below recurse through the driver. */
|
||||
static bool delete_walk_fd(int dirfd, const char* rel_path, const DeleteWalkContext* ctx,
|
||||
bool parent_deletable, bool* all_removed);
|
||||
|
||||
/* Descend into the child directory `name` of `dirfd`, walking it as part of the
|
||||
current operation. Returns false on a genuine open/walk failure; on success
|
||||
*child_all_removed reports whether the child removed everything it held (so
|
||||
the caller may rmdir it). */
|
||||
static bool delete_walk_child(int dirfd, const char* name, const char* child_rel,
|
||||
const DeleteWalkContext* ctx, bool deletable,
|
||||
bool* child_all_removed) {
|
||||
*child_all_removed = false;
|
||||
int childfd = openat(dirfd, name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (childfd < 0)
|
||||
return errno == ENOENT;
|
||||
bool ok = delete_walk_fd(childfd, child_rel, ctx, deletable, child_all_removed);
|
||||
close(childfd);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Classify every entry up front (the verdict does not depend on processing
|
||||
order) so the ordered passes below can act on it. Sets shielded[]/is_extra[]
|
||||
and reports through *local_survives whether anything in this directory stays
|
||||
in place. Returns false on a path-construction failure. */
|
||||
static bool delete_walk_classify(const char* rel_path, const DeleteDirEntry* entries, size_t count,
|
||||
const DeleteWalkContext* ctx, bool deletable, bool at_root,
|
||||
bool* shielded, bool* is_extra, bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
/* A --delay-updates run keeps its staging directory as a direct child of
|
||||
the receive root, and basis-dir snapshots live below it too. Their
|
||||
contents are not manifest entries, so descending into them would delete
|
||||
every staged / basis file as an "extra". Only the staging name (a
|
||||
top-level-only prefix) and the basis prefixes are protected: a nested
|
||||
destination directory that happens to be called .fastsync-stage is
|
||||
ordinary content. */
|
||||
if (path_under_skip_prefix(child_rel, at_root, ctx->skips, ctx->skip_count)) {
|
||||
shielded[i] = true;
|
||||
*local_survives = true;
|
||||
} else if (delete_protect_verdict(ctx->protect, child_rel, entries[i].name,
|
||||
entries[i].is_dir) == FILTER_ACTION_PROTECT) {
|
||||
/* A first-match protect rule shields the extra; for a directory the whole
|
||||
subtree is shielded (rsync prunes an excluded directory), so do not
|
||||
descend. */
|
||||
shielded[i] = true;
|
||||
*local_survives = true;
|
||||
} else if (entries[i].is_dir) {
|
||||
bool child_synced = ctx->dirs && path_index_contains(ctx->dirs, child_rel);
|
||||
is_extra[i] = deletable && !child_synced && !keep_is_dir(ctx->keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
*local_survives = true;
|
||||
} else {
|
||||
is_extra[i] = deletable && !keep_is_file(ctx->keep, child_rel);
|
||||
if (!is_extra[i])
|
||||
*local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Pass 1: extraneous subdirectories, descending. Recurses into each and, when
|
||||
the child removed everything it held, records or removes it and charges the
|
||||
budget. */
|
||||
static bool delete_walk_extra_dirs(int dirfd, const char* rel_path, const DeleteDirEntry* entries,
|
||||
size_t dir_count, const DeleteWalkContext* ctx, bool deletable,
|
||||
const bool* is_extra, bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = 0; i < dir_count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
bool child_all_removed = false;
|
||||
if (!delete_walk_child(dirfd, entries[i].name, child_rel, ctx, deletable, &child_all_removed))
|
||||
ok = false;
|
||||
if (child_all_removed && deletable) {
|
||||
if (ctx->state->mode == DELETE_WALK_MODE_LIST) {
|
||||
/* Record the directory with rsync's trailing slash. */
|
||||
char* copy = with_trailing_slash(child_rel);
|
||||
if (!copy) {
|
||||
ok = false;
|
||||
} else if (!array_list_add(ctx->state->out, copy)) {
|
||||
free(copy);
|
||||
ok = false;
|
||||
} else {
|
||||
(*ctx->state->recorded)++;
|
||||
}
|
||||
} else if (ctx->state->budget->deleted >= ctx->state->budget->max_delete) {
|
||||
ctx->state->budget->limit_hit = true;
|
||||
ctx->state->budget->skipped++;
|
||||
*local_survives = true;
|
||||
} else if (unlinkat(dirfd, entries[i].name, AT_REMOVEDIR) != 0) {
|
||||
/* ENOENT: already gone (fine). ENOTEMPTY/EEXIST: the directory still
|
||||
holds entries the walker leaves in place (a protected excluded
|
||||
prefix, a kept file the manifest protects, a symlink); rsync leaves
|
||||
such a directory behind, so this is not an error. Only genuine I/O
|
||||
failures abort the deletion. */
|
||||
if (errno != ENOENT && errno != ENOTEMPTY && errno != EEXIST)
|
||||
ok = false;
|
||||
*local_survives = true;
|
||||
} else {
|
||||
ctx->state->budget->deleted++;
|
||||
/* rsync reports a removed directory with a trailing slash. */
|
||||
if (ctx->state->observer) {
|
||||
char* with_slash = with_trailing_slash(child_rel);
|
||||
if (with_slash) {
|
||||
ctx->state->observer(ctx->state->observer_context, with_slash, DELETE_ENTRY_DIR);
|
||||
free(with_slash);
|
||||
} else {
|
||||
ctx->state->observer(ctx->state->observer_context, child_rel, DELETE_ENTRY_DIR);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
*local_survives = true;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Pass 2: extraneous files, descending. */
|
||||
static bool delete_walk_extra_files(int dirfd, const char* rel_path, const DeleteDirEntry* entries,
|
||||
size_t dir_count, size_t count, const DeleteWalkContext* ctx,
|
||||
const bool* is_extra, bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = dir_count; i < count; i++) {
|
||||
if (!is_extra[i])
|
||||
continue;
|
||||
if (ctx->state->mode == DELETE_WALK_MODE_LIST) {
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
char* copy = str_dup(child_rel);
|
||||
if (!copy || !array_list_add(ctx->state->out, copy)) {
|
||||
free(copy);
|
||||
ok = false;
|
||||
} else {
|
||||
(*ctx->state->recorded)++;
|
||||
}
|
||||
free(child_rel);
|
||||
} else if (ctx->state->budget->deleted >= ctx->state->budget->max_delete) {
|
||||
ctx->state->budget->limit_hit = true;
|
||||
ctx->state->budget->skipped++;
|
||||
*local_survives = true;
|
||||
} else if (unlinkat(dirfd, entries[i].name, 0) != 0) {
|
||||
if (errno != ENOENT)
|
||||
ok = false;
|
||||
*local_survives = true;
|
||||
} else {
|
||||
ctx->state->budget->deleted++;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (child_rel) {
|
||||
if (ctx->state->observer)
|
||||
ctx->state->observer(ctx->state->observer_context, child_rel,
|
||||
delete_entry_type_of_mode(entries[i].mode));
|
||||
char* escaped_path = output_escape(child_rel, log_get_8_bit_output());
|
||||
fprintf(stderr, " Deleted: %s\n", escaped_path ? escaped_path : "<allocation failed>");
|
||||
free(escaped_path);
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Pass 3: kept subdirectories, ascending (rsync descends into these only after
|
||||
the parent's own extras have been handled). */
|
||||
static bool delete_walk_kept_dirs(int dirfd, const char* rel_path, const DeleteDirEntry* entries,
|
||||
size_t dir_count, const DeleteWalkContext* ctx, bool deletable,
|
||||
const bool* is_extra, const bool* shielded,
|
||||
bool* local_survives) {
|
||||
bool ok = true;
|
||||
for (size_t i = dir_count; i-- > 0;) {
|
||||
if (is_extra[i] || shielded[i])
|
||||
continue;
|
||||
char* child_rel = path_cat((char*)rel_path, entries[i].name);
|
||||
if (!child_rel) {
|
||||
ok = false;
|
||||
continue;
|
||||
}
|
||||
bool child_all_removed = false;
|
||||
if (!delete_walk_child(dirfd, entries[i].name, child_rel, ctx, deletable, &child_all_removed))
|
||||
ok = false;
|
||||
/* A kept/synchronized directory is never removed. */
|
||||
*local_survives = true;
|
||||
free(child_rel);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Remove the extras directly inside the directory open on `dirfd` (DELETE mode)
|
||||
or record the paths that WOULD be removed (LIST mode), recursing into every
|
||||
child directory so kept content below a synchronized prefix is reached.
|
||||
`all_removed` reports whether every child entry was removed (so the caller may
|
||||
rmdir this directory). A child directory is never removed when it is itself a
|
||||
synchronized directory or holds kept content; with a dirs index supplied,
|
||||
direct children of a non-synchronized directory are never extras at all (they
|
||||
are left in place but still descended into). Symlinks are unlinked like any
|
||||
other non-directory extra (never followed).
|
||||
|
||||
Entries are processed in rsync's order (extraneous subdirectories in
|
||||
descending name order, then extraneous files, then kept subdirectories in
|
||||
ascending order) rather than readdir() order, so `--max-delete` leaves the
|
||||
same survivors and the `--info=del`/dry-run line order matches rsync. */
|
||||
static bool delete_walk_fd(int dirfd, const char* rel_path, const DeleteWalkContext* ctx,
|
||||
bool parent_deletable, bool* all_removed) {
|
||||
DeleteDirEntry* entries = NULL;
|
||||
size_t count = 0;
|
||||
bool collect_ok = true;
|
||||
if (!delete_dir_entries_collect(dirfd, &entries, &count, &collect_ok))
|
||||
return false;
|
||||
bool operation_ok = collect_ok;
|
||||
bool local_survives = false;
|
||||
bool* shielded = calloc(count ? count : 1, sizeof(bool));
|
||||
bool* is_extra = calloc(count ? count : 1, sizeof(bool));
|
||||
if (!shielded || !is_extra) {
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
return false;
|
||||
}
|
||||
/* A directory is deletable when it or ANY ancestor is synchronized; the
|
||||
`parent_deletable` flag carries that down the recursion so dest-only
|
||||
directories below a synchronized root are removed wholesale. */
|
||||
bool deletable = parent_deletable || is_synced_dir(ctx->dirs, rel_path);
|
||||
bool at_root = rel_path[0] == '\0';
|
||||
|
||||
/* Reproduce rsync's traversal order: extraneous subdirectories in descending
|
||||
name order, then extraneous files in descending name order, and kept
|
||||
subdirectories only afterwards (ascending). Sorting up front also fixes the
|
||||
identity of the survivors under a partial --max-delete. */
|
||||
if (count > 1)
|
||||
qsort(entries, count, sizeof(*entries), delete_dir_entry_cmp_desc);
|
||||
size_t dir_count = 0;
|
||||
while (dir_count < count && entries[dir_count].is_dir)
|
||||
dir_count++;
|
||||
|
||||
if (!delete_walk_classify(rel_path, entries, count, ctx, deletable, at_root, shielded, is_extra,
|
||||
&local_survives))
|
||||
operation_ok = false;
|
||||
if (!delete_walk_extra_dirs(dirfd, rel_path, entries, dir_count, ctx, deletable, is_extra,
|
||||
&local_survives))
|
||||
operation_ok = false;
|
||||
if (!delete_walk_extra_files(dirfd, rel_path, entries, dir_count, count, ctx, is_extra,
|
||||
&local_survives))
|
||||
operation_ok = false;
|
||||
if (!delete_walk_kept_dirs(dirfd, rel_path, entries, dir_count, ctx, deletable, is_extra,
|
||||
shielded, &local_survives))
|
||||
operation_ok = false;
|
||||
|
||||
free(shielded);
|
||||
free(is_extra);
|
||||
delete_dir_entries_free(entries, count);
|
||||
*all_removed = !local_survives;
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
/* Open the receive root following the same authorized-root confinement the
|
||||
walker uses, or dest_root directly when no authorized root is installed. */
|
||||
static int open_destination_root(const char* dest_root) {
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
if (root_fd >= 0) {
|
||||
if (utils_get_authorized_root_path())
|
||||
return utils_open_authorized_destination(dest_root);
|
||||
if (dest_root == NULL)
|
||||
return dup(root_fd);
|
||||
return -1;
|
||||
}
|
||||
return open(dest_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
}
|
||||
|
||||
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, ArrayList* out, size_t* count_out) {
|
||||
if (count_out)
|
||||
*count_out = 0;
|
||||
if (!manifest || !out)
|
||||
return false;
|
||||
PathIndex keep;
|
||||
if (!build_keep_index(manifest, &keep))
|
||||
return false;
|
||||
PathIndex dirs;
|
||||
bool have_dirs = synced_dirs != NULL;
|
||||
if (have_dirs &&
|
||||
!path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) {
|
||||
path_index_free(&keep);
|
||||
return false;
|
||||
}
|
||||
int rootfd = open_destination_root(dest_root);
|
||||
if (rootfd < 0) {
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
return false;
|
||||
}
|
||||
bool all_removed = false;
|
||||
size_t recorded = 0;
|
||||
DeleteWalkState state = {.mode = DELETE_WALK_MODE_LIST,
|
||||
.budget = NULL,
|
||||
.out = out,
|
||||
.recorded = &recorded,
|
||||
.observer = NULL,
|
||||
.observer_context = NULL};
|
||||
DeleteWalkContext ctx = {.keep = &keep,
|
||||
.dirs = have_dirs ? &dirs : NULL,
|
||||
.state = &state,
|
||||
.skips = skips,
|
||||
.skip_count = skip_count,
|
||||
.protect = protect};
|
||||
bool ok = delete_walk_fd(rootfd, "", &ctx, false, &all_removed);
|
||||
if (close(rootfd) != 0)
|
||||
ok = false;
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
if (count_out)
|
||||
*count_out = recorded;
|
||||
return ok;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect,
|
||||
size_t* deleted_out, size_t* skipped_out,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (deleted_out)
|
||||
*deleted_out = 0;
|
||||
if (skipped_out)
|
||||
*skipped_out = 0;
|
||||
if (!manifest)
|
||||
return DELETE_WALK_ERROR;
|
||||
/* Index the keep-set (and the synchronized-dir set, when supplied) once so
|
||||
membership is answered in O(path length) instead of scanning every entry
|
||||
for every destination entry. */
|
||||
PathIndex keep;
|
||||
if (!build_keep_index(manifest, &keep))
|
||||
return DELETE_WALK_ERROR;
|
||||
PathIndex dirs;
|
||||
bool have_dirs = synced_dirs != NULL;
|
||||
if (have_dirs &&
|
||||
!path_index_build(&dirs, (const char* const*)synced_dirs->items, (size_t)synced_dirs->size)) {
|
||||
path_index_free(&keep);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
int rootfd = open_destination_root(dest_root);
|
||||
if (rootfd < 0) {
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
return DELETE_WALK_ERROR;
|
||||
}
|
||||
DeleteBudget budget = {.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
bool all_removed = false;
|
||||
DeleteWalkState state = {.mode = DELETE_WALK_MODE_DELETE,
|
||||
.budget = &budget,
|
||||
.out = NULL,
|
||||
.recorded = NULL,
|
||||
.observer = observer,
|
||||
.observer_context = observer_context};
|
||||
DeleteWalkContext ctx = {.keep = &keep,
|
||||
.dirs = have_dirs ? &dirs : NULL,
|
||||
.state = &state,
|
||||
.skips = skips,
|
||||
.skip_count = skip_count,
|
||||
.protect = protect};
|
||||
bool ok = delete_walk_fd(rootfd, "", &ctx, false, &all_removed);
|
||||
if (close(rootfd) != 0)
|
||||
ok = false;
|
||||
path_index_free(&keep);
|
||||
if (have_dirs)
|
||||
path_index_free(&dirs);
|
||||
if (deleted_out)
|
||||
*deleted_out = budget.deleted;
|
||||
if (skipped_out)
|
||||
*skipped_out = budget.skipped;
|
||||
if (!ok)
|
||||
return DELETE_WALK_ERROR;
|
||||
return budget.limit_hit ? DELETE_WALK_LIMIT_REACHED : DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, size_t* deleted_out,
|
||||
size_t* skipped_out) {
|
||||
return delete_extras_limited_observed(dest_root, manifest, synced_dirs, max_delete, skips,
|
||||
skip_count, protect, deleted_out, skipped_out, NULL, NULL);
|
||||
}
|
||||
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest) {
|
||||
return delete_extras_limited(dest_root, manifest, NULL, SIZE_MAX, NULL, 0, NULL, NULL, NULL) ==
|
||||
DELETE_WALK_OK;
|
||||
}
|
||||
|
||||
/* Build the delete-walk protection prefix for one basis directory. The walker
|
||||
compares paths relative to the receive root, so a relative entry is already
|
||||
in the right form; an absolute entry that lies below the root is converted to
|
||||
its root-relative form, and one outside the root returns NULL (the walk
|
||||
cannot reach it, and it is not protected data beneath the root). Exposed so
|
||||
tests can exercise the root-of-"/" child mapping directly. */
|
||||
char* delete_basis_relative(const Config* config, const char* path) {
|
||||
if (!path)
|
||||
return NULL;
|
||||
if (path[0] != '/')
|
||||
return str_dup(path);
|
||||
const char* root = config->receive_root_directory;
|
||||
if (!root || root[0] != '/')
|
||||
return NULL;
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(path, root, root_len) != 0)
|
||||
return NULL;
|
||||
if (root_len == 1) {
|
||||
/* `root` is "/" (the only single-character absolute root): every absolute
|
||||
path is below it, and the child relative form is everything after the
|
||||
leading '/'. */
|
||||
if (path[1] == '\0')
|
||||
return NULL; /* identical to the root, not a child */
|
||||
return str_dup(path + 1);
|
||||
}
|
||||
if (path[root_len] != '/')
|
||||
return NULL; /* identical or a sibling sharing a name prefix */
|
||||
return str_dup(path + root_len + 1);
|
||||
}
|
||||
|
||||
bool delete_skips_build(const Config* config, const ArrayList* protected_paths,
|
||||
const ArrayList* size_skipped, bool basis_root_relative,
|
||||
DeleteSkipSet* out) {
|
||||
if (!out)
|
||||
return false;
|
||||
out->entries = NULL;
|
||||
out->owned_prefixes = NULL;
|
||||
out->count = 0;
|
||||
out->owned_count = 0;
|
||||
if (!config)
|
||||
return false;
|
||||
int protected_count = protected_paths ? protected_paths->size : 0;
|
||||
int size_skipped_count = size_skipped ? size_skipped->size : 0;
|
||||
int count =
|
||||
(config->delay_updates ? 1 : 0) + config->basis_count + protected_count + size_skipped_count;
|
||||
if (count == 0)
|
||||
return true;
|
||||
out->entries = calloc((size_t)count, sizeof(DeleteSkipEntry));
|
||||
if (!out->entries)
|
||||
return false;
|
||||
if (basis_root_relative && config->basis_count > 0) {
|
||||
out->owned_prefixes = calloc((size_t)config->basis_count, sizeof(char*));
|
||||
if (!out->owned_prefixes) {
|
||||
free(out->entries);
|
||||
out->entries = NULL;
|
||||
return false;
|
||||
}
|
||||
out->owned_count = config->basis_count;
|
||||
}
|
||||
int idx = 0;
|
||||
if (config->delay_updates) {
|
||||
/* Protect this transfer's actual (per-run unique) staging directory. The
|
||||
runtime name is only known to the receiver-side context; fall back to the
|
||||
reserved prefix for a context that was never created (e.g. a dry run). */
|
||||
const char* staging_name = (config->delay_context && config->delay_context->staging_name)
|
||||
? config->delay_context->staging_name
|
||||
: DELAY_UPDATES_STAGING_DIR;
|
||||
out->entries[idx].prefix = staging_name;
|
||||
out->entries[idx].top_level_only = true;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < config->basis_count; i++) {
|
||||
const char* prefix = config->basis_dirs[i].path;
|
||||
if (basis_root_relative) {
|
||||
/* An absolute basis outside the receive root is unreachable by this walk,
|
||||
so it contributes no protection prefix (and no slot). */
|
||||
char* relative = delete_basis_relative(config, config->basis_dirs[i].path);
|
||||
if (!relative)
|
||||
continue;
|
||||
out->owned_prefixes[i] = relative;
|
||||
prefix = relative;
|
||||
}
|
||||
out->entries[idx].prefix = prefix;
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < protected_count; i++) {
|
||||
out->entries[idx].prefix = (const char*)protected_paths->items[i];
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < size_skipped_count; i++) {
|
||||
out->entries[idx].prefix = (const char*)size_skipped->items[i];
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
out->count = idx;
|
||||
return true;
|
||||
}
|
||||
|
||||
void delete_skips_free(DeleteSkipSet* set) {
|
||||
if (!set)
|
||||
return;
|
||||
if (set->owned_prefixes) {
|
||||
for (int i = 0; i < set->owned_count; i++)
|
||||
free(set->owned_prefixes[i]);
|
||||
}
|
||||
free(set->owned_prefixes);
|
||||
free(set->entries);
|
||||
set->entries = NULL;
|
||||
set->owned_prefixes = NULL;
|
||||
set->count = 0;
|
||||
set->owned_count = 0;
|
||||
}
|
||||
@@ -1,198 +0,0 @@
|
||||
#ifndef DELETE_H
|
||||
#define DELETE_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "filter.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Delete engine.
|
||||
*
|
||||
* This module owns destination-relative delete traversal: the ordered directory
|
||||
* walker that reproduces rsync's extraneous-entry order, the skip-prefix
|
||||
* protection set shared by every delete pass, and the read-only enumeration
|
||||
* that mirrors the walker for -n/--dry-run. The budgeted manifest commit
|
||||
* (delete_commit.c) and the per-directory delete plans (delete_plan.c) are
|
||||
* built on the primitives exported here. */
|
||||
|
||||
/* Result of a bounded extra-file deletion run. */
|
||||
typedef enum {
|
||||
/* Every extra entry was removed (or there were none). */
|
||||
DELETE_WALK_OK = 0,
|
||||
/* The numeric cap for this run was reached before every extra was removed.
|
||||
The walker removed exactly the entries the cap allowed and skipped (without
|
||||
removing) the rest, matching rsync's partial --max-delete behavior. */
|
||||
DELETE_WALK_LIMIT_REACHED,
|
||||
/* A traversal or unlink failure aborted the deletion (partial removal is
|
||||
possible, mirroring the delete pass). */
|
||||
DELETE_WALK_ERROR
|
||||
} DeleteWalkResult;
|
||||
|
||||
/* Receiver-side delete-protection rules for one walk. `base_rules` is the
|
||||
* command-line rule set the config frame carried (owner "" rules); `dir_rules`
|
||||
* is the received per-directory rule set (rules carrying their owner directory
|
||||
* and no-inherit flag). Either may be NULL. */
|
||||
typedef struct {
|
||||
const FilterRuleList* base_rules;
|
||||
const FilterRuleList* dir_rules;
|
||||
/* When non-NULL and non-empty, a destination entry whose name ends with this
|
||||
suffix is protected from deletion. rsync never treats a --backup file as
|
||||
an extra, so a backup created at --delay-updates publication (or a
|
||||
pre-existing one) survives the delete-after pass. */
|
||||
const char* backup_suffix;
|
||||
} DeleteProtectRules;
|
||||
|
||||
/* rsync's first-match-wins receiver verdict for one candidate extra: the
|
||||
* per-directory chain is evaluated first (the containing directory's rules,
|
||||
* then each ancestor's, then the receive root's), then the base rules. Returns
|
||||
* FILTER_ACTION_PROTECT when the entry is shielded by a receiver-side exclude,
|
||||
* FILTER_ACTION_RISK when an include explicitly leaves it at risk, or
|
||||
* FILTER_ACTION_NONE when no rule matched. */
|
||||
FilterAction delete_protect_verdict(const DeleteProtectRules* protect, const char* rel_path,
|
||||
const char* leaf, bool is_dir);
|
||||
|
||||
/* The backup suffix the delete walker must shield from deletion, or NULL when
|
||||
--backup is inactive or the configured suffix is unusable (empty, or holding
|
||||
a path separator). Matches the suffix file_save uses for backups. */
|
||||
const char* delete_backup_suffix(const Config* config);
|
||||
|
||||
/* One protected entry for the delete walker. When top_level_only is true the
|
||||
prefix is skipped only as a DIRECT child of dest_root (the --delay-updates
|
||||
staging directory, which must not hide genuine extras inside a nested
|
||||
destination directory that happens to share the staging name); otherwise the
|
||||
prefix is skipped at any depth (the --compare-dest/--copy-dest/--link-dest
|
||||
basis trees, and the sender-side protected filter-excluded prefixes, which
|
||||
are never destination content). */
|
||||
typedef struct {
|
||||
const char* prefix;
|
||||
bool top_level_only;
|
||||
} DeleteSkipEntry;
|
||||
|
||||
/* A built skip-prefix set. `entries`/`count` are what path_under_skip_prefix()
|
||||
consumes. `owned_prefixes` holds any prefix strings the builder had to
|
||||
allocate (root-relative basis-dir conversions); it is NULL when every prefix
|
||||
is borrowed from the config or the caller's lists. Release with
|
||||
delete_skips_free(). */
|
||||
typedef struct {
|
||||
DeleteSkipEntry* entries;
|
||||
char** owned_prefixes;
|
||||
int count;
|
||||
int owned_count;
|
||||
} DeleteSkipSet;
|
||||
|
||||
/* True when child_rel is, or lies below, one of the protected entries (a prefix
|
||||
"a" protects "a" and "a/b/c" but not "ab"; top_level_only entries protect
|
||||
only DIRECT children of the destination root, i.e. child_rel has no '/'). */
|
||||
bool path_under_skip_prefix(const char* child_rel, bool at_root, const DeleteSkipEntry* skips,
|
||||
int skip_count);
|
||||
|
||||
/* One destination-directory entry collected up front so the delete walkers can
|
||||
reproduce rsync's traversal order instead of readdir() order. rsync processes
|
||||
a directory's extraneous subdirectories first (descending name, depth-first),
|
||||
then its extraneous files (descending name), and only afterwards descends into
|
||||
its kept subdirectories (ascending name). */
|
||||
typedef struct {
|
||||
char* name;
|
||||
bool is_dir;
|
||||
/* The entry's full st_mode from the AT_SYMLINK_NOFOLLOW stat, so a delete
|
||||
observer can classify a removed non-directory as reg/link/special. */
|
||||
mode_t mode;
|
||||
} DeleteDirEntry;
|
||||
/* Collect the entries of the directory open on `dirfd` (excluding "." and ".."),
|
||||
stat'ing each with AT_SYMLINK_NOFOLLOW. On success *out is a malloc'd array of
|
||||
*count entries whose names the caller frees with delete_dir_entries_free().
|
||||
Returns false on an allocation/readdir failure; a vanished entry (ENOENT) is
|
||||
skipped, any other stat failure is reported through *operation_ok while the
|
||||
walk continues. */
|
||||
bool delete_dir_entries_collect(int dirfd, DeleteDirEntry** out, size_t* count, bool* operation_ok);
|
||||
void delete_dir_entries_free(DeleteDirEntry* entries, size_t count);
|
||||
/* Sort comparators: `_desc` orders subdirectories before files and each group by
|
||||
descending name (rsync's extraneous-entry order); `_asc` orders plain ascending
|
||||
name (rsync's kept-subdirectory order). */
|
||||
int delete_dir_entry_cmp_desc(const void* a, const void* b);
|
||||
int delete_dir_entry_cmp_asc(const void* a, const void* b);
|
||||
|
||||
/* Remove files/dirs/symlinks under dest_root that are not listed in manifest
|
||||
without ever descending into a protected prefix (see DeleteSkipEntry). When
|
||||
`synced_dirs` is non-NULL, extras are only removed directly inside a directory
|
||||
whose destination-relative path is an exact entry in that list (the receive
|
||||
root is the "." sentinel); directories outside the synchronized set are still
|
||||
descended into so kept content below a listed directory is preserved, but
|
||||
nothing in them is removed. A NULL `synced_dirs` keeps the legacy behavior of
|
||||
treating the whole destination tree as deletable. `max_delete` caps the
|
||||
number of removed entries (SIZE_MAX = unlimited): the walker removes up to the
|
||||
cap and returns DELETE_WALK_LIMIT_REACHED when more extras remained.
|
||||
`deleted_out`/`skipped_out` optionally receive the number of entries removed
|
||||
and the number skipped because of the cap. */
|
||||
DeleteWalkResult delete_extras_limited(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, size_t* deleted_out,
|
||||
size_t* skipped_out);
|
||||
|
||||
/* Entry kind of a removed path, reported to the delete observer so the receiver
|
||||
can build rsync's `--stats` `Number of deleted files` per-type breakdown. The
|
||||
four categories are a strict partition of every removed entry. */
|
||||
typedef enum {
|
||||
DELETE_ENTRY_REG = 0,
|
||||
DELETE_ENTRY_DIR,
|
||||
DELETE_ENTRY_LINK,
|
||||
DELETE_ENTRY_SPECIAL
|
||||
} DeleteEntryType;
|
||||
|
||||
/* Optional per-deletion observer: called for each destination-relative path
|
||||
actually removed (a file, symlink, or directory) with its entry kind, in
|
||||
removal order, so the receiver can stream rsync's `--info=del`/`--info=remove`
|
||||
lines and tally the per-type `--stats` counters. */
|
||||
typedef void (*DeletePathObserver)(void* context, const char* rel_path, DeleteEntryType type);
|
||||
|
||||
/* Classify a removed entry from its st_mode for the per-type delete counters. */
|
||||
DeleteEntryType delete_entry_type_of_mode(mode_t mode);
|
||||
|
||||
/* `delete_extras_limited_observed` is delete_extras_limited with an optional
|
||||
* observer; the observer is invoked only for entries truly removed. When
|
||||
* `protect` is non-NULL its receiver-side verdict is evaluated for every
|
||||
* candidate extra: a first-match PROTECT leaves the entry (and, for a
|
||||
* directory, its whole subtree) in place, while RISK/NONE fall through to the
|
||||
* ordinary skip-prefix/keep-set logic. */
|
||||
DeleteWalkResult delete_extras_limited_observed(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, size_t max_delete,
|
||||
const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect,
|
||||
size_t* deleted_out, size_t* skipped_out,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context);
|
||||
/* Read-only companion to delete_extras_limited: walk the destination exactly as
|
||||
the delete pass would and APPEND (strdup'd) destination-relative paths that
|
||||
WOULD be removed, without touching disk. Used for -n/--dry-run --delete
|
||||
would-delete reporting. Returns true on a clean walk; the caller owns the
|
||||
strings appended to `out` and receives their count in *count_out. */
|
||||
bool delete_extras_list(const char* dest_root, const ArrayList* manifest,
|
||||
const ArrayList* synced_dirs, const DeleteSkipEntry* skips, int skip_count,
|
||||
const DeleteProtectRules* protect, ArrayList* out, size_t* count_out);
|
||||
bool delete_extras(const char* dest_root, const ArrayList* manifest);
|
||||
|
||||
/* Build the delete walk's skip-prefix set from the config's --delay-updates
|
||||
staging directory, its --compare-dest/--copy-dest/--link-dest basis dirs, and
|
||||
the caller-supplied protection lists, in that order. `protected_paths` and
|
||||
`size_skipped` are borrowed (may be NULL); every entry in them is protected at
|
||||
any depth. The staging directory is protected only as a DIRECT child of the
|
||||
receive root. `basis_root_relative` selects how a basis path becomes a
|
||||
prefix: true converts an absolute path under the receive root to its
|
||||
root-relative form (the whole-tree commit walk; an unreachable path
|
||||
contributes no slot), false keeps the configured path verbatim (the
|
||||
per-directory plan walk). On success the caller releases `*out` with
|
||||
delete_skips_free(); returns false on allocation failure. */
|
||||
bool delete_skips_build(const Config* config, const ArrayList* protected_paths,
|
||||
const ArrayList* size_skipped, bool basis_root_relative,
|
||||
DeleteSkipSet* out);
|
||||
void delete_skips_free(DeleteSkipSet* set);
|
||||
|
||||
/* Convert one basis-directory path to the receive-root-relative protection
|
||||
prefix the delete walker uses (NULL when it lies outside the root). Exposed
|
||||
for unit tests of the root-of-"/" and normalization edge cases. */
|
||||
char* delete_basis_relative(const Config* config, const char* path);
|
||||
|
||||
#endif
|
||||
@@ -1,530 +0,0 @@
|
||||
#include <errno.h>
|
||||
#include <ctype.h>
|
||||
#include <dirent.h>
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/sysmacros.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "array_list.h"
|
||||
#include "charset.h"
|
||||
#include "chmod.h"
|
||||
#include "chunk.h"
|
||||
#include "compression.h"
|
||||
#include "config.h"
|
||||
#include "data.h"
|
||||
#include "delay_updates.h"
|
||||
#include "delete_commit.h"
|
||||
#include "delete_plan.h"
|
||||
#include "delta.h"
|
||||
#include "file.h"
|
||||
#include "format.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include "metadata.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include "xattr.h"
|
||||
|
||||
#define MAX_SERVER_DELETE_COUNT 100000U
|
||||
/* Retained cost of one delete-manifest entry beyond its path bytes: the
|
||||
ArrayList pointer slot plus an approximate malloc header/rounding for the
|
||||
heap copy. Charged against MAX_MANIFEST_BYTES so a frame full of tiny paths
|
||||
cannot retain far more than the byte budget (B5). */
|
||||
#define MANIFEST_ENTRY_OVERHEAD (sizeof(char*) + 16)
|
||||
|
||||
/* Read a delete-manifest frame (the STATUS_MANIFEST leading code has already
|
||||
been consumed): a keep-set entry count followed by that many
|
||||
destination-relative paths, then a protected-prefix count followed by that
|
||||
many destination-relative prefixes, then a missing-args count followed by that
|
||||
many destination-relative delete paths, then (protocol 2.23.0) a
|
||||
synchronized-directory count followed by that many destination-relative
|
||||
directory paths (the receive root is the "." sentinel). The frame is
|
||||
self-delimiting (the counts are authoritative), so the caller decides what to
|
||||
do next and continues reading the following STATUS_* frame. Every section is
|
||||
validated identically: an entry must be non-empty, relative and traversal-free
|
||||
and the aggregate length across ALL sections is capped by MAX_MANIFEST_BYTES
|
||||
(so the missing-args deletion requests are confined like the rest of the
|
||||
manifest). Returns an owned DeleteManifest, or NULL after sending STATUS_ERROR
|
||||
when the frame is malformed (bad count, empty/absolute path, path traversal,
|
||||
or an aggregate size beyond MAX_MANIFEST_BYTES). */
|
||||
static bool receive_manifest_section(int fd, ArrayList* list, size_t* manifest_bytes,
|
||||
size_t* manifest_entries) {
|
||||
int count;
|
||||
if (!receive_int(fd, &count)) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
if (count < 0 || count > MAX_MANIFEST_ENTRIES ||
|
||||
(size_t)count > MAX_MANIFEST_ENTRIES - *manifest_entries) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < count; i++) {
|
||||
char* s = receive_wire_str(fd);
|
||||
size_t entry_size = s ? strlen(s) + MANIFEST_ENTRY_OVERHEAD : 0;
|
||||
if (!s || s[0] == '\0' || s[0] == '/' || has_path_traversal(s) ||
|
||||
entry_size > MAX_MANIFEST_BYTES - *manifest_bytes ||
|
||||
(*manifest_bytes += entry_size) > MAX_MANIFEST_BYTES || !array_list_add(list, s)) {
|
||||
free(s);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
*manifest_entries += (size_t)count;
|
||||
return true;
|
||||
}
|
||||
|
||||
DeleteManifest* receive_manifest_entries(int fd) {
|
||||
DeleteManifest* manifest = calloc(1, sizeof(DeleteManifest));
|
||||
if (!manifest) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return NULL;
|
||||
}
|
||||
manifest->keeps = array_list_create(free);
|
||||
manifest->protected = array_list_create(free);
|
||||
manifest->missing = array_list_create(free);
|
||||
manifest->dirs = array_list_create(free);
|
||||
if (!manifest->keeps || !manifest->protected || !manifest->missing || !manifest->dirs) {
|
||||
delete_manifest_free(manifest);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return NULL;
|
||||
}
|
||||
size_t manifest_bytes = 0;
|
||||
size_t manifest_entries = 0;
|
||||
if (!receive_manifest_section(fd, manifest->keeps, &manifest_bytes, &manifest_entries) ||
|
||||
!receive_manifest_section(fd, manifest->protected, &manifest_bytes, &manifest_entries) ||
|
||||
!receive_manifest_section(fd, manifest->missing, &manifest_bytes, &manifest_entries) ||
|
||||
!receive_manifest_section(fd, manifest->dirs, &manifest_bytes, &manifest_entries)) {
|
||||
delete_manifest_free(manifest);
|
||||
return NULL;
|
||||
}
|
||||
/* Per-directory filter rules (protocol 2.30.0) follow the manifest sections
|
||||
* with their own bounded self-describing format. */
|
||||
if (!delete_filter_dir_rules_receive(fd, &manifest->per_dir_rules)) {
|
||||
delete_manifest_free(manifest);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return NULL;
|
||||
}
|
||||
return manifest;
|
||||
}
|
||||
|
||||
void delete_manifest_free(DeleteManifest* manifest) {
|
||||
if (!manifest)
|
||||
return;
|
||||
array_list_delete(manifest->keeps);
|
||||
array_list_delete(manifest->protected);
|
||||
array_list_delete(manifest->missing);
|
||||
array_list_delete(manifest->dirs);
|
||||
filter_rule_list_free(manifest->per_dir_rules);
|
||||
free(manifest);
|
||||
}
|
||||
|
||||
/* Shared --max-delete budget for one receiver-side deletion commit. Both the
|
||||
--delete-missing-args exact-path removals and the ordinary extras walk draw
|
||||
from the same tally, matching rsync (whose --max-delete counts every deleted
|
||||
file or directory). `max_delete` is SIZE_MAX for an unlimited budget. */
|
||||
typedef struct {
|
||||
size_t max_delete;
|
||||
size_t deleted;
|
||||
size_t skipped;
|
||||
bool limit_hit;
|
||||
} DeleteBudgetState;
|
||||
|
||||
/* Remove every destination entry under the receive root that is not in the
|
||||
keep-set, bounded by the shared budget (a smaller client --max-delete=NUM
|
||||
replaces the server hard bound; rsync deletes up to the bound and skips the
|
||||
rest). With --delay-updates the not-yet-published staging directory is a
|
||||
direct child of the receive root and must not be treated as a set of extras;
|
||||
the manifest's protected prefixes (paths excluded on the source), the
|
||||
size-pruned prefixes (--max-size/--min-size, always protected) and the
|
||||
alternate basis directories are never destination content and are skipped at
|
||||
any depth. Returns true unless a traversal/unlink error aborted the walk;
|
||||
the budget's limit_hit/skipped fields report a cap-stopped run. */
|
||||
static bool delete_extras_budgeted_observed(const Config* config, const DeleteManifest* manifest,
|
||||
DeleteBudgetState* budget, DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (!config || !manifest || !manifest->keeps)
|
||||
return false;
|
||||
fprintf(stderr, "Deleting files not in manifest...\n");
|
||||
/* Protected entries: the --delay-updates staging name (only as a DIRECT child
|
||||
of the receive root), the alternate basis directories and the sender-side
|
||||
protected prefixes (filter-excluded and size-pruned source mirrors), all at
|
||||
any depth. See delete_skips_build(). */
|
||||
DeleteSkipSet skips;
|
||||
if (!delete_skips_build(config, manifest->protected, NULL, true, &skips))
|
||||
return false;
|
||||
/* Clamp rather than subtract: an accounting bug where deleted already exceeds
|
||||
max_delete must never underflow into an effectively unlimited budget. */
|
||||
size_t remaining;
|
||||
if (budget->max_delete == SIZE_MAX)
|
||||
remaining = SIZE_MAX;
|
||||
else if (budget->deleted >= budget->max_delete)
|
||||
remaining = 0;
|
||||
else
|
||||
remaining = budget->max_delete - budget->deleted;
|
||||
size_t deleted = 0;
|
||||
size_t skipped = 0;
|
||||
DeleteProtectRules protect = {.base_rules = config->protect_rules,
|
||||
.dir_rules = manifest->per_dir_rules,
|
||||
.backup_suffix = delete_backup_suffix(config)};
|
||||
DeleteWalkResult result = delete_extras_limited_observed(
|
||||
config->receive_root_directory, manifest->keeps, manifest->dirs, remaining, skips.entries,
|
||||
skips.count, &protect, &deleted, &skipped, observer, observer_context);
|
||||
delete_skips_free(&skips);
|
||||
budget->deleted += deleted;
|
||||
budget->skipped += skipped;
|
||||
if (result == DELETE_WALK_LIMIT_REACHED) {
|
||||
budget->limit_hit = true;
|
||||
return true;
|
||||
}
|
||||
if (result != DELETE_WALK_OK) {
|
||||
log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool delete_extras_budgeted(const Config* config, const DeleteManifest* manifest,
|
||||
DeleteBudgetState* budget) {
|
||||
return delete_extras_budgeted_observed(config, manifest, budget, NULL, NULL);
|
||||
}
|
||||
|
||||
/* Prefixes every observed path with a fixed subtree root, so a nested walk
|
||||
(a recursively removed missing-arg directory) reports receive-root-relative
|
||||
names like the rest of the delete output. */
|
||||
typedef struct {
|
||||
DeletePathObserver inner;
|
||||
void* inner_context;
|
||||
const char* prefix;
|
||||
} PrefixedDeleteObserver;
|
||||
|
||||
static void prefixed_delete_observer(void* context, const char* rel, DeleteEntryType type) {
|
||||
PrefixedDeleteObserver* prefixed = context;
|
||||
if (!prefixed->inner || !rel)
|
||||
return;
|
||||
char* joined = path_cat((char*)prefixed->prefix, rel);
|
||||
if (joined) {
|
||||
prefixed->inner(prefixed->inner_context, joined, type);
|
||||
free(joined);
|
||||
}
|
||||
}
|
||||
|
||||
/* --delete-missing-args exact-path deletions: each destination mirror in
|
||||
manifest->missing is an explicit user request, so it is removed even when the
|
||||
ordinary extras walk (with its protected prefixes) would leave it alone. The
|
||||
--delay-updates staging directory and basis snapshots are receiver artifacts
|
||||
and stay protected exactly as in the extras walker. A regular file or
|
||||
symlink is unlinked, an empty directory removed, and a NON-empty directory is
|
||||
removed recursively only when --delete or --force is in effect (rsync parity:
|
||||
the man page says a non-empty directory mirror is only deleted with --force
|
||||
or --delete); otherwise it is left with a warning and the run continues. A
|
||||
mirror that does not exist is a no-op. Each removal draws from the shared
|
||||
--max-delete budget: once it is exhausted the remaining requests are skipped
|
||||
and counted. Returns false only on a genuine error (a confinement failure on
|
||||
a validated path or an I/O error), which fails the run. */
|
||||
|
||||
/* How one missing-args request leaves the driver loop. The original walker
|
||||
`continue`s past an invalid/protected/absent/budget-skipped request (without
|
||||
breaking) but stops after a request that ran to completion while an error is
|
||||
pending; NEXT/STOP preserve that control flow exactly. */
|
||||
typedef enum { MISSING_ARG_NEXT, MISSING_ARG_STOP } MissingArgStep;
|
||||
|
||||
/* Remove a NON-empty missing-args directory recursively (--delete/--force in
|
||||
effect): walk its contents through the budgeted extras walker so every removed
|
||||
file/dir counts toward --max-delete (rsync parity), then remove the now-empty
|
||||
directory itself, which costs one more budget unit. A run that hits the cap
|
||||
leaves the remaining entries in place. The observer is wrapped so the nested
|
||||
walk reports receive-root-relative paths. Sets the *removed and *ok outputs. */
|
||||
static void delete_nonempty_missing_dir(const char* full, const char* rel,
|
||||
DeleteBudgetState* budget, DeletePathObserver observer,
|
||||
void* observer_context, bool* removed, bool* ok) {
|
||||
ArrayList* no_keeps = array_list_create(free);
|
||||
/* Never let an accounting slip (deleted > max_delete) underflow the remaining
|
||||
budget into SIZE_MAX, which would grant unlimited deletions. */
|
||||
size_t remaining =
|
||||
budget->deleted >= budget->max_delete ? 0 : budget->max_delete - budget->deleted;
|
||||
size_t contents_deleted = 0;
|
||||
size_t contents_skipped = 0;
|
||||
PrefixedDeleteObserver nested = {observer, observer_context, rel};
|
||||
DeleteWalkResult walk =
|
||||
no_keeps ? delete_extras_limited_observed(full, no_keeps, NULL, remaining, NULL, 0, NULL,
|
||||
&contents_deleted, &contents_skipped,
|
||||
observer ? prefixed_delete_observer : NULL,
|
||||
observer ? &nested : NULL)
|
||||
: DELETE_WALK_ERROR;
|
||||
if (no_keeps)
|
||||
array_list_delete(no_keeps);
|
||||
budget->deleted += contents_deleted;
|
||||
budget->skipped += contents_skipped;
|
||||
if (walk == DELETE_WALK_LIMIT_REACHED) {
|
||||
budget->limit_hit = true;
|
||||
} else if (walk != DELETE_WALK_OK) {
|
||||
*ok = false;
|
||||
} else if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
} else if (file_remove_tree_secure(full)) {
|
||||
/* The shared `if (removed)` tail charges this directory exactly once;
|
||||
counting it here too would consume two budget units. */
|
||||
*removed = true;
|
||||
} else {
|
||||
*ok = false;
|
||||
}
|
||||
}
|
||||
|
||||
/* Remove one missing-args destination mirror. `skips` holds the receiver
|
||||
artifacts (staging directory, basis snapshots) that stay protected. Returns
|
||||
MISSING_ARG_STOP when the driver loop must stop (a completed removal left a
|
||||
genuine error pending) and MISSING_ARG_NEXT otherwise; *ok accumulates the
|
||||
overall success across the whole run. */
|
||||
static MissingArgStep delete_one_missing_arg(const Config* config, const char* rel,
|
||||
const DeleteSkipSet* skips, DeleteBudgetState* budget,
|
||||
DeletePathObserver observer, void* observer_context,
|
||||
bool* ok) {
|
||||
if (!rel || *rel == '\0' || *rel == '/' || has_path_traversal(rel)) {
|
||||
/* Defensive only: receive_manifest_entries already validated every
|
||||
section identically, so a controlled peer never reaches this branch. */
|
||||
log_message(LOG_LEVEL_ERROR, "invalid missing-args delete path");
|
||||
*ok = false;
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
bool at_root = strchr(rel, '/') == NULL;
|
||||
if (path_under_skip_prefix(rel, at_root, skips->entries, skips->count)) {
|
||||
char* escaped = output_escape(rel, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"missing-args path '%s' is protected (staging directory or basis snapshot); "
|
||||
"not deleting",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
char* full = path_cat(config->receive_root_directory, rel);
|
||||
if (!full) {
|
||||
*ok = false;
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(full, &leaf, false);
|
||||
if (parent_fd < 0) {
|
||||
/* The mirror's parent directory may itself not exist on the destination
|
||||
(a deeper missing entry whose leading directories were never created).
|
||||
That is a no-op -- there is nothing to delete -- matching
|
||||
file_remove_tree_secure's absent-path handling; only a genuine I/O
|
||||
error (EACCES, a symlink loop, ...) fails the run. */
|
||||
bool absent = errno == ENOENT || errno == ENOTDIR;
|
||||
free(full);
|
||||
free(leaf);
|
||||
if (!absent)
|
||||
*ok = false;
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
/* Already absent: nothing to delete (a no-op, not a deletion). */
|
||||
if (errno != ENOENT)
|
||||
*ok = false;
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
free(full);
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
/* An entry that exists is one deletion: skip it (and count it) when the
|
||||
shared --max-delete budget is already exhausted. */
|
||||
if (budget->deleted >= budget->max_delete) {
|
||||
budget->limit_hit = true;
|
||||
budget->skipped++;
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
free(full);
|
||||
return MISSING_ARG_NEXT;
|
||||
}
|
||||
bool removed = false;
|
||||
if (S_ISDIR(st.st_mode)) {
|
||||
if (unlinkat(parent_fd, leaf, AT_REMOVEDIR) == 0) {
|
||||
removed = true;
|
||||
} else if (errno == ENOTEMPTY || errno == EEXIST) {
|
||||
close(parent_fd);
|
||||
parent_fd = -1;
|
||||
free(leaf);
|
||||
leaf = NULL;
|
||||
if (config->use_delete || config->force_delete) {
|
||||
delete_nonempty_missing_dir(full, rel, budget, observer, observer_context, &removed, ok);
|
||||
} else {
|
||||
char* escaped = output_escape(rel, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"missing-args destination '%s' is a non-empty directory; use --force or "
|
||||
"--delete to remove it",
|
||||
escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
}
|
||||
} else if (errno != ENOENT) {
|
||||
*ok = false;
|
||||
}
|
||||
} else {
|
||||
if (unlinkat(parent_fd, leaf, 0) == 0) {
|
||||
removed = true;
|
||||
} else if (errno != ENOENT) {
|
||||
*ok = false;
|
||||
}
|
||||
}
|
||||
if (removed) {
|
||||
budget->deleted++;
|
||||
if (observer)
|
||||
observer(observer_context, rel, delete_entry_type_of_mode(st.st_mode));
|
||||
char* escaped = output_escape(rel, log_get_8_bit_output());
|
||||
fprintf(stderr, " Deleted: %s\n", escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
}
|
||||
if (parent_fd >= 0)
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
free(full);
|
||||
return *ok ? MISSING_ARG_NEXT : MISSING_ARG_STOP;
|
||||
}
|
||||
|
||||
static bool delete_missing_args_budgeted_observed(const Config* config,
|
||||
const DeleteManifest* manifest,
|
||||
DeleteBudgetState* budget,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (!config || !manifest)
|
||||
return false;
|
||||
if (!manifest->missing || manifest->missing->size == 0)
|
||||
return true;
|
||||
fprintf(stderr, "Deleting destination mirrors of missing source arguments...\n");
|
||||
/* The staging directory and basis snapshots stay protected exactly as in the
|
||||
extras walker (the missing-args path overrides the ordinary protected
|
||||
prefixes, so those are not passed here). */
|
||||
DeleteSkipSet skips;
|
||||
if (!delete_skips_build(config, NULL, NULL, true, &skips))
|
||||
return false;
|
||||
bool ok = true;
|
||||
for (int i = 0; i < manifest->missing->size; i++) {
|
||||
const char* rel = (const char*)manifest->missing->items[i];
|
||||
if (delete_one_missing_arg(config, rel, &skips, budget, observer, observer_context, &ok) ==
|
||||
MISSING_ARG_STOP)
|
||||
break;
|
||||
}
|
||||
delete_skips_free(&skips);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Public wrappers used outside the commit path (and by unit tests): no
|
||||
--max-delete budget. */
|
||||
bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest,
|
||||
ArrayList* out, size_t* count_out) {
|
||||
if (count_out)
|
||||
*count_out = 0;
|
||||
if (!config || !manifest || !manifest->keeps || !out)
|
||||
return false;
|
||||
DeleteSkipSet skips;
|
||||
if (!delete_skips_build(config, manifest->protected, NULL, true, &skips))
|
||||
return false;
|
||||
DeleteProtectRules protect = {.base_rules = config->protect_rules,
|
||||
.dir_rules = manifest->per_dir_rules,
|
||||
.backup_suffix = delete_backup_suffix(config)};
|
||||
bool ok = delete_extras_list(config->receive_root_directory, manifest->keeps, manifest->dirs,
|
||||
skips.entries, skips.count, &protect, out, count_out);
|
||||
delete_skips_free(&skips);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest) {
|
||||
DeleteBudgetState budget = {
|
||||
.max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
return delete_extras_budgeted(config, manifest, &budget);
|
||||
}
|
||||
|
||||
bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest) {
|
||||
DeleteBudgetState budget = {
|
||||
.max_delete = SIZE_MAX, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
return delete_missing_args_budgeted_observed(config, manifest, &budget, NULL, NULL);
|
||||
}
|
||||
|
||||
bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest,
|
||||
size_t max_delete, size_t* deleted, size_t* skipped,
|
||||
bool* limit_hit) {
|
||||
return manifest_delete_missing_args_limited_observed(config, manifest, max_delete, deleted,
|
||||
skipped, limit_hit, NULL, NULL);
|
||||
}
|
||||
|
||||
bool manifest_delete_missing_args_limited_observed(
|
||||
const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted,
|
||||
size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context) {
|
||||
DeleteBudgetState budget = {
|
||||
.max_delete = max_delete, .deleted = 0, .skipped = 0, .limit_hit = false};
|
||||
bool ok =
|
||||
delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context);
|
||||
if (deleted)
|
||||
*deleted = budget.deleted;
|
||||
if (skipped)
|
||||
*skipped = budget.skipped;
|
||||
if (limit_hit)
|
||||
*limit_hit = budget.limit_hit;
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Commit every deletion family the manifest carries. The --delete-missing-args
|
||||
exact-path deletions run FIRST: they are explicit user requests and must not
|
||||
be blocked by the extras walker's filter-exclusion protection (a protected
|
||||
leftover inside a missing-argument directory must not make that user-requested
|
||||
removal fail). The ordinary extras walk then runs when --delete is active.
|
||||
Both draw from one --max-delete budget; the result reports a cap-stopped
|
||||
(partial) commit distinctly so the client can exit 25 like rsync. */
|
||||
DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest) {
|
||||
return manifest_delete_all_counted(config, manifest, NULL);
|
||||
}
|
||||
|
||||
DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest,
|
||||
size_t* deleted) {
|
||||
return manifest_delete_all_observed(config, manifest, deleted, NULL, NULL);
|
||||
}
|
||||
|
||||
DeleteCommitResult manifest_delete_all_observed(const Config* config,
|
||||
const DeleteManifest* manifest, size_t* deleted,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context) {
|
||||
if (deleted)
|
||||
*deleted = 0;
|
||||
if (!config || !manifest)
|
||||
return DELETE_COMMIT_ERROR;
|
||||
/* Central no-mutation guard: a dry-run never deletes. No manifest is sent on
|
||||
the dry-run path, but a hostile/buggy peer could; treat it as a no-op so
|
||||
the receiver can never remove anything. */
|
||||
if (config->dry_run)
|
||||
return DELETE_COMMIT_OK;
|
||||
/* A client --max-delete=NUM smaller than the server's hard bound replaces it
|
||||
for this run; both still bound the commit. */
|
||||
bool user_limited =
|
||||
config->max_delete >= 0 && (size_t)config->max_delete < MAX_SERVER_DELETE_COUNT;
|
||||
DeleteBudgetState budget = {.max_delete = user_limited ? (size_t)config->max_delete
|
||||
: MAX_SERVER_DELETE_COUNT,
|
||||
.deleted = 0,
|
||||
.skipped = 0,
|
||||
.limit_hit = false};
|
||||
if (config->delete_missing_args &&
|
||||
!delete_missing_args_budgeted_observed(config, manifest, &budget, observer, observer_context))
|
||||
return DELETE_COMMIT_ERROR;
|
||||
if (config->use_delete &&
|
||||
!delete_extras_budgeted_observed(config, manifest, &budget, observer, observer_context))
|
||||
return DELETE_COMMIT_ERROR;
|
||||
if (deleted)
|
||||
*deleted = budget.deleted;
|
||||
if (budget.limit_hit) {
|
||||
if (user_limited) {
|
||||
log_message(LOG_LEVEL_ERROR, "Deletions stopped due to --max-delete limit (%zu skipped)",
|
||||
budget.skipped);
|
||||
} else {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"Deletions stopped due to the server deletion limit of %u (%zu skipped)",
|
||||
(unsigned)MAX_SERVER_DELETE_COUNT, budget.skipped);
|
||||
}
|
||||
return DELETE_COMMIT_LIMIT_REACHED;
|
||||
}
|
||||
return DELETE_COMMIT_OK;
|
||||
}
|
||||
@@ -1,113 +0,0 @@
|
||||
#ifndef DELETE_COMMIT_H
|
||||
#define DELETE_COMMIT_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "delete.h"
|
||||
#include "filter.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Delete-commit module: delete-manifest receive plus the budgeted extras and
|
||||
* --delete-missing-args walkers. These declarations are re-exported by the
|
||||
* file_receive.h facade. */
|
||||
|
||||
/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative
|
||||
paths the sender transferred/keeps) plus `protected`, destination-relative
|
||||
prefixes the sender asks the receiver never to delete (paths excluded on the
|
||||
source, protected at any depth). When --delete-excluded is given the sender
|
||||
transmits an empty protected list so excluded destination mirrors are treated
|
||||
as ordinary extras. With --delete-missing-args a third section (`missing`)
|
||||
carries the destination mirrors of explicitly-listed source entries that do
|
||||
not exist: each is an exact deletion request, independent of the ordinary
|
||||
extras walk (never blocked by the protected prefixes) and processed when the
|
||||
manifest is committed. */
|
||||
typedef struct DeleteManifest {
|
||||
ArrayList* keeps;
|
||||
ArrayList* protected;
|
||||
ArrayList* missing;
|
||||
/* Destination-relative paths of the directories the sender synchronized for
|
||||
this run. The extras walker only removes entries directly inside one of
|
||||
these (the receive root is the "." sentinel); `--files-from` runs therefore
|
||||
leave untransmitted directories and the unlisted parts of listed ones
|
||||
alone, matching rsync's "delete only in synchronized directories". */
|
||||
ArrayList* dirs;
|
||||
/* Per-directory filter rules the sender compiled while scanning (protocol
|
||||
2.30.0), each carrying its owner directory and no-inherit flag. The
|
||||
receiver evaluates them (deepest before ancestors, then the command-line
|
||||
base rules) against every candidate extra so a destination-only entry that
|
||||
matches ONLY a per-directory `.rsync-filter`/dir-merge rule is shielded.
|
||||
NULL when the sender transmitted none. */
|
||||
FilterRuleList* per_dir_rules;
|
||||
} DeleteManifest;
|
||||
|
||||
void delete_manifest_free(DeleteManifest* manifest);
|
||||
/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then
|
||||
protected count + protected prefixes, then missing count + missing paths,
|
||||
then synchronized-directory count + directory paths (self-delimiting; the
|
||||
leading STATUS_MANIFEST code has been consumed). Returns an owned
|
||||
DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */
|
||||
DeleteManifest* receive_manifest_entries(int fd);
|
||||
/* Remove destination entries under config->receive_root_directory that are not
|
||||
in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and
|
||||
protected-prefix skips). `--max-delete` and `--force` are honored here. The
|
||||
caller decides WHEN to run it based on the negotiated delete timing. Returns
|
||||
false (and the transfer fails) when the deletion cannot be committed. */
|
||||
bool manifest_delete_extras(const Config* config, const DeleteManifest* manifest);
|
||||
/* --delete-missing-args exact-path deletions: remove each destination mirror
|
||||
in `manifest->missing` (never blocked by the protected prefixes, staging dir
|
||||
and basis dirs excluded). A regular file/symlink is unlinked; an empty
|
||||
directory is removed; a NON-empty directory is removed recursively only when
|
||||
--delete or --force is in effect, otherwise it is left with a warning (rsync
|
||||
parity). A missing path is a no-op. Returns false only on a genuine
|
||||
confinement or I/O error (the run then fails); tolerated per-path cases are
|
||||
reported and skipped. */
|
||||
bool manifest_delete_missing_args(const Config* config, const DeleteManifest* manifest);
|
||||
/* Budgeted form of manifest_delete_missing_args for the per-directory delete
|
||||
session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited)
|
||||
and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set
|
||||
when the budget stopped the pass with entries left over. Returns false only
|
||||
on a genuine deletion error. */
|
||||
bool manifest_delete_missing_args_limited(const Config* config, const DeleteManifest* manifest,
|
||||
size_t max_delete, size_t* deleted, size_t* skipped,
|
||||
bool* limit_hit);
|
||||
/* Observer-aware form of manifest_delete_missing_args_limited: `observer` (may
|
||||
be NULL) is invoked for every destination-relative path truly removed. */
|
||||
bool manifest_delete_missing_args_limited_observed(
|
||||
const Config* config, const DeleteManifest* manifest, size_t max_delete, size_t* deleted,
|
||||
size_t* skipped, bool* limit_hit, DeletePathObserver observer, void* observer_context);
|
||||
/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's
|
||||
partial --max-delete result: the budget allowed some deletions and the rest
|
||||
were skipped (the run still stores all file data but the client exits 25). */
|
||||
typedef enum {
|
||||
DELETE_COMMIT_OK = 0,
|
||||
DELETE_COMMIT_LIMIT_REACHED,
|
||||
DELETE_COMMIT_ERROR
|
||||
} DeleteCommitResult;
|
||||
|
||||
/* Run every deletion family the manifest carries: the --delete-missing-args
|
||||
exact-path deletions first (user requests are not blocked by exclusion
|
||||
protection), then the ordinary extras walk when --delete is active. Both
|
||||
share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to
|
||||
do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget
|
||||
stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */
|
||||
DeleteCommitResult manifest_delete_all(const Config* config, const DeleteManifest* manifest);
|
||||
/* Like manifest_delete_all, but reports how many destination entries the commit
|
||||
removed (for the end-of-transfer wire stats). `deleted` may be NULL. */
|
||||
DeleteCommitResult manifest_delete_all_counted(const Config* config, const DeleteManifest* manifest,
|
||||
size_t* deleted);
|
||||
/* Observer-aware form of manifest_delete_all_counted: `observer` (may be NULL)
|
||||
is invoked for every destination-relative path truly removed. */
|
||||
DeleteCommitResult manifest_delete_all_observed(const Config* config,
|
||||
const DeleteManifest* manifest, size_t* deleted,
|
||||
DeletePathObserver observer,
|
||||
void* observer_context);
|
||||
|
||||
/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as
|
||||
the delete pass would and append (strdup'd) destination-relative paths that
|
||||
WOULD be removed to `out`, without touching disk. Uses the same staging-dir,
|
||||
basis-dir and protected-prefix skips as the real commit. Returns true on a
|
||||
clean walk; `*count_out` receives the number of paths appended. */
|
||||
bool manifest_would_delete_list(const Config* config, const DeleteManifest* manifest,
|
||||
ArrayList* out, size_t* count_out);
|
||||
|
||||
#endif
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,120 +0,0 @@
|
||||
#ifndef DELETE_PLAN_H
|
||||
#define DELETE_PLAN_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "delete.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Per-directory delete plans (protocol 2.24.0).
|
||||
*
|
||||
* rsync's --delete-during removes a directory's extras while the generator
|
||||
* processes that directory, and --delete-delay records the deletion list during
|
||||
* the scan but applies it only after a fully-successful transfer. FastSync has
|
||||
* no per-directory generator pass; instead the sender streams one plan per
|
||||
* source directory, in directory order, and the receiver applies it when it
|
||||
* arrives (during) or snapshots its extras and commits them at the end (delay).
|
||||
*
|
||||
* The sender side builds a plan set from the path-only pre-scan (it needs every
|
||||
* directory's complete direct-child list before the first data byte of that
|
||||
* directory). The receiver side is a session that carries the global protected
|
||||
* prefixes (filter-excluded and size-skipped source mirrors), the
|
||||
* --delete-missing-args exact deletions, the shared --max-delete budget and,
|
||||
* for --delete-delay, the snapshotted extras. */
|
||||
|
||||
/* Per-directory filter-rule block (protocol 2.30.0). The sender compiles the
|
||||
* source's per-directory merge rules as it scans and streams them so the
|
||||
* receiver can re-derive the receiver-side protect/risk verdicts for
|
||||
* destination-only entries. The wire format is a group count, then for each
|
||||
* directory group its relative owner path followed by that directory's rule
|
||||
* records (action, sides, anchored, dir-only, negate, no-inherit, pattern).
|
||||
* All bounds (MAX_FILTER_RULES, MAX_FILTER_BYTES, MAX_PROTECT_PATTERN_LEN) are
|
||||
* enforced on both sides; a malformed receive frame signals STATUS_ERROR and
|
||||
* returns false. Receive yields a flat FilterRuleList whose rules carry their
|
||||
* owner, or NULL when no rules were sent. */
|
||||
bool delete_filter_dir_rules_send(int fd, const FilterRuleList* rules);
|
||||
bool delete_filter_dir_rules_receive(int fd, FilterRuleList** out);
|
||||
|
||||
/* ---- Sender: plan builder ---- */
|
||||
|
||||
typedef struct DeletePlanSender DeletePlanSender;
|
||||
|
||||
DeletePlanSender* delete_plan_sender_create(void);
|
||||
void delete_plan_sender_destroy(DeletePlanSender* sender);
|
||||
/* Record one transmitted entry. `path` is the destination-relative wire path;
|
||||
* is_dir marks an explicit directory entry (--dirs, a -x mount point). */
|
||||
bool delete_plan_sender_add(DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Drop plans for directories outside `synced_dirs` (the --files-from
|
||||
* synchronization scope; pass NULL when a full recursive transfer synchronized
|
||||
* every directory). The receive root is the "." sentinel.
|
||||
*
|
||||
* `walk_root` scopes a general -R transfer: when non-NULL it is the
|
||||
* reconstructed destination prefix the run actually transferred, and only the
|
||||
* plan for that prefix (and directories below it) is ever transmitted, so the
|
||||
* prefix's parent-directory siblings are never walked. Pass NULL for a plain
|
||||
* recursive transfer and for --files-from. */
|
||||
void delete_plan_sender_finalize(DeletePlanSender* sender, const ArrayList* synced_dirs,
|
||||
const char* walk_root);
|
||||
/* True when no transmitted FILE entry was recorded (an ambiguous empty scan).
|
||||
Directory keep entries do not count, so an I/O error that hid every file
|
||||
still refuses to delete. */
|
||||
bool delete_plan_sender_empty(const DeletePlanSender* sender);
|
||||
/* Attach the global config sections advertised on the first plan frame. The
|
||||
* block is always transmitted by delete_plan_send_root(), on a config-only
|
||||
* carrier frame when the scope allows no directory plan. */
|
||||
void delete_plan_sender_set_config(DeletePlanSender* sender, const ArrayList* protected_prefixes,
|
||||
const ArrayList* size_skipped, const ArrayList* missing_args,
|
||||
const FilterRuleList* per_dir_rules);
|
||||
/* Send the root plan (even before any data, so root extras are handled like
|
||||
* rsync's first generator directory), after transmitting the per-run config
|
||||
* block on its own carrier frame. Returns -1 on I/O error. */
|
||||
int delete_plan_send_root(int fd, DeletePlanSender* sender);
|
||||
/* Send the plans for every ancestor of `path` (root-first) and, when is_dir,
|
||||
* for `path` itself; already-sent plans are skipped. */
|
||||
int delete_plan_send_for_path(int fd, DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Send the plan for every directory in `dirs` that has not been transmitted
|
||||
* yet. */
|
||||
int delete_plan_send_remaining(int fd, DeletePlanSender* sender, const ArrayList* dirs);
|
||||
/* Transmit the COMPLETE per-directory plan set in one pass, before any data
|
||||
* frame: the root plan (with the one-shot per-run config block on its carrier
|
||||
* frame) followed by every directory in `dirs`. Because the whole plan set is
|
||||
* known from the path-only pre-scan, sending it all up front means a
|
||||
* mid-transfer abort has already applied every planned removal, matching
|
||||
* rsync's generator (which runs ahead of its throttled sender). A completed
|
||||
* run is unaffected. `dirs` is the set of directories whose direct children
|
||||
* were enumerated (the scanner's plan_dirs sink), so a merely listed but
|
||||
* untraversed directory never gets a plan and its mirror is left intact.
|
||||
* Returns -1 on I/O error. */
|
||||
int delete_plan_send_all(int fd, DeletePlanSender* sender, const ArrayList* dirs);
|
||||
|
||||
/* ---- Receiver: delete session ---- */
|
||||
|
||||
typedef struct DeletePlanSession DeletePlanSession;
|
||||
|
||||
DeletePlanSession* delete_plan_session_create(const Config* config);
|
||||
void delete_plan_session_destroy(DeletePlanSession* session);
|
||||
/* Read one STATUS_DELETE_PLAN frame (the leading status already consumed) and
|
||||
* act on it. Returns 0 on success (including a dry-run/disabled no-op) and -1
|
||||
* after signalling STATUS_ERROR on a malformed frame or a deletion failure. */
|
||||
int delete_plan_session_receive(DeletePlanSession* session, const Config* config, int fd);
|
||||
/* Apply the deferred snapshot (--delete-delay) and the missing-args deletions.
|
||||
* Safe to call once; returns the commit outcome. */
|
||||
DeleteCommitResult delete_plan_session_commit(DeletePlanSession* session, const Config* config);
|
||||
/* True once the shared --max-delete budget stopped part of a deletion. */
|
||||
bool delete_plan_session_limit_reached(const DeletePlanSession* session);
|
||||
/* Number of destination entries the session actually removed, for the
|
||||
end-of-transfer stats. For --delete-delay this excludes a snapshotted entry
|
||||
that survived (e.g. a refilled directory that failed ENOTEMPTY), even though
|
||||
that entry already consumed --max-delete budget at snapshot time. */
|
||||
size_t delete_plan_session_deleted(const DeletePlanSession* session);
|
||||
/* Install an observer invoked for every destination-relative path the session
|
||||
truly removes (including the deferred --delete-delay commit), so the receiver
|
||||
can report rsync's `deleting PATH` lines through the terminal STATUS_STATS
|
||||
record. Pass NULL/0 to clear. */
|
||||
void delete_plan_session_set_delete_observer(DeletePlanSession* session,
|
||||
DeletePathObserver observer, void* context);
|
||||
|
||||
#endif
|
||||
+24
-356
@@ -1,12 +1,9 @@
|
||||
#include "delta.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include <errno.h>
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define XXH_STATIC_LINKING_ONLY
|
||||
#define XXH_IMPLEMENTATION
|
||||
@@ -31,21 +28,12 @@ uint32_t delta_xxhash32(const void* data, uint32_t len) {
|
||||
return XXH32(data, len, 0);
|
||||
}
|
||||
|
||||
uint32_t delta_xxhash32_seeded(const void* data, uint32_t len, uint32_t seed) {
|
||||
return XXH32(data, len, seed);
|
||||
}
|
||||
|
||||
uint64_t delta_xxhash64(const void* data, size_t len) {
|
||||
return XXH64(data, len, 0);
|
||||
}
|
||||
|
||||
DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size,
|
||||
uint32_t block_size) {
|
||||
return delta_signature_create_seeded(old_file_data, old_file_size, block_size, 0);
|
||||
}
|
||||
|
||||
DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_t old_file_size,
|
||||
uint32_t block_size, uint32_t seed) {
|
||||
if (old_file_data == NULL || old_file_size == 0 || block_size == 0)
|
||||
return NULL;
|
||||
|
||||
@@ -55,7 +43,7 @@ DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_
|
||||
|
||||
uint32_t block_count = (uint32_t)((old_file_size + block_size - 1) / block_size);
|
||||
|
||||
DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature));
|
||||
DeltaSignature* sig = malloc(sizeof(DeltaSignature));
|
||||
if (!sig)
|
||||
return NULL;
|
||||
|
||||
@@ -66,7 +54,7 @@ DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
sig->blocks = protocol_alloc((size_t)block_count * sizeof(DeltaBlockSig));
|
||||
sig->blocks = malloc((size_t)block_count * sizeof(DeltaBlockSig));
|
||||
if (!sig->blocks) {
|
||||
free(sig);
|
||||
return NULL;
|
||||
@@ -78,70 +66,12 @@ DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_
|
||||
uint32_t len =
|
||||
(uint32_t)((old_file_size - offset < block_size) ? (old_file_size - offset) : block_size);
|
||||
sig->blocks[i].adler32 = delta_adler32(data + offset, len);
|
||||
sig->blocks[i].xxhash = delta_xxhash32_seeded(data + offset, len, seed);
|
||||
sig->blocks[i].xxhash = delta_xxhash32(data + offset, len);
|
||||
}
|
||||
|
||||
return sig;
|
||||
}
|
||||
|
||||
/* Bounded read of exactly `len` bytes at `off`; retries on EINTR. */
|
||||
static bool pread_all(int fd, void* buf, size_t len, uint64_t off) {
|
||||
uint8_t* p = buf;
|
||||
size_t done = 0;
|
||||
while (done < len) {
|
||||
ssize_t n = pread(fd, p + done, len - done, (off_t)(off + done));
|
||||
if (n < 0 && errno == EINTR)
|
||||
continue;
|
||||
if (n <= 0)
|
||||
return false;
|
||||
done += (size_t)n;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
DeltaSignature* delta_signature_create_fd_seeded(int fd, uint64_t old_file_size,
|
||||
uint32_t block_size, uint32_t seed) {
|
||||
if (fd < 0 || old_file_size == 0 || block_size == 0 || block_size > DELTA_BLOCK_SIZE_MAX ||
|
||||
old_file_size > UINT32_MAX * (uint64_t)block_size)
|
||||
return NULL;
|
||||
uint32_t block_count = (uint32_t)((old_file_size + block_size - 1) / block_size);
|
||||
/* Bound the signature's own memory (block_count * sizeof(DeltaBlockSig)). */
|
||||
if (block_count == 0 || block_count > MAX_DELTA_BLOCKS)
|
||||
return NULL;
|
||||
DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature));
|
||||
if (!sig)
|
||||
return NULL;
|
||||
sig->file_size = old_file_size;
|
||||
sig->block_size = block_size;
|
||||
sig->block_count = block_count;
|
||||
sig->blocks = protocol_alloc((size_t)block_count * sizeof(DeltaBlockSig));
|
||||
if (!sig->blocks) {
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
uint8_t* block = malloc(block_size);
|
||||
if (!block) {
|
||||
free(sig->blocks);
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
for (uint32_t i = 0; i < block_count; i++) {
|
||||
uint64_t offset = (uint64_t)i * block_size;
|
||||
uint32_t len =
|
||||
(uint32_t)((old_file_size - offset < block_size) ? (old_file_size - offset) : block_size);
|
||||
if (!pread_all(fd, block, len, offset)) {
|
||||
free(block);
|
||||
free(sig->blocks);
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
sig->blocks[i].adler32 = delta_adler32(block, len);
|
||||
sig->blocks[i].xxhash = delta_xxhash32_seeded(block, len, seed);
|
||||
}
|
||||
free(block);
|
||||
return sig;
|
||||
}
|
||||
|
||||
Data* delta_signature_serialize(const DeltaSignature* sig) {
|
||||
if (!sig)
|
||||
return NULL;
|
||||
@@ -152,7 +82,7 @@ Data* delta_signature_serialize(const DeltaSignature* sig) {
|
||||
total > SIZE_MAX)
|
||||
return NULL;
|
||||
|
||||
uint8_t* buf = protocol_alloc((size_t)total);
|
||||
uint8_t* buf = malloc((size_t)total);
|
||||
if (!buf)
|
||||
return NULL;
|
||||
|
||||
@@ -181,7 +111,7 @@ DeltaSignature* delta_signature_deserialize(const Data* data) {
|
||||
const uint8_t* buf = (const uint8_t*)data->data;
|
||||
size_t pos = 0;
|
||||
|
||||
DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature));
|
||||
DeltaSignature* sig = malloc(sizeof(DeltaSignature));
|
||||
if (!sig)
|
||||
return NULL;
|
||||
|
||||
@@ -219,7 +149,7 @@ DeltaSignature* delta_signature_deserialize(const Data* data) {
|
||||
free(sig);
|
||||
return NULL;
|
||||
}
|
||||
sig->blocks = protocol_alloc((size_t)blocks_size);
|
||||
sig->blocks = malloc((size_t)blocks_size);
|
||||
if (!sig->blocks) {
|
||||
free(sig);
|
||||
return NULL;
|
||||
@@ -248,7 +178,7 @@ static bool ensure_capacity(DeltaInstruction** instrs, uint32_t* capacity, uint3
|
||||
if (*capacity > MAX_DELTA_INSTRUCTIONS / 2)
|
||||
return false;
|
||||
uint32_t new_cap = *capacity * 2;
|
||||
DeltaInstruction* tmp = protocol_realloc(*instrs, (size_t)new_cap * sizeof(DeltaInstruction));
|
||||
DeltaInstruction* tmp = realloc(*instrs, (size_t)new_cap * sizeof(DeltaInstruction));
|
||||
if (!tmp)
|
||||
return false;
|
||||
*instrs = tmp;
|
||||
@@ -265,7 +195,7 @@ static bool flush_literal(DeltaInstruction** instrs, uint32_t* capacity, uint32_
|
||||
uint32_t lit_len = (uint32_t)(end - start);
|
||||
if (!ensure_capacity(instrs, capacity, *count))
|
||||
return false;
|
||||
uint8_t* lit_data = protocol_alloc(lit_len);
|
||||
uint8_t* lit_data = malloc(lit_len);
|
||||
if (!lit_data)
|
||||
return false;
|
||||
memcpy(lit_data, data + start, lit_len);
|
||||
@@ -285,126 +215,8 @@ static void free_instructions(DeltaInstruction* instrs, uint32_t count) {
|
||||
free(instrs);
|
||||
}
|
||||
|
||||
/* Sentinel meaning "no signature block" in the lookup index chains. Block
|
||||
* counts are bounded well below UINT32_MAX, so it doubles as a null link. */
|
||||
#define DELTA_NO_BLOCK UINT32_MAX
|
||||
|
||||
/* Avalanche mix for the rolling checksum so blocks do not cluster in the
|
||||
* bucket table when the weak checksum has little entropy (e.g. all-zero or
|
||||
* patterned files). */
|
||||
static uint32_t delta_adler_mix(uint32_t h) {
|
||||
h ^= h >> 16;
|
||||
h *= 0x7feb352dU;
|
||||
h ^= h >> 15;
|
||||
h *= 0x846ca68bU;
|
||||
h ^= h >> 16;
|
||||
return h;
|
||||
}
|
||||
|
||||
/* Smallest power of two >= v. v must be non-zero. */
|
||||
static uint32_t delta_next_pow2(uint32_t v) {
|
||||
v--;
|
||||
v |= v >> 1;
|
||||
v |= v >> 2;
|
||||
v |= v >> 4;
|
||||
v |= v >> 8;
|
||||
v |= v >> 16;
|
||||
return v + 1;
|
||||
}
|
||||
|
||||
/* Build a hash index over sig->blocks keyed by the (mixed) rolling checksum.
|
||||
* All blocks sharing an Adler-32 value land in the same bucket; collisions
|
||||
* are chained through a single contiguous allocation:
|
||||
*
|
||||
* [0, bucket_count) heads (first block per bucket)
|
||||
* [bucket_count, 2*bucket_count) tails (last block per bucket)
|
||||
* [2*bucket_count, ...) per-block chain links
|
||||
*
|
||||
* Blocks are inserted in ascending index order so every bucket chain is
|
||||
* ordered exactly like the historical linear scan. Returns the base pointer
|
||||
* (also the heads array) or NULL when no index could be allocated; callers
|
||||
* then fall back to the linear scan. */
|
||||
static uint32_t* delta_build_index(const DeltaSignature* sig, uint32_t bucket_count) {
|
||||
if (sig->block_count == 0 || bucket_count == 0)
|
||||
return NULL;
|
||||
|
||||
size_t entries = (size_t)2 * bucket_count + sig->block_count;
|
||||
if (entries > SIZE_MAX / sizeof(uint32_t))
|
||||
return NULL;
|
||||
|
||||
uint32_t* index = protocol_alloc(entries * sizeof(uint32_t));
|
||||
if (!index)
|
||||
return NULL;
|
||||
|
||||
uint32_t* heads = index;
|
||||
uint32_t* tails = index + bucket_count;
|
||||
uint32_t* next = index + 2 * bucket_count;
|
||||
uint32_t mask = bucket_count - 1;
|
||||
|
||||
memset(heads, 0xFF, (size_t)bucket_count * sizeof(uint32_t));
|
||||
memset(tails, 0xFF, (size_t)bucket_count * sizeof(uint32_t));
|
||||
|
||||
for (uint32_t j = 0; j < sig->block_count; j++) {
|
||||
uint32_t b = delta_adler_mix(sig->blocks[j].adler32) & mask;
|
||||
if (heads[b] == DELTA_NO_BLOCK)
|
||||
heads[b] = j;
|
||||
else
|
||||
next[tails[b]] = j;
|
||||
tails[b] = j;
|
||||
next[j] = DELTA_NO_BLOCK;
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
/* Locate the signature block matching the byte window at new_data[i].
|
||||
*
|
||||
* Mirrors the original per-window behaviour exactly: only a full block_size
|
||||
* window can match, candidates are accepted only when the weak (Adler-32) and
|
||||
* strong (xxHash32) checksums both agree, and the lowest block index wins so
|
||||
* the emitted op stream is byte-identical to the linear scan. When heads is
|
||||
* non-NULL the candidate set is reached through the bucket index (expected
|
||||
* O(1) per window); otherwise an exact linear scan is used. */
|
||||
static uint32_t delta_find_match(const uint8_t* window, uint32_t window_len, uint32_t adler,
|
||||
bool full_window, const DeltaSignature* sig, const uint32_t* heads,
|
||||
const uint32_t* next, uint32_t mask, uint32_t seed) {
|
||||
if (!full_window || sig->block_count == 0)
|
||||
return DELTA_NO_BLOCK;
|
||||
|
||||
if (heads) {
|
||||
uint32_t b = delta_adler_mix(adler) & mask;
|
||||
uint32_t window_xxh = 0;
|
||||
bool have_xxh = false;
|
||||
for (uint32_t j = heads[b]; j != DELTA_NO_BLOCK; j = next[j]) {
|
||||
if (sig->blocks[j].adler32 != adler)
|
||||
continue;
|
||||
if (!have_xxh) {
|
||||
window_xxh = delta_xxhash32_seeded(window, window_len, seed);
|
||||
have_xxh = true;
|
||||
}
|
||||
if (window_xxh == sig->blocks[j].xxhash)
|
||||
return j;
|
||||
}
|
||||
return DELTA_NO_BLOCK;
|
||||
}
|
||||
|
||||
/* Fallback used when the index could not be allocated. */
|
||||
for (uint32_t j = 0; j < sig->block_count; j++) {
|
||||
if (sig->blocks[j].adler32 == adler) {
|
||||
uint32_t window_xxh = delta_xxhash32_seeded(window, window_len, seed);
|
||||
if (window_xxh == sig->blocks[j].xxhash)
|
||||
return j;
|
||||
}
|
||||
}
|
||||
return DELTA_NO_BLOCK;
|
||||
}
|
||||
|
||||
Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig,
|
||||
uint32_t block_size) {
|
||||
return delta_compute_seeded(new_file_data, new_file_size, sig, block_size, 0);
|
||||
}
|
||||
|
||||
Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
const DeltaSignature* sig, uint32_t block_size, uint32_t seed) {
|
||||
if (!new_file_data || !sig || !sig->blocks || new_file_size == 0 || block_size == 0 ||
|
||||
block_size > DELTA_BLOCK_SIZE_MAX || sig->block_size != block_size)
|
||||
return NULL;
|
||||
@@ -413,29 +225,10 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
|
||||
uint32_t capacity = 64;
|
||||
uint32_t count = 0;
|
||||
DeltaInstruction* instrs = protocol_alloc((size_t)capacity * sizeof(DeltaInstruction));
|
||||
DeltaInstruction* instrs = malloc((size_t)capacity * sizeof(DeltaInstruction));
|
||||
if (!instrs)
|
||||
return NULL;
|
||||
|
||||
/* Build a one-time bucket index over the signature blocks keyed by the weak
|
||||
* checksum. This turns the per-byte-window candidate lookup from an
|
||||
* O(block_count) linear scan into an expected O(1) probe, which dominates
|
||||
* the cost for large mostly-matching files (the diff steps one byte at a
|
||||
* time through changed regions). On allocation failure the probe falls back
|
||||
* to the original linear scan, so behaviour is unchanged under memory
|
||||
* pressure. */
|
||||
uint32_t* index = NULL;
|
||||
const uint32_t* chain_next = NULL;
|
||||
uint32_t mask = 0;
|
||||
if (sig->block_count > 0) {
|
||||
uint32_t bucket_count = delta_next_pow2(sig->block_count);
|
||||
index = delta_build_index(sig, bucket_count);
|
||||
if (index) {
|
||||
chain_next = index + 2 * bucket_count;
|
||||
mask = bucket_count - 1;
|
||||
}
|
||||
}
|
||||
|
||||
uint64_t literal_start = 0;
|
||||
bool has_literal = false;
|
||||
|
||||
@@ -470,13 +263,13 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
}
|
||||
|
||||
bool matched = false;
|
||||
uint32_t match_block = delta_find_match(new_data + i, window_len, adler, full_window, sig,
|
||||
index, chain_next, mask, seed);
|
||||
if (match_block != DELTA_NO_BLOCK) {
|
||||
for (uint32_t j = 0; j < sig->block_count; j++) {
|
||||
if (adler == sig->blocks[j].adler32 && full_window) {
|
||||
uint32_t xxh = delta_xxhash32(new_data + i, window_len);
|
||||
if (xxh == sig->blocks[j].xxhash) {
|
||||
if (has_literal) {
|
||||
if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, i)) {
|
||||
free_instructions(instrs, count);
|
||||
free(index);
|
||||
return NULL;
|
||||
}
|
||||
has_literal = false;
|
||||
@@ -484,11 +277,10 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
|
||||
if (!ensure_capacity(&instrs, &capacity, count)) {
|
||||
free_instructions(instrs, count);
|
||||
free(index);
|
||||
return NULL;
|
||||
}
|
||||
instrs[count].type = DELTA_INSTR_BLOCK_MATCH;
|
||||
instrs[count].match.block_index = match_block;
|
||||
instrs[count].match.block_index = j;
|
||||
instrs[count].match.block_offset = 0;
|
||||
instrs[count].match.length = window_len;
|
||||
count++;
|
||||
@@ -496,6 +288,9 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
i += window_len;
|
||||
rolling_valid = false;
|
||||
matched = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!matched) {
|
||||
@@ -507,8 +302,6 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
}
|
||||
}
|
||||
|
||||
free(index);
|
||||
|
||||
if (has_literal) {
|
||||
if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, new_file_size)) {
|
||||
free_instructions(instrs, count);
|
||||
@@ -516,7 +309,7 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
}
|
||||
}
|
||||
|
||||
Delta* delta = protocol_alloc(sizeof(Delta));
|
||||
Delta* delta = malloc(sizeof(Delta));
|
||||
if (!delta) {
|
||||
free_instructions(instrs, count);
|
||||
return NULL;
|
||||
@@ -562,7 +355,7 @@ Data* delta_serialize(const Delta* delta) {
|
||||
if (delta->delta_size > UINT64_MAX - header_size || header_size + delta->delta_size > SIZE_MAX)
|
||||
return NULL;
|
||||
uint64_t total = header_size + delta->delta_size;
|
||||
uint8_t* buf = protocol_alloc((size_t)total);
|
||||
uint8_t* buf = malloc((size_t)total);
|
||||
if (!buf)
|
||||
return NULL;
|
||||
|
||||
@@ -602,7 +395,7 @@ Delta* delta_deserialize(const Data* data) {
|
||||
const uint8_t* buf = (const uint8_t*)data->data;
|
||||
size_t pos = 0;
|
||||
|
||||
Delta* delta = protocol_alloc(sizeof(Delta));
|
||||
Delta* delta = malloc(sizeof(Delta));
|
||||
if (!delta)
|
||||
return NULL;
|
||||
|
||||
@@ -619,10 +412,9 @@ Delta* delta_deserialize(const Data* data) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
delta->instructions =
|
||||
delta->instruction_count == 0
|
||||
delta->instructions = delta->instruction_count == 0
|
||||
? NULL
|
||||
: protocol_alloc((size_t)delta->instruction_count * sizeof(DeltaInstruction));
|
||||
: malloc((size_t)delta->instruction_count * sizeof(DeltaInstruction));
|
||||
if (delta->instruction_count > 0 && !delta->instructions) {
|
||||
free(delta);
|
||||
return NULL;
|
||||
@@ -673,7 +465,7 @@ Delta* delta_deserialize(const Data* data) {
|
||||
free(delta);
|
||||
return NULL;
|
||||
}
|
||||
delta->instructions[i].literal.data = protocol_alloc(lit_len ? lit_len : 1);
|
||||
delta->instructions[i].literal.data = malloc(lit_len ? lit_len : 1);
|
||||
if (!delta->instructions[i].literal.data) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate %u bytes for literal data", lit_len);
|
||||
free_instructions(delta->instructions, i);
|
||||
@@ -700,7 +492,7 @@ void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta,
|
||||
delta->new_file_size > DELTA_MAX_FILE_SIZE || delta->new_file_size > SIZE_MAX)
|
||||
return NULL;
|
||||
|
||||
void* output = protocol_alloc(delta->new_file_size ? (size_t)delta->new_file_size : 1);
|
||||
void* output = malloc(delta->new_file_size ? (size_t)delta->new_file_size : 1);
|
||||
if (!output)
|
||||
return NULL;
|
||||
|
||||
@@ -747,130 +539,6 @@ void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta,
|
||||
return output;
|
||||
}
|
||||
|
||||
void* delta_apply_fd(int src_fd, uint64_t old_size, const Delta* delta, uint32_t block_size) {
|
||||
if (!delta || block_size == 0 || block_size > DELTA_BLOCK_SIZE_MAX ||
|
||||
(delta->instruction_count > 0 && !delta->instructions) || delta->new_file_size == 0 ||
|
||||
delta->new_file_size > SIZE_MAX)
|
||||
return NULL;
|
||||
|
||||
void* output = protocol_alloc((size_t)delta->new_file_size);
|
||||
if (!output)
|
||||
return NULL;
|
||||
|
||||
uint8_t* out = (uint8_t*)output;
|
||||
uint64_t out_pos = 0;
|
||||
|
||||
for (uint32_t i = 0; i < delta->instruction_count; i++) {
|
||||
if (delta->instructions[i].type == DELTA_INSTR_BLOCK_MATCH) {
|
||||
uint64_t src_offset = (uint64_t)delta->instructions[i].match.block_index * block_size;
|
||||
if (src_offset > UINT64_MAX - delta->instructions[i].match.block_offset) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
src_offset += delta->instructions[i].match.block_offset;
|
||||
uint32_t len = delta->instructions[i].match.length;
|
||||
|
||||
if (src_offset > old_size || (uint64_t)len > old_size - src_offset ||
|
||||
out_pos > delta->new_file_size || (uint64_t)len > delta->new_file_size - out_pos) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
if (!pread_all(src_fd, out + out_pos, len, src_offset)) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
out_pos += len;
|
||||
} else if (delta->instructions[i].type == DELTA_INSTR_LITERAL) {
|
||||
uint32_t len = delta->instructions[i].literal.length;
|
||||
if (out_pos > delta->new_file_size || (uint64_t)len > delta->new_file_size - out_pos) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(out + out_pos, delta->instructions[i].literal.data, len);
|
||||
out_pos += len;
|
||||
} else {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
if (out_pos != delta->new_file_size) {
|
||||
free(output);
|
||||
return NULL;
|
||||
}
|
||||
return output;
|
||||
}
|
||||
|
||||
bool delta_apply_to_fd(const void* old_data, int src_fd, uint64_t old_size, const Delta* delta,
|
||||
uint32_t block_size, int dst_fd) {
|
||||
if (!delta || (old_data == NULL && src_fd < 0) || block_size == 0 ||
|
||||
block_size > DELTA_BLOCK_SIZE_MAX || (delta->instruction_count > 0 && !delta->instructions))
|
||||
return false;
|
||||
const int chunk = 1 << 20;
|
||||
uint8_t* buf = malloc((size_t)chunk);
|
||||
if (!buf)
|
||||
return false;
|
||||
uint64_t out_pos = 0;
|
||||
bool ok = true;
|
||||
for (uint32_t i = 0; i < delta->instruction_count && ok; i++) {
|
||||
uint64_t src_offset = 0;
|
||||
uint64_t len = 0;
|
||||
const uint8_t* lit = NULL;
|
||||
if (delta->instructions[i].type == DELTA_INSTR_BLOCK_MATCH) {
|
||||
src_offset = (uint64_t)delta->instructions[i].match.block_index * block_size;
|
||||
if (src_offset > UINT64_MAX - delta->instructions[i].match.block_offset ||
|
||||
src_offset + delta->instructions[i].match.block_offset > old_size) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
src_offset += delta->instructions[i].match.block_offset;
|
||||
len = delta->instructions[i].match.length;
|
||||
if (len > old_size - src_offset) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
} else if (delta->instructions[i].type == DELTA_INSTR_LITERAL) {
|
||||
lit = delta->instructions[i].literal.data;
|
||||
len = delta->instructions[i].literal.length;
|
||||
} else {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (out_pos > delta->new_file_size || len > delta->new_file_size - out_pos) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
uint64_t done = 0;
|
||||
while (ok && done < len) {
|
||||
size_t want = (len - done) < (uint64_t)chunk ? (size_t)(len - done) : (size_t)chunk;
|
||||
if (lit) {
|
||||
memcpy(buf, lit + done, want);
|
||||
} else if (old_data) {
|
||||
memcpy(buf, (const uint8_t*)old_data + src_offset + done, want);
|
||||
} else if (!pread_all(src_fd, buf, want, src_offset + done)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
const uint8_t* p = buf;
|
||||
size_t written = 0;
|
||||
while (written < want) {
|
||||
ssize_t n = write(dst_fd, p + written, want - written);
|
||||
if (n < 0 && errno == EINTR)
|
||||
continue;
|
||||
if (n <= 0) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
written += (size_t)n;
|
||||
}
|
||||
done += want;
|
||||
}
|
||||
out_pos += len;
|
||||
}
|
||||
free(buf);
|
||||
return ok && out_pos == delta->new_file_size;
|
||||
}
|
||||
|
||||
void delta_destroy(Delta* delta) {
|
||||
if (!delta)
|
||||
return;
|
||||
|
||||
@@ -56,41 +56,15 @@ typedef struct {
|
||||
|
||||
DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size,
|
||||
uint32_t block_size);
|
||||
/* Seeded equivalent of delta_signature_create: the per-block strong (xxHash32)
|
||||
* checksum uses `seed` (the low 32 bits of --checksum-seed). Passing seed 0 is
|
||||
* identical to the unseeded function. */
|
||||
DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_t old_file_size,
|
||||
uint32_t block_size, uint32_t seed);
|
||||
/* Streaming equivalent of delta_signature_create_seeded: reads the basis blocks
|
||||
* from `fd` in bounded chunks, so an arbitrarily large basis can be signed
|
||||
* without materializing it. The signature itself is bounded (MAX_DELTA_BLOCKS
|
||||
* entries); an over-large basis returns NULL and the caller falls back to a
|
||||
* whole-file transfer. */
|
||||
DeltaSignature* delta_signature_create_fd_seeded(int fd, uint64_t old_file_size,
|
||||
uint32_t block_size, uint32_t seed);
|
||||
Data* delta_signature_serialize(const DeltaSignature* sig);
|
||||
DeltaSignature* delta_signature_deserialize(const Data* data);
|
||||
void delta_signature_destroy(DeltaSignature* sig);
|
||||
|
||||
Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig,
|
||||
uint32_t block_size);
|
||||
/* Seeded equivalent of delta_compute: the per-window strong (xxHash32) check
|
||||
* uses `seed` (the low 32 bits of --checksum-seed). The receiver's signature
|
||||
* must have been built with the same seed for matching. */
|
||||
Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
|
||||
const DeltaSignature* sig, uint32_t block_size, uint32_t seed);
|
||||
Data* delta_serialize(const Delta* delta);
|
||||
Delta* delta_deserialize(const Data* data);
|
||||
void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta, uint32_t block_size);
|
||||
/* Streaming equivalent of delta_apply: matched blocks are read from `src_fd` as
|
||||
* they are emitted, so the basis never has to be resident. */
|
||||
void* delta_apply_fd(int src_fd, uint64_t old_size, const Delta* delta, uint32_t block_size);
|
||||
/* Fully streamed reconstruction: matched blocks come from `old_data` (when
|
||||
* non-NULL) or are read from `src_fd`, and the reconstructed bytes are written
|
||||
* straight to `dst_fd` in bounded chunks, so a reconstructed file larger than
|
||||
* memory is never materialized. */
|
||||
bool delta_apply_to_fd(const void* old_data, int src_fd, uint64_t old_size, const Delta* delta,
|
||||
uint32_t block_size, int dst_fd);
|
||||
void delta_destroy(Delta* delta);
|
||||
|
||||
bool delta_should_attempt(uint64_t old_size, uint64_t new_size, uint64_t max_file_size);
|
||||
@@ -98,7 +72,6 @@ bool delta_is_worthwhile(const Delta* delta, uint64_t new_file_size);
|
||||
|
||||
uint32_t delta_adler32(const void* data, uint32_t len);
|
||||
uint32_t delta_xxhash32(const void* data, uint32_t len);
|
||||
uint32_t delta_xxhash32_seeded(const void* data, uint32_t len, uint32_t seed);
|
||||
uint64_t delta_xxhash64(const void* data, size_t len);
|
||||
|
||||
#endif
|
||||
|
||||
+76
-1847
File diff suppressed because it is too large
Load Diff
+7
-188
@@ -4,7 +4,6 @@
|
||||
#include "file_send.h"
|
||||
#include "file_receive.h"
|
||||
#include "file_types.h"
|
||||
#include "checksum.h"
|
||||
#include <stdbool.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/stat.h>
|
||||
@@ -15,206 +14,26 @@
|
||||
File* file_create(const char* path);
|
||||
void file_destroy(void* item);
|
||||
bool file_load_data(File* file);
|
||||
/* Compute the whole-file content digest of `file` with the negotiated
|
||||
* --checksum-choice algorithm and --checksum-seed. Writes the digest into
|
||||
* `out` (capacity `out_capacity`) and its length into `*out_len`. Returns
|
||||
* false on read/allocation failure or when the digest would not fit. */
|
||||
bool file_checksum(File* file, ChecksumAlgo algo, uint64_t seed, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len);
|
||||
bool file_checksum(File* file, uint64_t* checksum);
|
||||
size_t file_content_to_buffer(File* file);
|
||||
FileMetadata* file_metadata_create(const char* path, const struct stat* stats, bool capture_atime,
|
||||
bool capture_crtime);
|
||||
FileMetadata* file_metadata_create(const struct stat* stats);
|
||||
void file_metadata_destroy(void* metadata);
|
||||
/* --open-noatime process-wide sender policy; see file.c. */
|
||||
void file_set_open_noatime(bool enable);
|
||||
bool file_get_open_noatime(void);
|
||||
/* Capture the process umask ONCE, before any threads are created. Call this at
|
||||
* the very top of main() in both entry points so the cached value is read while
|
||||
* the process is still single-threaded: reading the umask needs a get+set round
|
||||
* trip (umask(0); umask(old)), which would race against receiver threads
|
||||
* creating files if it happened during the first write. Idempotent and safe to
|
||||
* call more than once. */
|
||||
void file_umask_capture(void);
|
||||
/* Process-wide umask, captured once (thread-safe). Used to derive the mode of
|
||||
* a brand-new destination like rsync: source_mode & 0777 & ~umask. Falls back
|
||||
* to file_umask_capture() (behind pthread_once) if capture was never called. */
|
||||
unsigned file_process_umask(void);
|
||||
/* Open `path` read-only for transfer, honouring --open-noatime when set. */
|
||||
int file_open_for_read(const char* path);
|
||||
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse);
|
||||
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave; rsync parity).
|
||||
* --munge-links is a RECEIVER-side rewrite: rsync prefixes every stored symlink
|
||||
* target with this marker, making the link unusable while the referenced
|
||||
* directory does not exist. A SENDER receiving a munged source strips it back
|
||||
* off before transmitting (so a munged tree round-trips through the receiver's
|
||||
* re-munging). */
|
||||
#define SYMLINK_MUNGE_PREFIX "/rsyncd-munged/"
|
||||
|
||||
char* file_symlink_munge(const char* target);
|
||||
/* rsync 3.4.1 unsafe_symlink(): true when `target` escapes the transfer tree
|
||||
* rooted at `link_path` (the symlink's transfer-relative path incl. its name).
|
||||
* Absolute/empty targets and targets climbing above the transfer root (via
|
||||
* "..") are unsafe, as are internal "/../" components and trailing "/..". */
|
||||
bool file_symlink_unsafe(const char* target, const char* link_path);
|
||||
/* True when a lexical target is relative and contains no ".." component, so it
|
||||
* can never escape the receive root once created beneath it. */
|
||||
bool file_symlink_target_contained(const char* target);
|
||||
/* Strip a leading SYMLINK_MUNGE_PREFIX from `target` (mutable, in place);
|
||||
* returns true when a marker was removed. */
|
||||
bool file_symlink_unmunge(char* target);
|
||||
/* Create a symlink at `path` -> `target`, confined below the authorized root
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). The link
|
||||
* value is copied verbatim (rsync -l); only the placement path is confined.
|
||||
* Returns false when a directory already occupies `path`. */
|
||||
bool file_symlink_at_secure(const char* path, const char* target);
|
||||
/* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing
|
||||
* symlink-to-directory to be followed as a directory. */
|
||||
void file_set_keep_dirlinks(bool enable);
|
||||
|
||||
/* --trust-sender receiver process-wide policy (Phase 5). When set, the
|
||||
* receiver trusts that the sender already produced a clean file list and skips
|
||||
* its own redundant up-front re-validation of incoming paths (the empty/".."
|
||||
* rejection and the escaping-symlink-target containment). The low-level
|
||||
* fd-relative confinement primitives below are deliberately NOT disabled by
|
||||
* this flag, so a hostile sender still cannot escape the authorized root. */
|
||||
void file_set_trust_sender(bool enable);
|
||||
bool file_get_trust_sender(void);
|
||||
/* A configured fd without a canonical identity deliberately rejects paths. */
|
||||
bool file_set_authorized_root(int fd, const char* canonical_path);
|
||||
|
||||
/* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */
|
||||
bool file_path_exists_secure(const char* path);
|
||||
bool file_stat_secure(const char* path, struct stat* st);
|
||||
bool file_destination_is_newer_secure(const char* path, const FileMetadata* metadata);
|
||||
int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs);
|
||||
/* Protocol 2.28.0 variant: also increments *dirs_created for every missing
|
||||
* parent directory this walk creates that lies strictly below `count_floor`
|
||||
* (a receive-root-relative path, or NULL to count all of them). */
|
||||
int file_open_secure_parent_counted(const char* path, char** leaf_out, bool create_dirs,
|
||||
unsigned* dirs_created, const char* count_floor);
|
||||
bool file_ensure_directory_secure(const char* path);
|
||||
bool file_directory_exists_secure(const char* path);
|
||||
bool file_rename_secure(const char* old_path, const char* new_path);
|
||||
/* Remove the whole directory tree at `path` (confined, symlink-safe). Used by
|
||||
--force to clear a non-empty destination directory that blocks an incoming
|
||||
regular file. See the .c for the exact success semantics. */
|
||||
bool file_remove_tree_secure(const char* path);
|
||||
/* Open a private 0700 directory (creating it on demand) that must live below
|
||||
the authorized root. Used for the --delay-updates staging directory. */
|
||||
int file_open_private_dir(const char* dir_path);
|
||||
|
||||
/* Open an existing --temp-dir scratch directory (relative or absolute; no
|
||||
creation). When an authorized receive root is configured the directory's
|
||||
REAL path (symlinks resolved) must lie within it, so a client-planted
|
||||
symlink cannot redirect receiver scratch files outside the sandbox; an
|
||||
in-root symlink to another filesystem is still allowed for rsync's EXDEV
|
||||
fallback. */
|
||||
int file_open_temp_dir(const char* dir_path);
|
||||
|
||||
/* The file_to_disk_secure* variants write a temporary copy in the destination
|
||||
directory and atomically rename it over `path`. temp_dir is a scratch
|
||||
directory (an absolute path, or one the caller already resolved against the
|
||||
destination root): when it is non-NULL the temporary copy is instead created
|
||||
there (with a name unique across the whole scratch directory) and atomically
|
||||
renamed into the destination directory once fully written and fsynced. When
|
||||
that rename/link fails with EXDEV (the scratch dir is on another filesystem)
|
||||
the write falls back to a non-atomic copy directly in the destination
|
||||
directory, matching rsync. Pass NULL for the same-directory behavior.
|
||||
--inplace writes never use temp_dir. */
|
||||
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
|
||||
FileAttrPolicy policy, const char* temp_dir);
|
||||
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
FileAttrPolicy policy, bool use_fsync, const char* temp_dir);
|
||||
/* With update enabled, an existing newer destination is left untouched. The
|
||||
check is descriptor-based for inplace writes; atomic replacement still has
|
||||
an unavoidable final rename race without filesystem locking. */
|
||||
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
bool inplace, bool sparse, const FileMetadata* metadata);
|
||||
bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
unsigned long long data_size, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
/* Receiver write-path variant that also applies per-file xattrs (-X/-A) and the
|
||||
* --fake-super stat xattr fd-relative before the final rename. `update` /
|
||||
* `no_replace` / `use_fsync` mirror the plain wrappers above; `keep_partial`
|
||||
* enables --partial best-effort retention of a failed write's temp. */
|
||||
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
|
||||
bool fake_super, bool keep_partial, const char* temp_dir);
|
||||
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
|
||||
(via a temp name + rename); fall back to a byte-identical local copy from
|
||||
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
|
||||
`metadata` is applied only on the copy fallback. `preallocate` applies to
|
||||
that copy fallback only (a hard-linked file shares the basis inode and is
|
||||
never re-allocated). */
|
||||
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
/* Like file_to_disk_secure_link, but the byte-copy fallback also applies the
|
||||
* per-file xattrs (-X/-A) and --fake-super stat xattr (fd-relative). On a
|
||||
* successful hard link no attributes are applied (the shared inode already
|
||||
* carries the basis's). */
|
||||
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
/* Streaming --copy-dest install: atomically materialize `path` by copying the
|
||||
* bytes of `basis_path` through a bounded buffer (no whole-file buffering, so
|
||||
* an arbitrarily large basis works), applying the SOURCE metadata and the
|
||||
* per-file xattrs / --fake-super record. `update` honors a newer destination;
|
||||
* a --temp-dir scratch location falls back to a direct write on EXDEV. */
|
||||
bool file_copy_basis_stream_attrs(const char* path, const char* basis_path,
|
||||
unsigned long long expected_size, bool preallocate,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
/* Receive one length-prefixed whole-file data frame, streaming the payload
|
||||
* through a bounded buffer when it (or its known logical size) exceeds
|
||||
* `stream_limit`. On success exactly one of *out_buffer / *out_spool is set:
|
||||
* - *out_buffer: the historical charged whole-buffer Data (caller destroys);
|
||||
* - *out_spool: an owned temp path holding the payload, installed through the
|
||||
* File's basis_copy field with File.data_spool set so file_destroy unlinks
|
||||
* it. The destination policy/metadata/atomic-store handling is then the
|
||||
* existing bounded-buffer basis install (file_copy_basis_stream_attrs).
|
||||
* `expected_size` (0 = unknown) is the logical size from the check frame;
|
||||
* `dest_path` locates the spool next to the destination; `compress` selects
|
||||
* incremental decompression. Returns false on any framing/I/O/size error. */
|
||||
bool file_receive_payload(int fd, bool compress, unsigned long long expected_size,
|
||||
const char* dest_path, unsigned long long stream_limit, Data** out_buffer,
|
||||
char** out_spool, unsigned long long* out_size);
|
||||
/* Create a confined spool temp file next to `dest_path`; returns an open write
|
||||
* fd and an owned absolute path (to be installed via File.basis_copy with
|
||||
* File.data_spool set, and unlinked by file_destroy). */
|
||||
int file_spool_for_payload(const char* dest_path, char** out_spool_path);
|
||||
/* Protocol 2.28.0 receiver-stat variants: like the two above but additionally
|
||||
* report through `dirs_created` (when non-NULL) how many parent directories the
|
||||
* confined secure walk had to create that lie strictly below `count_floor` (a
|
||||
* receive-root-relative prefix, or NULL for all). Used to reproduce rsync's
|
||||
* `Number of created files` directory count on a fresh destination. */
|
||||
bool file_to_disk_secure_attrs_counted(
|
||||
const char* path, const void* data, unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
bool keep_partial, const char* temp_dir, unsigned* dirs_created, const char* count_floor,
|
||||
uint32_t fake_super_rdev_major, uint32_t fake_super_rdev_minor);
|
||||
bool file_to_disk_secure_link_attrs_counted(const char* path, const char* basis_path,
|
||||
const void* data, unsigned long long data_size,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
FileAttrPolicy policy, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir, unsigned* dirs_created,
|
||||
const char* count_floor);
|
||||
/* The logical transfer root expressed receive-root-relative, or NULL when the
|
||||
* wire paths carry no mirror scaffolding above it. Caller frees non-NULL. */
|
||||
char* file_transfer_root_floor(const Config* config);
|
||||
unsigned long long data_size, bool sparse,
|
||||
const FileMetadata* metadata);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -1,44 +0,0 @@
|
||||
#ifndef FILE_ATTR_H
|
||||
#define FILE_ATTR_H
|
||||
|
||||
#include "config.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/*
|
||||
* Per-attribute receiver policy for applying a transmitted FileMetadata. This
|
||||
* is the split-out replacement for the former single use_metadata bundle: each
|
||||
* flag is applied independently, matching rsync's -p/-t/-o/-g/-E/-U semantics.
|
||||
* `use_metadata` remains the transport/presence gate (whether the metadata frame
|
||||
* travelled at all); this struct decides which attributes are ACTUALLY applied.
|
||||
*
|
||||
* It lives in its own header (rather than metadata.h) because xattr.h's
|
||||
* fake_super_restore_fd() takes one and metadata.h <-> file_types.h form an
|
||||
* include cycle that must not be entered from xattr.h.
|
||||
*
|
||||
* The mode leg is: perms wins over executability; an exec-bits-only change is
|
||||
* made only when perms is off; when neither is set the receiver deliberately
|
||||
* sets no source mode. file.c then substitutes the pre-existing destination
|
||||
* mode for a brand-new destination with metadata it uses the sanitized
|
||||
* source-mode-&-umask base (S_IWGRP|S_IWOTH cleared), and the fixed 0644
|
||||
* default only when no metadata is available at all, so a no--p overwrite
|
||||
* does not lose the destination's perms.
|
||||
*/
|
||||
typedef struct FileAttrPolicy {
|
||||
bool perms; /* config->preserve_perms: apply the source mode bits */
|
||||
bool times; /* config->preserve_times: apply the source mtime */
|
||||
bool atimes; /* config->preserve_atimes (-U): apply the source atime */
|
||||
bool executability; /* config->use_executability (-E): exec-bits-only mode */
|
||||
/* privilege_super_mode_permitted(): when false (SUPER_MODE_OFF / --no-super,
|
||||
or a daemon that did not grant `client owner = yes`), the setuid/setgid/
|
||||
sticky bits are stripped from every applied mode (source mode and any
|
||||
--chmod result) even under --perms. When true, rsync's exact semantics are
|
||||
preserved: -p copies the special bits and the kernel decides. */
|
||||
bool super_permitted;
|
||||
} FileAttrPolicy;
|
||||
|
||||
/* Build the per-attribute policy from a connection's Config. A NULL config
|
||||
* yields the all-off policy (no attribute application). */
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config);
|
||||
|
||||
#endif
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user