Release v2.26.0 #284
+12
-12
@@ -9,10 +9,10 @@ on:
|
||||
jobs:
|
||||
lint:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: clang-format check
|
||||
run: find src/ tests/ -name '*.c' -o -name '*.h' | xargs clang-format --dry-run --Werror
|
||||
@@ -26,11 +26,11 @@ jobs:
|
||||
# suite) run on merge to dev/main, so PR CI stays well under ~3 minutes.
|
||||
build-and-test:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
@@ -51,7 +51,7 @@ jobs:
|
||||
|
||||
sanitizers:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
strategy:
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
sanitizer: [address, undefined]
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build-${{ matrix.sanitizer }} -S . -DSANITIZER=${{ matrix.sanitizer }}
|
||||
@@ -72,12 +72,12 @@ jobs:
|
||||
|
||||
fuzz-build:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure (clang + fuzz)
|
||||
run: CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
|
||||
@@ -94,12 +94,12 @@ jobs:
|
||||
|
||||
coverage:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DENABLE_COVERAGE=ON
|
||||
@@ -118,12 +118,12 @@ jobs:
|
||||
|
||||
valgrind:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
needs: lint
|
||||
if: github.event_name == 'push'
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4
|
||||
|
||||
- name: Configure
|
||||
run: cmake -B build -S . -DSTRICT_WARNINGS=ON
|
||||
|
||||
@@ -8,3 +8,7 @@ build-*/
|
||||
build2/
|
||||
build3/
|
||||
build_docker2/
|
||||
|
||||
# Test/run artifacts
|
||||
root/
|
||||
test_partial_install_tmp/
|
||||
|
||||
@@ -128,7 +128,7 @@ Do not wait for the user to tell you CI failed — check proactively. The user s
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -92,7 +92,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,16 +27,19 @@ FetchContent_Declare(xxhash GIT_REPOSITORY https://github.com/Cyan4973/xxHash GI
|
||||
FetchContent_MakeAvailable(xxhash)
|
||||
|
||||
# Sanitizer option
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread none)
|
||||
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, undefined, none)")
|
||||
set_property(CACHE SANITIZER PROPERTY STRINGS address thread undefined none)
|
||||
if(SANITIZER STREQUAL "address")
|
||||
add_compile_options(-fsanitize=address -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=address)
|
||||
elseif(SANITIZER STREQUAL "thread")
|
||||
add_compile_options(-fsanitize=thread -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=thread)
|
||||
elseif(SANITIZER STREQUAL "undefined")
|
||||
add_compile_options(-fsanitize=undefined -fno-omit-frame-pointer -g)
|
||||
add_link_options(-fsanitize=undefined)
|
||||
elseif(NOT SANITIZER STREQUAL "none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, none")
|
||||
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, undefined, none")
|
||||
endif()
|
||||
|
||||
option(STRICT_WARNINGS "Enable strict warnings" OFF)
|
||||
@@ -52,6 +55,16 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
file(GLOB SHARED_SRCS "src/shared/*.c")
|
||||
@@ -61,15 +74,15 @@ file(GLOB TEST_SRCS "tests/*.c")
|
||||
|
||||
add_executable(server ${SERVER_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(server PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS})
|
||||
target_include_directories(client PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} src/client/scanner.c)
|
||||
target_include_directories(tests PRIVATE tests src/shared src/server src/client)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
target_link_libraries(tests PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY} ${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
```
|
||||
|
||||
### Source Layout
|
||||
@@ -82,19 +95,25 @@ tests/integration/ — Python pytest integration tests
|
||||
```
|
||||
|
||||
### Dependencies
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)`
|
||||
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)` (default compression codec)
|
||||
- **zlib** — found via `find_library(ZLIB_LIBRARY z)` (the `zlib`/`zlibx` codecs)
|
||||
- **lz4** — found via `find_library(LZ4_LIBRARY lz4)` (the `lz4` codec)
|
||||
- **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport)
|
||||
- **xxHash** — fetched via `FetchContent` from GitHub (delta transfer hashing, v0.8.3)
|
||||
- **xxHash** — fetched via `FetchContent` from the upstream repository (delta transfer hashing, v0.8.3)
|
||||
- **pthreads** — found via `find_package(Threads REQUIRED)`
|
||||
- **C11 standard** — required
|
||||
- **CMake 3.22+** — minimum version
|
||||
|
||||
The codec matrix (protocol 2.26.0) uses zstd/zlib/lz4 for compression and
|
||||
xxHash/OpenSSL for the `xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1` checksums
|
||||
(`none` needs no library); both codec families are negotiated per transfer.
|
||||
|
||||
## Conventions
|
||||
|
||||
- Use `file(GLOB ...)` for source collection (existing pattern).
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `${ZLIB_LIBRARY}`, `${LZ4_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
|
||||
- Include directories: `src/shared`, `src/server`, `src/client`, `tests` (for test target).
|
||||
- Sanitizer support: pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (live option in CMakeLists.txt).
|
||||
- Sanitizer support: pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (live option in CMakeLists.txt).
|
||||
- Build with `cmake -B build -S . && cmake --build build -j$(nproc)`.
|
||||
- For CI, dependencies are provided by the project's custom Docker image (repo-root `Dockerfile`, same image CI uses). For local development, use `nix-shell`. Never add `apt-get install` / `pip install` to CI workflows. See `AGENTS.md`.
|
||||
|
||||
@@ -105,7 +124,7 @@ tests/integration/ — Python pytest integration tests
|
||||
3. Add new dependencies with `find_package` or `find_library`.
|
||||
4. When adding a new executable target, follow the pattern of existing targets.
|
||||
5. When adding a new library (static/shared), use `add_library` and follow the project's naming.
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (matching CI's matrix strategy).
|
||||
6. For sanitizer builds, pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (matching CI's matrix strategy).
|
||||
7. Always verify the build compiles after changes.
|
||||
|
||||
## Sanitizer Configurations
|
||||
@@ -119,11 +138,9 @@ cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (race conditions)
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
For UndefinedBehaviorSanitizer (no `-DSANITIZER=undefined` option in CMakeLists.txt yet), use the manual flag approach:
|
||||
UndefinedBehaviorSanitizer uses the same built-in option:
|
||||
```bash
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=undefined -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=undefined"
|
||||
cmake -B build -S . -DSANITIZER=undefined
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
@@ -159,7 +176,7 @@ cmake -B build -S . -DCMAKE_BUILD_TYPE=RelWithDebInfo
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/server
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
./build/client
|
||||
./build/tests
|
||||
```
|
||||
@@ -187,7 +204,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -27,7 +27,7 @@ FastSync is a file synchronization tool (like rsync, but faster). It transfers f
|
||||
cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Server (TCP mode)
|
||||
./build/server
|
||||
./build/server -p 8080 --allow-unauthenticated
|
||||
|
||||
# Client (TCP mode)
|
||||
./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk
|
||||
@@ -37,13 +37,13 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
|
||||
# Run tests
|
||||
./build/tests # unit tests
|
||||
python3 test.py # integration tests
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # integration tests
|
||||
```
|
||||
|
||||
## Code Walkthrough
|
||||
|
||||
### Client Entry Point (`src/client/client_cli.c`)
|
||||
- Parses CLI arguments using `getopt_long`
|
||||
- Parses CLI arguments using a custom option-table parser (`OPTION_TABLE` in `src/client/client_cli.c`); there is no `getopt*` usage
|
||||
- Creates `Config` struct with all options
|
||||
- Detects SSH destinations (contains `:`)
|
||||
- Calls into `client_send.c` for the actual transfer
|
||||
@@ -109,7 +109,7 @@ Collection of files for batch transfer. Serialized with file count, then per-fil
|
||||
zstd streaming compression via `ZSTD_compressStream2`/`ZSTD_decompressStream`. Compression happens per-chunk in the sender stage. Level 1-22 (default 5). Streaming means memory usage stays bounded regardless of file size.
|
||||
|
||||
### "How does sendfile() work?"
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `-f` flag. Only works with TCP (not SSH, not compression).
|
||||
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `--sendfile` (long form only). Only works with TCP (not SSH, not compression).
|
||||
|
||||
### "How does incremental sync work?"
|
||||
Client sends file metadata (path, size, mtime) to server. Server checks if destination file has same size+mtime. If match, server responds `STATUS_OK` (skip). If mismatch, server responds `STATUS_NEXT` (send).
|
||||
@@ -138,7 +138,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -316,7 +316,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -14,10 +14,9 @@ Diagnose crashes, memory errors, hangs, and logic bugs. You use structured debug
|
||||
### Memory Errors
|
||||
```bash
|
||||
# AddressSanitizer (fast, recommended first)
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/client # or ./build/server
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/client # or ./build-asan/server -p 8080 --allow-unauthenticated
|
||||
|
||||
# Valgrind (slower, more thorough)
|
||||
valgrind --leak-check=full --show-leak-kinds=all --track-origins=yes \
|
||||
@@ -32,10 +31,9 @@ valgrind --tool=drd ./build/client ...
|
||||
|
||||
### Thread Sanitizer
|
||||
```bash
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=thread" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
|
||||
### GDB
|
||||
@@ -143,7 +141,7 @@ gprof ./build/client gmon.out
|
||||
|
||||
### Step 5: Verify
|
||||
- Run `./build/tests` (unit tests)
|
||||
- Run `python3 test.py` (integration tests)
|
||||
- Run `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` (integration tests)
|
||||
- Run under valgrind again to confirm clean
|
||||
- Test under ASan again
|
||||
|
||||
@@ -162,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -96,7 +96,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -16,12 +16,18 @@ Scan the codebase for patterns that suggest new feature opportunities. You ident
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_cli.c Entry point, OPTION_TABLE parser, config setup
|
||||
usage.c Usage/help text (authoritative CLI flag list)
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
change_list.c File change-list bookkeeping
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
server_cli.c Server option-table CLI parsing
|
||||
receiver.c Receiver-side file handling
|
||||
receiver_pipeline.c Receiver worker pipeline
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
@@ -32,40 +38,63 @@ src/shared/ Shared libraries (used by both client and server)
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_send.c/h Sender-side file transfer
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_list.c/h File list model
|
||||
file_store.c/h Destination file store
|
||||
array_list.c/h Dynamic array
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
batch.c/h Batch files (--write-batch/--read-batch)
|
||||
charset.c/h Filename charset conversion (--iconv)
|
||||
chmod.c/h Permission modification (--chmod)
|
||||
xattr.c/h Extended attributes
|
||||
hardlink.c/h Hard-link handling
|
||||
identity.c/h uid/gid mapping (--usermap/--groupmap/--chown)
|
||||
credentials.c/h Daemon credentials
|
||||
daemon_conf.c/h Daemon module configuration
|
||||
motd.c/h Daemon MOTD
|
||||
delay_updates.c/h Delayed update staging
|
||||
stop_condition.c/h Stop-after/stop-at handling
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
file_types.h Shared file type definitions
|
||||
```
|
||||
|
||||
### Existing CLI Flags (from client_cli.c)
|
||||
### Existing CLI Flags (authoritative source: `src/client/usage.c`)
|
||||
```
|
||||
--source-dir <dir> Source directory to sync (required)
|
||||
--dest-dir <dir> Destination directory on server (required)
|
||||
--host <host> Server hostname/IP (required)
|
||||
--port <port> Server TCP port
|
||||
--server-mode Listen as server
|
||||
--use-compression, -c Enable zstd compression
|
||||
--use-multithreading, -m Enable multithreaded transfer
|
||||
--use-sendfile, -s Use sendfile() zero-copy TCP
|
||||
--use-ssh, -S Use SSH transport
|
||||
--use-tls, -T Enable TLS encryption
|
||||
--cert <file> TLS certificate file
|
||||
--key <file> TLS key file
|
||||
--ca <file> TLS CA certificate file
|
||||
--insecure Skip TLS verification
|
||||
--bwlimit <bytes/s> Bandwidth limit
|
||||
--delete Delete files not in source
|
||||
--include <pattern> Include filter pattern
|
||||
--exclude <pattern> Exclude filter pattern
|
||||
--dry-run Print what would be transferred
|
||||
--save-to-disk Save transferred files to disk (for server tests)
|
||||
--source-dir <dir> Source directory
|
||||
--dest-dir <dir> Destination directory on server
|
||||
--server-host <ip> Server IP address (default: 127.0.0.1)
|
||||
--server-port <n> Server port (default: 8080); --port is an alias
|
||||
-c, --checksum Verify content by checksum instead of size+mtime
|
||||
-z, --compress [level] Enable compression (level 1-22, default 5)
|
||||
-j, --threads[=N] Enable multithreaded scanner/loader/sender pipeline
|
||||
--chunk-serialization Enable chunk serialization (long form only)
|
||||
--sendfile sendfile() zero-copy (TCP only; long form only)
|
||||
-s, --secluded-args Protect-args compatibility option (no effect)
|
||||
--tls Enable TLS encryption; --cert/--key/--ca give PEMs
|
||||
--bwlimit <KB/s> Bandwidth limit in kilobytes per second
|
||||
--delete Delete files on receiver not in source
|
||||
--incremental Skip files unchanged since last transfer
|
||||
--delta Delta transfer for changed files (needs --incremental)
|
||||
-f, --filter=RULE rsync-style filter rule (+/- include/exclude)
|
||||
--exclude <pattern> Exclude files matching pattern
|
||||
--include <pattern> Only include files matching pattern
|
||||
-m, --prune-empty-dirs Do not transfer empty directory entries
|
||||
-n, --dry-run Show what would be transferred
|
||||
--save-to-disk Write received files to disk
|
||||
--version Print version and exit
|
||||
--help Print help
|
||||
--help Show help
|
||||
```
|
||||
> Always confirm the current flags with `./build/client --help`; the table above
|
||||
> is a representative subset. `src/client/usage.c` is the authoritative list and
|
||||
> `OPTION_TABLE` in `src/client/client_cli.c` is the parser (there is no `getopt*`).
|
||||
|
||||
## Feature Scout Checklist
|
||||
|
||||
@@ -288,7 +317,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -21,7 +21,7 @@ Design integration tests that verify the full transfer pipeline works end-to-end
|
||||
- Multiple configurations (TCP, SSH, TLS, compression, multithreading)
|
||||
- Network shaping (LAN, WAN profiles)
|
||||
- Feature tests (dry run, archive, exclude, delete, incremental, bandwidth limit)
|
||||
- Run: `python3 -m pytest tests/ -v --tb=short`
|
||||
- Run: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
|
||||
### 3. New: Focused Integration Tests
|
||||
When adding new features or fixing bugs, write targeted integration tests.
|
||||
@@ -35,13 +35,14 @@ mkdir -p /tmp/fastsync_test/src
|
||||
echo "test content" > /tmp/fastsync_test/src/file.txt
|
||||
|
||||
# Start server
|
||||
./build/server &
|
||||
./build/server -p 8080 --allow-unauthenticated &
|
||||
SERVER_PID=$!
|
||||
sleep 0.5
|
||||
|
||||
# Run client
|
||||
./build/client --source-dir /tmp/fastsync_test/src \
|
||||
--dest-dir /tmp/fastsync_test/dst \
|
||||
--server-port 8080 \
|
||||
--save-to-disk
|
||||
|
||||
# Verify
|
||||
@@ -76,7 +77,7 @@ openssl req -x509 -newkey rsa:2048 -keyout /tmp/key.pem -out /tmp/cert.pem \
|
||||
### Pattern 4: Incremental Sync
|
||||
```bash
|
||||
# First sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
|
||||
# Modify source
|
||||
echo "updated" >> /tmp/src/file.txt
|
||||
@@ -89,14 +90,14 @@ echo "updated" >> /tmp/src/file.txt
|
||||
### Pattern 5: Delete Verification
|
||||
```bash
|
||||
# Initial sync
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk
|
||||
|
||||
# Add extra file to dest
|
||||
echo "extra" > /tmp/dst/.../extra.txt
|
||||
|
||||
# Sync with --delete
|
||||
./build/client --source-dir /tmp/src --dest-dir /tmp/dst \
|
||||
--save-to-disk --delete -M
|
||||
--save-to-disk --delete
|
||||
|
||||
# Verify extra.txt is gone
|
||||
test ! -f /tmp/dst/.../extra.txt
|
||||
@@ -115,7 +116,7 @@ The project uses Gitea Actions. Key jobs:
|
||||
jobs:
|
||||
new-job:
|
||||
runs-on: ubuntu-latest
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v7
|
||||
container: gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- name: Configure
|
||||
@@ -127,7 +128,7 @@ jobs:
|
||||
- name: Unit Tests
|
||||
run: ./build-${{ matrix.sanitizer }}/tests
|
||||
- name: Integration Tests
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/ -v --tb=short
|
||||
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
The symlink step is required because `tests/conftest.py` expects `./build` to exist.
|
||||
|
||||
@@ -135,7 +136,7 @@ The symlink step is required because `tests/conftest.py` expects `./build` to ex
|
||||
|
||||
After any code change:
|
||||
- [ ] Unit tests pass: `./build/tests`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/ -v --tb=short`
|
||||
- [ ] Integration tests pass: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`
|
||||
- [ ] Build clean: no warnings with `-Wall`
|
||||
- [ ] No memory errors: ASan clean
|
||||
- [ ] No thread errors: TSan clean (if threading involved)
|
||||
@@ -156,7 +157,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
---
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates GitHub issues from their findings.
|
||||
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates Gitea issues from their findings.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
@@ -12,7 +12,7 @@ You are the primary orchestrator agent. Your job is to:
|
||||
2. Decide which specialized sub-agents to dispatch for analysis
|
||||
3. Delegate analysis work using the task tool
|
||||
4. Receive structured findings from sub-agents
|
||||
5. Create GitHub issues from those findings using `gh issue create`
|
||||
5. Create Gitea issues from those findings using `tea issues create`
|
||||
6. Coordinate the overall analysis workflow end-to-end
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
@@ -97,7 +97,7 @@ First, read the repository structure to understand what exists:
|
||||
### Phase 2: Determine Analysis Scope
|
||||
Based on what the user requests or what needs attention:
|
||||
- **New features wanted?** → Dispatch `feature-scout` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-screener` sub-agent
|
||||
- **Security audit needed?** → Dispatch `security-auditor` sub-agent
|
||||
- **Code quality review?** → Dispatch `code-quality-guardian` sub-agent
|
||||
- **All of the above?** → Run all three in parallel
|
||||
|
||||
@@ -110,7 +110,7 @@ Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
```
|
||||
Task: Ask the security-screener agent to analyze the codebase.
|
||||
Task: Ask the security-auditor agent to analyze the codebase.
|
||||
Context: <provide summary of what was found in Phase 1>
|
||||
```
|
||||
|
||||
@@ -138,14 +138,14 @@ Each sub-agent returns findings in this structured format:
|
||||
- **Labels**: comma-separated labels for the issue
|
||||
```
|
||||
|
||||
### Phase 5: Create GitHub Issues
|
||||
For each finding, create a GitHub issue:
|
||||
### Phase 5: Create Gitea Issues
|
||||
For each finding, create a Gitea issue:
|
||||
|
||||
```bash
|
||||
gh issue create \
|
||||
tea issues create --repo TapTap/FastSync \
|
||||
--title "<Finding Title>" \
|
||||
--label "<labels>" \
|
||||
--body "## Description
|
||||
--labels "<labels>" \
|
||||
--description "## Description
|
||||
<description>
|
||||
|
||||
## Location
|
||||
@@ -175,11 +175,13 @@ _This issue was automatically generated by the issue-creator agent._"
|
||||
|
||||
### Duplicate Detection
|
||||
Before creating an issue:
|
||||
1. Check existing open issues: `gh issue list --state open --label "<label>"`
|
||||
2. Search for similar titles using `gh issue list --search "<keywords>"`
|
||||
1. Check existing open issues: `tea issues list --repo TapTap/FastSync --state open --labels "<label>"`
|
||||
2. Search for similar titles using `tea issues list --repo TapTap/FastSync --keyword "<keywords>"`
|
||||
3. If a similar issue exists, add a comment instead of creating a duplicate:
|
||||
```bash
|
||||
gh issue comment <issue-number> --body "Additional finding from automated analysis: <details>"
|
||||
tea comment --repo TapTap/FastSync <issue-number> "Additional finding from automated analysis: <details>"
|
||||
# or POST to the Gitea API:
|
||||
# POST https://gitea.tap-tap.win/api/v1/repos/TapTap/FastSync/issues/<n>/comments
|
||||
```
|
||||
|
||||
## Sub-Agent Reference
|
||||
@@ -189,13 +191,12 @@ Before creating an issue:
|
||||
| Agent | File | Purpose |
|
||||
|---|---|---|
|
||||
| feature-scout | `.opencode/agents/feature-scout.md` | Scans for feature opportunities |
|
||||
| security-screener | `.opencode/agents/security-screener.md` | Scans for security vulnerabilities |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits and vulnerability scans |
|
||||
| code-quality-guardian | `.opencode/agents/code-quality-guardian.md` | Scans for code quality improvements |
|
||||
| architect | `.opencode/agents/architect.md` | Architecture reviews |
|
||||
| c-reviewer | `.opencode/agents/c-reviewer.md` | C code correctness reviews |
|
||||
| debugger | `.opencode/agents/debugger.md` | Bug diagnosis |
|
||||
| refactorer | `.opencode/agents/refactorer.md` | Code refactoring |
|
||||
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits |
|
||||
| test-writer | `.opencode/agents/test-writer.md` | Test development |
|
||||
| perf-analyst | `.opencode/agents/perf-analyst.md` | Performance analysis |
|
||||
| protocol-designer | `.opencode/agents/protocol-designer.md` | Protocol design |
|
||||
@@ -242,7 +243,7 @@ tests/test_file.c — File tests
|
||||
tests/test_transport_tcp.c — TCP transport tests
|
||||
tests/test_transport_tls.c — TLS transport tests
|
||||
tests/test_array_list.c — Array list tests
|
||||
tests/pytest/ — Python integration tests
|
||||
tests/integration/ — Python pytest integration tests
|
||||
```
|
||||
|
||||
### Build & Config Files
|
||||
@@ -259,7 +260,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -56,9 +56,11 @@ DirectoryScanner → Queue(Scanner→Loader) → ChunkBuilder → Queue(Loader
|
||||
|
||||
### Benchmark Context
|
||||
|
||||
From README benchmarks (25MB mixed files, localhost):
|
||||
- Best config: `-m -c` (multithread + compression) → 0.20s, 11.2× faster than rsync
|
||||
- `sendfile()` bypasses userspace → ~2× faster on localhost
|
||||
Use the maintained benchmark tool — do not cite stale README numbers:
|
||||
- `python3 benchmark/bench.py` runs the repeatable throughput benchmark.
|
||||
- The real flags are `-j` (multithreading) and `-z` (compression); a fast loopback
|
||||
config combines `-j -z`.
|
||||
- `sendfile()` (via `--sendfile`) bypasses userspace → ~2× faster on localhost
|
||||
- Compression reduces wire data enough that transfer becomes latency-bound on WAN
|
||||
|
||||
## Output Format
|
||||
@@ -120,6 +122,7 @@ time ./build/client [args...]
|
||||
|
||||
# High precision
|
||||
perf stat -e task-clock ./build/client [args...]
|
||||
```
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
@@ -127,9 +130,8 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
```
|
||||
|
||||
@@ -91,7 +91,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -160,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -3,25 +3,64 @@ description: Audits FastSync for security vulnerabilities — TLS config, input
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are a security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
You are the security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport. This is the single canonical security agent.
|
||||
|
||||
## Your Role
|
||||
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices.
|
||||
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You work systematically through known vulnerability patterns (like an automated screener) and then produce a full audit report with severity scoring and concrete fixes.
|
||||
|
||||
## Attack Surface
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
|
||||
### Network Input Points
|
||||
1. **TCP server** (`src/server/server.c`) — accepts connections from any client
|
||||
2. **SSH transport** (`src/shared/transport_ssh.c`) — receives data via stdio pipe
|
||||
3. **Protocol parsing** (`src/shared/protocol.c`) — deserializes all incoming data
|
||||
4. **Config deserialization** (`src/shared/config.c`) — receives remote config
|
||||
5. **Chunk deserialization** (`src/shared/chunk.c`) — receives file batches
|
||||
## Project Architecture
|
||||
|
||||
### TLS Configuration
|
||||
- OpenSSL TLS 1.2+ via `src/shared/transport_tls.c`
|
||||
- Certificate/key loading, CA verification
|
||||
- SSL context setup, cipher suite selection
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
client_validation.c Destination/CLI validation
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
receiver.c Receiver-side file handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
file_receive.c/h Receiver-side file transfer
|
||||
file_store.c/h Destination file store
|
||||
delta.c/h Delta transfer algorithm
|
||||
checksum.c/h Whole-file/block checksums (xxHash, md5)
|
||||
filter.c/h rsync-style filter rules
|
||||
xattr.c/h Extended attributes
|
||||
identity.c/h uid/gid mapping
|
||||
credentials.c/h Daemon credentials
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` / `receiver.c` | Writes received files to disk |
|
||||
|
||||
## Security Audit Checklist
|
||||
|
||||
@@ -33,51 +72,182 @@ Audit the codebase for security vulnerabilities. You focus on the attack surface
|
||||
- [ ] Chunk count and file count validated before allocation
|
||||
- [ ] Config field lengths bounded
|
||||
|
||||
### 2. Buffer Safety
|
||||
- [ ] No `strcpy` — use `snprintf` or `strncpy` with null termination
|
||||
- [ ] `malloc` size calculations don't overflow (e.g., `count * sizeof(...)`)
|
||||
- [ ] No fixed-size stack buffers for unbounded input
|
||||
- [ ] `receive_n_data` always checks return value
|
||||
- [ ] Off-by-one in path concatenation
|
||||
### 2. Buffer Overflow Risks
|
||||
|
||||
### 3. Memory Safety in Error Paths
|
||||
- [ ] All error paths free allocated resources
|
||||
- [ ] No use-after-free on error paths
|
||||
- [ ] No double-free on error paths
|
||||
- [ ] Partial reads handled (don't use incomplete data)
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
|
||||
### 4. TLS/SSL Security
|
||||
- [ ] TLS 1.2 minimum enforced (no SSLv3, TLS 1.0, TLS 1.1)
|
||||
- [ ] Certificate verification enabled when CA provided
|
||||
- [ ] Certificate verification disabled only with explicit warning
|
||||
- [ ] Private key file permissions checked
|
||||
- [ ] No hardcoded certificates or keys
|
||||
- [ ] Cipher suites restricted to strong algorithms
|
||||
- [ ] SSL error codes checked after `SSL_read`/`SSL_write`
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
|
||||
### 5. Authentication & Authorization
|
||||
### 3. Path Traversal in File Operations
|
||||
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 4. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 5. TLS / SSL Security
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--ca`/warning
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
- [ ] **No hardcoded certificates or keys**
|
||||
- [ ] **SSL error codes checked after `SSL_read`/`SSL_write`**
|
||||
|
||||
### 6. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
- [ ] **All error paths free allocated resources** (no leaks / UAF / double-free)
|
||||
- [ ] **Partial reads handled** (don't use incomplete data)
|
||||
|
||||
### 7. Integer Overflow in Allocation
|
||||
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 8. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 9. Authentication & Authorization
|
||||
- [ ] SSH transport relies on SSH authentication (not custom auth)
|
||||
- [ ] No password/credential storage in plaintext
|
||||
- [ ] Server doesn't trust client-supplied paths blindly
|
||||
- [ ] Destination directory validated before writing
|
||||
|
||||
### 6. Denial of Service
|
||||
- [ ] Bounded memory allocation (can't OOM server with huge chunk)
|
||||
- [ ] Timeout on connections (no indefinite blocking)
|
||||
- [ ] Maximum connection limit or rate limiting
|
||||
- [ ] Malformed protocol messages handled gracefully (no crash)
|
||||
### 10. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
|
||||
### 7. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
### 11. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
|
||||
### 8. File System Security
|
||||
### 12. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 13. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 14. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
### 15. File System Security
|
||||
- [ ] Received file permissions validated (no SUID/SGID injection)
|
||||
- [ ] Symlink attack prevention (don't follow symlinks in destination)
|
||||
- [ ] Race conditions in file creation (TOCTOU)
|
||||
- [ ] Temporary file security (if any)
|
||||
|
||||
### 16. Cryptographic Practices
|
||||
- [ ] No custom crypto — uses OpenSSL only
|
||||
- [ ] No hardcoded keys, IVs, or salts
|
||||
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
|
||||
|
||||
## Common Vulnerability Patterns
|
||||
|
||||
### Format String Bugs
|
||||
@@ -118,9 +288,59 @@ receive_n_data(fd, buffer, expected_size);
|
||||
if (!receive_n_data(fd, buffer, expected_size)) { /* handle error */ }
|
||||
```
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
For each vulnerability found:
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Detailed Finding Fields
|
||||
|
||||
For each vulnerability found, also be prepared to report:
|
||||
1. **Location** — file:line
|
||||
2. **Severity** — critical / high / medium / low / informational
|
||||
3. **Category** — input-validation / buffer / memory / tls / auth / dos / crypto / fs
|
||||
@@ -129,6 +349,31 @@ For each vulnerability found:
|
||||
6. **Fix** — concrete code change
|
||||
7. **CVSS estimate** — rough severity score if exploitable
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### Audit Summary
|
||||
|
||||
Also provide a summary:
|
||||
```
|
||||
=== SECURITY AUDIT SUMMARY ===
|
||||
@@ -140,13 +385,30 @@ Low: <count>
|
||||
Informational: <count>
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -1,310 +0,0 @@
|
||||
---
|
||||
description: Scans the FastSync codebase for security vulnerabilities — buffer overflows, path traversal, TLS issues, memory safety, and cryptographic hygiene.
|
||||
mode: subagent
|
||||
---
|
||||
|
||||
You are a security screener for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
|
||||
|
||||
## Your Role
|
||||
|
||||
Scan the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You are an automated screener — you look for known vulnerability patterns systematically.
|
||||
|
||||
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
|
||||
|
||||
## Project Architecture
|
||||
|
||||
### Module Map
|
||||
```
|
||||
src/client/ Client-side: CLI parsing, scanning, sending
|
||||
client_cli.c Entry point, argument parsing, config setup
|
||||
client_send.c Transfer orchestration, pipeline management
|
||||
scanner.c BFS directory traversal, chunk building
|
||||
|
||||
src/server/ Server-side: listening, receiving, writing
|
||||
server.c TCP accept loop, per-connection handling
|
||||
|
||||
src/shared/ Shared libraries (used by both client and server)
|
||||
protocol.c/h Wire protocol: status codes, send/receive primitives
|
||||
compression.c/h zstd streaming compression/decompression
|
||||
chunk.c/h File grouping and batch serialization
|
||||
queue.c/h Thread-safe bounded queue (producer-consumer)
|
||||
config.c/h Runtime configuration, serialization, parsing
|
||||
data.c/h Generic buffer type (Data)
|
||||
metadata.c/h File metadata (mode, uid, gid, mtime)
|
||||
file.c/h File representation
|
||||
array_list.c/h Dynamic array
|
||||
transport_tcp.c/h TCP client/server with sendfile() zero-copy
|
||||
transport_ssh.c/h SSH transport with ControlMaster
|
||||
transport_tls.c/h TLS encryption via OpenSSL
|
||||
multiprocessing.c/h Fork-based concurrency
|
||||
log.c/h Logging utilities
|
||||
utils.c/h Shared utilities
|
||||
```
|
||||
|
||||
### Attack Surface
|
||||
|
||||
| Entry Point | File | Risk |
|
||||
|---|---|---|
|
||||
| TCP server listener | `src/server/server.c` | Externally reachable on network |
|
||||
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
|
||||
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
|
||||
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
|
||||
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
|
||||
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
|
||||
| File writer | `src/server/server.c` | Writes received files to disk |
|
||||
|
||||
## Security Screener Checklist
|
||||
|
||||
### 1. Buffer Overflow Risks
|
||||
Search for these dangerous patterns in all `.c` and `.h` files:
|
||||
|
||||
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
|
||||
```c
|
||||
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
|
||||
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
|
||||
char line[4096]; // SUSPICIOUS — what limits the line length?
|
||||
```
|
||||
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
|
||||
```bash
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
|
||||
```
|
||||
- [ ] **Unbounded `sprintf` to fixed buffer**
|
||||
```c
|
||||
char buf[256];
|
||||
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
|
||||
```
|
||||
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
|
||||
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
|
||||
```c
|
||||
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
|
||||
```
|
||||
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
|
||||
|
||||
### 2. Path Traversal in File Operations
|
||||
Check all paths constructed from received data:
|
||||
|
||||
- [ ] **Files constructed with client-provided filenames + destination directory**
|
||||
```c
|
||||
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
|
||||
```
|
||||
Check for `../` filtering:
|
||||
```bash
|
||||
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
|
||||
```
|
||||
- [ ] **`realpath()` usage** for path canonicalization
|
||||
- [ ] **Symlink following** — does the server follow symlinks in the destination?
|
||||
- [ ] **Null byte injection** — received filenames with embedded `\0`
|
||||
|
||||
### 3. Unchecked Return Values from Critical Functions
|
||||
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
|
||||
```bash
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
```
|
||||
For each match, verify NULL check exists before use.
|
||||
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
|
||||
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
|
||||
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
|
||||
- [ ] **`fopen()` / `open()`** return values not checked
|
||||
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
|
||||
|
||||
### 4. TLS / SSL Misconfiguration
|
||||
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
|
||||
```c
|
||||
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
|
||||
```
|
||||
- [ ] **Certificate verification disabled** without explicit `--insecure` flag
|
||||
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
|
||||
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
|
||||
- [ ] **Private key file permissions** not checked before loading
|
||||
- [ ] **Hostname verification** not performed on server certificate
|
||||
- [ ] **Session renegotiation** not limited (DoS vector)
|
||||
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
|
||||
|
||||
### 5. Memory Safety Issues
|
||||
- [ ] **Use-after-free** — object freed but pointer still used later
|
||||
- [ ] **Double-free** — `free()` called twice on same pointer
|
||||
- [ ] **Memory leaks** on error paths — allocated but not freed before return
|
||||
- [ ] **Integer overflow** in allocation size computation
|
||||
```c
|
||||
// DANGER: count * sizeof(Type) can overflow
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
|
||||
// SAFE:
|
||||
if (count > SIZE_MAX / sizeof(Element)) return NULL;
|
||||
void *arr = malloc(count * sizeof(Element));
|
||||
```
|
||||
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
|
||||
```c
|
||||
// BAD: leaks original pointer on failure
|
||||
buf = realloc(buf, new_size);
|
||||
|
||||
// GOOD:
|
||||
void *tmp = realloc(buf, new_size);
|
||||
if (!tmp) { free(buf); return NULL; }
|
||||
buf = tmp;
|
||||
```
|
||||
|
||||
### 6. Integer Overflow in Allocation
|
||||
Check all size calculations:
|
||||
|
||||
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
|
||||
- [ ] Allocations where size is multiplied by count
|
||||
```bash
|
||||
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
|
||||
```
|
||||
- [ ] Loop counters that could wrap (unsigned underflow)
|
||||
- [ ] Signed integer overflow in size checks
|
||||
|
||||
### 7. Format String Vulnerabilities
|
||||
- [ ] User-controlled data passed as format string
|
||||
```c
|
||||
printf(user_input); // VULNERABLE
|
||||
fprintf(stderr, user_input); // VULNERABLE
|
||||
syslog(LOG_INFO, user_input); // VULNERABLE
|
||||
|
||||
printf("%s", user_input); // SAFE
|
||||
```
|
||||
```bash
|
||||
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
```
|
||||
|
||||
### 8. TOCTOU Race Conditions
|
||||
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
|
||||
```c
|
||||
if (access(path, F_OK) == 0) { // CHECK
|
||||
fd = open(path, O_RDWR); // USE — file could have changed
|
||||
}
|
||||
```
|
||||
- [ ] `stat()` followed by `open()` with different permissions
|
||||
- [ ] Temporary file creation with predictable names
|
||||
|
||||
### 9. Insecure Temporary File Usage
|
||||
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
|
||||
- [ ] Temporary files created in world-writable directories
|
||||
- [ ] Temporary files not cleaned up on error paths
|
||||
- [ ] Predictable temp file names (race + symlink attack)
|
||||
|
||||
### 10. Hardcoded Secrets / Credentials
|
||||
- [ ] Hardcoded passwords, API keys, or tokens
|
||||
- [ ] Hardcoded TLS private keys or certificates
|
||||
- [ ] Hardcoded connection strings with embedded credentials
|
||||
- [ ] Test certificates/keys in source tree (should be documented if intentional)
|
||||
|
||||
### 11. Denial of Service Vectors
|
||||
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
|
||||
- Check `chunk.c` for chunk count limits
|
||||
- Check `protocol.c` for message size limits
|
||||
- Check `config.c` for config field size limits
|
||||
- [ ] **No connection limits** — server doesn't cap concurrent connections
|
||||
- [ ] **No timeouts** — connections can hang indefinitely
|
||||
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
|
||||
- [ ] **Repeated slow reads** — slow loris style attack
|
||||
- [ ] **Fork bomb** — server forks per connection without limit
|
||||
|
||||
### 12. Information Disclosure
|
||||
- [ ] Server sends detailed error messages to client (path disclosure, version info)
|
||||
- [ ] Debug logging enabled in production
|
||||
- [ ] Stack traces leaked to users
|
||||
- [ ] Timing side channels in authentication or comparison
|
||||
|
||||
## How to Scan
|
||||
|
||||
### Automated Pattern Search
|
||||
Run these searches across the codebase:
|
||||
|
||||
```bash
|
||||
# Buffer overflow risks
|
||||
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
|
||||
|
||||
# Fixed size stack buffers
|
||||
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
|
||||
|
||||
# Format string risks
|
||||
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
|
||||
|
||||
# Malloc without null check pattern
|
||||
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
|
||||
|
||||
# Integer overflow in allocation
|
||||
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
|
||||
|
||||
# Path construction
|
||||
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
|
||||
```
|
||||
|
||||
### Manual Code Review
|
||||
After automated scanning, manually review high-risk files:
|
||||
1. `src/shared/protocol.c` — all receive paths
|
||||
2. `src/shared/config.c` — deserialization logic
|
||||
3. `src/shared/chunk.c` — chunk parsing
|
||||
4. `src/shared/transport_tls.c` — TLS configuration
|
||||
5. `src/server/server.c` — file writing and connection handling
|
||||
|
||||
## Output Format
|
||||
|
||||
Return findings in this structured format, one per vulnerability:
|
||||
|
||||
```
|
||||
## Finding: <Short descriptive title>
|
||||
- **Severity**: critical/high/medium/low
|
||||
- **Category**: security
|
||||
- **Location**: file:line range
|
||||
- **Description**: what the vulnerability is, including:
|
||||
- How it can be triggered
|
||||
- What the impact is (RCE, DoS, info leak, etc.)
|
||||
- Whether it requires authentication
|
||||
- **Suggestion**: how to fix it, including concrete code changes
|
||||
- **Labels**: security, comma-separated additional labels
|
||||
```
|
||||
|
||||
### Example
|
||||
|
||||
```
|
||||
## Finding: Unchecked malloc in chunk deserialization allows OOM
|
||||
- **Severity**: high
|
||||
- **Category**: security
|
||||
- **Location**: src/shared/chunk.c:45-50
|
||||
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
|
||||
where `count` comes directly from the network. An attacker can send a crafted
|
||||
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
|
||||
to either fail (crash if unchecked) or allocate enormous memory (OOM).
|
||||
No authentication needed — the attack works on the initial connection.
|
||||
- **Suggestion**: Add bounds checking before allocation:
|
||||
```c
|
||||
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
|
||||
log_error("Invalid chunk file count: %u", count);
|
||||
return NULL;
|
||||
}
|
||||
```
|
||||
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
|
||||
- **Labels**: security, dos
|
||||
```
|
||||
|
||||
### No Findings
|
||||
If no security issues are found, return:
|
||||
```
|
||||
## No security findings
|
||||
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
|
||||
```
|
||||
|
||||
## Severity Guidelines
|
||||
|
||||
| Severity | Definition | Example |
|
||||
|---|---|---|
|
||||
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
|
||||
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
|
||||
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
|
||||
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
|
||||
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
|
||||
|
||||
## CI & Task Execution
|
||||
|
||||
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
|
||||
@@ -138,11 +138,9 @@ int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
|
||||
|
||||
Build for fuzzing:
|
||||
```bash
|
||||
cmake -B build-fuzz -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=fuzzer,address,undefined -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=fuzzer,address,undefined"
|
||||
CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
|
||||
cmake --build build-fuzz -j$(nproc)
|
||||
./build-fuzz/tests/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
./build-fuzz/fuzz_chunk_deserialize corpus/ -max_len=1048576
|
||||
```
|
||||
|
||||
### AFL++ Harness
|
||||
@@ -216,7 +214,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
|
||||
|
||||
## Branch Strategy
|
||||
|
||||
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
|
||||
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging.
|
||||
|
||||
## Dependency Installation
|
||||
|
||||
|
||||
@@ -38,16 +38,20 @@ dd if=/dev/urandom of=/tmp/fastsync_bench/src/large.bin bs=1M count=10 2>/dev/nu
|
||||
Test each configuration 3 times, record median:
|
||||
|
||||
```bash
|
||||
# Real FastSync flags: -z=compression, -j=multithreading,
|
||||
# --chunk-serialization, --sendfile (long form only). The old rsync-style
|
||||
# spellings -c/-m/-s/-f are NOT the same options (-c=--checksum,
|
||||
# -m=--prune-empty-dirs, -s=--secluded-args, -f=--filter) and must not be used.
|
||||
CONFIGS=(
|
||||
"Standard|"
|
||||
"Compression|-c"
|
||||
"Multithreading|-m"
|
||||
"MT+Compression|-m -c"
|
||||
"Chunk Serialization|-s"
|
||||
"MT+Compression+Chunk|-m -c -s"
|
||||
"Sendfile|-f"
|
||||
"Compression|-z"
|
||||
"Multithreading|-j"
|
||||
"MT+Compression|-j -z"
|
||||
"Chunk Serialization|-j -z --chunk-serialization"
|
||||
"Sendfile|--sendfile"
|
||||
)
|
||||
|
||||
PORT=18080
|
||||
for config in "${CONFIGS[@]}"; do
|
||||
IFS='|' read -r name flags <<< "$config"
|
||||
echo "=== $name ==="
|
||||
@@ -55,13 +59,14 @@ for config in "${CONFIGS[@]}"; do
|
||||
rm -rf /tmp/fastsync_bench/dst
|
||||
mkdir -p /tmp/fastsync_bench/dst
|
||||
|
||||
./build/server &
|
||||
./build/server -p "$PORT" --allow-unauthenticated &
|
||||
SERVER_PID=$!
|
||||
sleep 0.5
|
||||
|
||||
START=$(date +%s%N)
|
||||
./build/client --source-dir /tmp/fastsync_bench/src \
|
||||
--dest-dir /tmp/fastsync_bench/dst \
|
||||
--server-port "$PORT" \
|
||||
--save-to-disk $flags
|
||||
END=$(date +%s%N)
|
||||
|
||||
@@ -74,14 +79,21 @@ for config in "${CONFIGS[@]}"; do
|
||||
done
|
||||
```
|
||||
|
||||
### Step 4: Full Integration Benchmark (Optional)
|
||||
### Step 4: Full Benchmark Tool (Preferred)
|
||||
|
||||
The maintained benchmark tool is `benchmark/bench.py`. It handles building,
|
||||
data generation, network shaping (LAN/WAN profiles or custom `--delay`/`--jitter`/
|
||||
`--throughput`/`--loss`), rsync comparison, and JSON/table reporting:
|
||||
|
||||
For comprehensive benchmarking with network shaping:
|
||||
```bash
|
||||
python3 test.py --full
|
||||
python3 benchmark/bench.py --help
|
||||
python3 benchmark/bench.py --runs 5 --profiles unlimited
|
||||
python3 benchmark/bench.py --size-mb 100 --random-ratio 0.5 --output json
|
||||
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
|
||||
```
|
||||
|
||||
This tests LAN/WAN profiles, SSH, TLS, and compares against rsync.
|
||||
Network shaping needs root (`tc`/`netem` on `lo`). SSH and TLS coverage lives in
|
||||
the pytest integration suite, not the benchmark tool.
|
||||
|
||||
### Step 5: Report Results
|
||||
|
||||
@@ -93,12 +105,13 @@ Platform: <OS, CPU, network>
|
||||
Configuration | Run 1 | Run 2 | Run 3 | Median
|
||||
-----------------------|---------|---------|---------|--------
|
||||
Standard | 0.12s | 0.11s | 0.12s | 0.12s
|
||||
Compression (-c) | 0.09s | 0.08s | 0.09s | 0.09s
|
||||
Multithreading (-m) | 0.07s | 0.07s | 0.08s | 0.07s
|
||||
MT+Compression (-m -c) | 0.05s | 0.05s | 0.06s | 0.05s
|
||||
Sendfile (-f) | 0.04s | 0.04s | 0.04s | 0.04s
|
||||
Compression (-z) | 0.09s | 0.08s | 0.09s | 0.09s
|
||||
Multithreading (-j) | 0.07s | 0.07s | 0.08s | 0.07s
|
||||
MT+Compression (-j -z) | 0.05s | 0.05s | 0.06s | 0.05s
|
||||
Chunk Serialization (--chunk-serialization) | 0.05s | 0.04s | 0.05s | 0.05s
|
||||
Sendfile (--sendfile) | 0.04s | 0.04s | 0.04s | 0.04s
|
||||
|
||||
Best configuration: MT+Compression (-m -c)
|
||||
Best configuration: Sendfile (--sendfile)
|
||||
Throughput: <X> MB/s
|
||||
```
|
||||
|
||||
|
||||
@@ -32,23 +32,19 @@ Try to reproduce the issue with the exact command the user provides.
|
||||
|
||||
**Memory errors (first priority):**
|
||||
```bash
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
# or run the failing command
|
||||
```
|
||||
|
||||
**Thread errors:**
|
||||
```bash
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
|
||||
**Valgrind (if ASan doesn't find it):**
|
||||
@@ -108,13 +104,12 @@ cmake -B build -S . && cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
|
||||
# If integration test needed
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
|
||||
# Re-run under sanitizer to confirm fix
|
||||
rm -rf build
|
||||
cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
# reproduce the original failing command
|
||||
```
|
||||
|
||||
|
||||
@@ -19,7 +19,7 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log main..HEAD --oneline
|
||||
git log dev..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Clean build
|
||||
@@ -39,17 +39,13 @@ If the PR touches threading, memory management, or network code, also build with
|
||||
```bash
|
||||
# AddressSanitizer
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
|
||||
# ThreadSanitizer (if threading changes)
|
||||
rm -rf build-tsan
|
||||
cmake -B build-tsan -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
|
||||
cmake -B build-tsan -S . -DSANITIZER=thread
|
||||
cmake --build build-tsan -j$(nproc)
|
||||
./build-tsan/tests
|
||||
```
|
||||
@@ -91,10 +87,10 @@ If tests fail:
|
||||
### Step 6: Run integration tests (optional)
|
||||
|
||||
```bash
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
This runs the integration + benchmark suite. It takes longer — only run if the user asks or if unit tests pass.
|
||||
This runs the integration suite (benchmarking is `benchmark/bench.py`). It takes longer — only run if the user asks or if unit tests pass.
|
||||
|
||||
### Step 7: Fix and commit
|
||||
|
||||
|
||||
@@ -19,13 +19,13 @@ tea pr checkout <number>
|
||||
If already on a PR branch, verify with:
|
||||
```bash
|
||||
git branch --show-current
|
||||
git log main..HEAD --oneline
|
||||
git log dev..HEAD --oneline
|
||||
```
|
||||
|
||||
### Step 2: Get changed files
|
||||
|
||||
```bash
|
||||
git diff main --name-only -- '*.c' '*.h'
|
||||
git diff dev --name-only -- '*.c' '*.h'
|
||||
```
|
||||
|
||||
This gives the list of C source and header files changed in the PR.
|
||||
@@ -125,7 +125,7 @@ STYLE: <count>
|
||||
|
||||
If the user wants to post the review as a PR comment:
|
||||
```bash
|
||||
tea pr comment <number> --comment "<review report>"
|
||||
tea comment --repo TapTap/FastSync <number> "<review report>"
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -16,7 +16,7 @@ Ask the user or determine from context:
|
||||
- **Minor** (x.Y.0) — new features, backward compatible
|
||||
- **Patch** (x.y.Z) — bug fixes, no protocol changes
|
||||
|
||||
Current version: `PROTOCOL_VERSION "1.1.0"` in `src/shared/config.h`
|
||||
Current version: `PROTOCOL_VERSION "2.26.0"` in `src/shared/config.h`
|
||||
|
||||
### Step 2: Check Protocol Version
|
||||
|
||||
@@ -37,7 +37,7 @@ rm -rf build
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
python3 test.py
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
ALL tests must pass before release.
|
||||
@@ -46,12 +46,10 @@ ALL tests must pass before release.
|
||||
|
||||
```bash
|
||||
# ASan
|
||||
rm -rf build
|
||||
cmake -B build -S . \
|
||||
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
|
||||
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
|
||||
cmake --build build -j$(nproc)
|
||||
./build/tests
|
||||
rm -rf build-asan
|
||||
cmake -B build-asan -S . -DSANITIZER=address
|
||||
cmake --build build-asan -j$(nproc)
|
||||
./build-asan/tests
|
||||
```
|
||||
|
||||
### Step 5: Update README (If Needed)
|
||||
@@ -79,12 +77,23 @@ git commit -m "Release vX.Y.Z
|
||||
git tag -a vX.Y.Z -m "Release vX.Y.Z"
|
||||
```
|
||||
|
||||
### Step 8: Push
|
||||
### Step 8: Push and Open dev → main PR
|
||||
|
||||
`main` is protected and only receives changes via `dev` → `main` PRs (see AGENTS.md). Never push directly to `main`.
|
||||
|
||||
```bash
|
||||
git push origin main --tags
|
||||
# Push the release commit and tag to dev
|
||||
git push origin dev
|
||||
git push origin vX.Y.Z
|
||||
|
||||
# Open the dev → main release PR for review + CI
|
||||
tea pr create --repo TapTap/FastSync --head dev --base main \
|
||||
--title "Release vX.Y.Z" \
|
||||
--description "Release vX.Y.Z"
|
||||
```
|
||||
|
||||
Then wait for the full CI to pass and request review before the PR is merged to `main`.
|
||||
|
||||
### Step 9: Report
|
||||
|
||||
```
|
||||
|
||||
@@ -102,9 +102,9 @@ Informational: <count>
|
||||
...
|
||||
|
||||
=== VERDICT ===
|
||||
[PASS] No critical/high issues found
|
||||
[PASS] No critical/high-severity issues found
|
||||
— or —
|
||||
[FAIL] <N> critical/high issues must be fixed
|
||||
[FAIL] <N> critical/high-severity issues must be fixed
|
||||
```
|
||||
|
||||
## Rules
|
||||
|
||||
@@ -4,18 +4,19 @@ FastSync is a high-performance file synchronization system written in C11. It su
|
||||
|
||||
## Dependency installation
|
||||
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v10`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, and Node.js.
|
||||
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v11`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, Node.js, plus `rsync` 3.4.1 (with zstd/xxhash/lz4), `acl` and `attr` (setfacl/getfacl, setfattr/getfattr) for drop-in parity tests.
|
||||
|
||||
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
|
||||
|
||||
```bash
|
||||
# Use the prebuilt CI image directly (faster, guaranteed CI parity)
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v10
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v10 fastsync-ci:local
|
||||
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v11
|
||||
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v11 fastsync-ci:local
|
||||
|
||||
# Or build the image from the repo-root Dockerfile
|
||||
# (Note: the prebuilt :v10 image reflects the previous Dockerfile state;
|
||||
# rebuild from source to pick up any newly added packages like lcov/valgrind.)
|
||||
# (Note: the prebuilt :v11 image is built from the current Dockerfile and
|
||||
# includes rsync 3.4.1 plus acl/attr; rebuild from source after changing
|
||||
# the Dockerfile.)
|
||||
docker build -t fastsync-ci:local .
|
||||
|
||||
# Build, run unit tests, and run integration tests inside the container
|
||||
@@ -28,7 +29,7 @@ docker run --rm --user "$(id -u):$(id -g)" -v "$PWD:/workspace" \
|
||||
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load'
|
||||
```
|
||||
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash from GitHub via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
|
||||
|
||||
If a dependency is missing from the CI image, add it to the `Dockerfile` (and rebuild) rather than adding an install step to the CI workflow.
|
||||
|
||||
@@ -59,28 +60,28 @@ python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset o
|
||||
|
||||
## CI Workflow — Waiting for Results
|
||||
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Use `gh run watch` or similar to monitor CI status, then inspect logs on failure.
|
||||
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Monitor CI status via the Gitea API (see below) or `tea actions`, then inspect logs on failure.
|
||||
|
||||
## CI Troubleshooting
|
||||
|
||||
### If lint (clang-format) fails
|
||||
Run clang-format in the CI Docker image to match the exact CI version:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i'
|
||||
```
|
||||
|
||||
### If cppcheck fails
|
||||
Fix reported issues locally, then verify with:
|
||||
```bash
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \
|
||||
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 \
|
||||
sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/'
|
||||
```
|
||||
|
||||
### If integration tests fail
|
||||
Run locally before pushing:
|
||||
```bash
|
||||
python3 -m pytest tests/ -v --tb=short
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
## Branch Strategy
|
||||
@@ -165,7 +166,7 @@ This can be cron'd locally if desired (e.g., `crontab -e` with `opencode run`).
|
||||
## Is opencode a good option?
|
||||
|
||||
**Yes, for FastSync's needs.** The hybrid model works well:
|
||||
- opencode's 17 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- opencode's 16 specialized agents handle deep code analysis, fixes, tests, and reviews
|
||||
- The assistant orchestrates subagents, merges branches, and iterates on CI
|
||||
- You only review the final output
|
||||
|
||||
|
||||
+315
@@ -4,6 +4,321 @@ All notable changes to FastSync are documented here. Versions match
|
||||
`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must
|
||||
run the same version because the handshake is strict.
|
||||
|
||||
## [2.26.0] - 2026-09-17
|
||||
|
||||
### Added
|
||||
|
||||
- **Parity-completion wave.** Closed the remaining rsync-parity gaps against
|
||||
rsync 3.4.1 and reclassified the inherently non-rsync rows. It moved the wire
|
||||
protocol three times (`2.23.0 → 2.24.0 → 2.25.0 → 2.26.0`).
|
||||
- **Delete timing (2.24.0):** per-directory delete plans
|
||||
(`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`. An interrupted
|
||||
during-transfer has already removed the reached directories' extras, while a
|
||||
delayed transfer commits per directory only after the whole transfer
|
||||
succeeds (a late-created extra survives `--delete-delay` but not
|
||||
`--delete-after`). `-R --delete` is scoped to the transferred prefix; empty
|
||||
in-scope source directories survive; dry-run never deletes.
|
||||
- **Wire stats (2.25.0):** `STATUS_STATS` carries the receiver counters
|
||||
(matched data, deleted files) and the dry-run would-delete list. `--stats`
|
||||
prints rsync's protocol-independent lines; `--progress`/`-P` print per-file
|
||||
blocks; `--out-format` gains `%b` (wire bytes), `%c` (block-sum bytes) and
|
||||
`%C` (whole-file digest); `-n --delete` prints escaped `*deleting` lines in
|
||||
the sequential and `--threads` paths.
|
||||
- **Codecs (2.26.0):** `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/
|
||||
`none` checksums, with rsync-style `auto` negotiation (default `xxh128` +
|
||||
`zstd`) and exit-4 rejection of unknown names; the resolved `compression_algo`
|
||||
crosses the wire.
|
||||
- General `-R`/`--relative` (including the `/./` cut) and `--no-implied-dirs`;
|
||||
one-level `-d`/`--dirs` listing for `dir`, `dir/` and `.`; the full filter
|
||||
grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` and
|
||||
modifiers) with `-f` bound to `--filter`; a single `-F` transfers
|
||||
`.rsync-filter` and `-FF` excludes it.
|
||||
- Receiver-side `--chown`/`--usermap`/`--groupmap` TO-name resolution; absolute
|
||||
basis directories and a `--link-dest` relink of an up-to-date destination;
|
||||
a receiver-side `--ignore-existing` short-circuit before any payload;
|
||||
`--preallocate` now wins over `--sparse` via `fallocate(2)`.
|
||||
- Client quick wins: `--iconv=.`/`-`/`--no-iconv`, a lone `-h` prints help, an
|
||||
empty `--files-from` succeeds (exit 0), a broken referent under
|
||||
`-L`/`--copy-unsafe-links` exits 23, the full `--info`/`--debug`
|
||||
vocabularies, and the aliases `--ignore-non-existing`, `--protect-args`,
|
||||
`--msgs2stderr`.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.23.0 → 2.24.0` (delete plans),
|
||||
`2.24.0 → 2.25.0` (`STATUS_STATS` + `report_stats`), and
|
||||
`2.25.0 → 2.26.0` (codec negotiation + `md4`/`sha1`/`none`).
|
||||
- `--checksum-choice`/`--cc` now accepts `md4`, `sha1`, `none` and the two-name
|
||||
form; the negotiated whole-file default is `xxh128`.
|
||||
- `--compress-choice`/`--zc` now accepts `lz4`, `zlib`, `zlibx`.
|
||||
- `RSYNC_COMPAT.md` reclassifies the matrix: 9 already-parity rows to ✅, 17
|
||||
inherently non-rsync rows to ❌ (native daemon config/auth, batch, privileged
|
||||
xattr namespaces, and the safe-subset device/privilege flags), and the genuine
|
||||
fixes to ✅; new rows cover `--bwlimit`, `--partial`, `--partial-dir`,
|
||||
`--no-whole-file`, `--inc-recursive`/`--no-inc-recursive`, `--protect-args`
|
||||
and `--msgs2stderr`.
|
||||
- The client `--help` `--max-delete` text now describes the implemented partial
|
||||
semantics (delete up to N, skip the rest, exit 25).
|
||||
|
||||
### Notes
|
||||
|
||||
- Remaining documented divergences include the `--stats` per-type file-count
|
||||
breakdown, `%b`/`%c` being FastSync wire counts, `-n --delete` line ordering,
|
||||
the default `--delete` timing (delete-after, not rsync's delete-during),
|
||||
destination-only exclude protection (still sender-derived), `--temp-dir`
|
||||
absolute paths, basis-dir attribute re-application and the 256 MiB whole-file
|
||||
cap, `--fuzzy` tie-breaking, `--bwlimit=0`/decimal rates, `zlibx`==`zlib`, and
|
||||
recursive empty-directory creation.
|
||||
- Build: adds zlib and lz4 as link dependencies.
|
||||
|
||||
## [2.23.0] - 2026-09-16
|
||||
|
||||
### Added
|
||||
|
||||
- **Rsync-parity wave.** Closed the remaining CLI, filesystem, ownership,
|
||||
deletion, and output gaps against rsync 3.4.1.
|
||||
- Short options `-r` (`--recursive`), `-b` (`--backup`), `-L`
|
||||
(`--copy-links`), and `-B` (`--block-size`/`--delta-block`); rsync
|
||||
short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached/inline values
|
||||
(`--opt=value`, `-B1000`, `-essh`, `-MOPT`). A value that starts with `-`
|
||||
is not mistaken for a cluster.
|
||||
- `-c`/`--checksum` now implies the incremental checksum quick-check (and,
|
||||
like rsync, does not imply `-t`).
|
||||
- `--checksum-choice`/`--cc` accepts `xxh64`/`xxhash`/`xxh3`/`xxh128`/`md5`/
|
||||
`auto` and rejects `md4`/`sha1`/`none` and the two-name form by name;
|
||||
`--checksum-seed=0` (the default) is randomized per transfer and the chosen
|
||||
seed is sent to the receiver.
|
||||
- `--compress-choice`/`--zc` accepts `zstd`/`none`/`auto` and rejects
|
||||
`lz4`/`zlib`/`zlibx` by name; `--skip-compress` defaults to rsync 3.4.1's
|
||||
built-in suffix list; `--no-whole-file` is accepted.
|
||||
- `--timeout` defaults to 0 (disabled) and `--contimeout` to 60 s (both `0`
|
||||
disables), matching rsync; `--max-alloc=0` means no local limit.
|
||||
- `--temp-dir` is confined to the receive root (absolute/`..` rejected by the
|
||||
receiver) and an `EXDEV` install falls back to a non-atomic copy.
|
||||
- `--numeric-ids` is documented as a mapping modifier only;
|
||||
`--usermap`/`--groupmap` support inclusive `LOW-HIGH` ranges, `*`,
|
||||
empty-`FROM` (unnamed ids), and receiver-resolved `TO` names; `--chown`
|
||||
conflicts with a map on the same side are rejected.
|
||||
- `--fake-super` records the *resolved* owner (never a real chown) and replays
|
||||
mode/time; directory ownership and directory xattrs/ACLs are preserved.
|
||||
- `-l`/`--links` stores symlink targets verbatim (absolute and `..`-bearing
|
||||
included), matching rsync; `--safe-links`/`--copy-unsafe-links` are applied
|
||||
sender-side and `--munge-links` uses rsync's `/rsyncd-munged/` marker;
|
||||
`--trust-sender` no longer affects symlink targets.
|
||||
- `--specials` recreates unix sockets with `mknod(S_IFSOCK)` (so `-D` covers
|
||||
the full rsync node set).
|
||||
- Deletion: the manifest carries a synchronized-directory section so
|
||||
`--files-from` subsets no longer delete untransmitted paths;
|
||||
`--delete-excluded` leaves size-pruned mirrors protected; extraneous
|
||||
destination symlinks are unlinked (never followed); `--max-delete=N` is
|
||||
partial (delete up to N, skip the rest, exit 25) and `--delete-missing-args`
|
||||
removals draw from the same budget; `--force` is honored during
|
||||
`--delay-updates` publication.
|
||||
- `-x`/`--one-file-system` emits the mount-point directory entry; the
|
||||
`--include`/`--exclude` layers are an ordered first-match rule list.
|
||||
- `--chmod` is a faithful port of rsync 3.4.1 (numeric/symbolic, `D`/`F`/`X`,
|
||||
`s`/`t`, append semantics, no `-p` implication, no sanitization).
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.22.0 → 2.23.0`: the delete manifest gains a
|
||||
synchronized-directory section and the terminal status gains
|
||||
`STATUS_DELETE_LIMIT` (client exit 25 on a `--max-delete`-capped commit).
|
||||
- **The 2.22.0 mode-masking divergence is removed.** Under `-p` the source mode
|
||||
is copied exactly, including `S_IWGRP`/`S_IWOTH` and setuid/setgid/sticky;
|
||||
`--chmod` no longer implies `-p`. New files without `-p` still use
|
||||
`source_mode & ~umask` when metadata is present (else `0644`), and new
|
||||
directories without `-p` still use the `0755` creation default.
|
||||
- `--protocol=NUM` accepts only the current `2.23.0` version string.
|
||||
|
||||
### Notes
|
||||
|
||||
- The rsync-compatibility matrix (`RSYNC_COMPAT.md`) now classifies every row
|
||||
as **parity**, **caveat** (works with a documented divergence), or
|
||||
**divergent** (not supported/no-op/impossible), replacing the previous
|
||||
misleading "N implemented / 0 divergence" summary. Durable documented
|
||||
divergences remain: receiver-side symlink target containment is not enforced
|
||||
by default (verbatim storage is rsync parity; use `--safe-links`),
|
||||
`--temp-dir` rejects absolute/foreign-filesystem paths, `--copy-devices`
|
||||
reads a bounded `st_size`, a broken referent under `--copy-links` exits 0,
|
||||
new directories without `-p` use `0755`, `--stats` receiver-only counters are
|
||||
0, and `--password-file`/`--early-input`/`--hash-credentials`/`--iterations`
|
||||
and the batch format are FastSync-native.
|
||||
|
||||
## [2.22.0] - 2026-09-15
|
||||
|
||||
### Added
|
||||
|
||||
- **Per-attribute metadata preservation (protocol 2.22.0).** The former single
|
||||
metadata bundle is split into four independent, rsync-compatible flags:
|
||||
`-p/--perms`, `-t/--times`, `-o/--owner`, and `-g/--group`, each applied
|
||||
independently on the receiver, with negations `--no-perms`/`--no-times`/
|
||||
`--no-owner`/`--no-group` (short `--no-p`/`--no-t`/`--no-o`/`--no-g`) and
|
||||
`--no-preserve` clearing all four. `-a/--archive` is now full rsync
|
||||
`-rlptgoD` (owner and group included; their application stays
|
||||
privilege-gated). `-A/--acls` and `--chmod` imply `-p`, `-X/--xattrs` does
|
||||
not, `-E/--executability` sets only executability, and `-U`/`-N` do not imply
|
||||
`-t`. `--incremental`/`--delta` still auto-preserve perms+times unless the
|
||||
user explicitly negated them.
|
||||
- Receiver applies directory modes under `-p` (at the end of the transfer,
|
||||
alongside the deferred directory times) and symlink mode under `-p`; `-O`
|
||||
suppresses directory times only.
|
||||
|
||||
### Changed
|
||||
|
||||
- `PROTOCOL_VERSION` bumped `2.21.0 → 2.22.0`: the binary config frame gains
|
||||
four appended booleans (`preserve_perms`/`preserve_times`/`preserve_owner`/
|
||||
`preserve_group`) after `omit_link_times`. The fixed-width `FileMetadata`
|
||||
layout is unchanged; the receiver derives the metadata-frame gate
|
||||
(`use_metadata`) from the four attributes.
|
||||
|
||||
### Notes
|
||||
|
||||
- Documented divergences from rsync: a client-supplied mode never grants
|
||||
group/other write (`S_IWGRP|S_IWOTH` are stripped for files, directories,
|
||||
symlinks, and specials; rsync's `-p` preserves them exactly); a brand-new file
|
||||
without `-p` gets `source_mode & ~umask` (sanitized) when metadata is present,
|
||||
else the historical fixed `0644`; `--chmod` implies `-p` (rsync does not);
|
||||
`-o`/`-g` map by name on the receiver with a raw-numeric fallback (only
|
||||
numeric ids cross the wire); and a daemon module without `client owner = yes`
|
||||
does not refuse a plain `-a`/`-o`/`-g` but forces super-user activities off,
|
||||
applies no ownership, and logs a warning (explicit `--chown`/`--usermap`/
|
||||
`--groupmap`/`--numeric-ids`/`--copy-as`/`--super` are still refused).
|
||||
|
||||
## [2.21.0] - 2026-09-14
|
||||
|
||||
### Added
|
||||
|
||||
- Optional server→client rejection detail (protocol 2.21.0). A rejected
|
||||
operation may now carry a bounded human-readable reason via
|
||||
`STATUS_ERROR_DETAIL` instead of a bare `STATUS_ERROR`, so the client can
|
||||
report *why* the server refused (daemon module gate, config validation,
|
||||
receiver-side path/node validation). `receive_status()` transparently maps the
|
||||
new status back to `STATUS_ERROR` for every existing call site and captures
|
||||
the reason into a thread-local buffer exposed by `protocol_last_error()`. The
|
||||
detail body is always consumed, so the stream cannot desynchronize, and
|
||||
messages are sliced to `MAX_ERROR_DETAIL_BYTES` (4096) on send.
|
||||
- **Server-contacting `--dry-run` (protocol 2.21.0).** `--dry-run` now performs
|
||||
a real handshake with a remote/daemon receiver and reports exactly what WOULD
|
||||
change based on receiver state (existing destination files, mtimes, checksums,
|
||||
basis dirs). The wire config carries the dry-run intent (`Config.dry_run`) and
|
||||
the receiver answers each per-file check with `STATUS_DRY_RUN_TRANSFER` (would
|
||||
transfer) or `STATUS_OK` (already up to date); the sender prints the
|
||||
would-transfer set and its trailer without sending any file data. The receiver
|
||||
performs the normal read-only incremental decision but mutates nothing: no temp
|
||||
files, writes, renames, deletes, metadata/xattr/chown, or directory creation.
|
||||
A plain local destination (no explicit `--server-port`/remote) keeps the
|
||||
original client-side dry-run. Would-delete reporting for `--delete*` is
|
||||
deferred to a follow-up; dry-run never deletes.
|
||||
- Daemon `max connections per host` (per-source-IP concurrent cap, default 0 =
|
||||
unlimited), `auth lockout threshold` (default 10; 0 disables) and
|
||||
`auth lockout duration` (default 300 s) config keys.
|
||||
- `fastsync-server --allow-super` opt-in for a privileged standalone TCP server;
|
||||
without it a root standalone receiver forces super-user activities off (device
|
||||
nodes, `--write-devices`, ownership). The `--stdio` SSH argv is client-composed,
|
||||
so super activities always stay off there.
|
||||
|
||||
### Changed
|
||||
|
||||
- Config wire fields are now declared once in an X-macro table
|
||||
(`CONFIG_WIRE_FIELDS` in `src/shared/config.h`) that generates the struct
|
||||
members, defaults, and the send/receive sequence, removing the manual
|
||||
six-site field sync. Wire bytes and `PROTOCOL_VERSION` are unchanged.
|
||||
- `receive_incremental_check()` (the per-file `STATUS_CHECK` fast path) is split
|
||||
into small static helpers with a short linear orchestrator. Pure refactor: the
|
||||
wire byte stream and all cleanup are unchanged.
|
||||
- `authorized_root` state has a single owner (`utils.c`) with read accessors; the
|
||||
duplicated statics in `file.c` and the server were removed.
|
||||
- `Data` records its owning `ProtocolSession` so its memory charge is returned to
|
||||
the session that reserved it, regardless of the destroying thread.
|
||||
- The receiver pipeline moved out of `shared` into `server/receiver_pipeline.[ch]`;
|
||||
the build now uses explicit `fastsync_shared` / `fastsync_client_core` /
|
||||
`fastsync_server_core` targets instead of a GLOB, and the client no longer links
|
||||
server code.
|
||||
- The benchmark tool generates the requested random/compressible data mix
|
||||
accurately, verifies each transfer before recording it, computes correct
|
||||
percentiles, adds a MB/s column, handles `tc`/netem without requiring `sudo`
|
||||
when already root, builds into a dedicated `build-bench/` directory, and adds a
|
||||
`--warm` incremental-transfer mode.
|
||||
- The `nix-shell` dev environment provides the full toolchain (clang-format,
|
||||
cppcheck, pytest-xdist, OpenSSH, rsync, iproute2, valgrind, lcov) and no longer
|
||||
builds on entry.
|
||||
|
||||
### Security
|
||||
|
||||
- Enforce the daemon's per-module `max connections` cap (0 = unlimited) and add
|
||||
the shared per-source `max connections per host` cap plus a cross-process
|
||||
`auth lockout`. Because the listener forks one child per connection, the
|
||||
counters live in an anonymous shared mapping created before the accept loop and
|
||||
reclaimed by the parent's `SIGCHLD` handler, so the per-module, per-source and
|
||||
auth-failure state is shared across every child (including after `SIGKILL`). The
|
||||
per-source table has a bounded lifetime (expired/idle entries are reclaimed,
|
||||
with a rate-limited warning when genuinely full), and the occupancy counters are
|
||||
re-derived from the shared slot table on every child exit. Trusted loopback
|
||||
peers are exempt (they share one address); clients behind a shared NAT/proxy
|
||||
share a single per-host budget and lockout, which is documented.
|
||||
- Hardening from a full security audit:
|
||||
- Fail a truncated zstd frame instead of spinning forever (remote DoS).
|
||||
- Open receiver destination/basis/hard-link entries `O_NONBLOCK` so a
|
||||
client-planted FIFO cannot block a worker indefinitely.
|
||||
- Require a regular file before `--inplace` writes, closing a FIFO-hang and a
|
||||
raw-device write that bypassed the `--write-devices` gate.
|
||||
- Reject SSH destinations whose user/host begins with `-` and insert `--` before
|
||||
the host token, closing `-o ProxyCommand=…` argument injection (RCE).
|
||||
- Gate client `--force` recursive removal behind the server `--allow-delete`
|
||||
policy.
|
||||
- Reject empty `hosts allow`/`hosts deny`/`auth users` values instead of
|
||||
silently meaning "unrestricted".
|
||||
- Restrict TLS 1.2 to AEAD suites and set server cipher preference; load the
|
||||
private key TOCTOU-safely from an `O_NOFOLLOW` fd; verify IP literals against
|
||||
IP SANs; guard client-cert CN truncation.
|
||||
- Make `--dry-run` content-blind: it neither reads destination files nor
|
||||
hashes basis files, removing a 1-bit content oracle against `read only`
|
||||
modules.
|
||||
- Bound glob matching (iterative DP, no exponential backtracking) and bound
|
||||
line reads for filter/`--files-from`/pattern files.
|
||||
- Gate `system.posix_acl_*` xattrs on `--acls` and charge decompression/chunk
|
||||
allocations against the per-connection memory budget.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Pre-auth NULL dereference in `config_delete()` when an over-long
|
||||
`basis_count` (and the analogous count fields) was received and then failed
|
||||
validation; received counts are now validated before being published.
|
||||
- Leaked inherited `Data` in the forked compression-truncation unit test
|
||||
(valgrind definite leak).
|
||||
- `receive_status()` no longer loses a captured rejection reason when owed
|
||||
keepalives are drained.
|
||||
|
||||
## [2.20.0] - 2026-09-13
|
||||
|
||||
### Security
|
||||
|
||||
- Cap cumulative `DirTimeList` growth and bound pre-auth config-string memory
|
||||
(remote memory-exhaustion DoS).
|
||||
- Daemon host access control (`hosts allow`/`hosts deny`, IPv4/IPv6/CIDR),
|
||||
configurable global `max connections`, connection audit logging, and a
|
||||
bounded `auth failure delay` throttle. IPv4-mapped peers are normalized and
|
||||
invalid patterns are rejected at parse time (no silent fail-open).
|
||||
- Honor `--timeout` for protocol I/O and bound idle/session time to defeat
|
||||
keepalive slowloris; child-safe signal handling in the forked daemon.
|
||||
- Compiler/linker hardening (`_FORTIFY_SOURCE`, stack protector, PIE, RELRO)
|
||||
and pinned build dependencies.
|
||||
|
||||
### Fixed
|
||||
|
||||
- Use-after-free in the basis-dir oversize preflight.
|
||||
- Placeholder `Data` leaks, `missing_args` leak, scanner chunk leak.
|
||||
- Thread-safe logging; single fd owner and cleanup epilogue in the server
|
||||
handler.
|
||||
|
||||
### Performance
|
||||
|
||||
- Metadata now crosses the wire as one packed frame (protocol 2.20.0).
|
||||
- Delete keep-set and `--files-from` lookups indexed (O(n*m) → O(n)).
|
||||
- Reused per-thread zstd contexts; `TCP_NODELAY` by default.
|
||||
- Byte-bounded sender queues; removed a redundant scanner `stat()`.
|
||||
|
||||
## [2.19.0] - 2026-09-12
|
||||
|
||||
### Security
|
||||
|
||||
+212
-26
@@ -1,6 +1,6 @@
|
||||
cmake_minimum_required(VERSION 3.22)
|
||||
|
||||
project(FastFileTransfer VERSION 2.19.0)
|
||||
project(FastFileTransfer VERSION 2.26.0)
|
||||
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
set(CMAKE_C_STANDARD 11)
|
||||
@@ -38,11 +38,24 @@ if(ENABLE_COVERAGE)
|
||||
add_link_options(--coverage)
|
||||
endif()
|
||||
|
||||
# --- Build hardening option ---
|
||||
# Production hardening is applied to the shipping server/client binaries only,
|
||||
# and only when no sanitizer or coverage instrumentation is active: sanitizers
|
||||
# carry their own instrumentation, and _FORTIFY_SOURCE requires an optimising
|
||||
# build (never the -O0 used for coverage).
|
||||
option(ENABLE_HARDENING "Enable compiler/linker hardening for production targets" ON)
|
||||
set(HARDENING_ACTIVE OFF)
|
||||
if(ENABLE_HARDENING AND SANITIZER STREQUAL "none" AND NOT ENABLE_COVERAGE)
|
||||
set(HARDENING_ACTIVE ON)
|
||||
endif()
|
||||
|
||||
include(FetchContent)
|
||||
FetchContent_Declare(
|
||||
xxhash
|
||||
GIT_REPOSITORY https://github.com/Cyan4973/xxHash
|
||||
GIT_TAG v0.8.3
|
||||
# v0.8.3 is a lightweight tag pointing at this exact commit (no ^{} peel
|
||||
# entry); pin the commit SHA instead of the mutable tag.
|
||||
GIT_TAG e626a72bc2321cd320e953a0ccf1584cad60f363 # v0.8.3
|
||||
SOURCE_SUBDIR cmake_unofficial
|
||||
)
|
||||
FetchContent_MakeAvailable(xxhash)
|
||||
@@ -55,37 +68,194 @@ if(NOT ZSTD_LIBRARY)
|
||||
message(FATAL_ERROR "zstd library not found. Ensure it is in your nix-shell!")
|
||||
endif()
|
||||
|
||||
find_library(ZLIB_LIBRARY z)
|
||||
if(NOT ZLIB_LIBRARY)
|
||||
message(FATAL_ERROR "zlib library not found. Ensure zlib1g-dev / nix zlib is available!")
|
||||
endif()
|
||||
|
||||
find_library(LZ4_LIBRARY lz4)
|
||||
if(NOT LZ4_LIBRARY)
|
||||
message(FATAL_ERROR "lz4 library not found. Ensure liblz4-dev / nix lz4 is available!")
|
||||
endif()
|
||||
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
file(GLOB SHARED_SRCS "src/shared/*.c")
|
||||
set(FILE_STORE_SRCS "${CMAKE_CURRENT_SOURCE_DIR}/src/shared/file_store.c")
|
||||
list(REMOVE_ITEM SHARED_SRCS ${FILE_STORE_SRCS})
|
||||
file(GLOB SERVER_SRCS "src/server/*.c")
|
||||
set(SERVER_RECEIVER_SRCS src/server/receiver.c)
|
||||
file(GLOB CLIENT_SRCS "src/client/*.c")
|
||||
# --- Explicit source lists ---
|
||||
# The shared library is self-contained: it must never depend on the client or
|
||||
# server modules. In particular, the receiver pipeline (receive_thread /
|
||||
# write_thread) lives under src/server, not here, so the client executable can
|
||||
# link the shared library without pulling in any server code.
|
||||
set(SHARED_SRCS
|
||||
src/shared/array_list.c
|
||||
src/shared/batch.c
|
||||
src/shared/charset.c
|
||||
src/shared/checksum.c
|
||||
src/shared/chmod.c
|
||||
src/shared/chunk.c
|
||||
src/shared/compression.c
|
||||
src/shared/config.c
|
||||
src/shared/credentials.c
|
||||
src/shared/daemon_conf.c
|
||||
src/shared/daemon_limits.c
|
||||
src/shared/data.c
|
||||
src/shared/delay_updates.c
|
||||
src/shared/delete_plan.c
|
||||
src/shared/delta.c
|
||||
src/shared/file.c
|
||||
src/shared/file_list.c
|
||||
src/shared/file_receive.c
|
||||
src/shared/file_send.c
|
||||
src/shared/file_store.c
|
||||
src/shared/filter.c
|
||||
src/shared/format.c
|
||||
src/shared/hardlink.c
|
||||
src/shared/identity.c
|
||||
src/shared/log.c
|
||||
src/shared/metadata.c
|
||||
src/shared/motd.c
|
||||
src/shared/multiprocessing.c
|
||||
src/shared/protocol.c
|
||||
src/shared/queue.c
|
||||
src/shared/stop_condition.c
|
||||
src/shared/transport_ssh.c
|
||||
src/shared/transport_tcp.c
|
||||
src/shared/transport_tls.c
|
||||
src/shared/utils.c
|
||||
src/shared/xattr.c
|
||||
)
|
||||
|
||||
# Server implementation (no main): the receiver read/write pipeline plus the
|
||||
# CLI parser. The server executable adds its own main (server.c).
|
||||
set(SERVER_CORE_SRCS
|
||||
src/server/receiver.c
|
||||
src/server/receiver_pipeline.c
|
||||
src/server/server_cli.c
|
||||
)
|
||||
set(SERVER_MAIN_SRCS src/server/server.c)
|
||||
|
||||
# Client implementation (no main): everything except the CLI entry point.
|
||||
set(CLIENT_CORE_SRCS
|
||||
src/client/change_list.c
|
||||
src/client/client_send.c
|
||||
src/client/client_validation.c
|
||||
src/client/scanner.c
|
||||
src/client/usage.c
|
||||
)
|
||||
set(CLIENT_MAIN_SRCS src/client/client_cli.c)
|
||||
|
||||
# --- Library targets ---
|
||||
add_library(fastsync_shared STATIC ${SHARED_SRCS})
|
||||
target_include_directories(fastsync_shared PUBLIC src/shared)
|
||||
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
|
||||
add_library(fastsync_client_core STATIC ${CLIENT_CORE_SRCS})
|
||||
target_include_directories(fastsync_client_core PUBLIC src/client)
|
||||
target_link_libraries(fastsync_client_core PUBLIC fastsync_shared)
|
||||
|
||||
add_library(fastsync_server_core STATIC ${SERVER_CORE_SRCS})
|
||||
target_include_directories(fastsync_server_core PUBLIC src/server)
|
||||
target_link_libraries(fastsync_server_core PUBLIC fastsync_shared)
|
||||
|
||||
# --- Main executables ---
|
||||
add_executable(server ${SERVER_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS})
|
||||
target_include_directories(server PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
# The client links only the shared library and its own core; it deliberately
|
||||
# does NOT get src/server on its include path nor compile receiver.c.
|
||||
add_executable(server ${SERVER_MAIN_SRCS})
|
||||
target_link_libraries(server PRIVATE fastsync_server_core)
|
||||
|
||||
add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
|
||||
target_include_directories(client PRIVATE src/shared src/server src/client)
|
||||
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
add_executable(client ${CLIENT_MAIN_SRCS})
|
||||
target_link_libraries(client PRIVATE fastsync_client_core)
|
||||
|
||||
# --- Production hardening ---
|
||||
# Each compile flag is probed so a compiler/architecture that lacks it still
|
||||
# configures cleanly. _FORTIFY_SOURCE is guarded separately because it only
|
||||
# works in an optimising build. xxHash is a static archive built by
|
||||
# FetchContent, so it must be position-independent for the -pie link; the same
|
||||
# applies to the first-party static libraries linked into the -pie binaries.
|
||||
if(HARDENING_ACTIVE)
|
||||
set_target_properties(xxhash fastsync_shared fastsync_server_core fastsync_client_core
|
||||
PROPERTIES POSITION_INDEPENDENT_CODE ON)
|
||||
include(CheckCCompilerFlag)
|
||||
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
|
||||
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
|
||||
check_c_compiler_flag("${flag}" ${_harden_var})
|
||||
endforeach()
|
||||
check_c_compiler_flag("-D_FORTIFY_SOURCE=2" HARDEN_FORTIFY_SOURCE)
|
||||
foreach(target fastsync_shared fastsync_server_core fastsync_client_core server client)
|
||||
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
|
||||
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
|
||||
if(${_harden_var})
|
||||
target_compile_options(${target} PRIVATE ${flag})
|
||||
endif()
|
||||
endforeach()
|
||||
if(HARDEN_FORTIFY_SOURCE)
|
||||
target_compile_options(${target} PRIVATE -D_FORTIFY_SOURCE=2)
|
||||
endif()
|
||||
endforeach()
|
||||
foreach(target server client)
|
||||
target_link_options(${target} PRIVATE -pie -Wl,-z,relro -Wl,-z,now -Wl,-z,noexecstack)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
# --- Testing ---
|
||||
enable_testing()
|
||||
|
||||
# Common test libraries
|
||||
set(TEST_LIBS Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
set(TEST_INCLUDES tests src/shared src/server src/client)
|
||||
# --- Unit tests ---
|
||||
# The monolithic test binary exercises both client and server code, so it is
|
||||
# the one place that legitimately sees both include directories and links both
|
||||
# core libraries. client_cli.c is compiled here directly (with the test build
|
||||
# define) rather than linked from fastsync_client_core so its test-only shims
|
||||
# and the absence of main() are preserved.
|
||||
set(TEST_SRCS
|
||||
tests/runner.c
|
||||
tests/test_array_list.c
|
||||
tests/test_batch.c
|
||||
tests/test_change_list.c
|
||||
tests/test_checksum.c
|
||||
tests/test_chunk.c
|
||||
tests/test_client_cli.c
|
||||
tests/test_compression.c
|
||||
tests/test_config.c
|
||||
tests/test_credentials.c
|
||||
tests/test_daemon_conf.c
|
||||
tests/test_daemon_limits.c
|
||||
tests/test_data.c
|
||||
tests/test_delay_updates.c
|
||||
tests/test_delta.c
|
||||
tests/test_file.c
|
||||
tests/test_file_list.c
|
||||
tests/test_file_sendfile.c
|
||||
tests/test_format.c
|
||||
tests/test_fuzz_smoke.c
|
||||
tests/test_glob.c
|
||||
tests/test_hardlink.c
|
||||
tests/test_iconv.c
|
||||
tests/test_log.c
|
||||
tests/test_metadata.c
|
||||
tests/test_motd.c
|
||||
tests/test_multiprocessing.c
|
||||
tests/test_property.c
|
||||
tests/test_protocol.c
|
||||
tests/test_protocol_error.c
|
||||
tests/test_queue.c
|
||||
tests/test_receiver_timeout.c
|
||||
tests/test_robustness.c
|
||||
tests/test_scanner.c
|
||||
tests/test_server.c
|
||||
tests/test_server_cli.c
|
||||
tests/test_shared_utils.c
|
||||
tests/test_stop.c
|
||||
tests/test_stress.c
|
||||
tests/test_transport_ssh.c
|
||||
tests/test_transport_tcp.c
|
||||
tests/test_transport_tls.c
|
||||
tests/test_xattr.c
|
||||
)
|
||||
|
||||
# Monolithic test binary (backward compatible)
|
||||
file(GLOB TEST_SRCS "tests/test_*.c" "tests/runner.c")
|
||||
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS} src/client/scanner.c src/client/change_list.c src/client/client_cli.c src/client/client_validation.c src/client/usage.c src/server/server_cli.c)
|
||||
target_include_directories(tests PRIVATE ${TEST_INCLUDES})
|
||||
add_executable(tests ${TEST_SRCS} src/client/client_cli.c)
|
||||
target_include_directories(tests PRIVATE tests)
|
||||
target_compile_definitions(tests PRIVATE FASTSYNC_TEST_BUILD)
|
||||
target_link_libraries(tests PRIVATE ${TEST_LIBS})
|
||||
target_link_libraries(tests PRIVATE fastsync_server_core fastsync_client_core)
|
||||
add_test(NAME unit_all COMMAND tests)
|
||||
|
||||
# --- Fuzz targets (requires clang) ---
|
||||
@@ -94,13 +264,29 @@ if(ENABLE_FUZZ)
|
||||
if(NOT CMAKE_C_COMPILER_ID MATCHES "Clang")
|
||||
message(FATAL_ERROR "ENABLE_FUZZ requires Clang (compiler is ${CMAKE_C_COMPILER_ID})")
|
||||
endif()
|
||||
file(GLOB FUZZ_SRCS "tests/fuzz/*.c")
|
||||
set(FUZZ_SRCS
|
||||
tests/fuzz/fuzz_chunk_deserialize.c
|
||||
tests/fuzz/fuzz_compress_decompress.c
|
||||
tests/fuzz/fuzz_config_receive.c
|
||||
tests/fuzz/fuzz_delta_deserialize.c
|
||||
tests/fuzz/fuzz_delta_signature_deserialize.c
|
||||
tests/fuzz/fuzz_glob_match.c
|
||||
tests/fuzz/fuzz_identity_parse.c
|
||||
tests/fuzz/fuzz_manifest.c
|
||||
tests/fuzz/fuzz_metadata_from_buf.c
|
||||
tests/fuzz/fuzz_protocol_framing.c
|
||||
tests/fuzz/fuzz_xattr_block.c
|
||||
)
|
||||
# Compile the sources under test directly so libFuzzer's coverage
|
||||
# instrumentation sees them (static libraries would be uninstrumented).
|
||||
set(FUZZ_CORE_SRCS ${SHARED_SRCS} src/server/receiver.c src/server/receiver_pipeline.c)
|
||||
foreach(FUZZ_SRC ${FUZZ_SRCS})
|
||||
get_filename_component(FUZZ_NAME ${FUZZ_SRC} NAME_WE)
|
||||
add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE ${TEST_INCLUDES})
|
||||
add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${FUZZ_CORE_SRCS})
|
||||
target_include_directories(${FUZZ_NAME} PRIVATE tests src/shared src/server)
|
||||
target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer)
|
||||
target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined)
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE ${TEST_LIBS})
|
||||
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} ${ZLIB_LIBRARY}
|
||||
${LZ4_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
+15
-1
@@ -2,8 +2,22 @@ FROM ubuntu:24.04
|
||||
RUN apt-get update && apt-get install -y --no-install-recommends \
|
||||
gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \
|
||||
python3 python3-pip python3-venv openssl openssh-client \
|
||||
lcov valgrind clang libclang-rt-18-dev && \
|
||||
lcov valgrind clang libclang-rt-18-dev \
|
||||
acl attr zlib1g-dev liblz4-dev libxxhash-dev && \
|
||||
pip3 install --break-system-packages pytest pytest-xdist && \
|
||||
curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \
|
||||
apt-get install -y --no-install-recommends nodejs && \
|
||||
rm -rf /var/lib/apt/lists/*
|
||||
|
||||
# rsync is used as the reference implementation for drop-in parity tests.
|
||||
# Ubuntu 24.04 ships 3.2.7, so build the pinned 3.4.1 reference from source.
|
||||
ARG RSYNC_VERSION=3.4.1
|
||||
ARG RSYNC_SHA256=2924bcb3a1ed8b551fc101f740b9f0fe0a202b115027647cf69850d65fd88c52
|
||||
RUN curl -fsSL "https://download.samba.org/pub/rsync/src/rsync-${RSYNC_VERSION}.tar.gz" -o /tmp/rsync.tar.gz && \
|
||||
echo "${RSYNC_SHA256} /tmp/rsync.tar.gz" | sha256sum -c - && \
|
||||
tar -xzf /tmp/rsync.tar.gz -C /tmp && \
|
||||
cd "/tmp/rsync-${RSYNC_VERSION}" && \
|
||||
./configure --enable-zstd --enable-xxhash --enable-lz4 && \
|
||||
make -j"$(nproc)" && \
|
||||
make install && \
|
||||
rm -rf "/tmp/rsync-${RSYNC_VERSION}" /tmp/rsync.tar.gz
|
||||
|
||||
+67
@@ -0,0 +1,67 @@
|
||||
# FastSync — Session Handoff (2026-09-17)
|
||||
|
||||
## Current status
|
||||
- **Release `v2.21.0`** tagged (`919a729`, "Release v2.21.0"); full CI green
|
||||
(run 552: lint, build-and-test, ASan, UBSan, fuzz-build, coverage, valgrind).
|
||||
`dev` has the release commit plus later doc-only merges (a README refresh and
|
||||
this handoff).
|
||||
- **Release PR #284 (`dev` -> `main`)** open, CI green (run 553).
|
||||
`main` is protected: it needs review/approval to merge.
|
||||
https://gitea.tap-tap.win/TapTap/FastSync/pulls/284
|
||||
- **`PROTOCOL_VERSION` = `"2.26.0"`** (`src/shared/config.h`); CMake
|
||||
`project(FastFileTransfer VERSION 2.26.0)`.
|
||||
- Working tree clean; no wave worktrees remain.
|
||||
|
||||
## What landed this session
|
||||
1. **Wave 8 (refactors):** Config X-macro wire table; single-owner `authorized_root`;
|
||||
daemon per-module/per-host caps + cross-process auth lockout (`daemon_limits.[ch]`);
|
||||
`Data` charge returns to its owning `ProtocolSession`.
|
||||
2. **Wave 9 (protocol 2.21.0):** optional `STATUS_ERROR_DETAIL` rejection reasons;
|
||||
server-contacting `--dry-run` (`STATUS_DRY_RUN_TRANSFER`, receiver mutates nothing).
|
||||
3. **Security wave:** ran 5 parallel audits (wire parsing; daemon/transport/TLS/auth;
|
||||
receiver confinement; client/CLI/SSH; crypto/memory/limits). Fixed all HIGH and the
|
||||
confirmed MEDIUMs:
|
||||
- SSH `-o ProxyCommand=…` argument injection (RCE) — reject leading `-`, insert `--`.
|
||||
- Truncated zstd frame infinite CPU loop (remote DoS).
|
||||
- FIFO receiver opens lacked `O_NONBLOCK` (indefinite hang).
|
||||
- `--inplace` could write a FIFO/device (bypass of `--write-devices` gate).
|
||||
- `--force` not gated by server `--allow-delete`.
|
||||
- Privileged standalone server defaulted super activities on; added `--allow-super`
|
||||
(never honored with `--stdio`).
|
||||
- `--dry-run` content/hash oracle on `read only`/basis files removed.
|
||||
- Empty `hosts allow`/`deny`/`auth users` now rejected.
|
||||
- TLS: AEAD-only 1.2 + server preference, TOCTOU-safe key load, IP-SAN verify,
|
||||
CN-truncation guard. Glob backtracking bounded; line reads bounded; ACL xattrs
|
||||
gated on `--acls`; decompression/chunk memory charged; pre-auth `basis_count`
|
||||
NULL-deref fixed.
|
||||
4. **Tooling:** benchmark accuracy (data mix, verification, percentiles, `tc`,
|
||||
`build-bench/`, `--warm` mode); `shell.nix` full toolchain and no build-on-entry;
|
||||
docs state push-only / remote-source unsupported.
|
||||
5. **Preserve-attribute split (protocol 2.22.0)** landed on `feat/preserve-attr-split`: per-attribute `-p/-t/-o/-g` + `--no-*` negations, `-a` = `-rlptgoD`, and the 2.21.0 → 2.22.0 wire bump.
|
||||
6. **Rsync-parity wave (protocol 2.23.0)** on `feat/rsync-parity`: rsync short options/clustering/attached values (`-r`/`-b`/`-L`/`-B`, `-av`, `-aAX`, `-B1000`, `-essh`, `-MOPT`), `-c` checksum quick-check, `--checksum-choice`/`--compress-choice` validation and seed randomization, rsync timeout/max-alloc defaults, temp-dir confinement + `EXDEV` fallback, ownership/mapping parity (numeric-ids modifier, map ranges/`*`/empty-FROM, `--chown`+map conflicts, fake-super resolved-owner record), verbatim symlink storage with rsync `--safe-links`/`--munge-links`, socket recreation under `--specials`, `--chmod` 3.4.1 semantics, and delete scoping + `--max-delete` partial/exit-25. Wire: appended delete-manifest synchronized-directory section and `STATUS_DELETE_LIMIT`.
|
||||
7. **Parity-completion wave (protocol 2.24.0 → 2.26.0)** on `feat/parity-completion`: per-directory delete plans (`STATUS_DELETE_PLAN`) for `--delete-during`/`--delete-delay`; receiver `STATUS_STATS` counters feeding `--stats`/`--progress` and `--out-format %b/%c/%C`, plus `-n --delete` lines; `lz4`/`zlib`/`zlibx` compression and `md4`/`sha1`/`none` checksums with `auto` negotiation (default `xxh128`/`zstd`); general `-R`/`--no-implied-dirs`/`-d`; the full filter grammar (`merge`/`dir-merge`/`hide`/`show`/`protect`/`risk`/`clear` + modifiers) and corrected `-F`/`-FF`; receiver-side `--chown`/map TO-name resolution; absolute basis dirs + `--link-dest` relink; receiver-side `--ignore-existing` short-circuit; `--preallocate` over `--sparse` via `fallocate(2)`; `--iconv=.`/`-`/`--no-iconv`; lone `-h` help; aliases `--ignore-non-existing`/`--protect-args`/`--msgs2stderr`; and the full `--info`/`--debug` vocabulary. `RSYNC_COMPAT.md` reclassifies the matrix to 106 ✅ / 27 ⚠️ / 23 ❌.
|
||||
|
||||
## Next steps
|
||||
1. **Merge PR #284** (`dev` -> `main`) once reviewed (protected branch).
|
||||
2. **Deferred security items** (documented, not implemented):
|
||||
- Pre-auth config/daemon-auth handshake has no aggregate wall-clock deadline
|
||||
(per-message timeout only) — slowloris holds connection slots.
|
||||
- Per-source registry fails open when the shared table is full (per-module/global
|
||||
caps and host ACLs still apply); consider fail-closed or larger/evicting table.
|
||||
- SCRAM-like daemon auth has no TLS channel binding (and is not RFC 5802).
|
||||
- `cleanup()` signal handler calls non-async-signal-safe teardown; daemon `umask(0)`.
|
||||
- Wire protocol assumes homogeneous word size/endianness (lengths are native
|
||||
`size_t`) — document or move to fixed-width framing.
|
||||
3. **Out of scope / intentional:** pull (remote source) mode is **not** planned —
|
||||
FastSync is push-only; see `RSYNC_COMPAT.md#direction`.
|
||||
|
||||
## Key facts / commands
|
||||
- CI image: `gitea.tap-tap.win/taptap/fastsync-ci:v11` (alias `fastsync-ci:local`).
|
||||
- Build/test: `cmake -B build -S . -DSTRICT_WARNINGS=ON && cmake --build build -j$(nproc) && ./build/tests`
|
||||
then `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"`.
|
||||
- Dev shell: `nix-shell` (provides clang-format, cppcheck, pytest-xdist, openssh,
|
||||
rsync, iproute2, valgrind, lcov; does not build on entry).
|
||||
- Gitea API token: supplied out-of-band via the `TOKEN` environment variable; it is
|
||||
intentionally **not** recorded in this file.
|
||||
- CI polling: `GET /api/v1/repos/TapTap/FastSync/actions/runs?limit=N`, match `head_sha`,
|
||||
then `/actions/runs/<id>/jobs`.
|
||||
@@ -1,4 +1,4 @@
|
||||
#FastSync
|
||||
# FastSync
|
||||
|
||||
FastSync is a high-performance file synchronization tool designed to become a
|
||||
drop-in replacement for common `rsync` workflows. It keeps the familiar
|
||||
@@ -7,7 +7,7 @@ multithreading, streaming zstd compression, chunking, zero-copy TCP transfers,
|
||||
and native TCP/TLS transports.
|
||||
|
||||
The release version is FastSync's client/server protocol version (printed by
|
||||
`fastsync --version`); client and server must match. See
|
||||
`./build/client --version`); client and server must match. See
|
||||
[CHANGELOG.md](CHANGELOG.md) for the history.
|
||||
|
||||
The compatibility target is straightforward:
|
||||
@@ -28,7 +28,7 @@ FastSync uses a producer-consumer transfer pipeline and can combine several
|
||||
optimizations for large or high-latency transfers:
|
||||
|
||||
- Multithreaded scanning, loading, and sending.
|
||||
- Streaming zstd compression with levels 1 through 22.
|
||||
- Streaming compression (zstd by default, plus lz4/zlib/zlibx) with levels 1 through 22.
|
||||
- Configurable file chunking and compact chunk serialization.
|
||||
- `sendfile()` zero-copy transfers over TCP.
|
||||
- Batched incremental checks to reduce round trips.
|
||||
@@ -51,28 +51,51 @@ replacement for every rsync feature or protocol mode.
|
||||
- Rsync-style source and destination arguments.
|
||||
- SSH transport using `user@host:destination` paths below the remote authorized root.
|
||||
- TCP client/server transfers.
|
||||
- Dry runs, excludes, includes, size filters, backups, statistics, and
|
||||
bandwidth limiting.
|
||||
- Incremental size/mtime checks and optional xxHash64 content checks.
|
||||
- Dry runs (server-contacting since protocol 2.21.0 for server-routed targets),
|
||||
excludes, includes, size filters, backups, statistics, and bandwidth
|
||||
limiting.
|
||||
- Incremental size/mtime checks and optional content checks (`xxh128` by
|
||||
default, selectable with `--checksum-choice`).
|
||||
- FastSync-native delta transfer for changed files.
|
||||
- Optional mode and timestamp preservation.
|
||||
- Delete manifests with server-side delete authorization.
|
||||
- Temporary-file writes with atomic rename by default.
|
||||
- Path traversal checks and destination-root confinement.
|
||||
|
||||
### Not yet equivalent to rsync
|
||||
### Boundaries and documented divergences
|
||||
|
||||
The items below summarize FastSync's rsync compatibility status — recently
|
||||
closed gaps and the remaining known divergences. Each row of the detailed
|
||||
matrix is classified as parity, caveat, or divergent in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
- The FastSync wire protocol is not the rsync wire protocol.
|
||||
- SSH mode requires `fastsync-server` on the remote host.
|
||||
- Archive mode does not yet provide all of rsync's `-rlptgoD` behavior.
|
||||
- Symlink transfer is incomplete; link targets are not yet recreated in all
|
||||
modes.
|
||||
- Owner/group, ACL, xattr, and hard-link handling is incomplete or
|
||||
unavailable.
|
||||
- Archive mode covers rsync's `-rlptgoD` behavior — links, permissions, times,
|
||||
owner, group, devices, and special files — and does not imply compression or
|
||||
multithreading (see [Client](#client)). Ownership application is still
|
||||
privilege-gated: a receiver that cannot `chown` logs a warning and skips it.
|
||||
Under `-p` the source mode is copied exactly, including setuid/setgid/sticky
|
||||
and group/other-write bits (strict rsync parity; see
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md)).
|
||||
- Symlink transfer stores targets **verbatim** (`-l`/`--links`), including
|
||||
absolute and `..`-bearing targets, matching rsync. The receiver does not
|
||||
enforce a containment predicate by default; `--safe-links` drops unsafe
|
||||
targets on the sender, and `--munge-links` rewrites them with rsync's
|
||||
`/rsyncd-munged/` marker. `--trust-sender` does not affect symlink targets.
|
||||
A destination later consumed by a link-following tool can therefore follow a
|
||||
link outside the receive root — use `--safe-links` for untrusted sources.
|
||||
- Hard links (`-H`/`--hard-links`), extended attributes (`-X`/`--xattrs`), and
|
||||
POSIX ACLs (`-A`/`--acls`) are preserved; owner/group is applied through
|
||||
`-o`/`-g` (or an `-a`/`--archive` transfer), through the opt-in identity flags
|
||||
(`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`), and only when
|
||||
the receiver has permission. See
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md) for the exact semantics and documented
|
||||
divergences.
|
||||
- Device and special-file preservation is implemented with documented
|
||||
divergences: recreated device nodes require `CAP_MKNOD` on the receiver (a
|
||||
non-root receiver skips the entry), and sockets cannot be recreated (FIFOs
|
||||
are).
|
||||
non-root receiver skips the entry), while FIFOs **and unix sockets** are
|
||||
recreated (`--specials`).
|
||||
- Sparse-file hole preservation (`-S`, `--sparse`) is implemented receiver-side:
|
||||
long all-zero runs are written as holes (no wire change; the full file image
|
||||
is already in memory).
|
||||
@@ -80,76 +103,169 @@ replacement for every rsync feature or protocol mode.
|
||||
the write atomic (temp + rename). With `--partial`, a failed/interrupted write
|
||||
now retains the already-written temp at the destination path (best-effort) so
|
||||
a later `--append`/`--append-verify` run can resume it.
|
||||
- `--dirs` is not implemented. Its compatibility aliases `--old-dirs` and
|
||||
`--old-d` are recognized but rejected explicitly rather than silently using
|
||||
FastSync's recursive directory behavior.
|
||||
- `-d`/`--dirs` and its aliases `--old-dirs`/`--old-d` transfer the named
|
||||
directory entries without recursing into their contents.
|
||||
- Short-option names are now rsync-parity (Phase 7 Wave A): FastSync's former
|
||||
collisions were renamed (`-j`/`--threads`, `--preserve`, `--sendfile`,
|
||||
`--chunk-serialization`, `--timeout`, `--ssh-port`), so `-m`, `-M`, `-f`,
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync. See `RSYNC_COMPAT.md`.
|
||||
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync.
|
||||
- Short-option clustering (`-av`, `-aAX`, `-rlpt`) and attached values
|
||||
(`-B1000`, `-essh`, `-MOPT`, `--opt=value`) are accepted, matching rsync.
|
||||
- `-r`, `-b`, `-L`, and `-B` are parsed with the rsync short names.
|
||||
- `--stats` prints the counters FastSync can observe plus the receiver-only
|
||||
counters (`Matched data`, deleted files) reported over the wire; rsync's
|
||||
per-type `Number of files` breakdown is not reproduced. `--progress` prints
|
||||
rsync-style per-file blocks (without rsync's leading `./` line).
|
||||
- Codecs match rsync 3.4.1: `zstd`/`lz4`/`zlib`/`zlibx` compression and
|
||||
`xxh128`/`xxh3`/`xxh64`/`md5`/`md4`/`sha1`/`none` checksums, negotiated with
|
||||
`auto`; `zlibx` behaves as `zlib`, and the transfer checksum is not separately
|
||||
selectable.
|
||||
|
||||
The detailed flag matrix is maintained in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented,
|
||||
partial, alternate, and planned behavior.
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It reports each row as **parity**,
|
||||
**caveat** (works with a documented divergence), or **divergent** (not
|
||||
supported), rather than treating "parsed" as parity.
|
||||
|
||||
## Quick Start
|
||||
|
||||
### Build
|
||||
|
||||
```bash
|
||||
cmake -B build -S .
|
||||
cmake --build build -j$(nproc)
|
||||
```
|
||||
|
||||
This produces `./build/client` and `./build/server`. `compile_commands.json` is a symlink to `build/compile_commands.json` and is used by clangd/editor tooling; its target is generated by the build, so it dangles until the first build.
|
||||
|
||||
### Client
|
||||
|
||||
| Argument | Description |
|
||||
|----------|-------------|
|
||||
| Positional | `<source> <dest>` — automatic SSH detection if dest contains `:` |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime |
|
||||
| `-z, --compress [level]` | Enable streaming zstd compression (level 1–22, default 5) |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials (not compression/multithreading) |
|
||||
| `-j, --threads` | Multithreading mode |
|
||||
| `-c, --checksum` | Verify content by checksum instead of size+mtime (implies the incremental checksum quick-check) |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh128` (default), `xxh3`, `xxh64`/`xxhash`, `md5`, `md4`, `sha1`, `none`, or `auto` (plus rsync's two-name `transfer,pre-transfer` form) |
|
||||
| `-z, --compress [level]` | Enable streaming compression (default `zstd`; level 1–22, default 5) |
|
||||
| `--compress-choice <alg>` | Compression algorithm: `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, or `auto` |
|
||||
| `--skip-compress <list>` | Skip compression for suffixes (`/`- or `,`-separated); defaults to rsync 3.4.1's built-in suffix list |
|
||||
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated (not compression/multithreading) |
|
||||
| `-j, --threads[=N]` | Multithreading mode; `N` (1–256) sets the parallel scanner worker count, bare `-j`/`--threads` uses the default |
|
||||
| `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) |
|
||||
| `-r, --recursive` | Recurse into directories (FastSync is always recursive; accepted for rsync compatibility) |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents; aliases `--old-dirs`/`--old-d` |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root |
|
||||
| `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) |
|
||||
| `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) |
|
||||
| `--sendfile` | Sendfile zero-copy. Incompatible with compression / chunk serialization. TCP only. Long form only. |
|
||||
| `--preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) |
|
||||
| `-n, --dry-run` | Scan and print what would be transferred |
|
||||
| `-p, --perms` | Preserve permission bits (part of the metadata bundle) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show real-time transfer speed |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded) |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk) |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`) |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch) |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing; `--no-whole-file` clears it |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`) |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime |
|
||||
| `--preserve` | Preserve mode and mtime (`-p` + `-t`; add `-o`/`-g` for owner/group or `-U`/`--atimes` for atime; `-N`/`--crtimes` captures birth time but cannot apply it) |
|
||||
| `-U, --atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N, --crtimes` | Capture birth time; cannot be applied (documented divergence) |
|
||||
| `-p, --perms` | Preserve permission bits. Strict rsync parity: the source mode is copied exactly, including setuid/setgid/sticky and group/other-write bits |
|
||||
| `-t, --times` | Preserve modification times |
|
||||
| `-o, --owner` | Preserve the source owner (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `-g, --group` | Preserve the source group (privilege-gated; mapped by name on the receiver with a numeric fallback) |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group`, `--no-preserve` | Negate the per-attribute flags (short `--no-p`/`--no-t`/`--no-o`/`--no-g`; `--no-preserve` clears all four) |
|
||||
| `-E, --executability` | Preserve executable permission bits |
|
||||
| `-X, --xattrs` | Preserve user `user.*` extended attributes |
|
||||
| `-A, --acls` | Preserve POSIX ACLs |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax) |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership |
|
||||
| `--numeric-ids` | Apply source numeric uid/gid directly instead of mapping by name |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP] (requires a privileged receiver) |
|
||||
| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes) |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`) |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`) |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--include-from <file>` | Read include patterns from a file |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `-x, --one-file-system` | Do not cross filesystem boundaries; the mount-point directory entry is emitted (empty at the destination) without descending |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G; `0` = no local limit, matching rsync) |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`) |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win) |
|
||||
| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded). Scoped to the synchronized directories, so `--files-from` subsets are safe |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) |
|
||||
| `--delete-after` | Explicit delete-after timing (implies `--delete`) |
|
||||
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
|
||||
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
|
||||
| `--max-size <n>` | Skip files larger than n bytes |
|
||||
| `--min-size <n>` | Skip files smaller than n bytes |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G) |
|
||||
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `--existing` | Skip files not already present at the destination; update existing files normally. |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds (default: 30) |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 10) |
|
||||
| `--backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing) |
|
||||
| `-h, --human-readable` | Format transfer byte sizes with binary units |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected) |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer (`--force` is honored at publication) |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install; confined to the receive root (relative only), with an `EXDEV` non-atomic copy fallback |
|
||||
| `-n, --dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `-q, --quiet` | Suppress non-error output |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters (FastSync does not print rsync's leading `./` line) |
|
||||
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption |
|
||||
| `--stats` | Print transfer statistics at end (bytes, files, timing), including the receiver-only counters reported over the wire; rsync's per-type `Number of files` breakdown is not reproduced |
|
||||
| `-i, --itemize-changes` | Print an rsync-style per-file change line |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`) |
|
||||
| `--list-only` | List source files instead of transferring |
|
||||
| `--fsync` | Fsync every written file before publication |
|
||||
| `-h, --human-readable` | Format transfer byte/rate counts with rsync's decimal (base-1000) units |
|
||||
| `--max-depth <n>` | Maximum directory depth to recurse (0 = unlimited, default: 0) |
|
||||
| `--log-file <path>` | Write log messages to file instead of stderr |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server) |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server) |
|
||||
| `--source-dir <path>` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) |
|
||||
| `--dest-dir <path>` | Server destination directory (overrides `FASTSYNC_DEST_DIR`) |
|
||||
| `--save-to-disk` | Write received files to disk |
|
||||
| `--server-host <ip>` | Server IP address (default: `127.0.0.1`) |
|
||||
| `--server-port <n>` | Server port (default: `8080`) |
|
||||
| `--ssh-port <port>` | SSH port (default: 22) |
|
||||
| `-e, --rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments, e.g. `-e "ssh -p 2222"`) |
|
||||
| `-M, --remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable) |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address |
|
||||
| `-4, --ipv4` | Force IPv4 for destination resolution |
|
||||
| `-6, --ipv6` | Force IPv6 for destination resolution |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect (`TCP_NODELAY`, `SO_KEEPALIVE`, `SO_RCVBUF`, `SO_SNDBUF`, `SO_REUSEADDR`) |
|
||||
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
|
||||
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
|
||||
| `--timeout <sec>` | I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`) and the per-message protocol poll deadline. Default `0` = disabled (matching rsync); `0` disables it. `--no-timeout` is the negation. The value is not sent on the wire; the server side keeps its own safe floor. |
|
||||
| `--contimeout <sec>` | Connection timeout in seconds (default: 60, matching rsync); `0` disables it (`--no-contimeout` is the negation) |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes (a positive integer); whatever was already transferred is kept |
|
||||
| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`); an early stop skips the late `--delete` keep-set |
|
||||
| `-b, --backup` | Backup existing destination files before overwriting |
|
||||
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
|
||||
| `--tls` | Enable TLS encryption |
|
||||
| `--cert <path>` | TLS certificate file (PEM) |
|
||||
| `--key <path>` | TLS private key file (PEM) |
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--client-cn <name>` | TLS client certificate common name; mandatory with `--tls` (a TLS connection always verifies the client CN) |
|
||||
|
||||
The exhaustive rsync flag matrix is in [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
|
||||
**Per-message vs. connection timeouts.** `--timeout` bounds each individual protocol
|
||||
send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It
|
||||
does not, by itself, stop a peer that keeps sending well-formed frames forever. The
|
||||
receiver therefore also enforces two wall-clock (`CLOCK_MONOTONIC`) bounds on a
|
||||
connection: a **1 hour** idle limit and a **24 hour** overall session cap. Only
|
||||
frames that move real work (not `STATUS_KEEPALIVE`/`STATUS_ABORT` and not an
|
||||
empty `STATUS_CHECK_BATCH`/`STATUS_DIR_TIMES`) refresh the idle timestamp, so a
|
||||
peer cannot hold a connection slot by emitting cheap empty frames; a peer that
|
||||
fabricates minimal non-empty frames can still occupy a slot until the 24 hour
|
||||
cap, since no bound can require actual payload without risking a legitimate
|
||||
long operation. Both are deliberately generous so a legitimate long-running
|
||||
transfer is never aborted.
|
||||
|
||||
### Server
|
||||
|
||||
@@ -163,6 +279,7 @@ partial, alternate, and planned behavior.
|
||||
| `--ca <path>` | TLS CA certificate file for verification (PEM) |
|
||||
| `--destination-root <path>` | Authorized destination root (default: `.`) |
|
||||
| `--allow-delete` | Permit manifest deletion |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. **Rejected with `--stdio`** (the SSH remote argv is client-composed, so a client could otherwise pass it and defeat the secure default; operators exposing `fastsync-server --stdio` over SSH must use a forced command if the default must hold). No effect when not root. |
|
||||
| `--allow-unauthenticated` | Permit plaintext TCP clients. For an `auth users` module this opts in **loopback plaintext only**; remote auth still requires verified TLS, so the flag never permits remote plaintext auth. |
|
||||
| `-v, --verbose` | Enable debug logging |
|
||||
| `--help` | Show help |
|
||||
@@ -174,60 +291,64 @@ partial, alternate, and planned behavior.
|
||||
| `FASTSYNC_SOURCE_DIR` | — | Source directory fallback |
|
||||
| `FASTSYNC_DEST_DIR` | — | Destination directory fallback |
|
||||
| `FASTSYNC_SAVE_TO_DISK` | `false` | Disk persistence fallback |
|
||||
| `FASTSYNC_SSH_PORT` | `22` | Default SSH port |
|
||||
| `FASTSYNC_SERVER_HOST` | `127.0.0.1` | Default server host |
|
||||
| `FASTSYNC_SERVER_PORT` | `8080` | Default server port |
|
||||
| `FASTSYNC_TLS_CERT` | — | Default TLS certificate path |
|
||||
| `FASTSYNC_TLS_KEY` | — | Default TLS private key path |
|
||||
| `FASTSYNC_TLS_CA` | — | Default TLS CA certificate path |
|
||||
|
||||
## Implementation Details
|
||||
|
||||
### Data Structures
|
||||
1. **Chunk** — collection of files (~10 MB total by default)
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`;
|
||||
uid / gid are advisory wire fields and are never applied by the receiver;
|
||||
atime is unsupported
|
||||
4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`, `queue_size`.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement
|
||||
|
||||
1. **Chunk** — collection of files (~10 MB total by default).
|
||||
2. **File** — path, content (`Data`), optional `FileMetadata` pointer.
|
||||
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec` (plus
|
||||
atime/crtime fields). `uid`/`gid` are applied only through the opt-in
|
||||
identity path; atime is preserved with `-U`/`--atimes`; crtime is captured
|
||||
but cannot be set on the destination.
|
||||
4. **Config** — runtime parameters. Most cross the wire (TLS settings
|
||||
excluded); `backup` and `backup_dir` are in the serialized wire table, while
|
||||
`timeout`, `contimeout`, `quiet`, `stats`, `max_depth`, and `log_file` are
|
||||
client-only.
|
||||
5. **Queue** — thread-safe bounded queue with condition variables.
|
||||
6. **DirectoryScanner** — recursive BFS traversal with exclude and include
|
||||
pattern support, max-depth enforcement.
|
||||
|
||||
### Key Algorithms
|
||||
1. **File scanning** — BFS directory traversal;
|
||||
entries matched against exclude and include patterns,
|
||||
max - depth enforced 2. * *Chunking ** — files accumulated until `chunk_size` threshold,
|
||||
then flushed 3. *
|
||||
*Compression ** — streaming zstd
|
||||
via `ZSTD_compressStream2` / `ZSTD_decompressStream` 4. *
|
||||
*Network protocol ** — status -
|
||||
code - driven exchange with metadata packing,
|
||||
keep - alive,
|
||||
and abort support 5. * *Incremental check ** — client sends `STATUS_CHECK` + path + size +
|
||||
mtime and,
|
||||
with `--checksum`, XXH64 content checksum; server compares against destination. Can be batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with `nanosleep` throttling on 64 KB write chunks
|
||||
7. **Metadata restoration** — `chmod()`, `chown()`, `utimensat()` on the receiving side
|
||||
8. **`--delete`** — sender tracks all sent paths;
|
||||
receiver walks destination tree and removes unlisted files / directories 9. *
|
||||
*SSH transport *
|
||||
* — `socketpair()` + `fork()` + `execvp("ssh",
|
||||
...)` with `ControlMaster` and port support
|
||||
10. *
|
||||
*TLS transport ** — OpenSSL `SSL_CTX` with TLS
|
||||
1.2 minimum,
|
||||
mutual CA verification,
|
||||
transparent `SSL_read`/`SSL_write` via `io_set_ssl()` 11. *
|
||||
*Path traversal protection ** — `has_path_traversal()` rejects any file path
|
||||
containing `..` components,
|
||||
preventing directory escape attacks 12. *
|
||||
*Connection limiting ** — server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100)13. *
|
||||
*Keep
|
||||
- alive ** — idle connections receive periodic `STATUS_KEEPALIVE` to detect half
|
||||
- open TCP connections 14. * *Abort handling ** — `SIGINT` sets an abort flag; the next protocol operation sends `STATUS_ABORT` for clean server cleanup
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically renamed via `rename()`, preventing partial files
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir` (or same directory with `~` suffix) preserving the original
|
||||
|
||||
1. **File scanning** — BFS directory traversal; entries matched against exclude
|
||||
and include patterns, with max-depth enforced.
|
||||
2. **Chunking** — files accumulated until the `chunk_size` threshold (default
|
||||
10 MiB) is reached, then flushed.
|
||||
3. **Compression** — streaming zstd via `ZSTD_compressStream2()` /
|
||||
`ZSTD_decompressStream()`.
|
||||
4. **Network protocol** — status-code-driven exchange with metadata packing,
|
||||
keep-alive, and abort support.
|
||||
5. **Incremental check** — the client sends `STATUS_CHECK` + path + size +
|
||||
mtime and, with `--checksum`, a whole-file content checksum (`xxh128` by
|
||||
default; selectable via `--checksum-choice`/`--cc`, seeded by
|
||||
`--checksum-seed`); the server compares against the destination. Can be
|
||||
batched via `STATUS_CHECK_BATCH` for reduced round-trips.
|
||||
6. **Bandwidth limiting** — token-bucket algorithm with sleep throttling on
|
||||
64 KiB write chunks.
|
||||
7. **Metadata restoration** — mode via `chmod()`/`fchmod()`, times via
|
||||
`utimensat()`/`futimens()`, and ownership only with an identity flag via
|
||||
fd-relative `fchown()`/`fchownat()`.
|
||||
8. **`--delete`** — the sender tracks all sent paths; the receiver walks the
|
||||
destination tree and removes unlisted files and directories.
|
||||
9. **SSH transport** — `socketpair()` + `fork()` + `execvp("ssh", ...)` with
|
||||
`ControlMaster` and port support.
|
||||
10. **TLS transport** — OpenSSL `SSL_CTX` with TLS 1.2 minimum, mutual CA
|
||||
verification, and transparent `SSL_read()`/`SSL_write()` via
|
||||
`io_set_ssl()`.
|
||||
11. **Path traversal protection** — `has_path_traversal()` rejects any file
|
||||
path containing `..` components, preventing directory escape attacks.
|
||||
12. **Connection limiting** — the server tracks active connections and rejects
|
||||
new ones beyond `max_connections` (default 100).
|
||||
13. **Keep-alive** — idle connections receive periodic `STATUS_KEEPALIVE` to
|
||||
detect half-open TCP connections.
|
||||
14. **Abort handling** — `SIGINT` sets an abort flag; the next protocol
|
||||
operation sends `STATUS_ABORT` for clean server cleanup.
|
||||
15. **Atomic writes** — files are written to a `.tmp` suffix then atomically
|
||||
renamed via `rename()`, preventing partial files.
|
||||
16. **Backup** — before overwriting, existing files are moved to `--backup-dir`
|
||||
(or the same directory with a `~` suffix), preserving the original.
|
||||
|
||||
## Security Features
|
||||
|
||||
@@ -284,22 +405,34 @@ cmake --build build -j$(nproc)
|
||||
|
||||
### SSH transfer
|
||||
|
||||
The remote host must have `fastsync-server` available in `PATH`, or use
|
||||
The remote host must have `fastsync-server` available in `PATH` (install or
|
||||
copy the built `./build/server` there as `fastsync-server`), or use
|
||||
`--fastsync-server-path`. SSH starts `fastsync-server --stdio` in its remote
|
||||
working directory, so use a destination below that directory unless the
|
||||
remote server is otherwise configured with a matching authorized root.
|
||||
|
||||
The remote `--stdio` server argv is composed by the client, so it must never
|
||||
be trusted to opt a root receiver into super-user activities: `--allow-super`
|
||||
is rejected with `--stdio` and super stays off on that path. Operators
|
||||
exposing `fastsync-server --stdio` over SSH must use a forced command (e.g. an
|
||||
`authorized_keys` `command=` entry) if the default must hold.
|
||||
|
||||
```bash
|
||||
ssh user@host 'mkdir -p destination'
|
||||
./build/client /path/to/source user@host:destination
|
||||
```
|
||||
|
||||
FastSync is **push-only**: the source (first argument) is always a local
|
||||
directory and only the destination may be remote. A remote source such as
|
||||
`client user@host:src ./local` (a "pull") is intentionally not supported; see
|
||||
[RSYNC_COMPAT.md](RSYNC_COMPAT.md#direction).
|
||||
|
||||
### TCP transfer
|
||||
|
||||
Start the FastSync server:
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to -p 8080
|
||||
./build/server --destination-root /path/to -p 8080 --allow-unauthenticated
|
||||
```
|
||||
|
||||
Then run the client:
|
||||
@@ -314,8 +447,13 @@ Plain TCP requires the explicit `--allow-unauthenticated` server option. Use TLS
|
||||
authenticated network connections.
|
||||
|
||||
### TLS transfer
|
||||
|
||||
Server TLS requires `--cert`, `--key`, `--ca`, and `--client-cn`; the client
|
||||
requires `--cert`, `--key`, and `--ca`.
|
||||
|
||||
```bash
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem -p 8443
|
||||
./build/server --destination-root /path/to --tls --cert server.pem --key server-key.pem \
|
||||
--ca ca.pem --client-cn client -p 8443
|
||||
./build/client --tls --cert client.pem --key client-key.pem --ca ca.pem \
|
||||
--server-host example.com --server-port 8443 \
|
||||
--source-dir /path/to/source --dest-dir /path/to/destination \
|
||||
@@ -328,32 +466,32 @@ These examples show the intended rsync-style workflow. Options marked as
|
||||
FastSync-native are optional performance or transport extensions.
|
||||
|
||||
```bash
|
||||
#Basic synchronization
|
||||
# Basic synchronization
|
||||
./build/client /source/ /destination/
|
||||
|
||||
#Archive - style synchronization(current FastSync archive behavior)
|
||||
# Archive-style synchronization (current FastSync archive behavior)
|
||||
./build/client -a /source/ user@host:destination/
|
||||
|
||||
#Preview a transfer without changing the destination
|
||||
# Preview a transfer without changing the destination
|
||||
./build/client -n /source/ /destination/
|
||||
|
||||
#Exclude temporary and object files
|
||||
# Exclude temporary and object files
|
||||
./build/client --exclude '*.tmp' --exclude '*.o' \
|
||||
/source/ user@host:destination/
|
||||
|
||||
#Remove destination entries not present in the source
|
||||
# Remove destination entries not present in the source
|
||||
./build/client --delete /source/ user@host:destination/
|
||||
|
||||
#Skip unchanged files using size and modification time
|
||||
# Skip unchanged files using size and modification time
|
||||
./build/client --incremental /source/ user@host:destination/
|
||||
|
||||
#Verify content when size and time are not sufficient
|
||||
# Verify content when size and time are not sufficient
|
||||
./build/client --incremental --checksum /source/ user@host:destination/
|
||||
|
||||
#Preserve supported mode and timestamp metadata
|
||||
./build/client -M /source/ user@host:destination/
|
||||
# Preserve supported mode and timestamp metadata
|
||||
./build/client --preserve /source/ user@host:destination/
|
||||
|
||||
#Keep backups of overwritten destination files
|
||||
# Keep backups of overwritten destination files
|
||||
./build/client --backup --backup-dir backups \
|
||||
/source/ user@host:destination/
|
||||
```
|
||||
@@ -365,12 +503,12 @@ features without changing the meaning of ordinary compatibility options.
|
||||
|
||||
| Option | Purpose |
|
||||
|---|---|
|
||||
| `-j`, `--threads` | Enable the multithreaded scanner/loader/sender pipeline. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming zstd compression, levels 1-22. |
|
||||
| `--compress-level <n>` | Set the zstd compression level. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` and `none`. |
|
||||
| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. |
|
||||
| `-z [level]`, `--compress [level]` | Enable streaming compression (default `zstd`), levels 1-22. |
|
||||
| `--compress-level <n>` | Set the compression level. |
|
||||
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` (default), `lz4`, `zlib`, `zlibx`, `none`, and `auto`; `zlibx` behaves as `zlib`. |
|
||||
| `--zl <n>` | Alias for `--compress-level`. |
|
||||
| `--skip-compress <list>` | Skip compression for comma-separated suffixes; incompatible with `--chunk-serialization`. |
|
||||
| `--skip-compress <list>` | Skip compression for `/`- or `,`-separated suffixes; defaults to rsync 3.4.1's built-in list. Incompatible with `--chunk-serialization`. |
|
||||
| `--compress-threads <n>` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. |
|
||||
| `--chunk-size <bytes>` | Set the transfer chunk size. |
|
||||
| `--chunk-serialization` | Enable FastSync chunk serialization (long form only; `-s` is rsync's `--secluded-args`). |
|
||||
@@ -379,13 +517,13 @@ features without changing the meaning of ordinary compatibility options.
|
||||
| `--delta-block <bytes>` | Set the FastSync delta block size (`--block-size` is an alias). |
|
||||
| `--delta-max <bytes>` | Limit files eligible for FastSync delta transfer. |
|
||||
| `--server-host <host>` | Select the TCP server host. |
|
||||
| `--server-port <port>` | Select the TCP server port. |
|
||||
| `--server-port <port>` | Select the TCP server port (`--port <port>` and `--port=<port>` are rsync-friendly aliases). |
|
||||
| `--tls` | Enable TLS for TCP transport. |
|
||||
| `--bwlimit <KB/s>` | Apply token-bucket bandwidth limiting. |
|
||||
| `--progress` | Show transfer progress and throughput. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `--timeout <seconds>` | Set I/O timeout. |
|
||||
| `--contimeout <seconds>` | Set connection timeout. |
|
||||
| `--progress` | Show rsync-style per-file progress blocks from the receiver's wire counters (FastSync omits rsync's leading `./` line). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire; rsync's per-type `Number of files` breakdown is not reproduced. |
|
||||
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout. Default `0` = disabled (matching rsync); `0` disables it. |
|
||||
| `--contimeout <seconds>` | Connection timeout (default 60, matching rsync); `0` disables it. |
|
||||
|
||||
Short-option conflicts with rsync have been resolved for the CLI namespace
|
||||
(Phase 7): `-c` is now rsync's `--checksum`, `-m` is `--prune-empty-dirs`, `-M`
|
||||
@@ -394,7 +532,8 @@ is `--remote-option`, `-f` is `--filter`, `-s` is `--secluded-args`, `-p` is
|
||||
long-form-only or new shorts: multithreading is `-j`/`--threads`, metadata
|
||||
is `--preserve`, sendfile is `--sendfile`, chunk serialization is
|
||||
`--chunk-serialization`, timeout is `--timeout`, and SSH port is `--ssh-port`.
|
||||
`-a`/`--archive` is now real rsync archive (`-rlptgoD`).
|
||||
`-a`/`--archive` is now rsync archive `-rlptgoD` (owner/group implied, but the
|
||||
receiver still needs privilege to apply them).
|
||||
|
||||
`--secluded-args` (and its short form `-s`) is accepted as a compatibility
|
||||
no-op. It does not change FastSync's transport or protocol behavior, because
|
||||
@@ -406,42 +545,94 @@ remote SSH argv is already built injection-safe.
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials. |
|
||||
| `-n`, `--dry-run` | Scan and report without writing files. |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. |
|
||||
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, perms, times, owner, group, devices and specials; ownership application stays privilege-gated. |
|
||||
| `-n`, `--dry-run` | Report what would be transferred without mutating the destination. Since protocol 2.21.0 a server-routed target contacts the receiver and reports would-transfer based on receiver state; a plain local destination keeps the client-side scan. Never mutates or deletes. |
|
||||
| `--remove-source-files` | Remove regular source files after a successful transfer. |
|
||||
| `--incremental` | Skip files matching destination size and mtime. Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
|
||||
| `-c, --checksum` | Verify content by checksum (implies the incremental quick-check). Algorithm selectable with `--checksum-choice`. |
|
||||
| `--checksum-choice <alg>` | Whole-file checksum algorithm: `xxh64`/`xxhash` (default), `xxh3`, `xxh128`, `md5`, or `auto`. |
|
||||
| `--checksum-seed <n>` | Seed for the whole-file xxHash digest; an unset/`0` seed is randomized per transfer, matching rsync. |
|
||||
| `--size-only` | Skip incremental files matching in size, ignoring mtime. |
|
||||
| `-I, --ignore-times` | Transfer files even when size and mtime match. |
|
||||
| `-u, --update` | Skip files newer than the source on the receiver. |
|
||||
| `-W, --whole-file` | Transfer changed files without delta processing (`--no-whole-file` clears it). |
|
||||
| `-B <n>, --block-size <n>` | Delta block size in bytes (alias `--delta-block`). |
|
||||
| `-d, --dirs` | Transfer the named directory entries without recursing into their contents (aliases `--old-dirs`/`--old-d`). |
|
||||
| `-R, --relative` | Use rsync's relative path semantics (including the `/./` cut); with `--files-from`, preserve each listed entry's relative path below the destination root. |
|
||||
| `--files-from <file>` | Read the source file list from FILE (paths relative to the source root). |
|
||||
| `--delay-updates` | Put updated files into place only at the end of the transfer. |
|
||||
| `--compare-dest <dir>` | Extra comparison basis: unchanged files are not transferred (requires/implies `--incremental`). |
|
||||
| `--copy-dest <dir>` | Like `--compare-dest`, but copies the unchanged file from DIR into the destination. |
|
||||
| `--link-dest <dir>` | Like `--copy-dest`, but hard-links the unchanged file from DIR (repeatable; earlier DIRs win). |
|
||||
| `--preallocate` | Allocate destination file space up front (fail-fast on a full disk). |
|
||||
| `--append` | Resume a shorter destination by appending only its tail (prefix not verified; requires `--incremental`). |
|
||||
| `--append-verify` | Like `--append`, but verifies the retained prefix checksum first (falls back to a full transfer on mismatch). |
|
||||
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. Scoped to the synchronized directories, so `--files-from` subsets are safe. |
|
||||
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). |
|
||||
| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). |
|
||||
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). |
|
||||
| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). |
|
||||
| `--delete-excluded` | Also delete filter-excluded destination mirrors (size-pruned mirrors stay protected). |
|
||||
| `--max-delete <n>` | Delete at most n destination entries; the rest are skipped and the run exits 25 (partial), matching rsync. |
|
||||
| `--force` | Allow an incoming file/symlink to replace a destination directory (also during `--delay-updates` publication). |
|
||||
| `--exclude <pattern>` | Exclude matching paths. Repeatable. |
|
||||
| `--include <pattern>` | Include matching paths. Repeatable. |
|
||||
| `--exclude-from <file>` | Read exclude patterns from a file. |
|
||||
| `--include-from <file>` | Read include patterns from a file. |
|
||||
| `-f, --filter=RULE` | Add an rsync-style filter rule (`+`/`-`, `include`/`exclude`, `merge`/`.`, `dir-merge`/`:`, `hide`/`H`, `show`/`S`, `protect`/`P`, `risk`/`R`, `clear`/`!`, and modifiers; repeatable). |
|
||||
| `--max-size <bytes>` | Skip files larger than the limit. |
|
||||
| `--min-size <bytes>` | Skip files smaller than the limit. |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth;
|
||||
zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.|
|
||||
| `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` |
|
||||
Back up overwritten files.| | `--backup - dir<dir>` | Store backups under a separate directory.|
|
||||
| `--suffix<suffix>` | Set the backup filename suffix.| | `--partial` |
|
||||
Select partial - transfer handling. On failed/interrupted writes the
|
||||
already-written temp file is retained (best-effort) for resumption.|
|
||||
With `--partial --partial-dir <dir>`, completed files are written under the
|
||||
partial directory and installed atomically. | | `--partial - dir<dir>` |
|
||||
Set a relative partial - transfer directory below the server destination root.
|
||||
Use with `--partial`. |
|
||||
| `--max-alloc <SIZE>` | Maximum single allocation (binary units; default 1G; `0` = no local limit). |
|
||||
| `--max-depth <n>` | Limit recursive scanning depth; zero means unlimited. |
|
||||
| `-b, --backup` | Back up overwritten files. |
|
||||
| `-T, --temp-dir <dir>` | Scratch directory for temp files before the atomic install (confined to the receive root; `EXDEV` falls back to a non-atomic copy). |
|
||||
| `--backup-dir <dir>` | Store backups under a separate directory (requires `--backup`). |
|
||||
| `--suffix <suffix>` | Set the backup filename suffix (default: `~`). |
|
||||
| `--partial` | Select partial-transfer handling. On failed/interrupted writes the already-written temp file is retained (best-effort) for resumption. With `--partial --partial-dir <dir>`, completed files are written under the partial directory and installed atomically. |
|
||||
| `--partial-dir <dir>` | Set a relative partial-transfer directory below the server destination root. Use with `--partial`. |
|
||||
| `--inplace` | Write directly to the destination instead of using a temporary file. |
|
||||
| `--fsync` | Fsync every written file before publication. |
|
||||
| `--write-batch=FILE` | Run the normal live transfer and also emit a self-contained batch file of the source tree. |
|
||||
| `--only-write-batch=FILE` | Emit the batch file only (no destination, no server). |
|
||||
| `--read-batch=FILE` | Apply a batch file to the destination (no source, no server). |
|
||||
| `--stop-after=MINS` | Stop the transfer after MINS minutes; whatever was already transferred is kept. |
|
||||
| `--stop-at=TIME` | Stop at an absolute time (`HH:MM`, `HH:MM:SS`, or `now+N[smhd]`). An early stop skips the late `--delete` keep-set. |
|
||||
|
||||
### Metadata and links
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--preserve` | Preserve supported file metadata, currently mode and modification time (long form only). |
|
||||
| `-l`, `--links` | Request symlink preservation;
|
||||
link-target transfer remains incomplete. |
|
||||
| `--copy-links` | Copy symlink referents. |
|
||||
| `--safe-links` | Skip symlinks that point outside the transfer tree. |
|
||||
| `--preserve` | Preserve mode and mtime (long form only; equivalent to `-p` + `-t`). Add `-o`/`-g` for owner/group, `-U`/`--atimes` for atime, or an identity flag (`--chown`/`--usermap`/`--groupmap`/`--numeric-ids`/`--copy-as`) for mapped ownership. |
|
||||
| `-U`, `--atimes` | Preserve access times. Captured with the metadata payload; does not enable ownership. |
|
||||
| `-N`, `--crtimes` | Capture birth time and transmit it; it cannot be applied because no portable filesystem call can set a birth time (documented divergence). |
|
||||
| `-p`, `--perms` | Preserve permission bits. One of the four per-attribute preserve flags (with `-t`/`-o`/`-g`); under `-p` the source mode is copied exactly (setuid/setgid/sticky and group/other-write included), matching rsync. |
|
||||
| `-t`, `--times` | Preserve modification times. Independent of the other attributes; `-O`/`--omit-dir-times` suppresses directories only. |
|
||||
| `-o`, `--owner` | Preserve the source owner (uid). Mapped by name on the receiver with a raw-numeric fallback (only numeric ids cross the wire); application is privilege-gated. |
|
||||
| `-g`, `--group` | Preserve the source group (gid). Same name-mapping/numeric-fallback and privilege gating as `-o`. |
|
||||
| `--no-perms`, `--no-times`, `--no-owner`, `--no-group` | Negate each per-attribute flag (also `--no-p`/`--no-t`/`--no-o`/`--no-g`); `--no-preserve` clears all four. |
|
||||
| `-E`, `--executability` | Preserve executable permission bits. |
|
||||
| `-X`, `--xattrs` | Preserve user `user.*` extended attributes. |
|
||||
| `-A`, `--acls` | Preserve POSIX ACLs. |
|
||||
| `--chmod <changes>` | Modify transferred permissions (rsync syntax, including `D`/`F`/`X` selectors and `s`/`t`); does not imply `-p`. |
|
||||
| `--chown=USER:GROUP` | Override the ownership of transferred files (`USER:GROUP`, `USER`, or `:GROUP`); conflicts with `--usermap`/`--groupmap` on the same side. |
|
||||
| `--usermap=MAP` | Map usernames when applying ownership (`FROM:TO` rules; names, ids, `LOW-HIGH` ranges, `*`, empty-`FROM`). |
|
||||
| `--groupmap=MAP` | Map group names when applying ownership (same syntax as `--usermap`). |
|
||||
| `--numeric-ids` | Mapping modifier: apply the source numeric uid/gid directly instead of mapping by name (combine with `-o`/`-g`, `-a`, or a map). |
|
||||
| `--copy-as=USER[:GROUP]` | Force every written entry to USER[:GROUP]; requires a privileged receiver. |
|
||||
| `--fake-super` | Record the resolved owner plus mode/time in a reserved `user.fastsync.stat` xattr and replay mode/time; never performs a real chown. |
|
||||
| `--super` | Permit the receiver to attempt confined super-user activities (device nodes). |
|
||||
| `--no-super` | Forbid those super-user activities even when the receiver is root. |
|
||||
| `-l`, `--links` | Copy symlinks as symlinks; the target is stored verbatim (absolute and `..`-bearing targets included), matching rsync. |
|
||||
| `-L`, `--copy-links` | Copy symlink referents (a broken referent makes the run exit 23, matching rsync). |
|
||||
| `--safe-links` | Skip symlinks whose target points outside the transfer tree (applied on the sender). |
|
||||
| `--copy-unsafe-links` | Copy unsafe symlink referents. |
|
||||
| `--munge-links` | Rewrite stored symlink targets with rsync's `/rsyncd-munged/` marker. |
|
||||
| `-k`, `--copy-dirlinks` | Treat a symlink to a directory as a real directory on the sender. |
|
||||
| `-K`, `--keep-dirlinks` | Follow an existing destination symlink-to-directory (confined to the receive root). |
|
||||
| `-H`, `--hard-links` | Preserve hard-link relationships across the transfer. |
|
||||
| `-D` | Preserve device and special files (implies `--devices --specials`). |
|
||||
| `--devices` | Recreate device nodes on the destination (privileged; skipped without `CAP_MKNOD`). |
|
||||
| `--specials` | Recreate special files: FIFOs and unix sockets. |
|
||||
| `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). |
|
||||
|
||||
### Output and logging
|
||||
@@ -449,8 +640,12 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--progress` | Show live transfer progress. |
|
||||
| `--stats` | Print transfer statistics. |
|
||||
| `-q`, `--quiet` | Suppress non-error output. |
|
||||
| `--progress` | Show rsync-style per-file progress blocks (not rsync's leading `./` line). |
|
||||
| `--stats` | Print transfer statistics, including the receiver-only counters reported over the wire. |
|
||||
| `-i`, `--itemize-changes` | Print an rsync-style per-file change line. |
|
||||
| `--out-format=FORMAT` | Output format for changed files (`%f %n %l %b %M %%`). |
|
||||
| `--list-only` | List source files instead of transferring. |
|
||||
| `--log-file <path>` | Write log output to a file. |
|
||||
| `-V`, `--version` | Print the FastSync protocol version. |
|
||||
| `--help` | Print command usage. |
|
||||
@@ -460,33 +655,116 @@ link-target transfer remains incomplete. |
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--ssh-port <port>` | SSH port for the SSH transport (default: 22). Note the short `-p` is now rsync's `--perms`. |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode. |
|
||||
| `-e`, `--rsh <command>` | Remote shell to launch for the SSH transport (default: `ssh`; may include arguments). |
|
||||
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode (client-only; never crosses the wire). |
|
||||
| `--rsync-path <path>` | Alias for `--fastsync-server-path`. |
|
||||
| `-M`, `--remote-option=OPT` | Append OPT to the remote server invocation over SSH (repeatable; rejected for daemon/TCP destinations). |
|
||||
| `--trust-sender` | Receiver-local: trust the remote sender's file list and skip path re-validation (does not affect symlink targets). |
|
||||
| `--timeout <sec>` | Socket + per-message I/O timeout; default `0` = disabled. |
|
||||
| `--contimeout <sec>` | Connection timeout; default 60; `0` disables. |
|
||||
| `--source-dir <path>` | Set the source directory explicitly. |
|
||||
| `--dest-dir <path>` | Set the destination directory explicitly. |
|
||||
| `--save-to-disk` | Enable server-side disk persistence. |
|
||||
| `--server-host <host>` | TCP server address. |
|
||||
| `--server-port <port>` | TCP server port. |
|
||||
| `--tls` | Enable TLS. Requires `--cert` and `--key`. |
|
||||
| `--server-port <port>` | TCP server port. `--port <port>` / `--port=<port>` is an alias. |
|
||||
| `--address <ip>` | Bind the outgoing client socket to this source address. |
|
||||
| `-4`, `--ipv4` | Force IPv4 for destination resolution. |
|
||||
| `-6`, `--ipv6` | Force IPv6 for destination resolution. |
|
||||
| `--sockopts=OPTS` | Comma-separated OPT=VAL socket options applied before connect. |
|
||||
| `--tls` | Enable TLS. Requires `--cert`, `--key`, and `--ca`. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--ca <path>` | CA file for peer verification (always required with `--tls`). |
|
||||
|
||||
## Server Options
|
||||
|
||||
| Option | Description |
|
||||
|---|---|
|
||||
| `--stdio` | Serve one SSH connection over standard input/output. |
|
||||
| `-p <port>` | TCP listen port. |
|
||||
| `--daemon` | Run as a persistent daemon listener using a module config file; the daemon default port is 873 (unlike `-p`, which defaults to 8080). |
|
||||
| `--config=FILE` | Daemon config file (default: `~/.config/fastsync/fastsyncd.conf`, else `/etc/fastsyncd.conf`). Requires `--daemon`. |
|
||||
| `--dparam=KEY=VALUE` | Override one global config key on the command line. Requires `--daemon`. |
|
||||
| `--no-detach` | Stay in the foreground (default detaches to the background when running `--daemon`). |
|
||||
| `-p, --port <port>` | TCP listen port (default: 8080, range: 1–65535). |
|
||||
| `--tls` | Enable TLS. |
|
||||
| `--cert <path>` | TLS certificate file. |
|
||||
| `--key <path>` | TLS private key file. |
|
||||
| `--ca <path>` | CA file for peer verification. |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root;
|
||||
defaults to the current directory. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. |
|
||||
| `--cert <path>` | TLS certificate file (PEM). |
|
||||
| `--key <path>` | TLS private key file (PEM). |
|
||||
| `--ca <path>` | CA file for peer verification (PEM). |
|
||||
| `--client-cn <name>` | TLS client certificate CN; mandatory with `--tls` (the server verifies the client CN). |
|
||||
| `--destination-root <path>` | Confine received files to this server-side root; defaults to the current directory. |
|
||||
| `--address <addr>` | Bind the listening socket to this address. |
|
||||
| `-4`, `--ipv4` | Bind an IPv4 socket (default). |
|
||||
| `-6`, `--ipv6` | Bind an IPv6 socket. |
|
||||
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. This also gates `--force` (which can recursively replace/remove a destination directory tree). |
|
||||
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. Rejected with `--stdio` (the SSH remote argv is client-composed; use a forced command if the default must hold). No effect when not root. Daemon modules opt in per module with `client owner = yes`. |
|
||||
| `--trust-sender` | Trust the remote sender's file list: skip the receiver's up-front path-traversal re-validation (fewer checks, faster, potentially unsafe; off by default). It does not affect symlink targets, which are stored verbatim either way. |
|
||||
| `--no-super` | Operator veto: never attempt super-user activities (ownership, device nodes) even as root, and refuse any client `--copy-as`/`--super` request. |
|
||||
| `--allow-unauthenticated` | Permit plaintext/anonymous network clients; an auth-required module still accepts only opted-in loopback plaintext. |
|
||||
| `--iconv=LOCAL[,REMOTE]` | Declare this server's LOCAL charset for file-name conversion. |
|
||||
| `--password-file=FILE` | Credential store for modules that declare `auth users`. Requires `--daemon`. |
|
||||
| `--early-input=FILE` | Second credential store layered over `--password-file`. Requires `--daemon`. |
|
||||
| `--hash-credentials <file>` | Read `<file>`'s `user:password` lines and print PBKDF2 credential-store lines to stdout, then exit. Cannot be combined with `--daemon` or `--stdio`. |
|
||||
| `--iterations N` | PBKDF2 iteration count for `--hash-credentials` (default 600000, range 100000–10000000). Requires `--hash-credentials`. |
|
||||
| `-v`, `--verbose` | Enable debug logging. |
|
||||
| `--help` | Print server usage. |
|
||||
|
||||
### Daemon configuration
|
||||
|
||||
`fastsync-server --daemon --config FILE` reads a line-based module config (an
|
||||
implicit global section, then `[module]` sections). Besides `port`, `motd file`,
|
||||
and `address`, the global section accepts:
|
||||
|
||||
- `max connections = N` — global cap on concurrent connections, default 100. The
|
||||
listener enforces it; `0`, negative, and non-numeric values are parse errors.
|
||||
- `max connections per host = N` — cap on concurrent connections from a single
|
||||
source IP, default 0 (unlimited). Enforced across all forked connection
|
||||
children through a shared registry.
|
||||
- `auth failure delay = MS` — milliseconds to sleep after a failed
|
||||
authentication, default 500. `0` disables it and the value is capped at 5000,
|
||||
so online password guessing is rate-limited per connection. Successful auths
|
||||
are never delayed.
|
||||
- `auth lockout threshold = N` — number of failed authentications from one source
|
||||
IP before that source is locked out, default 10; `0` disables the lockout. The
|
||||
failure counter is shared across every connection child, so the lockout holds
|
||||
even when the next attempt is handled by a different forked child.
|
||||
- `auth lockout duration = SECONDS` — how long a locked-out source is refused
|
||||
(default 300). A locked-out client is refused before any SCRAM challenge is
|
||||
sent; a successful authentication clears the counter.
|
||||
- `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access
|
||||
patterns.
|
||||
|
||||
A `[module]` requires `path`, and may also set `read only`, `client owner`,
|
||||
`auth users`, `max connections` (0 = unlimited; enforced per module across all
|
||||
connection children), and its own `hosts allow`/`hosts deny`.
|
||||
|
||||
The per-host cap and the shared auth lockout identify a source by its numeric
|
||||
peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local
|
||||
client shares that one address, so counting or locking them out would let one
|
||||
local process deny service to all the others. The per-module and global
|
||||
`max connections` caps still apply to loopback. Because the key is the peer IP,
|
||||
`max connections per host` and `auth lockout` also cannot distinguish clients
|
||||
behind the same NAT, proxy, or reverse-proxy address — they share one budget and
|
||||
one lockout counter, so an over-aggressive lockout can affect unrelated users
|
||||
behind that address. Prefer TLS client certificates (`--client-cn`) plus
|
||||
`hosts allow`/`hosts deny` for per-client policy when clients share an address,
|
||||
and size `auth lockout threshold` accordingly.
|
||||
|
||||
The shared per-source table has a bounded lifetime: an entry with no live
|
||||
connection is reclaimed once its lockout has expired, or after it has been idle
|
||||
(300 s). If every entry is still live or locked, a new source is admitted without
|
||||
per-host accounting (fail open) and a rate-limited warning is logged; the
|
||||
per-module cap and host ACLs still apply. The occupancy counters are re-derived
|
||||
from the shared slot table after every child exit, so a child killed mid-transfer
|
||||
(or mid-registration) cannot leak a slot or an occupancy count.
|
||||
|
||||
Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR
|
||||
(`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs
|
||||
are rejected at parse time rather than silently never matching. A matching
|
||||
`hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of
|
||||
them is rejected; deny takes precedence over allow. The global list is checked
|
||||
before the module list, before authentication, and the connecting peer address
|
||||
(IPv4 or IPv6) appears in the connection and authentication audit log lines.
|
||||
|
||||
## Architecture
|
||||
|
||||
### Client
|
||||
@@ -511,7 +789,7 @@ defaults to the current directory. |
|
||||
|
||||
## Protocol and Security
|
||||
|
||||
FastSync protocol version `2.19.0` is shared by the client and server. The
|
||||
FastSync protocol version `2.26.0` is shared by the client and server. The
|
||||
current protocol is sender-driven and includes configuration negotiation,
|
||||
including the maximum allocation limit, incremental checks, checksums,
|
||||
manifests, keep-alives, abort handling, per-file remove-source results, and
|
||||
@@ -563,10 +841,9 @@ mandates `--client-cn`, so a TLS connection to an auth-required module always
|
||||
has its client CN verified (`--client-cn` matches the certificate's CN only, not
|
||||
a subjectAltName, which is acceptable for a private CA).
|
||||
|
||||
TLS provides encrypted TCP transport. Supplying `--ca` enables certificate
|
||||
verification; without it, traffic is encrypted but peer identity is not
|
||||
verified. Use certificate verification for deployments where authentication
|
||||
matters. The default TCP transport is not encrypted.
|
||||
TLS provides encrypted TCP transport. Both the client and the server require
|
||||
`--ca` together with `--tls`, so peer certificates are always verified
|
||||
(`SSL_VERIFY_PEER`, depth 4). The default TCP transport is not encrypted.
|
||||
|
||||
The receiver protects its destination root with path validation, `openat()`
|
||||
directory traversal, `O_NOFOLLOW`, temporary files, and atomic renames. Delete
|
||||
@@ -577,18 +854,27 @@ operations require the server's explicit `--allow-delete` policy.
|
||||
The project will reach the drop-in replacement goal in stages:
|
||||
|
||||
1. Correct rsync option meanings, including short options, combined options,
|
||||
and `--option=value` syntax.
|
||||
and `--option=value` syntax — **done** in the rsync-parity wave: `-r`/`-b`/
|
||||
`-L`/`-B`, short-option clustering (`-av`, `-aAX`, `-rlpt`), and attached
|
||||
values (`-B1000`, `-essh`, `-MOPT`) all parse.
|
||||
2. Add differential tests that compare FastSync and rsync contents, metadata,
|
||||
links, deletes, filters, dry runs, and exit codes.
|
||||
3. Make `-a` implement the expected recursive, links, permissions, times,
|
||||
owner/group, and supported special-file behavior.
|
||||
4. Complete symlink, sparse-file, metadata, delete-policy, and resumable-write
|
||||
semantics.
|
||||
links, deletes, filters, dry runs, and exit codes — **done** for the
|
||||
completion wave's scope; the tests live in `tests/integration/` and skip
|
||||
cleanly when rsync is unavailable.
|
||||
3. `-a` implements full rsync `-rlptgoD`; under `-p` the source mode is copied
|
||||
exactly (no masking). Ownership application stays privilege-gated, as in
|
||||
rsync.
|
||||
4. Symlink (verbatim storage), sparse-file, metadata, delete-policy (including
|
||||
`--max-delete` partial + exit 25, per-directory `--delete-during`/
|
||||
`--delete-delay`), codecs, and resumable-write semantics are implemented;
|
||||
remaining work is the documented edge cases, which the **Parity Completion
|
||||
Wave** section of `RSYNC_COMPAT.md` enumerates honestly.
|
||||
5. Add rsync remote-shell and daemon protocol interoperability.
|
||||
6. Keep FastSync performance options as negotiated, optional extensions.
|
||||
|
||||
The exhaustive implementation matrix and compatibility notes are in
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md).
|
||||
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md); each row is classified as parity, caveat,
|
||||
or divergent.
|
||||
|
||||
## Testing
|
||||
|
||||
@@ -601,7 +887,7 @@ Run the unit test binary:
|
||||
Run the Python integration suite:
|
||||
|
||||
```bash
|
||||
python3 -m pytest tests/
|
||||
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"
|
||||
```
|
||||
|
||||
For stricter local validation:
|
||||
@@ -625,10 +911,13 @@ rsync protocol or filesystem-semantic compatibility.
|
||||
|
||||
## Performance Guidance
|
||||
|
||||
- Use `-m` for workloads with many files or enough CPU parallelism.
|
||||
- Use `-c` or `-z` when network bandwidth is more constrained than CPU.
|
||||
- Use `-j`/`--threads` for workloads with many files or enough CPU parallelism
|
||||
(`-m` is `--prune-empty-dirs`).
|
||||
- Use `-z` when network bandwidth is more constrained than CPU (`-c` is
|
||||
`--checksum`, not a bandwidth option).
|
||||
- Tune `--chunk-size` for file sizes, memory limits, and network latency.
|
||||
- Use `-f` for large uncompressed TCP transfers where zero-copy I/O helps.
|
||||
- Use `--sendfile` for large uncompressed TCP transfers where zero-copy I/O
|
||||
helps (`-f` is `--filter`).
|
||||
- Use `--incremental` to avoid retransmitting unchanged files.
|
||||
- Use `--delta` for changed files when both endpoints are FastSync peers.
|
||||
- Use `--bwlimit` when sharing a link with other traffic.
|
||||
|
||||
+602
-268
File diff suppressed because it is too large
Load Diff
+412
-67
@@ -3,19 +3,24 @@
|
||||
|
||||
Compares FastSync configs against rsync (no compression) and rsync+zstd.
|
||||
Data is ~75% random/incompressible and ~25% structured/compressible by default,
|
||||
controllable via --random-ratio.
|
||||
controllable via --random-ratio. Transfers are verified by default (source and
|
||||
destination must match) so a fast-but-broken copy is never counted.
|
||||
|
||||
Usage:
|
||||
python3 benchmark/bench.py
|
||||
python3 benchmark/bench.py --runs 5 --profiles lan wan
|
||||
python3 benchmark/bench.py --random-ratio 0.5 --size-mb 50
|
||||
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
|
||||
python3 benchmark/bench.py --warm --runs 3
|
||||
python3 benchmark/bench.py --output json
|
||||
"""
|
||||
import argparse
|
||||
import filecmp
|
||||
import json
|
||||
import math
|
||||
import os
|
||||
import random
|
||||
import shlex
|
||||
import shutil
|
||||
import socket
|
||||
import statistics
|
||||
@@ -25,8 +30,11 @@ import tempfile
|
||||
import time
|
||||
|
||||
PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, "build")
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server")]
|
||||
DEFAULT_BUILD_DIR = "build-bench"
|
||||
# Populated by configure_build_dirs(); default to the dedicated bench dir so
|
||||
# importing this module never depends on the user's existing build/ tree.
|
||||
BUILD_DIR = os.path.join(PROJECT_ROOT, DEFAULT_BUILD_DIR)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
BENCH_DIR = os.path.join(PROJECT_ROOT, "bench_data")
|
||||
|
||||
@@ -44,10 +52,11 @@ NETWORK_PROFILES = {
|
||||
|
||||
FASTSYNC_CONFIGS = [
|
||||
{"name": "fastsync", "flags": [], "tool": "fastsync"},
|
||||
{"name": "fastsync -c", "flags": ["-c"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m", "flags": ["-m"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m -c", "flags": ["-m", "-c"], "tool": "fastsync"},
|
||||
{"name": "fastsync -m -c -s", "flags": ["-m", "-c", "-s"], "tool": "fastsync"},
|
||||
{"name": "fastsync -z", "flags": ["-z"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j", "flags": ["-j"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j -z", "flags": ["-j", "-z"], "tool": "fastsync"},
|
||||
{"name": "fastsync -j -z --chunk-serialization", "flags": ["-j", "-z", "--chunk-serialization"], "tool": "fastsync"},
|
||||
{"name": "fastsync --sendfile", "flags": ["--sendfile"], "tool": "fastsync"},
|
||||
]
|
||||
|
||||
RSYNC_CONFIGS = [
|
||||
@@ -56,6 +65,7 @@ RSYNC_CONFIGS = [
|
||||
{"name": "rsync -z --zstd", "flags": ["-z", "--zc", "zstd"],"tool": "rsync"},
|
||||
]
|
||||
|
||||
|
||||
class RsyncDaemon:
|
||||
"""Manages an rsync daemon for network-fair benchmarking."""
|
||||
|
||||
@@ -117,6 +127,12 @@ STRUCTURED_FILES = {
|
||||
"nested/another.txt": b"another nested file\n" * 50,
|
||||
}
|
||||
|
||||
# Repeated text used to synthesize genuinely compressible filler of any size.
|
||||
COMPRESSIBLE_TEXT = (
|
||||
b"FastSync benchmark payload: the quick brown fox jumps over the lazy dog. "
|
||||
b"0123456789 ABCDEFGHIJKLMNOPQRSTUVWXYZ abcdefghijklmnopqrstuvwxyz\n"
|
||||
)
|
||||
|
||||
|
||||
class Progress:
|
||||
"""Simple progress bar with ETA."""
|
||||
@@ -151,35 +167,127 @@ class Progress:
|
||||
sys.stderr.flush()
|
||||
|
||||
|
||||
def write_compressible(path, nbytes):
|
||||
"""Write exactly nbytes of highly compressible, repeated text content."""
|
||||
if nbytes <= 0:
|
||||
return
|
||||
block = COMPRESSIBLE_TEXT * (max(1, 8192 // len(COMPRESSIBLE_TEXT)) + 1)
|
||||
remaining = nbytes
|
||||
with open(path, "wb") as f:
|
||||
while remaining > 0:
|
||||
piece = block if remaining >= len(block) else block[:remaining]
|
||||
f.write(piece)
|
||||
remaining -= len(piece)
|
||||
|
||||
|
||||
def generate_bench_data(source_dir, size_mb=25, random_ratio=0.75):
|
||||
"""Generate test data. ~random_ratio is incompressible, rest is structured."""
|
||||
"""Generate test data honouring the requested random/compressible split.
|
||||
|
||||
Exactly ``random_ratio * target`` bytes are incompressible random data and
|
||||
the remainder is genuinely compressible structured/repeated content. The
|
||||
measured byte counts are returned so callers can report the real mix.
|
||||
"""
|
||||
if os.path.exists(source_dir):
|
||||
shutil.rmtree(source_dir)
|
||||
os.makedirs(source_dir)
|
||||
|
||||
target = size_mb * 1024 * 1024
|
||||
structured_budget = int(target * (1 - random_ratio))
|
||||
written = 0
|
||||
random_budget = int(target * random_ratio)
|
||||
compressible_budget = target - random_budget
|
||||
compressible_written = 0
|
||||
random_written = 0
|
||||
files = 0
|
||||
|
||||
# A handful of fixed, human-meaningful files (directories, small files, a
|
||||
# binary blob) as long as they fit inside the compressible budget.
|
||||
for rel_path, content in STRUCTURED_FILES.items():
|
||||
if written >= structured_budget:
|
||||
if compressible_written + len(content) > compressible_budget:
|
||||
break
|
||||
full_path = os.path.join(source_dir, rel_path)
|
||||
os.makedirs(os.path.dirname(full_path), exist_ok=True)
|
||||
with open(full_path, "wb") as f:
|
||||
f.write(content)
|
||||
written += len(content)
|
||||
compressible_written += len(content)
|
||||
files += 1
|
||||
|
||||
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
|
||||
# Fill the rest of the compressible share with generated repeated content.
|
||||
if compressible_written < compressible_budget:
|
||||
os.makedirs(os.path.join(source_dir, "compressible"), exist_ok=True)
|
||||
i = 0
|
||||
while written < target:
|
||||
chunk_size = min(5 * 1024 * 1024, target - written)
|
||||
with open(os.path.join(source_dir, f"bulk/file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk_size))
|
||||
written += chunk_size
|
||||
while compressible_written < compressible_budget:
|
||||
chunk = min(1024 * 1024, compressible_budget - compressible_written)
|
||||
write_compressible(os.path.join(source_dir, "compressible", f"text_{i}.dat"), chunk)
|
||||
compressible_written += chunk
|
||||
files += 1
|
||||
i += 1
|
||||
|
||||
return written
|
||||
# Incompressible share.
|
||||
if random_written < random_budget:
|
||||
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
|
||||
i = 0
|
||||
while random_written < random_budget:
|
||||
chunk = min(5 * 1024 * 1024, random_budget - random_written)
|
||||
with open(os.path.join(source_dir, "bulk", f"file_{i}.dat"), "wb") as f:
|
||||
f.write(random.randbytes(chunk))
|
||||
random_written += chunk
|
||||
files += 1
|
||||
i += 1
|
||||
|
||||
return {
|
||||
"total_bytes": compressible_written + random_written,
|
||||
"compressible_bytes": compressible_written,
|
||||
"random_bytes": random_written,
|
||||
"files": files,
|
||||
}
|
||||
|
||||
|
||||
def list_relative_files(root):
|
||||
"""Return the set of file paths (relative to root) under a directory."""
|
||||
found = set()
|
||||
for dirpath, _dirnames, filenames in os.walk(root):
|
||||
for name in filenames:
|
||||
full = os.path.join(dirpath, name)
|
||||
found.add(os.path.relpath(full, root))
|
||||
return found
|
||||
|
||||
|
||||
def verify_transfer(source_dir, dest_dir):
|
||||
"""Recursively check dest matches source (paths, sizes, content).
|
||||
|
||||
Returns (ok, detail). Content is compared byte-for-byte, never hashed, so
|
||||
collisions are impossible. This is intentionally not part of the timing.
|
||||
"""
|
||||
if not os.path.isdir(dest_dir):
|
||||
return False, "destination directory missing"
|
||||
src_files = list_relative_files(source_dir)
|
||||
dst_files = list_relative_files(dest_dir)
|
||||
if src_files != dst_files:
|
||||
missing = src_files - dst_files
|
||||
extra = dst_files - src_files
|
||||
return False, f"path set mismatch (missing {len(missing)}, extra {len(extra)})"
|
||||
for rel in sorted(src_files):
|
||||
src = os.path.join(source_dir, rel)
|
||||
dst = os.path.join(dest_dir, rel)
|
||||
if os.path.getsize(src) != os.path.getsize(dst):
|
||||
return False, f"size mismatch: {rel}"
|
||||
if not filecmp.cmp(src, dst, shallow=False):
|
||||
return False, f"content mismatch: {rel}"
|
||||
return True, ""
|
||||
|
||||
|
||||
def percentile(values, pct):
|
||||
"""Linear-interpolation percentile (matches numpy's default method)."""
|
||||
if not values:
|
||||
return None
|
||||
ordered = sorted(values)
|
||||
if len(ordered) == 1:
|
||||
return ordered[0]
|
||||
rank = (len(ordered) - 1) * (pct / 100.0)
|
||||
low = math.floor(rank)
|
||||
high = math.ceil(rank)
|
||||
if low == high:
|
||||
return ordered[int(rank)]
|
||||
return ordered[low] + (ordered[high] - ordered[low]) * (rank - low)
|
||||
|
||||
|
||||
def find_free_port():
|
||||
@@ -207,18 +315,45 @@ def wait_proc(proc, timeout=5):
|
||||
proc.wait()
|
||||
|
||||
|
||||
def _tc_base_cmd():
|
||||
"""Return the command prefix for tc, honouring root vs sudo."""
|
||||
tc = shutil.which("tc")
|
||||
if not tc:
|
||||
raise RuntimeError(
|
||||
"tc (iproute2) not found in PATH; install iproute2 to use network profiles")
|
||||
if os.geteuid() == 0:
|
||||
return [tc]
|
||||
sudo = shutil.which("sudo")
|
||||
if sudo:
|
||||
return [sudo, tc]
|
||||
raise RuntimeError(
|
||||
"applying network limits requires root or sudo; "
|
||||
"re-run as root or install sudo")
|
||||
|
||||
|
||||
def _run_tc(args, check=True):
|
||||
return subprocess.run(_tc_base_cmd() + args, check=check, capture_output=True)
|
||||
|
||||
|
||||
def netem_apply(delay=None, jitter=None, throughput=None, loss=None):
|
||||
"""Apply tc/netem rules to loopback. Pass None to skip a parameter."""
|
||||
netem_reset()
|
||||
cmd = ["sudo", "tc", "qdisc", "add", "dev", "lo", "root", "netem"]
|
||||
params = []
|
||||
if throughput:
|
||||
cmd += ["rate", throughput]
|
||||
params += ["rate", throughput]
|
||||
if delay:
|
||||
cmd += ["delay", delay, jitter or "0ms"]
|
||||
params += ["delay", delay, jitter or "0ms"]
|
||||
if loss:
|
||||
cmd += ["loss", loss]
|
||||
if len(cmd) > 6:
|
||||
subprocess.run(cmd, check=True, capture_output=True)
|
||||
params += ["loss", loss]
|
||||
if not params:
|
||||
return
|
||||
try:
|
||||
_run_tc(["qdisc", "add", "dev", "lo", "root", "netem"] + params)
|
||||
except subprocess.CalledProcessError as exc:
|
||||
detail = exc.stderr.decode(errors="replace").strip() if exc.stderr else str(exc)
|
||||
raise RuntimeError(f"failed to apply network profile via tc/netem: {detail}") from exc
|
||||
except RuntimeError:
|
||||
raise
|
||||
|
||||
|
||||
def netem_apply_profile(profile_name):
|
||||
@@ -235,7 +370,11 @@ def netem_apply_profile(profile_name):
|
||||
|
||||
|
||||
def netem_reset():
|
||||
subprocess.run("sudo tc qdisc del dev lo root".split(), capture_output=True)
|
||||
"""Best-effort removal of any loopback qdisc. Always safe to call."""
|
||||
try:
|
||||
_run_tc(["qdisc", "del", "dev", "lo", "root"], check=False)
|
||||
except Exception:
|
||||
pass
|
||||
|
||||
|
||||
def run_fastsync(source_dir, dest_dir, flags, port):
|
||||
@@ -252,8 +391,10 @@ def run_fastsync(source_dir, dest_dir, flags, port):
|
||||
duration = time.monotonic() - start
|
||||
if result.returncode == 0:
|
||||
return duration
|
||||
sys.stderr.write(f" fastsync failed (exit {result.returncode}): "
|
||||
f"{result.stderr.strip()[:500]}\n")
|
||||
except subprocess.TimeoutExpired:
|
||||
pass
|
||||
sys.stderr.write(" fastsync timed out after 120s\n")
|
||||
return None
|
||||
|
||||
|
||||
@@ -269,8 +410,10 @@ def run_rsync(source_dir, dest_dir, flags, rsync_daemon=None):
|
||||
duration = time.monotonic() - start
|
||||
if result.returncode == 0:
|
||||
return duration
|
||||
sys.stderr.write(f" rsync failed (exit {result.returncode}): "
|
||||
f"{result.stderr.strip()[:500]}\n")
|
||||
except subprocess.TimeoutExpired:
|
||||
pass
|
||||
sys.stderr.write(" rsync timed out after 120s\n")
|
||||
return None
|
||||
|
||||
|
||||
@@ -282,7 +425,81 @@ def run_transfer(config, source_dir, dest_dir, port=None, rsync_daemon=None):
|
||||
return run_fastsync(source_dir, dest_dir, config["flags"], port)
|
||||
|
||||
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=None):
|
||||
def apply_incremental_changes(source_dir, target_bytes):
|
||||
"""Add and modify a few files so a warm transfer has real work to do.
|
||||
|
||||
Returns a mutation record (changed byte count plus enough data to revert
|
||||
and re-apply it) so every warm run can start from a pristine source.
|
||||
"""
|
||||
modified_n = 3
|
||||
added_n = 2
|
||||
per_file = max(4096, target_bytes // (modified_n + added_n))
|
||||
modified = {}
|
||||
added = {}
|
||||
changed = 0
|
||||
|
||||
existing = sorted(list_relative_files(source_dir))
|
||||
if existing:
|
||||
step = max(1, len(existing) // modified_n)
|
||||
for rel in existing[::step][:modified_n]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
original_size = os.path.getsize(path)
|
||||
with open(path, "ab") as f:
|
||||
f.write(random.randbytes(per_file))
|
||||
modified[rel] = (original_size, per_file)
|
||||
changed += per_file
|
||||
|
||||
for i in range(added_n):
|
||||
os.makedirs(os.path.join(source_dir, "incremental"), exist_ok=True)
|
||||
rel = os.path.join("incremental", f"new_{i}.dat")
|
||||
write_compressible(os.path.join(source_dir, rel), per_file)
|
||||
added[rel] = per_file
|
||||
changed += per_file
|
||||
|
||||
return {"changed": changed, "modified": modified, "added": added}
|
||||
|
||||
|
||||
def revert_incremental_changes(source_dir, mutation):
|
||||
"""Undo apply_incremental_changes so the source is pristine again."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (original_size, _appended) in mutation["modified"].items():
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
with open(path, "r+b") as f:
|
||||
f.truncate(original_size)
|
||||
for rel in mutation["added"]:
|
||||
path = os.path.join(source_dir, rel)
|
||||
if os.path.exists(path):
|
||||
os.remove(path)
|
||||
|
||||
|
||||
def reapply_incremental_changes(source_dir, mutation):
|
||||
"""Re-apply a mutation after an untimed pristine seed transfer."""
|
||||
if not mutation:
|
||||
return
|
||||
for rel, (_original_size, appended) in mutation["modified"].items():
|
||||
with open(os.path.join(source_dir, rel), "ab") as f:
|
||||
f.write(random.randbytes(appended))
|
||||
for rel, size in mutation["added"].items():
|
||||
write_compressible(os.path.join(source_dir, rel), size)
|
||||
|
||||
|
||||
def expected_received_root(dest_dir, source_dir, tool):
|
||||
"""Where a tool places transferred files inside dest_dir.
|
||||
|
||||
FastSync mirrors the absolute source path under dest_dir (see the
|
||||
integration suite's get_dest_received_dir); rsync copies the source tree
|
||||
contents directly into dest_dir.
|
||||
"""
|
||||
if tool == "rsync":
|
||||
return dest_dir
|
||||
return os.path.join(dest_dir, os.path.abspath(source_dir).lstrip(os.sep))
|
||||
|
||||
|
||||
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
|
||||
measure_bytes, verify=True, warm=False, mutation=None,
|
||||
progress=None):
|
||||
"""Run benchmark for all configs, returns list of results."""
|
||||
is_limited = profile_name != "unlimited"
|
||||
has_rsync = any(c["tool"] == "rsync" for c in configs)
|
||||
@@ -298,7 +515,10 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
results = []
|
||||
for config in configs:
|
||||
times = []
|
||||
invalid = 0
|
||||
for run_idx in range(runs):
|
||||
if warm:
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
if os.path.exists(dest_dir):
|
||||
shutil.rmtree(dest_dir)
|
||||
os.makedirs(dest_dir, exist_ok=True)
|
||||
@@ -306,15 +526,33 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
port = find_free_port()
|
||||
server = None
|
||||
try:
|
||||
if config["tool"] == "fastsync":
|
||||
if config["tool"] == "fastsync" or warm:
|
||||
server = subprocess.Popen(
|
||||
SERVER_CMD + ["-p", str(port)],
|
||||
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
|
||||
)
|
||||
wait_for_port(port)
|
||||
|
||||
if warm:
|
||||
seed = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if seed is None:
|
||||
invalid += 1
|
||||
sys.stderr.write(" warm-mode seeding failed; run not counted\n")
|
||||
continue
|
||||
reapply_incremental_changes(source_dir, mutation)
|
||||
|
||||
t = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
|
||||
if t is not None:
|
||||
if t is None:
|
||||
invalid += 1
|
||||
elif verify:
|
||||
root = expected_received_root(dest_dir, source_dir, config["tool"])
|
||||
ok, detail = verify_transfer(source_dir, root)
|
||||
if ok:
|
||||
times.append(t)
|
||||
else:
|
||||
invalid += 1
|
||||
sys.stderr.write(f" verification FAILED ({detail}); run not counted\n")
|
||||
else:
|
||||
times.append(t)
|
||||
finally:
|
||||
if server:
|
||||
@@ -327,15 +565,22 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
"config": config["name"],
|
||||
"tool": config["tool"],
|
||||
"profile": profile_name,
|
||||
"warm": warm,
|
||||
"runs": len(times),
|
||||
"invalid": invalid,
|
||||
"times": [round(t, 4) for t in times],
|
||||
}
|
||||
if times:
|
||||
entry["p50"] = round(statistics.median(times), 4)
|
||||
entry["p95"] = round(sorted(times)[int(len(times) * 0.95)], 4) if len(times) > 1 else entry["p50"]
|
||||
p50 = percentile(times, 50)
|
||||
p95 = percentile(times, 95)
|
||||
entry["p50"] = round(p50, 4)
|
||||
entry["p95"] = round(p95, 4)
|
||||
entry["min"] = round(min(times), 4)
|
||||
entry["max"] = round(max(times), 4)
|
||||
entry["stdev"] = round(statistics.stdev(times), 4) if len(times) > 1 else 0.0
|
||||
if measure_bytes:
|
||||
entry["throughput_mbps"] = round(
|
||||
(measure_bytes / (1024 * 1024)) / p50, 3)
|
||||
results.append(entry)
|
||||
return results
|
||||
finally:
|
||||
@@ -345,44 +590,59 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=No
|
||||
netem_reset()
|
||||
|
||||
|
||||
def print_table(results, total_bytes, random_ratio):
|
||||
def print_table(results, measure_bytes, stats, warm):
|
||||
"""Print results as a human-readable table grouped by profile."""
|
||||
profiles = {}
|
||||
for r in results:
|
||||
profiles.setdefault(r["profile"], []).append(r)
|
||||
|
||||
total = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total * 100 if total else 0
|
||||
rand_pct = stats["random_bytes"] / total * 100 if total else 0
|
||||
|
||||
for profile, entries in profiles.items():
|
||||
params = NETWORK_PROFILES.get(profile, {})
|
||||
print(f"\n{'=' * 85}")
|
||||
print(f"\n{'=' * 95}")
|
||||
print(f" Profile: {profile.upper()}")
|
||||
if params.get("rate"):
|
||||
print(f" Network: {params['rate']}, {params['delay']} +/- {params['jitter']}, loss {params['loss']}")
|
||||
else:
|
||||
print(f" Network: unlimited")
|
||||
print(f" Data: {total_bytes / (1024*1024):.1f} MB ({random_ratio*100:.0f}% random, {(1-random_ratio)*100:.0f}% compressible)")
|
||||
print(f"{'=' * 85}")
|
||||
print(f" Data: {total / (1024*1024):.1f} MB "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)")
|
||||
if warm:
|
||||
print(f" Mode: warm (incremental) — measured {measure_bytes / (1024*1024):.2f} MB "
|
||||
f"changed after an untimed full seed")
|
||||
else:
|
||||
print(" Mode: cold (full copy)")
|
||||
print(f"{'=' * 95}")
|
||||
|
||||
fs_entries = [e for e in entries if e.get("tool") == "fastsync"]
|
||||
rsync_entries = [e for e in entries if e.get("tool") == "rsync"]
|
||||
|
||||
header = (f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} "
|
||||
f"{'stdev':>8} {'MB/s':>9} {'runs':>5} {'bad':>4}")
|
||||
rule = (f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} "
|
||||
f"{'-' * 8} {'-' * 9} {'-' * 5} {'-' * 4}")
|
||||
|
||||
if fs_entries:
|
||||
print(f"\n FastSync:")
|
||||
print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
print(header)
|
||||
print(rule)
|
||||
for e in sorted(fs_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if rsync_entries:
|
||||
print(f"\n rsync:")
|
||||
print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
|
||||
print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
|
||||
print(header)
|
||||
print(rule)
|
||||
for e in sorted(rsync_entries, key=lambda x: x.get("p50", 999)):
|
||||
_print_entry(e)
|
||||
|
||||
if params.get("rate_bps") and fs_entries and rsync_entries:
|
||||
fs_best = min((e["p50"] for e in fs_entries if "p50" in e), default=None)
|
||||
rsync_best = min((e["p50"] for e in rsync_entries if "p50" in e), default=None)
|
||||
theoretical = total_bytes / params["rate_bps"]
|
||||
theoretical = measure_bytes / params["rate_bps"]
|
||||
if fs_best and rsync_best:
|
||||
print(f"\n Theoretical max (line rate): {theoretical:.4f}s")
|
||||
print(f" FastSync best: {fs_best:.4f}s ({theoretical/fs_best:.2f}x vs line rate)")
|
||||
@@ -392,10 +652,43 @@ def print_table(results, total_bytes, random_ratio):
|
||||
|
||||
def _print_entry(e):
|
||||
if "p50" in e:
|
||||
print(f" {e['config']:<25} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} {e['runs']:>5}")
|
||||
tp = f"{e['throughput_mbps']:.2f}" if "throughput_mbps" in e else "N/A"
|
||||
print(f" {e['config']:<38} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
|
||||
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} "
|
||||
f"{tp:>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
else:
|
||||
print(f" {e['config']:<25} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {e['runs']:>5}")
|
||||
print(f" {e['config']:<38} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} "
|
||||
f"{'N/A':>8} {'N/A':>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
|
||||
|
||||
|
||||
def configure_build_dirs(build_dir):
|
||||
"""Install the selected build directory and derived binary paths."""
|
||||
global BUILD_DIR, SERVER_CMD, CLIENT_CMD
|
||||
if not os.path.isabs(build_dir):
|
||||
build_dir = os.path.join(PROJECT_ROOT, build_dir)
|
||||
BUILD_DIR = os.path.abspath(build_dir)
|
||||
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
|
||||
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
|
||||
|
||||
|
||||
def build_project():
|
||||
"""Configure (Release) and build into the dedicated bench build dir."""
|
||||
if shutil.which("cmake") is None:
|
||||
sys.stderr.write("cmake not found in PATH; cannot build\n")
|
||||
sys.exit(1)
|
||||
os.makedirs(BUILD_DIR, exist_ok=True)
|
||||
configure = ["cmake", "-B", BUILD_DIR, "-S", PROJECT_ROOT,
|
||||
"-DCMAKE_BUILD_TYPE=Release"]
|
||||
result = subprocess.run(configure, capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("CMake configure failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
jobs = str(os.cpu_count() or 1)
|
||||
result = subprocess.run(["cmake", "--build", BUILD_DIR, "-j", jobs],
|
||||
capture_output=True, text=True)
|
||||
if result.returncode != 0:
|
||||
sys.stderr.write("Build failed:\n" + result.stdout + result.stderr + "\n")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def main():
|
||||
@@ -412,12 +705,20 @@ Custom network limits (--delay/--jitter/--throughput) override profiles.
|
||||
|
||||
Data mix:
|
||||
Default is ~75%% random/incompressible + ~25%% structured/compressible,
|
||||
reflecting typical real-world file sets.
|
||||
reflecting typical real-world file sets. The actual mix is measured and
|
||||
reported. Transfers are verified (destination must match source) unless
|
||||
--no-verify is given.
|
||||
|
||||
Warm mode:
|
||||
--warm seeds the destination with an untimed full copy of a pristine base,
|
||||
then measures only the incremental transfer after modifying a few files.
|
||||
|
||||
Examples:
|
||||
%(prog)s --profiles wan --runs 5
|
||||
%(prog)s --throughput 50mbit --delay 30ms --jitter 5ms
|
||||
%(prog)s --random-ratio 0.5 --size-mb 100
|
||||
%(prog)s --warm --runs 3 --no-rsync
|
||||
%(prog)s --dry-run --size-mb 4 --random-ratio 0.25
|
||||
""")
|
||||
parser.add_argument("--runs", type=int, default=3,
|
||||
help="Number of runs per config (default: 3)")
|
||||
@@ -425,7 +726,8 @@ Examples:
|
||||
choices=list(NETWORK_PROFILES.keys()),
|
||||
help="Predefined network profiles (default: unlimited)")
|
||||
parser.add_argument("--configs", nargs="+", default=None,
|
||||
help="Custom FastSync config flags")
|
||||
help="Custom FastSync config flags (shell-quoted, e.g. "
|
||||
"\"-j -z --chunk-serialization\")")
|
||||
parser.add_argument("--size-mb", type=int, default=25,
|
||||
help="Test data size in MB (default: 25)")
|
||||
parser.add_argument("--random-ratio", type=float, default=0.75,
|
||||
@@ -440,6 +742,14 @@ Examples:
|
||||
help="Custom packet loss (e.g. 1%%)")
|
||||
parser.add_argument("--no-rsync", action="store_true",
|
||||
help="Skip rsync comparison")
|
||||
parser.add_argument("--no-verify", action="store_true",
|
||||
help="Skip source/destination verification after each run")
|
||||
parser.add_argument("--warm", action="store_true",
|
||||
help="Incremental mode: seed dest first, measure only changes")
|
||||
parser.add_argument("--build-dir", default=DEFAULT_BUILD_DIR,
|
||||
help=f"Build directory (default: {DEFAULT_BUILD_DIR})")
|
||||
parser.add_argument("--dry-run", action="store_true",
|
||||
help="Only generate data and report its composition, then exit")
|
||||
parser.add_argument("--progress", action="store_true",
|
||||
help="Show progress bar with ETA")
|
||||
parser.add_argument("--output", choices=["table", "json"], default="table",
|
||||
@@ -448,12 +758,48 @@ Examples:
|
||||
help="Don't clean up test data")
|
||||
args = parser.parse_args()
|
||||
|
||||
# Build
|
||||
print("Building...")
|
||||
if os.system(f"cmake -B {BUILD_DIR} -S {PROJECT_ROOT} > /dev/null 2>&1") != 0:
|
||||
print("CMake configure failed"); sys.exit(1)
|
||||
if os.system(f"cmake --build {BUILD_DIR} -j$(nproc) > /dev/null 2>&1") != 0:
|
||||
print("Build failed"); sys.exit(1)
|
||||
if not 0.0 <= args.random_ratio <= 1.0:
|
||||
parser.error("--random-ratio must be between 0.0 and 1.0")
|
||||
if args.size_mb <= 0:
|
||||
parser.error("--size-mb must be positive")
|
||||
|
||||
configure_build_dirs(args.build_dir)
|
||||
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
stats = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
total_bytes = stats["total_bytes"]
|
||||
comp_pct = stats["compressible_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
rand_pct = stats["random_bytes"] / total_bytes * 100 if total_bytes else 0
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB in {stats['files']} files "
|
||||
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)",
|
||||
file=sys.stderr)
|
||||
|
||||
if args.dry_run:
|
||||
print(f"size_mb={args.size_mb} random_ratio={args.random_ratio:.4f} "
|
||||
f"total_bytes={stats['total_bytes']} "
|
||||
f"compressible_bytes={stats['compressible_bytes']} "
|
||||
f"random_bytes={stats['random_bytes']} files={stats['files']}")
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
return
|
||||
|
||||
# Warm mode: keep a pristine base copy, then mutate the live source.
|
||||
base_dir = None
|
||||
measure_bytes = total_bytes
|
||||
mutation = None
|
||||
if args.warm:
|
||||
change_target = max(64 * 1024, min(int(total_bytes * 0.01), 4 * 1024 * 1024))
|
||||
mutation = apply_incremental_changes(source_dir, change_target)
|
||||
measure_bytes = mutation["changed"]
|
||||
revert_incremental_changes(source_dir, mutation)
|
||||
print(f"Warm mode: each run seeds a full copy, then measures "
|
||||
f"{measure_bytes / (1024*1024):.3f} MB of add/change deltas", file=sys.stderr)
|
||||
|
||||
# Build (Release: benchmarking a debug build is meaningless)
|
||||
print(f"Building (Release) into {BUILD_DIR}...", file=sys.stderr)
|
||||
build_project()
|
||||
|
||||
# Determine active profile for display
|
||||
has_custom_net = args.delay or args.jitter or args.throughput or args.loss
|
||||
@@ -473,18 +819,10 @@ Examples:
|
||||
else:
|
||||
profiles_to_run = args.profiles or ["unlimited"]
|
||||
|
||||
# Generate data
|
||||
source_dir = os.path.join(BENCH_DIR, "source")
|
||||
dest_dir = os.path.join(BENCH_DIR, "dest")
|
||||
total_bytes = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
|
||||
compressible_pct = (1 - args.random_ratio) * 100
|
||||
random_pct = args.random_ratio * 100
|
||||
print(f"Generated {total_bytes / (1024*1024):.1f} MB "
|
||||
f"({random_pct:.0f}% random, {compressible_pct:.0f}% compressible)")
|
||||
|
||||
# Build config list
|
||||
# Build config list (shlex so quoted/space-separated flags survive)
|
||||
if args.configs:
|
||||
fastsync_configs = [{"name": c, "flags": c.split(), "tool": "fastsync"} for c in args.configs]
|
||||
fastsync_configs = [{"name": c, "flags": shlex.split(c), "tool": "fastsync"}
|
||||
for c in args.configs]
|
||||
else:
|
||||
fastsync_configs = list(FASTSYNC_CONFIGS)
|
||||
|
||||
@@ -496,14 +834,21 @@ Examples:
|
||||
total_runs = len(configs) * args.runs * len(profiles_to_run)
|
||||
progress = Progress(total_runs, "Benchmarking") if args.progress else None
|
||||
if progress:
|
||||
print(f"Running {total_runs} transfers...")
|
||||
print(f"Running {total_runs} transfers...", file=sys.stderr)
|
||||
|
||||
all_results = []
|
||||
try:
|
||||
for profile in profiles_to_run:
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile, progress)
|
||||
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile,
|
||||
measure_bytes, verify=not args.no_verify,
|
||||
warm=args.warm, mutation=mutation,
|
||||
progress=progress)
|
||||
all_results.extend(results)
|
||||
except RuntimeError as exc:
|
||||
sys.stderr.write(f"error: {exc}\n")
|
||||
sys.exit(1)
|
||||
finally:
|
||||
netem_reset()
|
||||
if not args.keep_data:
|
||||
shutil.rmtree(BENCH_DIR, ignore_errors=True)
|
||||
|
||||
@@ -511,7 +856,7 @@ Examples:
|
||||
if args.output == "json":
|
||||
print(json.dumps(all_results, indent=2))
|
||||
else:
|
||||
print_table(all_results, total_bytes, args.random_ratio)
|
||||
print_table(all_results, measure_bytes, stats, args.warm)
|
||||
print()
|
||||
|
||||
|
||||
|
||||
@@ -3,26 +3,59 @@
|
||||
}:
|
||||
|
||||
pkgs.mkShell {
|
||||
# Development shell for FastSync. Provides the host-side toolchain needed to
|
||||
# build, lint, unit-test, integration-test and benchmark the project.
|
||||
# It deliberately does NOT build on entry: run the CMake commands in README.md
|
||||
# (or use the CI Docker image for exact CI parity).
|
||||
nativeBuildInputs = with pkgs; [
|
||||
# build
|
||||
gcc
|
||||
cmake
|
||||
gnumake
|
||||
pkg-config
|
||||
# lint / static analysis (matches CI)
|
||||
clang-tools # clang-format
|
||||
cppcheck
|
||||
# tests
|
||||
(python3.withPackages (ps: with ps; [ pytest pytest-xdist psutil ]))
|
||||
openssh # SSH transport integration tests
|
||||
# debugging
|
||||
gdb
|
||||
valgrind
|
||||
# coverage
|
||||
lcov
|
||||
# benchmark tooling
|
||||
rsync
|
||||
iproute2 # tc/netem for network shaping
|
||||
# misc
|
||||
git
|
||||
curl
|
||||
nodejs
|
||||
nixpkgs-fmt
|
||||
docker
|
||||
tea
|
||||
];
|
||||
|
||||
buildInputs = with pkgs; [
|
||||
zstd
|
||||
zlib
|
||||
lz4
|
||||
openssl
|
||||
(python3.withPackages (ps: with ps; [ pytest ]))
|
||||
];
|
||||
|
||||
# The CMake configure step fetches xxHash via FetchContent, which needs
|
||||
# network access; NIX_ENFORCE_PURITY must be off so the sandbox does not block.
|
||||
NIX_ENFORCE_PURITY = 0;
|
||||
|
||||
shellHook = ''
|
||||
export NIX_ENFORCE_PURITY=0
|
||||
cmake -B build
|
||||
# Make an existing build tree available on PATH, but never build here.
|
||||
if [ -d "$PWD/build" ]; then
|
||||
export PATH="$PWD/build:$PATH"
|
||||
fi
|
||||
echo "FastSync dev shell ready."
|
||||
echo " Build: cmake -B build -S . && cmake --build build -j\$(nproc)"
|
||||
echo " Unit: ./build/tests"
|
||||
echo " CI parity: docker run --rm --user \"\$(id -u):\$(id -g)\" -v \"\$PWD:/workspace\" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v11 ..."
|
||||
'';
|
||||
}
|
||||
|
||||
+455
-154
@@ -1,5 +1,7 @@
|
||||
#include "change_list.h"
|
||||
#include "checksum.h"
|
||||
#include "utils.h"
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
@@ -7,21 +9,7 @@
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
/* Itemize code emitted for a transferred regular file.
|
||||
*
|
||||
* Layout (rsync-compatible 11-char item): `>f` marks a regular file that was
|
||||
* transferred to the remote host; the trailing nine markers are, in order,
|
||||
* c(hecksum) s(ize) t(ime) p(erms) o(wner) g(roup) u(ser/acl) a(ttrs) x(attrs).
|
||||
* Every marker is `+` (FastSync does not compare each attribute on the
|
||||
* receiving side, so a sent file is reported as fully updated). Files that
|
||||
* are already up to date print no line at all, matching rsync's single -i
|
||||
* which only itemizes changes.
|
||||
*
|
||||
* Because the scanner only yields regular-file transfer candidates, `>d`
|
||||
* (directory) lines are never produced; directories are not transferred as
|
||||
* items by FastSync. */
|
||||
#define ITEMIZE_SENT_FILE ">f+++++++++"
|
||||
#include <unistd.h>
|
||||
|
||||
typedef struct {
|
||||
char* data;
|
||||
@@ -80,103 +68,14 @@ static bool strbuf_append(StrBuf* buf, const char* text) {
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool strbuf_append_ull(StrBuf* buf, unsigned long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
static bool strbuf_append_longlong(StrBuf* buf, long long value) {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%lld", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
return strbuf_append(buf, digits);
|
||||
}
|
||||
|
||||
bool change_list_enabled(const Config* config) {
|
||||
return config != NULL && (config->itemize_changes || config->out_format != NULL ||
|
||||
(config->log_file != NULL && config->log_file_format != NULL));
|
||||
}
|
||||
|
||||
char* change_render_itemize(const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
const char* code = event->is_directory ? ">d+++++++++" : ITEMIZE_SENT_FILE;
|
||||
StrBuf line = {0};
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append(&line, " ") &&
|
||||
strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
/* ---- Itemize code ---- */
|
||||
|
||||
static const char* leaf_name(const char* path) {
|
||||
if (path == NULL)
|
||||
return "";
|
||||
const char* slash = strrchr(path, '/');
|
||||
return slash != NULL && slash[1] != '\0' ? slash + 1 : path;
|
||||
}
|
||||
|
||||
char* change_render_format(const char* format, const ChangeEvent* event) {
|
||||
if (format == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = strbuf_append(&line, leaf_name(event->path));
|
||||
break;
|
||||
case 'l':
|
||||
ok = strbuf_append_ull(&line, event->size);
|
||||
break;
|
||||
case 'b':
|
||||
ok = strbuf_append_ull(&line, event->bytes_sent);
|
||||
break;
|
||||
case 'M':
|
||||
ok = strbuf_append_longlong(&line, (long long)event->mtime_sec);
|
||||
break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* Format a mode as an `ls -l` permission string, e.g. `-rw-r--r--`. */
|
||||
/* Format the permission bits as an `ls -l` string, e.g. `-rw-r--r--`. */
|
||||
static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[0] = S_ISDIR(mode) ? 'd'
|
||||
: S_ISLNK(mode) ? 'l'
|
||||
@@ -198,29 +97,94 @@ static void mode_to_ls_string(mode_t mode, char out[11]) {
|
||||
out[10] = '\0';
|
||||
}
|
||||
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime,
|
||||
const char* path) {
|
||||
char permission[11];
|
||||
mode_to_ls_string(mode, permission);
|
||||
char date[32];
|
||||
struct tm broken_down;
|
||||
if (localtime_r(&mtime, &broken_down) != NULL) {
|
||||
if (strftime(date, sizeof(date), "%Y/%m/%d %H:%M:%S", &broken_down) == 0)
|
||||
snprintf(date, sizeof(date), "?");
|
||||
} else {
|
||||
snprintf(date, sizeof(date), "?");
|
||||
static char itemize_type_char(const ChangeEvent* event) {
|
||||
if (event->is_directory)
|
||||
return 'd';
|
||||
if (event->is_symlink)
|
||||
return 'L';
|
||||
if (event->is_special) {
|
||||
if (S_ISCHR(event->mode) || S_ISBLK(event->mode))
|
||||
return 'D';
|
||||
return 'S';
|
||||
}
|
||||
return 'f';
|
||||
}
|
||||
|
||||
static bool times_match(const Config* config, const ChangeEvent* event) {
|
||||
if (!event->dest.known || !event->dest.existed)
|
||||
return false;
|
||||
if (event->mtime_sec == event->dest.mtime_sec)
|
||||
return event->mtime_nsec == event->dest.mtime_nsec;
|
||||
long long delta = (long long)event->mtime_sec - (long long)event->dest.mtime_sec;
|
||||
if (delta < 0)
|
||||
delta = -delta;
|
||||
return delta <= (long long)config->modify_window;
|
||||
}
|
||||
|
||||
/* Fill the 11-character itemize code (10 chars + NUL). `created` means the
|
||||
* destination entry did not exist, so every attribute marker is `+`. */
|
||||
static void itemize_code(const Config* config, const ChangeEvent* event, char code[12]) {
|
||||
bool known = event->dest.known;
|
||||
bool created = !known || !event->dest.existed;
|
||||
char update;
|
||||
if (event->is_hardlink)
|
||||
update = 'h';
|
||||
else if (created)
|
||||
update = (event->is_directory || event->is_symlink || event->is_special) ? 'c' : '>';
|
||||
else
|
||||
update = '>';
|
||||
code[0] = update;
|
||||
code[1] = itemize_type_char(event);
|
||||
if (created) {
|
||||
for (int i = 0; i < 9; i++)
|
||||
code[2 + i] = '+';
|
||||
code[11] = '\0';
|
||||
return;
|
||||
}
|
||||
bool size_diff = event->size != event->dest.size;
|
||||
bool time_diff = !times_match(config, event);
|
||||
bool perms_diff = (event->mode & 07777) != (event->dest.mode & 07777);
|
||||
bool owner_diff = event->uid != (uid_t)event->dest.uid;
|
||||
bool group_diff = event->gid != (gid_t)event->dest.gid;
|
||||
code[2] = '.'; /* checksum: no destination digest available */
|
||||
code[3] = size_diff ? 's' : '.';
|
||||
code[4] = time_diff ? 't' : '.';
|
||||
code[5] = (config->preserve_perms && perms_diff) ? 'p' : '.';
|
||||
code[6] = (config->preserve_owner && owner_diff) ? 'o' : '.';
|
||||
code[7] = (config->preserve_group && group_diff) ? 'g' : '.';
|
||||
code[8] = '.'; /* reserved */
|
||||
code[9] = '.'; /* acl: not compared */
|
||||
code[10] = '.';
|
||||
code[11] = '\0';
|
||||
}
|
||||
|
||||
/* rsync %n: the transfer-relative name, with a trailing slash for directories. */
|
||||
static bool append_name(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (!strbuf_append(buf, event->name != NULL ? event->name : ""))
|
||||
return false;
|
||||
if (event->is_directory && (event->name == NULL || event->name[0] == '\0' ||
|
||||
event->name[strlen(event->name) - 1] != '/'))
|
||||
return strbuf_append_char(buf, '/');
|
||||
return true;
|
||||
}
|
||||
|
||||
/* rsync %L: " -> target" for a symlink, " => target" for a hard link, else "". */
|
||||
static bool append_link_suffix(StrBuf* buf, const ChangeEvent* event) {
|
||||
if (event->is_symlink && event->symlink_target != NULL)
|
||||
return strbuf_append(buf, " -> ") && strbuf_append(buf, event->symlink_target);
|
||||
if (event->is_hardlink && event->hardlink_target != NULL)
|
||||
return strbuf_append(buf, " => ") && strbuf_append(buf, event->hardlink_target);
|
||||
return true;
|
||||
}
|
||||
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event) {
|
||||
if (event == NULL || event->decision != CHANGE_SENT)
|
||||
return str_dup("");
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
StrBuf line = {0};
|
||||
char size_field[32];
|
||||
int written = snprintf(size_field, sizeof(size_field), "%llu", size);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, size_field) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, date) && strbuf_append_char(&line, ' ') &&
|
||||
strbuf_append(&line, path != NULL ? path : "");
|
||||
bool ok = strbuf_append(&line, code) && strbuf_append_char(&line, ' ') &&
|
||||
append_name(&line, event) && append_link_suffix(&line, event);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
@@ -228,6 +192,238 @@ char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- --out-format / --log-file-format ---- */
|
||||
|
||||
/* rsync 3.4.1's `%C` uses the negotiated transfer checksum; with the default
|
||||
* "auto" choice on both ends that is xxh128. FastSync's internal XXH64 default
|
||||
* is not an rsync algorithm, so map it to xxh128 for parity. */
|
||||
static ChecksumAlgo out_format_checksum_algo(const Config* config) {
|
||||
switch ((ChecksumAlgo)config->checksum_algo) {
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
return CHECKSUM_ALGO_MD5;
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return CHECKSUM_ALGO_XXH3;
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
return CHECKSUM_ALGO_XXH128;
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
default:
|
||||
return CHECKSUM_ALGO_XXH128;
|
||||
}
|
||||
}
|
||||
|
||||
/* Render a digest as rsync's sum_as_hex: for xxh128 the HIGH 64-bit half is
|
||||
* printed before the low half; every other algorithm prints its bytes in order. */
|
||||
static void digest_to_hex(ChecksumAlgo algo, const uint8_t* digest, size_t len, char* out) {
|
||||
if (algo == CHECKSUM_ALGO_XXH128 && len == 16) {
|
||||
uint64_t low = 0;
|
||||
uint64_t high = 0;
|
||||
memcpy(&low, digest, sizeof(low));
|
||||
memcpy(&high, digest + 8, sizeof(high));
|
||||
snprintf(out, len * 2 + 1, "%016llx%016llx", (unsigned long long)high, (unsigned long long)low);
|
||||
return;
|
||||
}
|
||||
static const char hex[] = "0123456789abcdef";
|
||||
for (size_t i = 0; i < len; i++) {
|
||||
out[i * 2] = hex[(digest[i] >> 4) & 0xf];
|
||||
out[i * 2 + 1] = hex[digest[i] & 0xf];
|
||||
}
|
||||
out[len * 2] = '\0';
|
||||
}
|
||||
|
||||
static bool format_uses_checksum(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0')
|
||||
break;
|
||||
if (token == 'C')
|
||||
return true;
|
||||
p += 2;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Fill event->checksum/checksum_known for a transferred regular file. A
|
||||
* non-regular entry (or a hard-link sibling) leaves checksum_known false, which
|
||||
* renders as spaces like rsync. */
|
||||
static void fill_event_checksum(const Config* config, const File* file, ChangeEvent* event) {
|
||||
if (file == NULL || file->is_dir || file->is_symlink || file->is_special ||
|
||||
(file->link_group != 0 && !file->link_first))
|
||||
return;
|
||||
if (!format_uses_checksum(config->out_format) && !format_uses_checksum(config->log_file_format))
|
||||
return;
|
||||
if (file->path == NULL)
|
||||
return;
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
uint8_t digest[CHECKSUM_MAX_DIGEST_LEN];
|
||||
size_t len = 0;
|
||||
/* rsync's %C is the transfer checksum, which is always seeded with 0 (it is
|
||||
* independent of --checksum-seed, as rsync 3.4.1 demonstrates). */
|
||||
if (!checksum_digest_file(algo, 0, file->path, digest, sizeof(digest), &len))
|
||||
return;
|
||||
digest_to_hex(algo, digest, len, event->checksum);
|
||||
event->checksum_known = true;
|
||||
}
|
||||
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event) {
|
||||
if (format == NULL || event == NULL)
|
||||
return NULL;
|
||||
StrBuf line = {0};
|
||||
bool ok = true;
|
||||
for (const char* p = format; *p != '\0' && ok;) {
|
||||
if (*p != '%') {
|
||||
ok = strbuf_append_char(&line, *p);
|
||||
p++;
|
||||
continue;
|
||||
}
|
||||
char token = p[1];
|
||||
if (token == '\0') {
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
}
|
||||
switch (token) {
|
||||
case '%':
|
||||
ok = strbuf_append_char(&line, '%');
|
||||
break;
|
||||
case 'i': {
|
||||
if (event->deleted) {
|
||||
/* rsync's ITEM_DELETED itemize code: `*deleting ` (11 chars). */
|
||||
ok = strbuf_append(&line, "*deleting ");
|
||||
break;
|
||||
}
|
||||
char code[12];
|
||||
itemize_code(config, event, code);
|
||||
ok = strbuf_append(&line, code);
|
||||
break;
|
||||
}
|
||||
case 'f':
|
||||
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
|
||||
break;
|
||||
case 'n':
|
||||
ok = append_name(&line, event);
|
||||
break;
|
||||
case 'L':
|
||||
ok = append_link_suffix(&line, event);
|
||||
break;
|
||||
case 'l': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->size);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'b': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_sent);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'c': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", event->bytes_read);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'C': {
|
||||
if (event->checksum_known) {
|
||||
ok = strbuf_append(&line, event->checksum);
|
||||
} else {
|
||||
/* rsync pads a non-regular / untransferred entry with spaces. */
|
||||
ChecksumAlgo algo = out_format_checksum_algo(config);
|
||||
int width = checksum_digest_len(algo) * 2;
|
||||
for (int i = 0; i < width && ok; i++)
|
||||
ok = strbuf_append_char(&line, ' ');
|
||||
}
|
||||
} break;
|
||||
case 'M': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(event->mtime_sec, true, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 't': {
|
||||
char when[32];
|
||||
if (format_rsync_datetime(time(NULL), false, when, sizeof(when)))
|
||||
ok = strbuf_append(&line, when);
|
||||
} break;
|
||||
case 'o':
|
||||
ok = strbuf_append(&line, "send");
|
||||
break;
|
||||
case 'p': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%ld", (long)getpid());
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'B': {
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
ok = strbuf_append(&line, permission + 1);
|
||||
} break;
|
||||
case 'U': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->uid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
case 'G': {
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%u", (unsigned)event->gid);
|
||||
ok = written >= 0 && (size_t)written < sizeof(digits) && strbuf_append(&line, digits);
|
||||
} break;
|
||||
default:
|
||||
/* Unknown escape sequences are preserved verbatim. */
|
||||
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
|
||||
break;
|
||||
}
|
||||
p += 2;
|
||||
}
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
if (line.data == NULL) {
|
||||
line.data = str_dup("");
|
||||
if (!line.data)
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- --list-only ---- */
|
||||
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event) {
|
||||
(void)config;
|
||||
if (event == NULL)
|
||||
return NULL;
|
||||
char permission[11];
|
||||
mode_to_ls_string(event->mode, permission);
|
||||
char date[32];
|
||||
if (!format_rsync_datetime(event->mtime_sec, false, date, sizeof(date)))
|
||||
snprintf(date, sizeof(date), "?");
|
||||
StrBuf line = {0};
|
||||
char size_field[40];
|
||||
char grouped[32];
|
||||
if (!format_big_num(event->size, false, grouped, sizeof(grouped))) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
int written = snprintf(size_field, sizeof(size_field), "%15s", grouped);
|
||||
if (written < 0 || (size_t)written >= sizeof(size_field)) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
const char* name = event->name != NULL && event->name[0] != '\0' ? event->name : ".";
|
||||
bool ok = strbuf_append(&line, permission) && strbuf_append(&line, size_field) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, date) &&
|
||||
strbuf_append_char(&line, ' ') && strbuf_append(&line, name);
|
||||
if (!ok) {
|
||||
strbuf_free(&line);
|
||||
return NULL;
|
||||
}
|
||||
return line.data;
|
||||
}
|
||||
|
||||
/* ---- Event emission ---- */
|
||||
|
||||
static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_output) {
|
||||
char* escaped = output_escape(line, eight_bit_output);
|
||||
if (escaped != NULL) {
|
||||
@@ -247,15 +443,16 @@ void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
bool to_stdout = config->itemize_changes || config->out_format != NULL;
|
||||
bool to_log = config->log_file != NULL && config->log_file_format != NULL;
|
||||
if (to_stdout) {
|
||||
char* line = config->out_format != NULL ? change_render_format(config->out_format, event)
|
||||
: change_render_itemize(event);
|
||||
char* line = config->out_format != NULL
|
||||
? change_render_format(config->out_format, config, event)
|
||||
: change_render_itemize(config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(stdout, line, config->eight_bit_output);
|
||||
free(line);
|
||||
}
|
||||
}
|
||||
if (to_log) {
|
||||
char* line = change_render_format(config->log_file_format, event);
|
||||
char* line = change_render_format(config->log_file_format, config, event);
|
||||
if (line != NULL) {
|
||||
print_escaped_line(config->log_file, line, config->eight_bit_output);
|
||||
free(line);
|
||||
@@ -266,9 +463,6 @@ void change_emit(const Config* config, const ChangeEvent* event) {
|
||||
static bool format_uses_mtime(const char* format) {
|
||||
if (format == NULL)
|
||||
return false;
|
||||
/* Mirror change_render_format's tokenizer: "%%" is a literal percent (so
|
||||
* "%%M" does NOT expand %M) and unknown "%X" escapes consume both chars.
|
||||
* This keeps the optional stat() fallback below in step with the renderer. */
|
||||
for (const char* p = format; *p != '\0';) {
|
||||
if (*p != '%') {
|
||||
p++;
|
||||
@@ -284,46 +478,153 @@ static bool format_uses_mtime(const char* format) {
|
||||
return false;
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
/* Relative path of an entry below the transfer root (no leading slash). Uses
|
||||
* the sender-side send_path override when present (bare-relative -R layout). */
|
||||
static char* relative_name(const Config* config, const File* file) {
|
||||
const char* full = file_wire_path(file);
|
||||
if (file->send_path != NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL || full == NULL)
|
||||
return str_dup(full != NULL ? full : "");
|
||||
size_t root_len = strlen(root);
|
||||
while (root_len > 1 && root[root_len - 1] == '/')
|
||||
root_len--;
|
||||
if (strncmp(root, full, root_len) == 0) {
|
||||
if (full[root_len] == '\0')
|
||||
return str_dup("");
|
||||
if (full[root_len] == '/')
|
||||
return str_dup(full + root_len + 1);
|
||||
}
|
||||
return str_dup(full);
|
||||
}
|
||||
|
||||
/* rsync %f long form: the source argument as typed (leading '/' removed,
|
||||
* trailing '/' removed, leading "./" removed) joined to the relative name. */
|
||||
static char* display_name(const Config* config, const char* name) {
|
||||
const char* root = config->send_directory;
|
||||
if (root == NULL)
|
||||
return str_dup(name != NULL ? name : "");
|
||||
const char* p = root;
|
||||
while (*p == '/')
|
||||
p++;
|
||||
if (p[0] == '.' && p[1] == '/')
|
||||
p += 2;
|
||||
size_t root_len = strlen(p);
|
||||
while (root_len > 0 && p[root_len - 1] == '/')
|
||||
root_len--;
|
||||
size_t name_len = name != NULL ? strlen(name) : 0;
|
||||
if (root_len == 0 && name_len == 0)
|
||||
return str_dup("");
|
||||
char* out = malloc(root_len + (root_len > 0 && name_len > 0 ? 1 : 0) + name_len + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
size_t offset = 0;
|
||||
if (root_len > 0) {
|
||||
memcpy(out, p, root_len);
|
||||
offset = root_len;
|
||||
}
|
||||
if (root_len > 0 && name_len > 0)
|
||||
out[offset++] = '/';
|
||||
if (name_len > 0)
|
||||
memcpy(out + offset, name, name_len);
|
||||
out[offset + name_len] = '\0';
|
||||
return out;
|
||||
}
|
||||
|
||||
static void fill_event_from_file(const Config* config, const File* file, ChangeEvent* event,
|
||||
char** name_out, char** path_out) {
|
||||
char* name = relative_name(config, file);
|
||||
char* path = display_name(config, name);
|
||||
event->name = name;
|
||||
event->path = path;
|
||||
*name_out = name;
|
||||
*path_out = path;
|
||||
if (file->metadata != NULL) {
|
||||
event->mtime_sec = file->metadata->mtime_sec;
|
||||
event->mtime_nsec = file->metadata->mtime_nsec;
|
||||
event->mode = file->metadata->mode;
|
||||
event->uid = file->metadata->uid;
|
||||
event->gid = file->metadata->gid;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0) {
|
||||
event->mtime_sec = st.st_mtime;
|
||||
event->mtime_nsec = st.st_mtim.tv_nsec;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
/* The displayed path is the one transmitted (with -R + --files-from this is
|
||||
the bare relative destination path); the metadata fallback below still
|
||||
stats the local absolute path. */
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = false;
|
||||
event.is_symlink = false;
|
||||
event.is_special = false;
|
||||
event.is_hardlink = false;
|
||||
event.size = file->data != NULL ? file->data->size : 0;
|
||||
/* FastSync has no wire-byte counter yet, so %b reports the source length
|
||||
* that had to be delivered (always equal to %l); the actual bytes written
|
||||
* to the socket (compressed/delta) are not measured. */
|
||||
event.bytes_sent = event.size;
|
||||
if (file->metadata != NULL) {
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
|
||||
/* Best-effort fallback for %M when no metadata was captured (no -M): the
|
||||
* path is stat()ed just to fill the field, and any failure leaves 0. */
|
||||
struct stat st;
|
||||
if (file->path != NULL && stat(file->path, &st) == 0)
|
||||
event.mtime_sec = st.st_mtime;
|
||||
event.dest = file->dest_state;
|
||||
if (file->is_symlink) {
|
||||
event.is_symlink = true;
|
||||
event.symlink_target = file->symlink_target;
|
||||
event.size = file->symlink_target != NULL ? strlen(file->symlink_target) : 0;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->is_special) {
|
||||
event.is_special = true;
|
||||
event.bytes_sent = 0;
|
||||
} else if (file->link_group != 0 && !file->link_first) {
|
||||
event.is_hardlink = true;
|
||||
event.hardlink_target = file->hardlink_target;
|
||||
event.bytes_sent = 0;
|
||||
} else {
|
||||
event.bytes_sent = bytes_sent;
|
||||
/* rsync's %c is the block-checksum bytes received for the file. Even a
|
||||
* whole-file transfer (no basis; --append/--inplace included) receives
|
||||
* rsync's 16-byte sum header, so rsync reports 16; a dry run transfers
|
||||
* nothing and reports 0. FastSync's whole-file path has no sum header, so
|
||||
* report rsync's value for parity. With delta enabled the real received
|
||||
* bytes are kept, but FastSync's signature framing differs from rsync's so
|
||||
* those stay numerically divergent. */
|
||||
bool delta_active = config->use_delta && !config->whole_file;
|
||||
event.bytes_read = (!config->dry_run && !delta_active) ? 16 : bytes_read;
|
||||
}
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL) {
|
||||
fill_event_checksum(config, file, &event);
|
||||
change_emit(config, &event);
|
||||
}
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
void change_emit_file_sent(const Config* config, const File* file) {
|
||||
if (file == NULL)
|
||||
return;
|
||||
unsigned long long payload = file->data != NULL ? file->data->size : 0;
|
||||
change_emit_file_sent_bytes(config, file, payload, 0);
|
||||
}
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
void change_emit_dir_sent(const Config* config, const File* file) {
|
||||
if (file == NULL || !change_list_enabled(config))
|
||||
return;
|
||||
ChangeEvent event;
|
||||
memset(&event, 0, sizeof(event));
|
||||
event.path = file_wire_path(file);
|
||||
event.decision = CHANGE_SENT;
|
||||
event.is_directory = true;
|
||||
event.size = 0;
|
||||
event.bytes_sent = 0;
|
||||
if (file->metadata != NULL)
|
||||
event.mtime_sec = file->metadata->mtime_sec;
|
||||
event.dest = file->dest_state;
|
||||
char* name = NULL;
|
||||
char* path = NULL;
|
||||
fill_event_from_file(config, file, &event, &name, &path);
|
||||
if (name != NULL && path != NULL)
|
||||
change_emit(config, &event);
|
||||
free(name);
|
||||
free(path);
|
||||
}
|
||||
|
||||
+51
-24
@@ -2,7 +2,9 @@
|
||||
#define CHANGE_LIST_H
|
||||
|
||||
#include "config.h"
|
||||
#include "checksum.h"
|
||||
#include "file_types.h"
|
||||
#include "format.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
@@ -26,42 +28,59 @@ typedef enum {
|
||||
} ChangeDecision;
|
||||
|
||||
typedef struct {
|
||||
const char* path; /* full source path */
|
||||
const char* path; /* long-form display path (rsync %f) */
|
||||
const char* name; /* transfer-relative path (rsync %n), no trailing slash */
|
||||
ChangeDecision decision;
|
||||
bool is_directory;
|
||||
bool is_symlink;
|
||||
bool is_special;
|
||||
bool is_hardlink; /* a hard-link sibling (linked, no data sent) */
|
||||
bool deleted; /* a would-delete report (-n --delete); no source file */
|
||||
const char* symlink_target;
|
||||
const char* hardlink_target;
|
||||
unsigned long long size; /* source file length in bytes */
|
||||
/* The number of bytes reported for a sent file. FastSync has no wire-byte
|
||||
* counter, so this is always the source length (== size / %l); actual
|
||||
* post-compression/delta bytes on the wire are not counted. */
|
||||
unsigned long long bytes_sent;
|
||||
time_t mtime_sec; /* 0 when unknown */
|
||||
unsigned long long bytes_sent; /* wire bytes actually transferred (rsync %b) */
|
||||
unsigned long long bytes_read; /* wire bytes read back for this file (rsync %c) */
|
||||
/* rsync %C: whole-file checksum hex for a transferred regular file. Only
|
||||
* filled when the active format uses %C (checksum_known == false otherwise,
|
||||
* which renders as spaces like rsync for non-regular entries). */
|
||||
bool checksum_known;
|
||||
char checksum[CHECKSUM_MAX_DIGEST_LEN * 2 + 1];
|
||||
time_t mtime_sec;
|
||||
long mtime_nsec;
|
||||
mode_t mode;
|
||||
uid_t uid;
|
||||
gid_t gid;
|
||||
/* Receiver-reported pre-transfer destination state (OutputDestState.known is
|
||||
* false when no report was requested/received). */
|
||||
OutputDestState dest;
|
||||
} ChangeEvent;
|
||||
|
||||
/* True when any output mode is active and per-file events matter. */
|
||||
bool change_list_enabled(const Config* config);
|
||||
|
||||
/* Render the rsync-style itemize line for a transferred file:
|
||||
* `>f+++++++++ <path>`
|
||||
* The 11-char code is `>f` (regular file transferred to the remote host)
|
||||
* followed by c/s/t/p/o/g/u/a/x markers that are all `+` (value will be set
|
||||
* / differs) because FastSync does not separately compare checksums, size,
|
||||
* mtime, perms, owner, group, uid, acl, or xattr on the receiving side, so a
|
||||
* sent file is reported as fully updated. Up-to-date files print no line
|
||||
* (rsync single `-i` only shows changes). Caller frees the result. */
|
||||
char* change_render_itemize(const ChangeEvent* event);
|
||||
/* Render the rsync-style itemize line for a transferred item
|
||||
* (`%i %n%L`): `>f+++++++++ sub/b.txt`. Caller frees the result. */
|
||||
char* change_render_itemize(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Expand an --out-format/--log-file-format template. Tokens:
|
||||
* %f full source path %b "bytes sent" == the source length (%l);
|
||||
* %n leaf (base) name actual post-compression/delta wire bytes
|
||||
* %l file length in bytes are not counted
|
||||
* %M mtime in whole seconds %% a literal percent sign
|
||||
/* Expand an --out-format/--log-file-format template. Supported tokens:
|
||||
* %i itemize code %n transfer-relative name (dir: trailing /)
|
||||
* %f long display path %l file length in bytes
|
||||
* %b wire bytes transferred %c block-checksum bytes received (rsync: 16
|
||||
* for a whole-file transfer, 0 for a dry run)
|
||||
* %C whole-file checksum hex (xxh128 by default; spaces for non-regular)
|
||||
* %M mtime (YYYY/MM/DD-HH:MM:SS)
|
||||
* %t current time %o operation ("send"/"del.")
|
||||
* %p pid %B permission bits without the type char
|
||||
* %U uid %G gid
|
||||
* %L " -> target" / " => target" %% a literal percent sign
|
||||
* Unknown %X sequences are preserved verbatim. Caller frees the result. */
|
||||
char* change_render_format(const char* format, const ChangeEvent* event);
|
||||
char* change_render_format(const char* format, const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Render one --list-only long-listing entry:
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 <path>`
|
||||
* `-rw-r--r-- 12 2026/09/06 10:00:00 sub/b.txt`
|
||||
* (ls -l style columns; mtime in the local time zone). Caller frees it. */
|
||||
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime, const char* path);
|
||||
char* change_render_list_line(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Emit an event to every active destination:
|
||||
* stdout: --itemize-changes line, or the --out-format expansion when set;
|
||||
@@ -69,7 +88,15 @@ char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime
|
||||
* CHANGE_UP_TO_DATE events produce no output. */
|
||||
void change_emit(const Config* config, const ChangeEvent* event);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent. */
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent. `bytes_sent`
|
||||
* is the process-wide wire-byte delta for this file (rsync's %b) and `bytes_read`
|
||||
* the received bytes used for the delta handshake; pass 0 when unknown. For a
|
||||
* whole-file transfer %c is pinned to rsync's 16-byte sum header regardless. */
|
||||
void change_emit_file_sent_bytes(const Config* config, const File* file,
|
||||
unsigned long long bytes_sent, unsigned long long bytes_read);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for a file the client just sent, deriving
|
||||
* the wire byte counts from the source payload length. */
|
||||
void change_emit_file_sent(const Config* config, const File* file);
|
||||
|
||||
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
|
||||
|
||||
+1773
-512
File diff suppressed because it is too large
Load Diff
+1337
-245
File diff suppressed because it is too large
Load Diff
@@ -4,10 +4,24 @@
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "transport_tcp.h"
|
||||
#include <signal.h>
|
||||
#include <stdbool.h>
|
||||
|
||||
int send_chunk(Client* client, Chunk* chunk, Config* config);
|
||||
/* Set ONLY by the client's SIGINT/SIGTERM handler (async-signal-safe: the
|
||||
* handler stores 1 and does nothing else). The send loops poll it via
|
||||
* client_abort_pending() and, when set, best-effort send STATUS_ABORT so the
|
||||
* receiver can clean up before the client exits. */
|
||||
extern volatile sig_atomic_t client_abort_requested;
|
||||
bool client_abort_pending(void);
|
||||
/* Arm/disarm abort handling around the network phase. While disarmed, a
|
||||
* SIGINT/SIGTERM takes the default action (immediate termination) so local-only
|
||||
* modes are not left unresponsive. Defined in client_cli.c. */
|
||||
void client_set_abort_armed(bool armed);
|
||||
|
||||
/* Both sender entry points BORROW `config` for the duration of the call; they
|
||||
* never free it, and the caller retains ownership (freeing it with
|
||||
* config_delete() once the call returns). */
|
||||
int send_files(Config* config);
|
||||
/* Takes ownership only when *config is set to NULL on return. */
|
||||
int send_files_multithreaded(Config** config);
|
||||
/* Phase 6 residual-batch (client-only). See client_send.c. */
|
||||
int write_batch_from_source(const Config* config, const char* batch_path);
|
||||
|
||||
+27
-105
@@ -1,6 +1,4 @@
|
||||
#include "client_validation.h"
|
||||
#include "charset.h"
|
||||
#include "delay_updates.h"
|
||||
#include "log.h"
|
||||
#include "usage.h"
|
||||
#include "utils.h"
|
||||
@@ -24,6 +22,19 @@ bool validate_config(const Config* config) {
|
||||
"--write-batch, --only-write-batch, and --read-batch are mutually exclusive");
|
||||
return false;
|
||||
}
|
||||
/* A dry-run of a local batch apply is not meaningful: --read-batch bypasses
|
||||
the client-side scan/server decision entirely, so dry-run would have no
|
||||
wire state to report (and must not be used as a mutation escape hatch).
|
||||
--only-write-batch likewise never contacts a receiver. --write-batch DOES
|
||||
run a live transfer but additionally mutates the filesystem by emitting the
|
||||
batch file, so a dry-run must not write it either. Reject all three up
|
||||
front instead of silently ignoring --dry-run. */
|
||||
if (config->dry_run && (read_batch || only_write_batch || write_batch)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--dry-run cannot be combined with --read-batch, --only-write-batch, or "
|
||||
"--write-batch; a dry-run must not mutate anything, including batch files");
|
||||
return false;
|
||||
}
|
||||
if (read_batch) {
|
||||
if (!config->receive_root_directory) {
|
||||
log_message(LOG_LEVEL_ERROR, "--read-batch requires a destination directory");
|
||||
@@ -41,17 +52,6 @@ bool validate_config(const Config* config) {
|
||||
print_usage();
|
||||
return false;
|
||||
}
|
||||
if (config_has_basis(config) && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--compare-dest/--copy-dest/--link-dest require per-file incremental checks and "
|
||||
"cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s "
|
||||
"(chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->compression_threads > 0 && !config->use_compression) {
|
||||
log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-c or -z)");
|
||||
return false;
|
||||
@@ -60,8 +60,14 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport");
|
||||
return false;
|
||||
}
|
||||
if (config->use_incremental && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR, "--incremental is not supported with -s (chunk serialization)");
|
||||
/* -M/--remote-option appends an option to the REMOTE server's argv, which
|
||||
* only exists on the SSH (user@host:path) transport. A daemon
|
||||
* (host::module/path) or local TCP destination has no remote command line,
|
||||
* so the option would be silently ignored; reject it by name instead. */
|
||||
if (config->remote_option_count > 0 && config->transport != TRANSPORT_SSH) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"-M/--remote-option is only valid with the SSH transport (user@host:path); it "
|
||||
"cannot be used with a daemon (host::module/path) or local TCP destination");
|
||||
return false;
|
||||
}
|
||||
/* -4 and -6 are mutually exclusive: a socket address family cannot be both. */
|
||||
@@ -69,64 +75,6 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive");
|
||||
return false;
|
||||
}
|
||||
if (config->skip_compress_set && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--skip-compress cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->use_delta && !config->whole_file && !config->use_incremental) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta requires --incremental");
|
||||
return false;
|
||||
}
|
||||
if (config->use_delta && !config->whole_file && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->use_delta && !config->whole_file && config->use_sendfile) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -f (sendfile)");
|
||||
return false;
|
||||
}
|
||||
/* --append / --append-verify resume a shorter existing destination by
|
||||
transmitting only the tail. The resume needs the per-file STATUS_CHECK
|
||||
handshake (so the dest length is learned), which chunk serialization -s
|
||||
disables; and whole-file is the opposite intent (send everything), so the
|
||||
two would silently make the resume pointless. Both are rejected up front
|
||||
rather than silently degrading to a full transfer. */
|
||||
if ((config->append || config->append_verify) && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--append/--append-verify require the per-file incremental check and cannot be "
|
||||
"combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if ((config->append || config->append_verify) && config->whole_file) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--append/--append-verify are incompatible with --whole-file (which forces a "
|
||||
"full transfer)");
|
||||
return false;
|
||||
}
|
||||
/* --hard-links/-H transmits each later group member as a dedicated per-file
|
||||
STATUS_HARDLINK frame, which chunk serialization -s does not support; and a
|
||||
hard-links sibling carries no payload, so the tail-resume of --append is
|
||||
meaningless for it. Both combinations are rejected up front rather than
|
||||
silently degrading. */
|
||||
if (config->preserve_hard_links && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--hard-links/-H cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
/* -X/-A ride the per-file metadata frame; the buffer-based chunk-serialization
|
||||
wire format does not carry the xattr block, so the pair is rejected up front
|
||||
(mirroring -H + -s) rather than silently dropping attributes. */
|
||||
if ((config->preserve_xattrs || config->preserve_acls) && config->use_chunk_serialization) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--xattrs/-X and --acls/-A cannot be combined with -s (chunk serialization)");
|
||||
return false;
|
||||
}
|
||||
if (config->preserve_hard_links && (config->append || config->append_verify)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--hard-links/-H cannot be combined with --append/--append-verify");
|
||||
return false;
|
||||
}
|
||||
if (config->log_file_format && !config->log_file) {
|
||||
log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file");
|
||||
return false;
|
||||
@@ -146,28 +94,12 @@ bool validate_config(const Config* config) {
|
||||
log_message(LOG_LEVEL_ERROR, "sending daemon credentials to a non-local server requires --tls");
|
||||
return false;
|
||||
}
|
||||
if (config->delay_updates && config->inplace) {
|
||||
log_message(LOG_LEVEL_ERROR, "--delay-updates does not work with --inplace");
|
||||
return false;
|
||||
}
|
||||
if (config->delay_updates && delay_updates_staging_name_conflict(config->backup_dir)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--backup-dir is reserved when --delay-updates is active (used for the internal "
|
||||
"staging directory)");
|
||||
return false;
|
||||
}
|
||||
if (!config_has_valid_delete_timing(config)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--delete-before/--delete-during/--delete-delay/--delete-after select the delete "
|
||||
"timing; at most one may be given and each implies --delete");
|
||||
return false;
|
||||
}
|
||||
/* --iconv: reject a malformed CONVERT_SPEC or an unsupported charset name at
|
||||
startup (a probe iconv_open is attempted), so a typo'd charset never fails
|
||||
the run mid-transfer with per-file errors. */
|
||||
if (!charset_spec_valid(config->iconv_spec)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv requires LOCAL[,REMOTE] charset names supported by iconv");
|
||||
/* Every cross-field invariant the receiver enforces lives in one shared
|
||||
predicate so the client and the server can never disagree. The client
|
||||
reports the specific reason here, before any network I/O. */
|
||||
const char* invariants_error = config_invariants_error(config);
|
||||
if (invariants_error) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s", invariants_error);
|
||||
return false;
|
||||
}
|
||||
/* --protocol: FastSync has exactly one wire format, so the forced version
|
||||
@@ -180,15 +112,5 @@ bool validate_config(const Config* config) {
|
||||
PROTOCOL_VERSION);
|
||||
return false;
|
||||
}
|
||||
/* --copy-as pushes the source ids through the metadata path (it implies
|
||||
--preserve). A later --no-preserve would clear use_metadata, leaving the
|
||||
transfer with nothing to chown while the receiver gate would still pass.
|
||||
Refuse the combination up front rather than silently chowning nothing. */
|
||||
if (config->copy_as_set && !config->use_metadata) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--copy-as requires metadata preservation and cannot be combined with "
|
||||
"--no-preserve");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
+824
-279
File diff suppressed because it is too large
Load Diff
+62
-58
@@ -14,6 +14,10 @@
|
||||
#include <sys/types.h>
|
||||
#include <threads.h>
|
||||
|
||||
/* Upper bound on the configurable parallel scanner worker count (--threads=N):
|
||||
* keeps one transfer from spawning an unbounded pool on a very large machine. */
|
||||
#define MAX_SCANNER_THREADS 256
|
||||
|
||||
typedef struct {
|
||||
bool use_metadata;
|
||||
/* Phase 4 metadata capture: -U/--atimes and -N/--crtimes tell the scanner to
|
||||
@@ -59,8 +63,23 @@ typedef struct {
|
||||
const FileListSet* file_list; /* --files-from allow-set, or NULL */
|
||||
const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */
|
||||
bool per_dir_filters; /* -F: read .rsync-filter per directory */
|
||||
/* --delete-excluded: per-directory plain rules become sender-only, so they no
|
||||
longer protect the receiver from deletion. */
|
||||
bool delete_excluded;
|
||||
/* -FF: also exclude the per-directory filter files themselves from the
|
||||
transfer (single -F transfers them). */
|
||||
bool exclude_per_dir_filter_files;
|
||||
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
|
||||
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
|
||||
/* -R/--relative outside --files-from: the destination-relative path prefix
|
||||
* reconstructed from the source spec (rsync's '/./' cut point), or NULL when
|
||||
* -R is off or --files-from is in use (the bare-relative path then comes from
|
||||
* the listed entry). Borrowed read-only; owned by client_send. */
|
||||
const char* relative_prefix;
|
||||
/* --list-only: emit an is_dir File for every traversed directory (the listing
|
||||
* includes directory entries, matching rsync). Client-only; never set on a
|
||||
* real transfer, which relies on implicit parent creation. */
|
||||
bool list_dirs;
|
||||
/* --prune-empty-dirs (long only): in --dirs mode an empty source directory's
|
||||
explicit entry is omitted from the transfer file list (so nothing is
|
||||
created at the destination and it can be pruned by --delete); explicitly
|
||||
@@ -69,17 +88,39 @@ typedef struct {
|
||||
bool prune_empty_dirs;
|
||||
/* Delete-excluded protection sink (optional): when non-NULL the scanner
|
||||
* appends the destination-relative path of every entry it prunes because a
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules, the legacy
|
||||
* --exclude/--include layer, and --max-size/--min-size). The sender turns
|
||||
* this list into the manifest's protected prefixes so `--delete` leaves the
|
||||
* destination mirror of excluded source paths alone (rsync's default), and
|
||||
* empties it when --delete-excluded opts back into deleting them. NOT
|
||||
* recorded for --files-from subset pruning (whose delete semantics stay
|
||||
* keep-set-only) or for -R/--files-from relative wire paths. When
|
||||
* `excluded_mutex` is non-NULL it is taken around every append (the parallel
|
||||
* scanner shares one list across its worker threads). */
|
||||
* USER SELECTION rule excluded it (--filter/-C/per-dir rules and the legacy
|
||||
* --exclude/--include layer). The sender turns this list into the manifest's
|
||||
* protected prefixes so `--delete` leaves the destination mirror of excluded
|
||||
* source paths alone (rsync's default), and drops it when --delete-excluded
|
||||
* opts back into deleting them. NOT recorded for --files-from subset pruning
|
||||
* (whose delete semantics derive from the synchronized-directory set) or for
|
||||
* -R/--files-from relative wire paths. When `excluded_mutex` is non-NULL it
|
||||
* is taken around every append (the parallel scanner shares one list across
|
||||
* its worker threads). */
|
||||
ArrayList* excluded_paths;
|
||||
mtx_t* excluded_mutex;
|
||||
/* Size-prune protection sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every entry it skipped because of
|
||||
* --max-size/--min-size. rsync never deletes a size-skipped source mirror,
|
||||
* even under --delete-excluded, so the sender always transmits this list as
|
||||
* protected prefixes (unlike excluded_paths, which --delete-excluded drops).
|
||||
* Guarded by `excluded_mutex` like excluded_paths. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Synchronized-directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it is about to traverse
|
||||
* that lies inside a --files-from listed directory (or of every traversed
|
||||
* directory when there is no list). The sender sends this set with the delete
|
||||
* manifest so the receiver confines its extras walk to synchronized
|
||||
* directories, exactly like rsync; the receive root is the "." sentinel.
|
||||
* Guarded by `excluded_mutex`. */
|
||||
ArrayList* synced_dirs;
|
||||
/* Delete-plan directory sink (optional): when non-NULL the scanner appends
|
||||
* the destination-relative path of every directory it traverses (except the
|
||||
* receive root). The per-directory --delete-during/--delete-delay plan
|
||||
* builder uses this to keep an empty in-scope source directory (rsync keeps
|
||||
* it) and to emit its plan after the data stream, when no file frame would
|
||||
* otherwise trigger it. Guarded by `excluded_mutex`. */
|
||||
ArrayList* plan_dirs;
|
||||
/* --ignore-errors: an unreadable directory during the scan is recorded as an
|
||||
* I/O error and skipped instead of aborting the scan. Client-only. */
|
||||
bool ignore_io_errors;
|
||||
@@ -117,37 +158,16 @@ typedef struct {
|
||||
typedef struct FilterNode FilterNode;
|
||||
|
||||
typedef struct {
|
||||
/* Scan inputs, copied once at create time. Everything that is also a
|
||||
ScannerOptions field lives here (with the normalized chunk_size); only
|
||||
scanner-owned bookkeeping stays as direct members below. */
|
||||
ScannerOptions options;
|
||||
Queue* directories;
|
||||
DIR* current_dir;
|
||||
char* current_path;
|
||||
bool use_metadata;
|
||||
bool preserve_atimes;
|
||||
bool preserve_crtimes;
|
||||
bool preserve_xattrs;
|
||||
bool preserve_acls;
|
||||
unsigned long long chunk_size;
|
||||
char** exclude_patterns;
|
||||
int exclude_count;
|
||||
char** include_patterns;
|
||||
int include_count;
|
||||
unsigned long long max_size;
|
||||
unsigned long long min_size;
|
||||
int max_depth;
|
||||
int current_depth;
|
||||
bool follow_symlinks;
|
||||
bool copy_links;
|
||||
bool safe_links;
|
||||
bool copy_unsafe_links;
|
||||
bool copy_dirlinks;
|
||||
bool munge_links;
|
||||
bool checksum;
|
||||
bool one_file_system;
|
||||
dev_t root_dev;
|
||||
bool failed;
|
||||
/* Phase 4 special/devices (see ScannerOptions). */
|
||||
bool preserve_devices;
|
||||
bool preserve_specials;
|
||||
bool copy_devices;
|
||||
/* Phase 2 (files-from / filter layer). */
|
||||
char* root_path; /* transfer root (fs path) for rel computation */
|
||||
char* current_rel; /* rel path of the open directory ("" == root) */
|
||||
@@ -155,40 +175,17 @@ typedef struct {
|
||||
FilterNode* seed_node; /* inherited context of the seed dir, or NULL */
|
||||
FilterNode* current_node; /* filter context of the open directory */
|
||||
ArrayList* filter_nodes; /* owned FilterNode arena (may be NULL) */
|
||||
const FileListSet* file_list;
|
||||
const FilterRuleList* base_filters;
|
||||
bool per_dir_filters;
|
||||
/* --dirs / -R state for the directory-entry generator (dirs_mode replaces
|
||||
/* --dirs / -R state for the directory-entry generator (options.dirs replaces
|
||||
the recursive scan). */
|
||||
bool dirs_mode;
|
||||
bool relative_mode; /* file_list && relative: send bare relative wire paths */
|
||||
bool prune_empty_dirs;
|
||||
bool dirs_root_emitted;
|
||||
int list_index;
|
||||
ArrayList* dirs_batch; /* owned when non-NULL */
|
||||
unsigned long long dirs_batch_size;
|
||||
/* Excluded-path sink (see ScannerOptions). `excluded_mutex` is shared across
|
||||
parallel worker threads. */
|
||||
ArrayList* excluded_paths;
|
||||
mtx_t* excluded_mutex;
|
||||
/* --ignore-errors: continue past unreadable directories (records io_error). */
|
||||
bool ignore_io_errors;
|
||||
/* --ignore-missing-args: --dirs listed-but-missing entries are skipped, not
|
||||
fatal (see ScannerOptions.ignore_missing_args). */
|
||||
bool ignore_missing_args;
|
||||
/* A directory could not be opened (I/O error, e.g. EACCES). With
|
||||
--ignore-errors the scan continues past it and the caller decides what to
|
||||
do; `failed` is reserved for fatal errors that always abort the scan. */
|
||||
bool io_error;
|
||||
/* --hard-links (-H): shared link-group detection table (see ScannerOptions).
|
||||
NULL when -H is off. */
|
||||
HardLinkTable* hardlinks;
|
||||
/* Phase 6: sender stop deadline (from ScannerOptions). */
|
||||
const StopCondition* stop_condition;
|
||||
/* P7 Wave D directory-time capture (see ScannerOptions). */
|
||||
bool capture_dir_times;
|
||||
ArrayList* dir_entries;
|
||||
mtx_t* dir_entries_mutex;
|
||||
} DirectoryScanner;
|
||||
|
||||
typedef struct {
|
||||
@@ -234,6 +231,13 @@ bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entr
|
||||
* "/". Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_path_relative(const char* root, const char* fs_path);
|
||||
|
||||
/* -R/--relative destination-relative prefix reconstructed from a source spec:
|
||||
* the path after rsync's first '.' path component (the '/./' cut point), with
|
||||
* leading/trailing slashes removed, or the whole spec (normalized) when there
|
||||
* is no cut. Returns "" for the receive root, or NULL when `spec` is NULL or
|
||||
* allocation fails. Exposed so tests can exercise the mapping directly. */
|
||||
char* scanner_relative_prefix(const char* spec);
|
||||
|
||||
ParallelScanner* parallel_scanner_create_with_options(const char* root_directory,
|
||||
const ScannerOptions* options,
|
||||
ProtocolSession* allocation_session);
|
||||
|
||||
+135
-73
@@ -2,6 +2,7 @@
|
||||
#include <stdio.h>
|
||||
#include <delta.h>
|
||||
#include <chunk.h>
|
||||
#include "scanner.h"
|
||||
|
||||
void print_usage(void) {
|
||||
printf("Usage:\n");
|
||||
@@ -19,11 +20,16 @@ void print_usage(void) {
|
||||
printf("Options:\n");
|
||||
printf(" -c, --checksum Verify content by checksum instead of size+mtime\n");
|
||||
printf(" -z, --compress [level] Enable compression (level 1-22, default 5)\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, metadata,\n");
|
||||
printf(" devices and specials (not compression/multithreading)\n");
|
||||
printf(" -a, --archive rsync archive mode (-rlptgoD): links, perms, times,\n");
|
||||
printf(" owner, group, devices and specials; not\n");
|
||||
printf(" compression/multithreading\n");
|
||||
printf(" -r, --recursive Recurse into directories (FastSync is always recursive)\n");
|
||||
printf(" -n, --dry-run Show what would be transferred\n");
|
||||
printf(" --remove-source-files Remove regular source files after successful transfer\n");
|
||||
printf(" -p, --perms Preserve permission bits (part of the metadata bundle)\n");
|
||||
printf(" -p, --perms Preserve permission bits\n");
|
||||
printf(" -t, --times Preserve modification times\n");
|
||||
printf(" -o, --owner Preserve owner (uid)\n");
|
||||
printf(" -g, --group Preserve group (gid)\n");
|
||||
printf(" --ssh-port <port> SSH port (default: 22)\n");
|
||||
printf(" -e, --rsh <command> Remote shell to launch on the client for the SSH\n");
|
||||
printf(" transport (default: ssh). The command may include\n");
|
||||
@@ -53,24 +59,26 @@ void print_usage(void) {
|
||||
printf(" Emit the batch file only (no destination, no server)\n");
|
||||
printf(" --read-batch=FILE Apply the batch file to the destination (no source, no\n");
|
||||
printf(" server); takes only the destination as an argument\n");
|
||||
printf(" NOTE: the FastSync batch format is NOT interoperable with rsync's batch\n");
|
||||
printf(" files (different container format); do not mix the two tools.\n");
|
||||
printf(" --delete Delete files on receiver not in source\n");
|
||||
printf(" (default timing: delete only after the whole\n");
|
||||
printf(" transfer has succeeded)\n");
|
||||
printf(" --delete-before Delete extras before the transfer starts\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-during Delete extras once the keep-set manifest is known,\n");
|
||||
printf(" before the data is applied (implies --delete)\n");
|
||||
printf(" --delete-during Delete a directory's extras as that directory is\n");
|
||||
printf(" processed (implies --delete)\n");
|
||||
printf(" --del Alias for --delete-during\n");
|
||||
printf(" --delete-delay Delete extras only after a successful transfer\n");
|
||||
printf(" (implies --delete)\n");
|
||||
printf(" --delete-delay Record the extras during the scan but remove them\n");
|
||||
printf(" only after a successful transfer (implies --delete)\n");
|
||||
printf(" --delete-after Delete only after the whole transfer succeeded\n");
|
||||
printf(" (the default --delete timing; implies --delete)\n");
|
||||
printf(" --delete-excluded Also delete destination files that were excluded on\n");
|
||||
printf(" the source (default protects them, matching rsync)\n");
|
||||
printf(" --max-delete=NUM Never delete more than NUM destination entries per run;\n");
|
||||
printf(" if the extras would exceed NUM, nothing is deleted and\n");
|
||||
printf(" the run fails with a clear error (implies --delete only\n");
|
||||
printf(" when used with it)\n");
|
||||
printf(" --max-delete=NUM Delete at most NUM destination entries per run; if the\n");
|
||||
printf(" extras exceed NUM, the rest are skipped and the run is\n");
|
||||
printf(" reported as partial (exit 25, matching rsync). Only\n");
|
||||
printf(" applies together with --delete\n");
|
||||
printf(" --ignore-errors Continue (and still delete) when a source directory is\n");
|
||||
printf(" unreadable during the scan, instead of aborting with no\n");
|
||||
printf(" deletion\n");
|
||||
@@ -99,20 +107,23 @@ void print_usage(void) {
|
||||
printf(" parent directory is not itself listed\n");
|
||||
printf(" --mkpath Create the destination root directory on the server when it\n");
|
||||
printf(" does not exist yet\n");
|
||||
printf(" --exclude <pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file> Read include patterns from file\n");
|
||||
printf(" --exclude <pattern>, --exclude=<pattern> Exclude files matching pattern\n");
|
||||
printf(" --include <pattern>, --include=<pattern> Only include files matching pattern\n");
|
||||
printf(" --exclude-from <file>, --exclude-from=<file> Read exclude patterns from file\n");
|
||||
printf(" --include-from <file>, --include-from=<file> Read include patterns from file\n");
|
||||
printf(" --files-from <file> Read the source file list from FILE (paths relative to the "
|
||||
"source root)\n");
|
||||
printf(" -0, --from0 Entries in --files-from are NUL-delimited\n");
|
||||
printf(" -f, --filter=RULE rsync-style filter rule (+/- include/exclude; repeatable;\n");
|
||||
printf(" both --filter=RULE and the -f RULE / -f=RULE short forms work)\n");
|
||||
printf(" -f, --filter=RULE rsync-style filter rule: exclude/- include/+ hide/H show/S\n");
|
||||
printf(" protect/P risk/R merge/. dir-merge/: clear/! with modifiers\n");
|
||||
printf(" (repeatable; --filter=RULE and -f RULE / -f=RULE both work)\n");
|
||||
printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n");
|
||||
printf(" -F Apply per-directory .rsync-filter files during the scan\n");
|
||||
printf(" -F Apply per-directory .rsync-filter files; repeated -FF also\n");
|
||||
printf(" excludes the .rsync-filter files themselves\n");
|
||||
printf(" --max-size <n> Skip files larger than n bytes\n");
|
||||
printf(" --min-size <n> Skip files smaller than n bytes\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G)\n");
|
||||
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G; 0 = no limit,\n");
|
||||
printf(" matching rsync)\n");
|
||||
printf(" --incremental Skip files unchanged since last transfer\n");
|
||||
printf(" --size-only Skip incremental files matching in size, ignoring mtime\n");
|
||||
printf(" -I, --ignore-times Transfer files even when size and mtime match\n");
|
||||
@@ -127,37 +138,54 @@ void print_usage(void) {
|
||||
printf(" --link-dest <dir> Like --copy-dest, but hard-links the unchanged file from DIR\n");
|
||||
printf(" into the destination (repeatable; earlier DIRs win)\n");
|
||||
printf(" --checksum-choice, --cc <alg> Whole-file checksum algorithm for --incremental/\n");
|
||||
printf(" --checksum compares (xxh64/xxhash or md5; default xxh64 with\n");
|
||||
printf(" seed 0). The seed comes from --checksum-seed\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash64 digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. The\n");
|
||||
printf(" digest algorithm and seed must match on sender and receiver\n");
|
||||
printf(" --checksum compares. Accepted: xxh128 (default), xxh3, xxh64\n");
|
||||
printf(" (aka xxhash), md5, md4, sha1, or none. A two-name\n");
|
||||
printf(" 'transfer,pre-transfer' form is accepted like rsync; 'none' as\n");
|
||||
printf(" the pre-transfer algorithm is rejected with --checksum\n");
|
||||
printf(" --checksum-seed <num> Seed for the whole-file xxHash digest (and the delta\n");
|
||||
printf(" block strong hash, low 32 bits); md5 ignores the seed. A seed\n");
|
||||
printf(" of 0 (the default) is randomized per transfer, exactly like\n");
|
||||
printf(" rsync, and the chosen seed is sent to the receiver\n");
|
||||
printf(" --delta Delta transfer for changed files (requires --incremental)\n");
|
||||
printf(" -W, --whole-file Transfer changed files without delta processing\n");
|
||||
printf(" --no-whole-file rsync spelling that clears -W/--whole-file\n");
|
||||
printf(" -y, --fuzzy Use a similar-named file already in the destination\n");
|
||||
printf(" directory as the delta basis when the destination has no\n");
|
||||
printf(" usable file at the exact path (saves bandwidth; implies\n");
|
||||
printf(" --incremental and --delta; inert with --whole-file,\n");
|
||||
printf(" --no-delta, or --no-incremental)\n");
|
||||
printf(" --no-fuzzy Disable --fuzzy\n");
|
||||
printf(" --delta-block <n>, --block-size <n>\n");
|
||||
printf(" -B <n>, --block-size <n>, --delta-block <n>\n");
|
||||
printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT);
|
||||
printf(" --delta-max <n> Max file size for delta transfer (default: %llu)\n",
|
||||
DELTA_MAX_FILE_SIZE);
|
||||
printf(" -j, --threads Enable multithreading\n");
|
||||
printf(" -j, --threads[=N] Enable the multithreaded scanner/loader/sender\n");
|
||||
printf(" pipeline; N (1-%d) sets the parallel scanner worker\n",
|
||||
MAX_SCANNER_THREADS);
|
||||
printf(" count (bare -j/--threads uses the default)\n");
|
||||
printf(" --chunk-serialization Enable chunk serialization (long form only)\n");
|
||||
printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n");
|
||||
printf(" SSH argv is already built injection-safe)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm (default: zstd)\n");
|
||||
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only;\n");
|
||||
printf(" -f is bound to --filter, not --sendfile)\n");
|
||||
printf(" --compress-choice <alg> Compression algorithm: zstd (default), lz4, zlib,\n");
|
||||
printf(" zlibx, none, or auto\n");
|
||||
printf(" --zc <alg> Alias for --compress-choice\n");
|
||||
printf(" -v, --verbose Enable debug logging\n");
|
||||
printf(" -q, --quiet Suppress non-error output\n");
|
||||
printf(" --debug=FLAGS Fine-grained debug logging (use --debug=help for flags)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,misc,skip,stats,all,none\n");
|
||||
printf(" none suppresses info even with --verbose\n");
|
||||
printf(" --preserve Preserve file metadata (long form only)\n");
|
||||
printf(" --info=FLAGS Fine-grained info: copy,name,misc,skip,stats,all,none\n");
|
||||
printf(" (use --info=help for flags; none suppresses --verbose)\n");
|
||||
printf(" --preserve Preserve permissions and times (= -pt; long form only)\n");
|
||||
printf(" --no-perms Negate -p/--perms\n");
|
||||
printf(" --no-times Negate -t/--times\n");
|
||||
printf(" --no-owner Negate -o/--owner\n");
|
||||
printf(" --no-group Negate -g/--group\n");
|
||||
printf(" --no-preserve Disable metadata preservation (negates --preserve)\n");
|
||||
printf(" -E, --executability Preserve executable permission bits\n");
|
||||
printf(" -U, --atimes Preserve access times\n");
|
||||
printf(" -N, --crtimes Capture birth time; cannot be applied (documented\n");
|
||||
printf(" divergence)\n");
|
||||
printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n");
|
||||
printf(" privileged security.*/trusted.* namespaces are\n");
|
||||
printf(" never captured or applied)\n");
|
||||
@@ -172,28 +200,32 @@ void print_usage(void) {
|
||||
printf(" (char/block device-node creation, --write-devices)\n");
|
||||
printf(" within the confined receive root. Never elevates\n");
|
||||
printf(" privileges and never bypasses confinement; ownership\n");
|
||||
printf(" is still applied only with an explicit identity flag\n");
|
||||
printf(" (--numeric-ids/--chown/--usermap/--groupmap/--copy-as)\n");
|
||||
printf(" is still applied only with -o/--owner, -g/--group, or an\n");
|
||||
printf(" explicit identity flag (--chown/--usermap/--groupmap/\n");
|
||||
printf(" --copy-as); --numeric-ids only changes how ids map\n");
|
||||
printf(" --no-super Forbid those super-user activities even when the\n");
|
||||
printf(" receiver is running as root\n");
|
||||
printf(" --chmod <changes> Modify transferred permissions (rsync syntax)\n");
|
||||
printf(" --numeric-ids Do not map uid/gid by name: use the source numeric\n");
|
||||
printf(" ids directly when applying ownership\n");
|
||||
printf(
|
||||
" --chmod <changes> Modify new/transferred permissions (rsync syntax; implies no -p)\n");
|
||||
printf(" --numeric-ids Map uid/gid by id instead of by name (a modifier, not\n");
|
||||
printf(" an ownership request: combine with -o/-g or a map)\n");
|
||||
printf(" --usermap=MAP Map usernames when applying ownership: comma-separated\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM/TO are names\n");
|
||||
printf(" (resolved on the source machine), * (match any /\n");
|
||||
printf(" current user), or @N numeric ids. e.g. *:nobody\n");
|
||||
printf(" FROM:TO rules, first match wins. FROM is a name (from\n");
|
||||
printf(" the source), an id, an inclusive LOW-HIGH range, *\n");
|
||||
printf(" (any id), or empty (ids with no name). TO is an id, *\n");
|
||||
printf(" (current user), or a name resolved on the receiver.\n");
|
||||
printf(" e.g. 0-99:nobody,*:normal (cannot mix with --chown)\n");
|
||||
printf(" --groupmap=MAP Map group names when applying ownership (same syntax)\n");
|
||||
printf(" --chown=USER:GROUP Override the ownership of transferred files. Forms:\n");
|
||||
printf(" USER:GROUP, USER (owner only), :GROUP (group only); a\n");
|
||||
printf(" value of * means the current/root user as appropriate.\n");
|
||||
printf(" Names resolve on the source machine; @N for numerics.\n");
|
||||
printf(" (Metadata is enabled with --preserve; -M now means\n");
|
||||
printf(" rsync's --remote-option.)\n");
|
||||
printf(" (Implies owner/group metadata; -M now means rsync's\n");
|
||||
printf(" --remote-option.)\n");
|
||||
printf(" --copy-as=USER[:GROUP] Force every written entry (files, dirs, symlinks\n");
|
||||
printf(" and special nodes) to USER[:GROUP], resolved on the\n");
|
||||
printf(" source machine like --chown. Requires a privileged\n");
|
||||
printf(" (root) receiver and implies --preserve; an\n");
|
||||
printf(" (root) receiver and implies owner/group metadata; an\n");
|
||||
printf(" unprivileged receiver refuses the transfer. Never\n");
|
||||
printf(" switches process credentials (safe-subset; see\n");
|
||||
printf(" RSYNC_COMPAT.md). A daemon refuses it.\n");
|
||||
@@ -203,10 +235,12 @@ void print_usage(void) {
|
||||
printf(" --save-to-disk Write received files to disk\n");
|
||||
printf(" --server-host <ip> Server IP address (default: 127.0.0.1)\n");
|
||||
printf(" --server-port <n> Server port (default: 8080)\n");
|
||||
printf(" --port <n> Alias for --server-port\n");
|
||||
printf(" --password-file <f> Authenticate a host::module/path daemon destination.\n");
|
||||
printf(" The file's first user:password line supplies the\n");
|
||||
printf(" username and password (only a SHA-256 digest of the\n");
|
||||
printf(" password is sent; keep the file mode 0600)\n");
|
||||
printf(" FastSync-native SCRAM/PBKDF2 credential scheme (NOT\n");
|
||||
printf(" rsync's --password-file): the file's first user:password\n");
|
||||
printf(" line supplies the username and password; no password or\n");
|
||||
printf(" reusable digest is sent (keep the file mode 0600)\n");
|
||||
printf(" --no-motd Suppress display of the daemon's MOTD (the server\n");
|
||||
printf(" still sends it; the client just does not show it)\n");
|
||||
printf(" --bwlimit <KB/s> Bandwidth limit in kilobytes per second\n");
|
||||
@@ -214,55 +248,69 @@ void print_usage(void) {
|
||||
printf(" --cert <path> TLS certificate file (PEM)\n");
|
||||
printf(" --key <path> TLS private key file (PEM)\n");
|
||||
printf(" --ca <path> TLS CA certificate file (PEM)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 30; long form only)\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 10)\n");
|
||||
printf(" --timeout <sec> I/O timeout in seconds (default: 0 = disabled, matching\n");
|
||||
printf(" rsync). 0 disables it; --no-timeout is the same\n");
|
||||
printf(" --contimeout <sec> Connection timeout in seconds (default: 60, matching\n");
|
||||
printf(" rsync); 0 disables it (--no-contimeout)\n");
|
||||
printf(" --stop-after=MINS Stop the transfer after MINS minutes (a positive\n");
|
||||
printf(" integer); whatever was already transferred is kept\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time: HH:MM, HH:MM:SS, or\n");
|
||||
printf(" now+N[smhd] (a time already in the past stops the\n");
|
||||
printf(" transfer immediately; client-only). An early stop\n");
|
||||
printf(" skips the late --delete keep-set so it cannot delete\n");
|
||||
printf(" source mirrors that were not yet scanned\n");
|
||||
printf(" --stop-at=TIME Stop at an absolute time. Accepts rsync's date form\n");
|
||||
printf(" (Y-M-DTh:m, Y/M/DTh:m, abbreviable fields such as 12-31,\n");
|
||||
printf(" 14:00, :59, 1) plus FastSync's HH:MM[:SS] and now+N[smhd]\n");
|
||||
printf(" (a time already in the past stops the transfer\n");
|
||||
printf(" immediately; client-only). An early stop skips the late\n");
|
||||
printf(" --delete keep-set so it cannot delete source mirrors that\n");
|
||||
printf(" were not yet scanned\n");
|
||||
printf(" --address <ip> Bind the outgoing client socket to this source address\n");
|
||||
printf(" -4, --ipv4 Force IPv4 for destination resolution\n");
|
||||
printf(" -6, --ipv6 Force IPv6 for destination resolution\n");
|
||||
printf(" --sockopts=OPTS Comma-separated OPT=VAL socket options applied before connect:\n");
|
||||
printf(" TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR\n");
|
||||
printf(" --backup Backup existing files before overwriting\n");
|
||||
printf(" -b, --backup Backup existing files before overwriting\n");
|
||||
printf(" --backup-dir <dir> Directory for backups (requires --backup)\n");
|
||||
printf(" --suffix <str> Backup suffix (default: ~)\n");
|
||||
printf(" --stats Print transfer statistics at end\n");
|
||||
printf(" -i, --itemize-changes Print an rsync-style per-file change line\n");
|
||||
printf(" --out-format=FORMAT Output format for changed files (%%f %%n %%l %%b %%M %%%%)\n");
|
||||
printf(" --out-format=FORMAT Output format (%%f %%n %%l %%b %%c %%C %%i %%M %%%%)\n");
|
||||
printf(" --list-only List source files instead of transferring\n");
|
||||
printf(" --log-file-format=FORMAT Per-file log line format (needs --log-file)\n");
|
||||
printf(" -h, --human-readable Print byte sizes in human-readable form\n");
|
||||
printf(" --max-depth <n> Maximum directory depth (0=unlimited)\n");
|
||||
printf(" -x, --one-file-system Do not cross filesystem boundaries\n");
|
||||
printf(" --log-file <path> Write log messages to file\n");
|
||||
printf(" --log-file <path>, --log-file=<path> Write log messages to file\n");
|
||||
printf(" --stderr=MODE Route logging to stderr: errors or all\n");
|
||||
printf(" --partial Keep partial files on interrupted transfer\n");
|
||||
printf(" --partial-dir <dir> Directory for partial files\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install\n");
|
||||
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install.\n");
|
||||
printf(" Confined to the receive root: a relative dir resolves below\n");
|
||||
printf(" it and an absolute/traversal dir is rejected. The dir must\n");
|
||||
printf(" already exist; a different filesystem falls back to a\n");
|
||||
printf(" non-atomic copy instead of aborting\n");
|
||||
printf(" --fastsync-server-path <path>\n");
|
||||
printf(" Path to fastsync-server on remote (default: fastsync-server)\n");
|
||||
printf(" --old-args Accepted for rsync CLI compatibility; no effect (the\n");
|
||||
printf(" remote server path is always safely quoted now)\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation over SSH\n");
|
||||
printf(" (repeatable; each value is single-quote-escaped on the remote\n");
|
||||
printf(" command line; empty values and values with control characters\n");
|
||||
printf(" are rejected; -M OPT, -M=OPT and --remote-option=OPT work)\n");
|
||||
printf(" --trust-sender Trust the remote sender's file list: the receiver skips its\n");
|
||||
printf(" own up-front path-traversal/containment re-validation of the\n");
|
||||
printf(" incoming file list (fewer checks, faster, potentially unsafe).\n");
|
||||
printf(" Local receiver policy: never sent to the peer, off by default\n");
|
||||
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation. SSH\n");
|
||||
printf(" transport ONLY (user@host:path): a daemon (host::module) or\n");
|
||||
printf(" local TCP destination rejects it (no remote command line to\n");
|
||||
printf(" append to). Repeatable; each value is single-quote-escaped on\n");
|
||||
printf(" the remote command line; empty values and values with control\n");
|
||||
printf(" characters are rejected; -M OPT, -M=OPT and\n");
|
||||
printf(" --remote-option=OPT work\n");
|
||||
printf(" --trust-sender RECEIVER-LOCAL policy: trust the remote sender's file list\n");
|
||||
printf(" and skip the receiver's own up-front path-traversal/\n");
|
||||
printf(" containment re-validation of the incoming list (fewer checks,\n");
|
||||
printf(" faster, potentially unsafe). It is never sent to the peer, so\n");
|
||||
printf(" for a push it must be enabled on the receiving SERVER\n");
|
||||
printf(" (fastsync-server --trust-sender) or forwarded with\n");
|
||||
printf(" -M--trust-sender; the client flag alone has no effect\n");
|
||||
printf(" -l, --links Copy symlinks as symlinks\n");
|
||||
printf(" --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks that point outside transfer tree\n");
|
||||
printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n");
|
||||
printf(" -L, --copy-links Transform symlinks into referent files\n");
|
||||
printf(" --safe-links Skip symlinks whose target points outside the tree\n");
|
||||
printf(" --copy-unsafe-links Copy unsafe symlinks (outside tree) as referent files\n");
|
||||
printf(" -k, --copy-dirlinks Transform symlinks to directories into real dirs\n");
|
||||
printf(" -K, --keep-dirlinks Keep an existing symlink-to-dir as that dir\n");
|
||||
printf(" --munge-links Munge symlink targets on the wire (sender)\n");
|
||||
printf(" --munge-links Munge stored symlink targets (/rsyncd-munged/) on the receiver\n");
|
||||
printf(" -H, --hard-links Preserve hard-link relationships across the transfer\n");
|
||||
printf(" -S, --sparse Handle sparse files efficiently\n");
|
||||
printf(
|
||||
@@ -270,8 +318,7 @@ void print_usage(void) {
|
||||
printf(
|
||||
" --devices Recreate device nodes on the destination (privileged; skipped when\n");
|
||||
printf(" the receiver lacks CAP_MKNOD)\n");
|
||||
printf(" --specials Recreate special files (FIFOs) on the destination (sockets "
|
||||
"skipped)\n");
|
||||
printf(" --specials Recreate special files (FIFOs, sockets) on the destination\n");
|
||||
printf(" --copy-devices Copy a source device's content as a regular file instead\n");
|
||||
printf(" --write-devices Write received data into an existing destination device node\n");
|
||||
printf(" --inplace Update files in-place (no temp+rename)\n");
|
||||
@@ -284,7 +331,9 @@ void print_usage(void) {
|
||||
printf(" --fsync Fsync every written file before publication\n");
|
||||
printf(" --compress-level <n> Compression level (default: 5)\n");
|
||||
printf(" --zl <n> Alias for --compress-level\n");
|
||||
printf(" --skip-compress=LIST Skip compression for comma-separated suffixes\n");
|
||||
printf(" --skip-compress=LIST Skip compression for suffixes in LIST (separated by\n");
|
||||
printf(" '/' as in rsync, or ','); a leading dot is optional. The\n");
|
||||
printf(" default is rsync 3.4.1's built-in skip-compress list\n");
|
||||
printf(" --compress-threads <n> Compression worker threads (requires zstd threaded support)\n");
|
||||
printf(" --no-OPTION Disable a supported boolean option\n");
|
||||
printf(" --help Show this help\n");
|
||||
@@ -292,7 +341,20 @@ void print_usage(void) {
|
||||
}
|
||||
|
||||
void print_debug_usage(void) {
|
||||
printf("Supported debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n");
|
||||
printf("Emitting debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): ACL,BACKUP,BIND,CHDIR,\n");
|
||||
printf("CONNECT,CMD,DEL,DELTASUM,DUP,EXIT,FILTER,FLIST,FUZZY,GENR,HASH,HLINK,\n");
|
||||
printf("ICONV,NSTR,OWN,RECV,SEND,TIME.\n");
|
||||
printf("Flags may be comma-separated, for example: --debug=io,proto\n");
|
||||
printf("Other rsync debug flags are unsupported and rejected.\n");
|
||||
printf("An optional level suffix is accepted (e.g. --debug=io2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
|
||||
void print_info_usage(void) {
|
||||
printf("Emitting info flags: COPY,NAME,MISC,SKIP,STATS,ALL,NONE\n");
|
||||
printf("Also accepted for rsync CLI parity (silent): BACKUP,DEL,FLIST,MOUNT,\n");
|
||||
printf("NONREG,PROGRESS,REMOVE,SYMSAFE.\n");
|
||||
printf("Flags may be comma-separated, for example: --info=name,stats\n");
|
||||
printf("An optional level suffix is accepted (e.g. --info=stats2); level 0\n");
|
||||
printf("silences that item. Unknown names are rejected.\n");
|
||||
}
|
||||
|
||||
@@ -3,5 +3,6 @@
|
||||
|
||||
void print_usage(void);
|
||||
void print_debug_usage(void);
|
||||
void print_info_usage(void);
|
||||
|
||||
#endif
|
||||
|
||||
+306
-37
@@ -3,6 +3,7 @@
|
||||
#include "charset.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "delay_updates.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
@@ -11,7 +12,9 @@
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code) {
|
||||
if (!outcomes)
|
||||
@@ -42,17 +45,49 @@ void receiver_outcomes_destroy(ReceiverOutcomes* outcomes) {
|
||||
/* End-of-transfer success frame. When --remove-source-files was negotiated
|
||||
each processed data file is acknowledged first (STATUS_NEXT = written,
|
||||
STATUS_OK = skipped) so the sender never removes a source the receiver did
|
||||
not actually store. The frame always ends with a plain STATUS_OK. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes) {
|
||||
not actually store. The frame ends with `final_status` (STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped). */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status) {
|
||||
if (!config->remove_source_files)
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
size_t count = outcomes ? outcomes->count : 0;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
Status per_file = outcomes->entries[i] == FILE_SAVE_WRITTEN ? STATUS_NEXT : STATUS_OK;
|
||||
if (!send_status(fd, per_file))
|
||||
return false;
|
||||
}
|
||||
return send_status(fd, STATUS_OK);
|
||||
return send_status(fd, final_status);
|
||||
}
|
||||
|
||||
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
|
||||
const struct ArrayList* would_delete) {
|
||||
if (!config->report_stats)
|
||||
return true;
|
||||
ReceiverStats local;
|
||||
memset(&local, 0, sizeof(local));
|
||||
const ReceiverStats* out = stats ? stats : &local;
|
||||
size_t count = would_delete ? (size_t)would_delete->size : 0;
|
||||
if (count > (size_t)MAX_MANIFEST_ENTRIES)
|
||||
count = MAX_MANIFEST_ENTRIES;
|
||||
ReceiverStats record = *out;
|
||||
record.would_delete_count = count;
|
||||
if (!send_status(fd, STATUS_STATS) || !format_stats_send(fd, &record) ||
|
||||
!send_int(fd, (int)count))
|
||||
return false;
|
||||
for (size_t i = 0; i < count; i++) {
|
||||
const char* path = (const char*)would_delete->items[i];
|
||||
if (!send_wire_str(fd, path ? path : ""))
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Add a delete commit's tally to the sink's end-of-transfer wire counters (when
|
||||
the sink reports them). Runs on the receiving thread, so no locking. */
|
||||
static void receiver_tally_deleted(const ReceiverSink* sink, size_t deleted) {
|
||||
if (sink && sink->stats && deleted > 0)
|
||||
sink->stats->deleted_files += deleted;
|
||||
}
|
||||
|
||||
static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) {
|
||||
@@ -153,8 +188,95 @@ static bool receiver_process_batch(Config* config, int file_descriptor) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* ---- Anti-slowloris connection bounds ----
|
||||
* A legitimate transfer either streams data frames continuously or, when it
|
||||
* must pause, sends STATUS_KEEPALIVE so the peer sees the connection is alive.
|
||||
* An attacker can therefore squat on a connection slot indefinitely by sending
|
||||
* only keepalives under the per-message timeout. Two CLOCK_MONOTONIC bounds
|
||||
* defeat that without ever punishing a real transfer:
|
||||
*
|
||||
* MAX_SESSION_IDLE_SEC (1 h): the longest a stream may make no forward
|
||||
* progress. Data/status frames count as progress and refresh the timer;
|
||||
* keepalives do not. One hour is far longer than any real pause between
|
||||
* data frames, yet small enough to reap a slowloris well before the 24 h
|
||||
* session cap.
|
||||
*
|
||||
* MAX_SESSION_WALL_SEC (24 h): an absolute ceiling on one connection's
|
||||
* lifetime as defense-in-depth against a trickle of progress frames that
|
||||
* resets the idle timer just below its limit. Larger than any plausible
|
||||
* single transfer while still bounding resource occupancy.
|
||||
*
|
||||
* Both are wall-clock deltas, so the per-message poll timeout (60 s by default,
|
||||
* or --timeout) can never fool them, and both the single-threaded and the -m
|
||||
* receiver paths (receiver_process_pending) share the same logic. */
|
||||
#define MAX_SESSION_IDLE_SEC 3600u
|
||||
#define MAX_SESSION_WALL_SEC 86400u
|
||||
|
||||
static unsigned int g_max_session_idle_sec = MAX_SESSION_IDLE_SEC;
|
||||
static unsigned int g_max_session_wall_sec = MAX_SESSION_WALL_SEC;
|
||||
|
||||
void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec) {
|
||||
g_max_session_idle_sec = idle_sec;
|
||||
g_max_session_wall_sec = wall_sec;
|
||||
}
|
||||
|
||||
void receiver_reset_time_limits(void) {
|
||||
g_max_session_idle_sec = MAX_SESSION_IDLE_SEC;
|
||||
g_max_session_wall_sec = MAX_SESSION_WALL_SEC;
|
||||
}
|
||||
|
||||
bool receiver_time_limit_exceeded(const struct timespec* session_start,
|
||||
const struct timespec* last_progress,
|
||||
const struct timespec* now) {
|
||||
if (!session_start || !last_progress || !now)
|
||||
return false;
|
||||
if (now->tv_sec - session_start->tv_sec >= (time_t)g_max_session_wall_sec)
|
||||
return true;
|
||||
if (now->tv_sec - last_progress->tv_sec >= (time_t)g_max_session_idle_sec)
|
||||
return true;
|
||||
return false;
|
||||
}
|
||||
|
||||
/* A frame proves forward progress only when it cannot be fabricated for free.
|
||||
* KEEPALIVE/ABORT are pure liveness, and CHECK_BATCH/DIR_TIMES may carry zero
|
||||
* entries, so a peer must not be able to hold a connection slot forever by
|
||||
* merely emitting empty frames. */
|
||||
static bool status_counts_as_progress(Status status) {
|
||||
switch (status) {
|
||||
case STATUS_KEEPALIVE:
|
||||
case STATUS_ABORT:
|
||||
case STATUS_CHECK_BATCH:
|
||||
case STATUS_DIR_TIMES:
|
||||
return false;
|
||||
default:
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/* Refresh the progress timestamp for a forward-moving frame and enforce the
|
||||
* bounds above. Returns false when the connection must be dropped; the
|
||||
* terminal STATUS_ERROR is sent only when the sink owns error reporting (the
|
||||
* -m sink sets send_error=false so the main thread emits exactly one). */
|
||||
static bool receiver_note_status(const struct timespec* session_start,
|
||||
struct timespec* last_progress, Status status, int file_descriptor,
|
||||
const ReceiverSink* sink) {
|
||||
struct timespec now;
|
||||
if (clock_gettime(CLOCK_MONOTONIC, &now) != 0)
|
||||
now = *last_progress;
|
||||
if (status_counts_as_progress(status))
|
||||
*last_progress = now;
|
||||
if (!receiver_time_limit_exceeded(session_start, last_progress, &now))
|
||||
return true;
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"Receive session exceeded its time bound (idle %us / total %us); aborting connection",
|
||||
g_max_session_idle_sec, g_max_session_wall_sec);
|
||||
if (!sink || sink->send_error)
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
return false;
|
||||
}
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) {
|
||||
return receiver_process_pending(config, file_descriptor, sink, NULL);
|
||||
return receiver_process_pending(config, file_descriptor, sink, NULL, NULL);
|
||||
}
|
||||
|
||||
/* Runs the whole receive loop. The delete manifest may legitimately arrive
|
||||
@@ -168,19 +290,36 @@ int receiver_process(Config* config, int file_descriptor, const ReceiverSink* si
|
||||
the whole transfer succeeded. See receiver_process_pending() for how the -m
|
||||
receiver defers that commit until its disk writer has drained. */
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest) {
|
||||
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans) {
|
||||
Status status;
|
||||
if (!receive_status(file_descriptor, &status))
|
||||
return -1;
|
||||
/* Wall-clock (=CLOCK_MONOTONIC) anti-slowloris bookkeeping. session_start is
|
||||
* fixed for the whole connection; last_progress is refreshed by every frame
|
||||
* that is not a keepalive/abort. */
|
||||
struct timespec session_start;
|
||||
struct timespec last_progress;
|
||||
clock_gettime(CLOCK_MONOTONIC, &session_start);
|
||||
last_progress = session_start;
|
||||
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
|
||||
return -1;
|
||||
bool early_delete = config_delete_timing_early(config);
|
||||
bool per_dir_delete = config_delete_timing_per_dir(config);
|
||||
/* Parked keep-set for the late/commit timing. Every exit path below frees it
|
||||
exactly once; the only exception is the successful FINISHED handoff, which
|
||||
transfers ownership to *pending_manifest (used by the -m receiver). */
|
||||
DeleteManifest* deferred_manifest = NULL;
|
||||
/* Per-directory delete session for --delete-during/--delete-delay. During the
|
||||
loop it applies plans inline (during) or snapshots their extras (delay); on
|
||||
a successful FINISHED it is either committed here or handed to
|
||||
*pending_plans so the -m caller commits after its disk writer drained. */
|
||||
DeletePlanSession* plan_session = NULL;
|
||||
bool delete_limit_noted = false;
|
||||
while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK ||
|
||||
status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH ||
|
||||
status == STATUS_MKDIR || status == STATUS_MANIFEST || status == STATUS_HARDLINK ||
|
||||
status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES) {
|
||||
status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES ||
|
||||
status == STATUS_DELETE_PLAN) {
|
||||
if (status == STATUS_KEEPALIVE) {
|
||||
if (!send_status(file_descriptor, STATUS_KEEPALIVE))
|
||||
goto fail;
|
||||
@@ -191,10 +330,19 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
goto fail;
|
||||
}
|
||||
if (status == STATUS_CHECK) {
|
||||
bool skipped;
|
||||
File* file = receive_incremental_check(file_descriptor, config, &skipped);
|
||||
if (!skipped && (!file || !sink->store_file(file, sink->context)))
|
||||
bool skipped = false;
|
||||
bool would_transfer = false;
|
||||
File* file = receive_incremental_check_ex(file_descriptor, config, &skipped, &would_transfer);
|
||||
if (config->dry_run) {
|
||||
/* Server-contacting --dry-run: the reply has already been sent
|
||||
(STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and
|
||||
nothing may be stored. Both flags false means a genuine protocol
|
||||
error (STATUS_ERROR already sent or sent by receive_error below). */
|
||||
if (!skipped && !would_transfer)
|
||||
goto receive_error;
|
||||
} else if (!skipped && (!file || !sink->store_file(file, sink->context))) {
|
||||
goto receive_error;
|
||||
}
|
||||
} else if (status == STATUS_CHUNK) {
|
||||
Chunk* chunk = receive_chunk_data(file_descriptor, config);
|
||||
if (!chunk || !receiver_process_chunk(chunk, sink))
|
||||
@@ -226,26 +374,47 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
DeleteManifest* manifest = receive_manifest_entries(file_descriptor);
|
||||
if (!manifest)
|
||||
goto fail; /* receive_manifest_entries already sent STATUS_ERROR */
|
||||
if (early_delete) {
|
||||
/* --delete-before / --delete-during: the manifest is authoritative the
|
||||
moment it arrives, before any file data. Delete now and acknowledge
|
||||
so the sender only starts streaming once the deletion committed (or
|
||||
failed). This is the rsync delete-before/delete-during window: a
|
||||
later transfer failure does not restore these deletions. */
|
||||
bool deletion_ok = (config->use_delete || config->delete_missing_args)
|
||||
? manifest_delete_all(config, manifest)
|
||||
: true;
|
||||
if (config->dry_run) {
|
||||
/* Server-contacting --dry-run mutates nothing, so a keep-set manifest
|
||||
is consumed and discarded. The early-delete mode still needs its ACK
|
||||
so a sender blocked on the delete handshake is not left hanging.
|
||||
When would-delete reporting is armed, enumerate (read-only) the
|
||||
destination extras so the terminal STATUS_STATS frame can list them. */
|
||||
if (config->use_delete && sink->would_delete) {
|
||||
size_t count = 0;
|
||||
if (!manifest_would_delete_list(config, manifest, sink->would_delete, &count))
|
||||
log_message(LOG_LEVEL_WARNING, "dry-run: could not enumerate would-delete paths");
|
||||
}
|
||||
delete_manifest_free(manifest);
|
||||
if (!deletion_ok) {
|
||||
if (early_delete && !send_status(file_descriptor, STATUS_OK))
|
||||
goto fail;
|
||||
goto next_status;
|
||||
}
|
||||
if (early_delete) {
|
||||
/* --delete-before: the whole-tree manifest is authoritative the moment
|
||||
it arrives, before any file data. Delete now and acknowledge so the
|
||||
sender only starts streaming once the deletion committed (or failed).
|
||||
A later transfer failure does not restore these deletions. A
|
||||
--max-delete-capped commit still succeeds and the transfer proceeds;
|
||||
the terminal success frame reports the cap. */
|
||||
size_t deleted = 0;
|
||||
DeleteCommitResult deletion = (config->use_delete || config->delete_missing_args)
|
||||
? manifest_delete_all_counted(config, manifest, &deleted)
|
||||
: DELETE_COMMIT_OK;
|
||||
receiver_tally_deleted(sink, deleted);
|
||||
delete_manifest_free(manifest);
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
if (!send_status(file_descriptor, STATUS_OK))
|
||||
goto fail;
|
||||
} else if (config->use_delete || config->delete_missing_args) {
|
||||
/* Plain --delete / --delete-after / --delete-delay and the
|
||||
--delete-missing-args exact-path deletions: hold the manifest and
|
||||
commit it only after STATUS_FINISHED. */
|
||||
/* Plain --delete / --delete-after and the --delete-missing-args
|
||||
exact-path deletions: hold the manifest and commit it only after
|
||||
STATUS_FINISHED. The per-directory modes never send this frame. */
|
||||
if (deferred_manifest) {
|
||||
log_message(LOG_LEVEL_ERROR, "Received a second delete manifest");
|
||||
delete_manifest_free(deferred_manifest);
|
||||
@@ -259,6 +428,23 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
delete_manifest_free(manifest);
|
||||
}
|
||||
goto next_status;
|
||||
} else if (status == STATUS_DELETE_PLAN) {
|
||||
if (!per_dir_delete) {
|
||||
log_message(LOG_LEVEL_ERROR, "Received a per-directory delete plan without a per-dir "
|
||||
"delete timing");
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (!plan_session)
|
||||
plan_session = delete_plan_session_create(config);
|
||||
if (!plan_session || delete_plan_session_receive(plan_session, config, file_descriptor) != 0)
|
||||
goto fail;
|
||||
if (delete_plan_session_limit_reached(plan_session) && !delete_limit_noted &&
|
||||
sink->note_delete_limit) {
|
||||
sink->note_delete_limit(sink->context);
|
||||
delete_limit_noted = true;
|
||||
}
|
||||
goto next_status;
|
||||
} else {
|
||||
File* file = file_receive(config, file_descriptor);
|
||||
if (!file) {
|
||||
@@ -271,6 +457,8 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
next_status:
|
||||
if (!receive_status(file_descriptor, &status))
|
||||
goto receive_error;
|
||||
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
|
||||
goto fail;
|
||||
}
|
||||
if (status != STATUS_FINISHED) {
|
||||
log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status");
|
||||
@@ -290,13 +478,45 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
*pending_manifest = deferred_manifest;
|
||||
deferred_manifest = NULL;
|
||||
} else {
|
||||
bool deletion_ok = manifest_delete_all(config, deferred_manifest);
|
||||
size_t deleted = 0;
|
||||
DeleteCommitResult deletion =
|
||||
manifest_delete_all_counted(config, deferred_manifest, &deleted);
|
||||
receiver_tally_deleted(sink, deleted);
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
if (!deletion_ok) {
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (deletion == DELETE_COMMIT_LIMIT_REACHED && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
}
|
||||
}
|
||||
/* Per-directory deletion: --delete-during already applied each plan inline, so
|
||||
this only finishes the missing-args deletions; --delete-delay committed
|
||||
nothing yet and applies its decompressed snapshot here. The -m receiver
|
||||
hands the session to its caller instead, which commits after the disk
|
||||
writer drained. */
|
||||
if (plan_session) {
|
||||
if (pending_plans) {
|
||||
*pending_plans = plan_session;
|
||||
plan_session = NULL;
|
||||
} else if (config->dry_run) {
|
||||
/* Central dry-run no-op: never commit a deletion for a -n run. */
|
||||
delete_plan_session_destroy(plan_session);
|
||||
plan_session = NULL;
|
||||
} else {
|
||||
DeleteCommitResult deletion = delete_plan_session_commit(plan_session, config);
|
||||
bool limit = delete_plan_session_limit_reached(plan_session);
|
||||
receiver_tally_deleted(sink, delete_plan_session_deleted(plan_session));
|
||||
delete_plan_session_destroy(plan_session);
|
||||
plan_session = NULL;
|
||||
if (deletion == DELETE_COMMIT_ERROR) {
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
goto fail;
|
||||
}
|
||||
if (limit && !delete_limit_noted && sink->note_delete_limit)
|
||||
sink->note_delete_limit(sink->context);
|
||||
}
|
||||
}
|
||||
if (sink->send_success) {
|
||||
@@ -311,11 +531,14 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
|
||||
|
||||
fail:
|
||||
/* Failure exits that must not (or already did) report a STATUS_ERROR. The
|
||||
parked keep-set is dropped: never commit a deletion for a failed stream. */
|
||||
parked keep-set/session is dropped: never commit a deletion for a failed
|
||||
stream. */
|
||||
if (deferred_manifest) {
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
}
|
||||
if (plan_session)
|
||||
delete_plan_session_destroy(plan_session);
|
||||
return -1;
|
||||
|
||||
receive_error:
|
||||
@@ -323,6 +546,8 @@ receive_error:
|
||||
delete_manifest_free(deferred_manifest);
|
||||
deferred_manifest = NULL;
|
||||
}
|
||||
if (plan_session)
|
||||
delete_plan_session_destroy(plan_session);
|
||||
if (sink->send_error)
|
||||
send_status(file_descriptor, STATUS_ERROR);
|
||||
return -1;
|
||||
@@ -337,29 +562,49 @@ typedef struct {
|
||||
after the whole transfer (and its delete/publication phases) has run so a
|
||||
child write never clobbers a directory mtime. */
|
||||
DirTimeList dir_times;
|
||||
/* Set when a --max-delete commit was capped; the terminal frame then carries
|
||||
STATUS_DELETE_LIMIT so the sender exits 25 like rsync. */
|
||||
bool delete_limit_reached;
|
||||
/* End-of-transfer wire counters (protocol 2.25.0) and the -n/--dry-run
|
||||
--delete would-delete path list collected while processing the manifest. */
|
||||
ReceiverStats stats;
|
||||
ArrayList* would_delete;
|
||||
} ReceiverSaveContext;
|
||||
|
||||
static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
FileSaveResult result = FILE_SAVE_ERROR;
|
||||
if (!context->config->save_to_disk) {
|
||||
if (context->config->dry_run) {
|
||||
/* Defense in depth: a dry-run receiver mutates nothing even if a data
|
||||
frame reaches the sink (the sender is not supposed to send one). */
|
||||
result = FILE_SAVE_SKIPPED;
|
||||
} else if (!context->config->save_to_disk) {
|
||||
/* Nothing is stored; report the file as not-written so a
|
||||
--remove-source-files sender keeps its source. */
|
||||
result = FILE_SAVE_SKIPPED;
|
||||
} else {
|
||||
result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
|
||||
}
|
||||
/* A directory's times are deferred, never applied inline: collect the
|
||||
metadata now and apply it at the end. -O/--omit-dir-times is honored by
|
||||
dir_time_list_apply's caller (see receiver_send_success_frame). */
|
||||
/* Wire-stats tally: bytes reconstructed from the basis file (delta matches)
|
||||
count as matched data in the end-of-transfer report. */
|
||||
if (result != FILE_SAVE_ERROR && file->matched_bytes > 0)
|
||||
context->stats.matched_data += file->matched_bytes;
|
||||
/* A directory's metadata is deferred, never applied inline: collect it now
|
||||
and apply it at the end. -O/--omit-dir-times and --preserve_perms/-times
|
||||
are honored by dir_metadata_list_apply's caller (see
|
||||
receiver_send_success_frame). */
|
||||
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
context->config->use_metadata && !context->config->omit_dir_times &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (result != FILE_SAVE_ERROR && context->config->remove_source_files && !file->is_dir &&
|
||||
!file->is_special && !file->skip &&
|
||||
/* A dry-run receiver mutates nothing AND records no per-file outcomes: a
|
||||
hostile dry-run client that streamed data frames anyway must not be able to
|
||||
grow `outcomes` without bound (receiver_outcomes_append reallocs uncharged)
|
||||
or force a per-frame ack. */
|
||||
if (!context->config->dry_run && result != FILE_SAVE_ERROR &&
|
||||
context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip &&
|
||||
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
return false;
|
||||
@@ -368,8 +613,20 @@ static bool receiver_save_file(File* file, void* context_pointer) {
|
||||
return result != FILE_SAVE_ERROR;
|
||||
}
|
||||
|
||||
static void receiver_note_delete_limit(void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static bool receiver_send_success_frame(int fd, void* context_pointer) {
|
||||
ReceiverSaveContext* context = context_pointer;
|
||||
Status final_status = context->delete_limit_reached ? STATUS_DELETE_LIMIT : STATUS_OK;
|
||||
if (!receiver_send_stats_frame(fd, context->config, &context->stats, context->would_delete))
|
||||
return false;
|
||||
/* Server-contacting --dry-run: nothing was staged or written, so there is
|
||||
nothing to publish and no directory times to stamp. */
|
||||
if (context->config->dry_run)
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
/* --delay-updates: the whole protocol stream (including manifest/delete
|
||||
handling, which ran inside receiver_process) has succeeded and every
|
||||
staged file was fully written. Publish them atomically now, before the
|
||||
@@ -385,18 +642,30 @@ static bool receiver_send_success_frame(int fd, void* context_pointer) {
|
||||
phases have committed, so it is finally safe to stamp directory times.
|
||||
This runs after the deferred deletion because receiver_process commits it
|
||||
before calling this success frame. */
|
||||
dir_time_list_apply(&context->dir_times, context->config->receive_root_directory);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes);
|
||||
dir_metadata_list_apply(&context->dir_times, context->config->receive_root_directory,
|
||||
context->config);
|
||||
return receiver_send_final_success(fd, context->config, &context->outcomes, final_status);
|
||||
}
|
||||
|
||||
int receiver_receive_files(Config* config, int file_descriptor) {
|
||||
ReceiverSaveContext context = {.config = config, .outcomes = {0}};
|
||||
dir_time_list_init(&context.dir_times);
|
||||
ReceiverSink sink = {receiver_save_file, &context, true, true, receiver_send_success_frame};
|
||||
context.would_delete = array_list_create(free);
|
||||
if (!context.would_delete)
|
||||
return -1;
|
||||
ReceiverSink sink = {receiver_save_file,
|
||||
&context,
|
||||
true,
|
||||
true,
|
||||
receiver_send_success_frame,
|
||||
receiver_note_delete_limit,
|
||||
&context.stats,
|
||||
context.would_delete};
|
||||
int ret = receiver_process(config, file_descriptor, &sink);
|
||||
if (ret != 0 && config->delay_updates && config->delay_context)
|
||||
delay_updates_cleanup(config->delay_context);
|
||||
receiver_outcomes_destroy(&context.outcomes);
|
||||
dir_time_list_free(&context.dir_times);
|
||||
array_list_delete(context.would_delete);
|
||||
return ret;
|
||||
}
|
||||
|
||||
+53
-5
@@ -2,8 +2,12 @@
|
||||
#define RECEIVER_H
|
||||
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include <stdbool.h>
|
||||
#include <time.h>
|
||||
|
||||
typedef bool (*ReceiverFileSink)(File* file, void* context);
|
||||
|
||||
@@ -19,6 +23,12 @@ typedef struct {
|
||||
|
||||
typedef bool (*ReceiverSuccessFrame)(int fd, void* context);
|
||||
|
||||
/* Records that a --max-delete commit stopped with extras left over, so the
|
||||
caller's terminal success frame can carry STATUS_DELETE_LIMIT instead of
|
||||
STATUS_OK. The commit runs on the receiver thread, so the flag is stored in
|
||||
the sink's own context rather than in a shared global. */
|
||||
typedef void (*ReceiverNoteDeleteLimit)(void* context);
|
||||
|
||||
typedef struct {
|
||||
ReceiverFileSink store_file;
|
||||
void* context;
|
||||
@@ -26,23 +36,61 @@ typedef struct {
|
||||
bool send_success;
|
||||
/* Emits the end-of-transfer success frame. When the sender requested
|
||||
--remove-source-files this includes one per-file status per processed
|
||||
data file followed by the final STATUS_OK; otherwise just STATUS_OK. */
|
||||
data file followed by the final status; otherwise just the final status. */
|
||||
ReceiverSuccessFrame send_success_frame;
|
||||
/* Optional; may be NULL when the sink has no --max-delete handling. */
|
||||
ReceiverNoteDeleteLimit note_delete_limit;
|
||||
/* Optional end-of-transfer wire counters (protocol 2.25.0). When non-NULL
|
||||
and the wire config carries report_stats, the success frame is preceded by
|
||||
a STATUS_STATS record; `would_delete` (optional, receiver-owned strings)
|
||||
carries the -n/--dry-run --delete path list. */
|
||||
ReceiverStats* stats;
|
||||
struct ArrayList* would_delete;
|
||||
} ReceiverSink;
|
||||
|
||||
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code);
|
||||
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes);
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes);
|
||||
|
||||
/* Send the terminal success frame. `final_status` is usually STATUS_OK, or
|
||||
STATUS_DELETE_LIMIT when a --max-delete commit was capped. */
|
||||
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes,
|
||||
Status final_status);
|
||||
|
||||
/* Emit STATUS_STATS (a fixed ReceiverStats record plus, when `would_delete` is
|
||||
non-NULL, a count and that many wire strings) when the wire config requested
|
||||
report_stats. A no-op otherwise. */
|
||||
bool receiver_send_stats_frame(int fd, const Config* config, const ReceiverStats* stats,
|
||||
const struct ArrayList* would_delete);
|
||||
|
||||
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink);
|
||||
/* receiver_process with an escape hatch for the commit-style (late) deletion:
|
||||
when `pending_manifest` is non-NULL the receiver does NOT delete at
|
||||
STATUS_FINISHED itself; instead it stores the owned keep-set manifest there
|
||||
(leaving *pending_manifest untouched on early modes/errors) so the caller can
|
||||
commit the deletion only after its disk writer has fully drained. Pass NULL
|
||||
to keep the default behaviour (delete before the success frame). */
|
||||
commit the deletion only after its disk writer has fully drained. Likewise,
|
||||
when `pending_plans` is non-NULL the --delete-delay per-directory session is
|
||||
handed to the caller instead of being committed at STATUS_FINISHED. Pass NULL
|
||||
for either to keep the default behaviour (delete before the success frame). */
|
||||
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
|
||||
DeleteManifest** pending_manifest);
|
||||
DeleteManifest** pending_manifest, DeletePlanSession** pending_plans);
|
||||
int receiver_receive_files(Config* config, int file_descriptor);
|
||||
|
||||
/* ---- Connection time bounds (anti-slowloris) ----
|
||||
* receiver_process_pending() aborts a connection that makes no forward progress
|
||||
* (only STATUS_KEEPALIVE/STATUS_ABORT frames) beyond a wall-clock idle limit,
|
||||
* and enforces a hard cap on the whole session. Both are CLOCK_MONOTONIC
|
||||
* deltas, independent of the per-message poll deadline, so a 60 s (or
|
||||
* --timeout) receive window can never reset them. Defaults are deliberately
|
||||
* generous (see MAX_SESSION_IDLE_SEC / MAX_SESSION_WALL_SEC in receiver.c). */
|
||||
|
||||
/* Test seam: override the idle/session wall-clock limits (0 = abort on the
|
||||
* next status). Always restore with receiver_reset_time_limits(). */
|
||||
void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec);
|
||||
void receiver_reset_time_limits(void);
|
||||
/* Pure predicate over explicit monotonic timestamps, exposed so the bound is
|
||||
* unit-testable without sleeping. True when either the idle or the overall
|
||||
* session limit has elapsed. */
|
||||
bool receiver_time_limit_exceeded(const struct timespec* session_start,
|
||||
const struct timespec* last_progress, const struct timespec* now);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,290 @@
|
||||
#include "receiver_pipeline.h"
|
||||
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <threads.h>
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue,
|
||||
int file_descriptor, SSL* ssl) {
|
||||
PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver));
|
||||
if (context == NULL)
|
||||
return NULL;
|
||||
context->config = config;
|
||||
context->queue = queue;
|
||||
context->file_descriptor = file_descriptor;
|
||||
context->ssl = ssl;
|
||||
context->outcomes.entries = NULL;
|
||||
context->outcomes.count = 0;
|
||||
context->outcomes.capacity = 0;
|
||||
dir_time_list_init(&context->dir_times);
|
||||
protocol_session_init(&context->session, file_descriptor, file_descriptor);
|
||||
protocol_session_set_ssl(&context->session, ssl);
|
||||
context->receiver_done = false;
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->deferred_manifest = NULL;
|
||||
context->deferred_plans = NULL;
|
||||
context->delete_limit_reached = false;
|
||||
memset(&context->stats, 0, sizeof(context->stats));
|
||||
context->would_delete = NULL;
|
||||
atomic_init(&context->cancelled, false);
|
||||
int init = 0;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_full) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_empty) != thrd_success)
|
||||
goto fail;
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
context->would_delete = array_list_create(free);
|
||||
if (!context->would_delete)
|
||||
goto fail;
|
||||
return context;
|
||||
|
||||
fail:
|
||||
log_perror("Error initializing synchronization objects");
|
||||
if (init >= 3)
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
if (init >= 2)
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&context->mutex);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
|
||||
config_delete(context->config);
|
||||
if (context->deferred_manifest)
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
if (context->deferred_plans)
|
||||
delete_plan_session_destroy(context->deferred_plans);
|
||||
queue_destroy(context->queue);
|
||||
receiver_outcomes_destroy(&context->outcomes);
|
||||
dir_time_list_free(&context->dir_times);
|
||||
if (context->would_delete)
|
||||
array_list_delete(context->would_delete);
|
||||
mtx_destroy(&context->mutex);
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
free(context);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes) {
|
||||
if (context == NULL)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
context->max_queue_bytes = max_bytes;
|
||||
context->queued_bytes = 0;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes) {
|
||||
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
if (released_bytes >= context->queued_bytes)
|
||||
context->queued_bytes = 0;
|
||||
else
|
||||
context->queued_bytes -= released_bytes;
|
||||
cnd_signal(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) {
|
||||
if (context == NULL || file == NULL)
|
||||
return false;
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
mtx_lock(&context->mutex);
|
||||
while (!atomic_load(&context->cancelled)) {
|
||||
bool blocked_by_count = queue_is_full(context->queue);
|
||||
bool blocked_by_budget = false;
|
||||
if (context->max_queue_bytes > 0) {
|
||||
size_t budget = context->max_queue_bytes;
|
||||
size_t used = context->queued_bytes;
|
||||
if (used >= budget) {
|
||||
blocked_by_budget = true;
|
||||
} else if (file_bytes > budget - used) {
|
||||
/* A single payload larger than the whole budget (not possible with
|
||||
the per-file receive cap) is only admitted to an empty pipeline so
|
||||
the wait can never deadlock. */
|
||||
blocked_by_budget = used != 0;
|
||||
}
|
||||
}
|
||||
if (!blocked_by_count && !blocked_by_budget)
|
||||
break;
|
||||
cnd_wait(&context->condition_not_full, &context->mutex);
|
||||
}
|
||||
if (atomic_load(&context->cancelled)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (!queue_enqueue(context->queue, file)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
context->queued_bytes += file_bytes;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool receiver_enqueue_file(File* file, void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
if (file && file->matched_bytes > 0) {
|
||||
mtx_lock(&context->mutex);
|
||||
context->stats.matched_data += file->matched_bytes;
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
return pipeline_context_receiver_enqueue_file(context, file);
|
||||
}
|
||||
|
||||
/* Early delete modes (--delete-before/--delete-during) commit the manifest
|
||||
inside receiver_process_pending on this thread; record a capped commit so
|
||||
server.c's terminal frame can report STATUS_DELETE_LIMIT. The plain bool is
|
||||
safe: receive_thread writes it before the main thread joins the thread. */
|
||||
static void receiver_pipeline_note_delete_limit(void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
context->delete_limit_reached = true;
|
||||
}
|
||||
|
||||
static void receiver_thread_fail(PipelineContextReceiver* context) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
int receive_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
int file_descriptor = context->file_descriptor;
|
||||
const Config* config = context->config;
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
ReceiverSink sink = {receiver_enqueue_file,
|
||||
context,
|
||||
false,
|
||||
false,
|
||||
NULL,
|
||||
receiver_pipeline_note_delete_limit,
|
||||
&context->stats,
|
||||
context->would_delete};
|
||||
if (receiver_process_pending((Config*)config, file_descriptor, &sink, &context->deferred_manifest,
|
||||
&context->deferred_plans) != 0) {
|
||||
receiver_thread_fail(context);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
mtx_lock(&context->mutex);
|
||||
context->receiver_done = true;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
|
||||
int write_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
bool save_to_disk = context->config->save_to_disk;
|
||||
char* root_directory = str_dup(context->config->receive_root_directory);
|
||||
mtx_unlock(&context->mutex);
|
||||
if (save_to_disk && !root_directory) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
File* file =
|
||||
queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty,
|
||||
&context->condition_not_full, &context->receiver_done);
|
||||
if (file == NULL) {
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
FileSaveResult result = FILE_SAVE_SKIPPED;
|
||||
/* Server-contacting --dry-run: never write. The receiver thread does not
|
||||
enqueue anything on the dry-run path, but this keeps the writer thread
|
||||
provably mutation-free if a data frame ever reached it. */
|
||||
bool dry_run = context->config->dry_run;
|
||||
if (save_to_disk && !dry_run) {
|
||||
result = file_save_to_disk_full(root_directory, file, context->config);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
}
|
||||
/* P7 Wave D: a directory's times are never applied inline (a later child
|
||||
write would clobber them); accumulate the metadata here and let the
|
||||
caller apply it once every writer has drained. */
|
||||
if (!dry_run && result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(context->config) &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
/* Record the per-file outcome so a --remove-source-files sender learns
|
||||
which sources were actually written versus skipped on the receiver.
|
||||
Explicit directory entries and recreated device/special nodes have no
|
||||
source and are never acknowledged (mirrors receiver.c). */
|
||||
if (!dry_run && context->config->remove_source_files && !file->is_dir && !file->is_special &&
|
||||
!file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
#ifndef RECEIVER_PIPELINE_H
|
||||
#define RECEIVER_PIPELINE_H
|
||||
|
||||
#include <stdatomic.h>
|
||||
#include <stdbool.h>
|
||||
#include <threads.h>
|
||||
|
||||
#include "config.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "receiver.h"
|
||||
#include <openssl/ssl.h>
|
||||
|
||||
typedef struct PipelineContextReceiver {
|
||||
Queue* queue;
|
||||
Config* config;
|
||||
int file_descriptor;
|
||||
SSL* ssl;
|
||||
ProtocolSession session;
|
||||
ReceiverOutcomes outcomes;
|
||||
mtx_t mutex;
|
||||
cnd_t condition_not_full;
|
||||
cnd_t condition_not_empty;
|
||||
bool receiver_done;
|
||||
atomic_bool cancelled;
|
||||
/* Aggregate payload bytes that have been received but not yet released by
|
||||
the disk writer (queued or in the writer's hand). Guarded by `mutex`.
|
||||
When `max_queue_bytes` is non-zero the receiver blocks before enqueuing
|
||||
once this total would exceed it, so decompressed/copied file payloads
|
||||
buffered ahead of a slow disk writer respect the per-connection memory
|
||||
budget instead of growing without bound. */
|
||||
size_t queued_bytes;
|
||||
size_t max_queue_bytes;
|
||||
/* Keep-set manifest for the commit-style (late) deletion
|
||||
(--delete/--delete-after/--delete-delay). receive_thread parses the whole
|
||||
protocol stream but hands the manifest here instead of deleting while the
|
||||
disk writer may still be draining; the caller (server.c) commits the
|
||||
deletion after both threads have joined, so no extra is removed unless the
|
||||
transfer truly succeeded. NULL in the early delete modes (which delete at
|
||||
the manifest). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* Per-directory delete session for --delete-delay: receive_thread snapshots
|
||||
each plan's extras as it arrives and hands the session here instead of
|
||||
committing while the disk writer may still be draining; server.c commits it
|
||||
after both threads joined. NULL for every other timing. */
|
||||
DeletePlanSession* deferred_plans;
|
||||
/* Set by server.c when the deferred delete commit hit the --max-delete
|
||||
budget; the terminal success frame then carries STATUS_DELETE_LIMIT
|
||||
(rsync exit 25) while the transfer itself still succeeds. */
|
||||
bool delete_limit_reached;
|
||||
/* P7 Wave D: directory metadata collected by write_thread from received
|
||||
directory entries. Only write_thread mutates it (before it joins); the
|
||||
caller (server.c) applies it after the delete/delay-updates phase. */
|
||||
DirTimeList dir_times;
|
||||
/* End-of-transfer wire counters (protocol 2.25.0). receive_thread accumulates
|
||||
matched_data under `mutex`; server.c adds the delete-commit tallies after
|
||||
both threads join and emits the STATUS_STATS frame. */
|
||||
ReceiverStats stats;
|
||||
/* -n/--dry-run --delete would-delete path list, collected by receive_thread
|
||||
and reported in the STATUS_STATS frame. */
|
||||
struct ArrayList* would_delete;
|
||||
} PipelineContextReceiver;
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
|
||||
int file_descriptor, SSL* ssl);
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context);
|
||||
/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes);
|
||||
/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue
|
||||
is full by element count or when adding `file` would push queued_bytes over
|
||||
the configured byte limit; waits until the disk writer releases bytes.
|
||||
Takes ownership of `file` on success and destroys it on failure/cancel. */
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file);
|
||||
/* Account for `released_bytes` of payload memory that has been freed by the
|
||||
disk writer, unblocking a receiver that is waiting on the byte limit. */
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes);
|
||||
int receive_thread(void* pipeline_context);
|
||||
int write_thread(void* pipeline_context);
|
||||
|
||||
#endif
|
||||
+628
-243
File diff suppressed because it is too large
Load Diff
+32
-2
@@ -179,6 +179,8 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
opts->trust_sender = true;
|
||||
} else if (arg_is(argv[i], "--no-super")) {
|
||||
opts->no_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-super")) {
|
||||
opts->allow_super = true;
|
||||
} else if (arg_is(argv[i], "--allow-unauthenticated")) {
|
||||
opts->allow_unauthenticated = true;
|
||||
} else if (arg_has_value(argv[i], "--iconv", &inline_value)) {
|
||||
@@ -190,14 +192,20 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
inline_value = argv[++i];
|
||||
}
|
||||
opts->iconv_spec = inline_value;
|
||||
} else if (arg_is(argv[i], "-p")) {
|
||||
} else if (arg_is(argv[i], "-p") || arg_has_value(argv[i], "--port", &inline_value)) {
|
||||
if (inline_value) {
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(inline_value, &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
} else {
|
||||
if (i + 1 >= argc) {
|
||||
set_error(err, err_size, "missing argument for -p");
|
||||
set_error(err, err_size, "missing argument for %s", argv[i]);
|
||||
return -1;
|
||||
}
|
||||
opts->port_set = true;
|
||||
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
|
||||
return -1;
|
||||
}
|
||||
} else {
|
||||
if (arg_has_value(argv[i], "--config", &inline_value)) {
|
||||
if (!inline_value) {
|
||||
@@ -259,6 +267,28 @@ int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err,
|
||||
set_error(err, err_size, "--hash-credentials cannot be combined with --daemon or --stdio");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->no_super) {
|
||||
set_error(err, err_size, "--allow-super and --no-super are mutually exclusive");
|
||||
return -1;
|
||||
}
|
||||
if (opts->allow_super && opts->daemon_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is for a locally-launched standalone TCP server; daemon modules opt "
|
||||
"in per module with 'client owner = yes'");
|
||||
return -1;
|
||||
}
|
||||
/* --stdio is the SSH transport: the remote server argv is composed by the
|
||||
* CLIENT (directly and via --remote-option), so a client could otherwise pass
|
||||
* --allow-super to a root --stdio receiver and defeat the C3 secure default.
|
||||
* Never honor it there; the super mode stays forced OFF. An operator who
|
||||
* must keep the historical permissive behavior over SSH has to launch the
|
||||
* receiver through a forced command, not via client-composed argv. */
|
||||
if (opts->allow_super && opts->stdio_mode) {
|
||||
set_error(err, err_size,
|
||||
"--allow-super is not accepted with --stdio (the remote argv is client-composed; "
|
||||
"use a forced command if the default must hold)");
|
||||
return -1;
|
||||
}
|
||||
if (opts->hash_iterations_set && opts->hash_credentials_file == NULL) {
|
||||
set_error(err, err_size, "--iterations requires --hash-credentials");
|
||||
return -1;
|
||||
|
||||
@@ -45,6 +45,17 @@ typedef struct ServerCliOptions {
|
||||
* device-node creation) even when running as root. Applies to --stdio and
|
||||
* --daemon alike; also makes the server refuse any client --copy-as. */
|
||||
bool no_super; /* --no-super */
|
||||
/* --allow-super: locally-launched standalone TCP listener opt-in that keeps
|
||||
* the historical permissive behavior for a PRIVILEGED (root) receiver.
|
||||
* Without it a root standalone server forces SUPER_MODE_OFF, so a client
|
||||
* --devices / --write-devices / --super / ownership request cannot make it
|
||||
* create device nodes, write raw devices, or apply client-chosen ownership.
|
||||
* It is rejected for --stdio: that path's remote argv is composed by the
|
||||
* client (directly and via --remote-option), so it must never opt a root
|
||||
* receiver back into super mode. Non-root receivers are unaffected (the
|
||||
* kernel refuses the confined attempts). The daemon path instead uses the
|
||||
* per-module `client owner = yes` opt-in. */
|
||||
bool allow_super; /* --allow-super */
|
||||
/* --iconv=CONVERT_SPEC: the server's own LOCAL charset declaration. The
|
||||
* client's full spec rides the wire config frame anyway; when the server is
|
||||
* started with its own --iconv, its LOCAL half overrides the local charset
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
#include "log.h"
|
||||
#include "array_list.h"
|
||||
#include "protocol.h"
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -39,6 +40,8 @@ void array_list_delete(ArrayList* array_list) {
|
||||
static bool array_list_extend(ArrayList* array_list) {
|
||||
if (array_list == NULL)
|
||||
return false;
|
||||
if (array_list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_capacity = array_list->capacity * 2;
|
||||
if (new_capacity == 0)
|
||||
new_capacity = INITIAL_ARRAY_SIZE;
|
||||
|
||||
+53
-18
@@ -2,6 +2,7 @@
|
||||
#include "data.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "identity.h"
|
||||
#include "log.h"
|
||||
#include <errno.h>
|
||||
#include <stdlib.h>
|
||||
@@ -10,11 +11,15 @@
|
||||
|
||||
/* Serialization metadata mode for the batch stream, captured from the config at
|
||||
* batch_write_header time. The header persists it into the file so a batch is
|
||||
* self-describing: batch_read_apply re-reads it from the file (not from the
|
||||
* reading config), so a batch written with -M is applied identically by an
|
||||
* invoking process regardless of its own -M setting. The batch driver is a
|
||||
* single sequential scan pass within one thread, so this module-level flag is
|
||||
* safe. */
|
||||
* self-describing about whether per-entry metadata was CAPTURED in the stream:
|
||||
* batch_read_apply re-reads it from the file (not from the reading config) to
|
||||
* decode the chunk records correctly. Which attributes are actually APPLIED,
|
||||
* however, comes from the INVOKING process's per-attribute config (the
|
||||
* FileAttrPolicy and the dir-metadata gate), so a batch written with -M is NOT
|
||||
* automatically applied identically by an invoking process with a different
|
||||
* -p/-t/-o/-g: --read-batch must be invoked with the same -p/-t/-o/-g as the
|
||||
* write side (rsync requires the same options). The batch driver is a single
|
||||
* sequential scan pass within one thread, so this module-level flag is safe. */
|
||||
static bool batch_metadata_mode = false;
|
||||
|
||||
static bool write_all_bytes(int fd, const void* data, size_t size) {
|
||||
@@ -91,22 +96,37 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
if (fd < 0 || dest_root == NULL || dest_root[0] == '\0')
|
||||
return -1;
|
||||
|
||||
/* Directory metadata is deferred to the end of the apply (a child write would
|
||||
* otherwise clobber its parent's mtime/mode). The batch header's single
|
||||
* metadata bit only says whether metadata is present in the stream; which
|
||||
* attributes are APPLIED comes from the invoking process's config, so
|
||||
* --read-batch must be invoked with the same -p/-t/-o/-g as the write side
|
||||
* (rsync requires the same options). The identity snapshot is activated so
|
||||
* -o/-g and the explicit ownership flags can apply. */
|
||||
DirTimeList dir_times;
|
||||
dir_time_list_init(&dir_times);
|
||||
int result = -1;
|
||||
if (!identity_set_active(config)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not activate the identity policy");
|
||||
goto done;
|
||||
}
|
||||
|
||||
char magic[BATCH_MAGIC_LEN];
|
||||
bool eof = false;
|
||||
if (!read_exact(fd, magic, BATCH_MAGIC_LEN, &eof) || eof ||
|
||||
memcmp(magic, BATCH_MAGIC, BATCH_MAGIC_LEN) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad magic)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char version;
|
||||
if (!read_exact(fd, &version, 1, &eof) || eof || version != BATCH_FORMAT_VERSION) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad or missing format version)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
unsigned char mode;
|
||||
if (!read_exact(fd, &mode, 1, &eof) || eof || (mode != 0 && mode != 1)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad metadata flag)");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
bool use_metadata = mode == 1;
|
||||
|
||||
@@ -114,47 +134,62 @@ int batch_read_apply(int fd, const Config* config, const char* dest_root) {
|
||||
unsigned long long length;
|
||||
if (!read_exact(fd, &length, sizeof(length), &eof)) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated length prefix");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (eof)
|
||||
break; /* clean end of stream */
|
||||
if (length == 0 || length > BATCH_MAX_RECORD) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected record length %llu (valid range 1..%llu)",
|
||||
length, (unsigned long long)BATCH_MAX_RECORD);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
char* record = (char*)malloc((size_t)length);
|
||||
if (record == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: could not allocate a %llu-byte record", length);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
if (!read_exact(fd, record, (size_t)length, &eof) || eof) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: truncated chunk record");
|
||||
free(record);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
Data* data = data_create(record, (size_t)length);
|
||||
if (data == NULL)
|
||||
return -1; /* data_create frees `record` on failure */
|
||||
goto done; /* data_create frees `record` on failure */
|
||||
Chunk* chunk = chunk_deserialize(data, use_metadata);
|
||||
data_destroy(data);
|
||||
if (chunk == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "batch: rejected malformed chunk record");
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
File* file = chunk->items[i];
|
||||
chunk->items[i] = NULL;
|
||||
if (file == NULL)
|
||||
continue;
|
||||
FileSaveResult result = file_save_to_disk_full(dest_root, file, config);
|
||||
FileSaveResult save = file_save_to_disk_full(dest_root, file, config);
|
||||
/* Accumulate directory metadata (when it applies) before the File is
|
||||
* destroyed; applied once the whole stream has been consumed. */
|
||||
if (save != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
dir_metadata_should_capture(config) &&
|
||||
!dir_time_list_add(&dir_times, file->path, file->metadata, file->xattrs)) {
|
||||
file_destroy(file);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
chunk_destroy(chunk);
|
||||
return -1;
|
||||
goto done;
|
||||
}
|
||||
file_destroy(file);
|
||||
if (save == FILE_SAVE_ERROR) {
|
||||
chunk_destroy(chunk);
|
||||
goto done;
|
||||
}
|
||||
}
|
||||
chunk_destroy(chunk);
|
||||
}
|
||||
return 0;
|
||||
dir_metadata_list_apply(&dir_times, dest_root, config);
|
||||
result = 0;
|
||||
|
||||
done:
|
||||
identity_clear_active();
|
||||
dir_time_list_free(&dir_times);
|
||||
return result;
|
||||
}
|
||||
+328
-17
@@ -1,12 +1,170 @@
|
||||
#include "checksum.h"
|
||||
#include <fcntl.h>
|
||||
#include <openssl/evp.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* delta.c owns the single XXH_IMPLEMENTATION that provides the xxHash symbols
|
||||
* for the whole binary; this TU only needs the declarations. */
|
||||
* for the whole binary; this TU only needs the declarations. The streaming
|
||||
* state structs and XXH3_update are exposed only with XXH_STATIC_LINKING_ONLY. */
|
||||
#define XXH_STATIC_LINKING_ONLY
|
||||
#include <xxhash.h>
|
||||
|
||||
/* ---------------------------------------------------------------------------
|
||||
* Self-contained MD4 (RFC 1320). OpenSSL's MD4 lives in the legacy provider
|
||||
* and is not guaranteed present, so FastSync carries its own implementation to
|
||||
* keep --checksum-choice=md4 working on every build.
|
||||
* ------------------------------------------------------------------------- */
|
||||
|
||||
typedef struct {
|
||||
uint32_t state[4];
|
||||
uint64_t bit_count;
|
||||
uint8_t buffer[64];
|
||||
size_t buffer_len;
|
||||
} Md4Ctx;
|
||||
|
||||
static uint32_t md4_rotl(uint32_t x, int n) {
|
||||
return (x << n) | (x >> (32 - n));
|
||||
}
|
||||
|
||||
static void md4_transform(uint32_t state[4], const uint8_t block[64]) {
|
||||
uint32_t x[16];
|
||||
for (int i = 0; i < 16; i++)
|
||||
x[i] = (uint32_t)block[i * 4] | ((uint32_t)block[i * 4 + 1] << 8) |
|
||||
((uint32_t)block[i * 4 + 2] << 16) | ((uint32_t)block[i * 4 + 3] << 24);
|
||||
|
||||
uint32_t a = state[0], b = state[1], c = state[2], d = state[3];
|
||||
|
||||
#define F(x, y, z) (((x) & (y)) | (~(x) & (z)))
|
||||
#define G(x, y, z) (((x) & (y)) | ((x) & (z)) | ((y) & (z)))
|
||||
#define H(x, y, z) ((x) ^ (y) ^ (z))
|
||||
#define ROUND1(a, b, c, d, k, s) a = md4_rotl(a + F(b, c, d) + x[k], s)
|
||||
#define ROUND2(a, b, c, d, k, s) a = md4_rotl(a + G(b, c, d) + x[k] + 0x5a827999u, s)
|
||||
#define ROUND3(a, b, c, d, k, s) a = md4_rotl(a + H(b, c, d) + x[k] + 0x6ed9eba1u, s)
|
||||
|
||||
ROUND1(a, b, c, d, 0, 3);
|
||||
ROUND1(d, a, b, c, 1, 7);
|
||||
ROUND1(c, d, a, b, 2, 11);
|
||||
ROUND1(b, c, d, a, 3, 19);
|
||||
ROUND1(a, b, c, d, 4, 3);
|
||||
ROUND1(d, a, b, c, 5, 7);
|
||||
ROUND1(c, d, a, b, 6, 11);
|
||||
ROUND1(b, c, d, a, 7, 19);
|
||||
ROUND1(a, b, c, d, 8, 3);
|
||||
ROUND1(d, a, b, c, 9, 7);
|
||||
ROUND1(c, d, a, b, 10, 11);
|
||||
ROUND1(b, c, d, a, 11, 19);
|
||||
ROUND1(a, b, c, d, 12, 3);
|
||||
ROUND1(d, a, b, c, 13, 7);
|
||||
ROUND1(c, d, a, b, 14, 11);
|
||||
ROUND1(b, c, d, a, 15, 19);
|
||||
|
||||
ROUND2(a, b, c, d, 0, 3);
|
||||
ROUND2(d, a, b, c, 4, 5);
|
||||
ROUND2(c, d, a, b, 8, 9);
|
||||
ROUND2(b, c, d, a, 12, 13);
|
||||
ROUND2(a, b, c, d, 1, 3);
|
||||
ROUND2(d, a, b, c, 5, 5);
|
||||
ROUND2(c, d, a, b, 9, 9);
|
||||
ROUND2(b, c, d, a, 13, 13);
|
||||
ROUND2(a, b, c, d, 2, 3);
|
||||
ROUND2(d, a, b, c, 6, 5);
|
||||
ROUND2(c, d, a, b, 10, 9);
|
||||
ROUND2(b, c, d, a, 14, 13);
|
||||
ROUND2(a, b, c, d, 3, 3);
|
||||
ROUND2(d, a, b, c, 7, 5);
|
||||
ROUND2(c, d, a, b, 11, 9);
|
||||
ROUND2(b, c, d, a, 15, 13);
|
||||
|
||||
ROUND3(a, b, c, d, 0, 3);
|
||||
ROUND3(d, a, b, c, 8, 9);
|
||||
ROUND3(c, d, a, b, 4, 11);
|
||||
ROUND3(b, c, d, a, 12, 15);
|
||||
ROUND3(a, b, c, d, 2, 3);
|
||||
ROUND3(d, a, b, c, 10, 9);
|
||||
ROUND3(c, d, a, b, 6, 11);
|
||||
ROUND3(b, c, d, a, 14, 15);
|
||||
ROUND3(a, b, c, d, 1, 3);
|
||||
ROUND3(d, a, b, c, 9, 9);
|
||||
ROUND3(c, d, a, b, 5, 11);
|
||||
ROUND3(b, c, d, a, 13, 15);
|
||||
ROUND3(a, b, c, d, 3, 3);
|
||||
ROUND3(d, a, b, c, 11, 9);
|
||||
ROUND3(c, d, a, b, 7, 11);
|
||||
ROUND3(b, c, d, a, 15, 15);
|
||||
|
||||
#undef F
|
||||
#undef G
|
||||
#undef H
|
||||
#undef ROUND1
|
||||
#undef ROUND2
|
||||
#undef ROUND3
|
||||
|
||||
state[0] += a;
|
||||
state[1] += b;
|
||||
state[2] += c;
|
||||
state[3] += d;
|
||||
}
|
||||
|
||||
static void md4_init(Md4Ctx* ctx) {
|
||||
ctx->state[0] = 0x67452301u;
|
||||
ctx->state[1] = 0xefcdab89u;
|
||||
ctx->state[2] = 0x98badcfeu;
|
||||
ctx->state[3] = 0x10325476u;
|
||||
ctx->bit_count = 0;
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
|
||||
static void md4_update(Md4Ctx* ctx, const uint8_t* data, size_t len) {
|
||||
ctx->bit_count += (uint64_t)len * 8;
|
||||
while (len > 0) {
|
||||
size_t space = sizeof(ctx->buffer) - ctx->buffer_len;
|
||||
size_t take = len < space ? len : space;
|
||||
memcpy(ctx->buffer + ctx->buffer_len, data, take);
|
||||
ctx->buffer_len += take;
|
||||
data += take;
|
||||
len -= take;
|
||||
if (ctx->buffer_len == sizeof(ctx->buffer)) {
|
||||
md4_transform(ctx->state, ctx->buffer);
|
||||
ctx->buffer_len = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
static void md4_final(Md4Ctx* ctx, uint8_t out[16]) {
|
||||
uint64_t bit_count = ctx->bit_count;
|
||||
uint8_t pad = 0x80;
|
||||
md4_update(ctx, &pad, 1);
|
||||
uint8_t zero = 0;
|
||||
while (ctx->buffer_len != 56)
|
||||
md4_update(ctx, &zero, 1);
|
||||
uint8_t length_le[8];
|
||||
for (int i = 0; i < 8; i++)
|
||||
length_le[i] = (uint8_t)((bit_count >> (8 * i)) & 0xff);
|
||||
md4_update(ctx, length_le, sizeof(length_le));
|
||||
for (int i = 0; i < 4; i++) {
|
||||
out[i * 4] = (uint8_t)(ctx->state[i] & 0xff);
|
||||
out[i * 4 + 1] = (uint8_t)((ctx->state[i] >> 8) & 0xff);
|
||||
out[i * 4 + 2] = (uint8_t)((ctx->state[i] >> 16) & 0xff);
|
||||
out[i * 4 + 3] = (uint8_t)((ctx->state[i] >> 24) & 0xff);
|
||||
}
|
||||
}
|
||||
|
||||
/* One-shot EVP digest (md5/sha1). Returns false when OpenSSL refuses. */
|
||||
static bool evp_digest(const EVP_MD* md, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
static const uint8_t empty = 0;
|
||||
const void* input = data ? data : ∅
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_Digest(input, size, out, &digest_len, md, NULL) != 1)
|
||||
return false;
|
||||
if (digest_len > out_capacity)
|
||||
return false;
|
||||
*out_len = digest_len;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
@@ -14,39 +172,158 @@ bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t
|
||||
if (data == NULL && size != 0)
|
||||
return false;
|
||||
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64: {
|
||||
uint64_t digest = XXH64(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5) {
|
||||
case CHECKSUM_ALGO_XXH3: {
|
||||
uint64_t digest = XXH3_64bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_XXH128: {
|
||||
XXH128_hash_t digest = XXH3_128bits_withSeed(data, size, seed);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
/* md5 takes no seed; the caller's seed is deliberately ignored (documented
|
||||
* in RSYNC_COMPAT.md). OpenSSL's one-shot EVP_Digest needs a non-NULL
|
||||
* buffer even for an empty input, so map a NULL data + size==0 to an empty
|
||||
* buffer. */
|
||||
static const uint8_t empty = 0;
|
||||
const void* input = data ? data : ∅
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_Digest(input, size, out, &digest_len, EVP_md5(), NULL) != 1)
|
||||
return false;
|
||||
if (digest_len > out_capacity)
|
||||
return false;
|
||||
*out_len = digest_len;
|
||||
* in RSYNC_COMPAT.md). */
|
||||
return evp_digest(EVP_md5(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_MD4: {
|
||||
Md4Ctx ctx;
|
||||
md4_init(&ctx);
|
||||
md4_update(&ctx, (const uint8_t*)data, size);
|
||||
md4_final(&ctx, out);
|
||||
*out_len = 16;
|
||||
return true;
|
||||
}
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
/* sha1 takes no seed; the caller's seed is deliberately ignored. */
|
||||
return evp_digest(EVP_sha1(), data, size, out, out_capacity, out_len);
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
/* No checksum requested: an empty digest is the successful result. */
|
||||
*out_len = 0;
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len) {
|
||||
if (!path || !out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
|
||||
return false;
|
||||
|
||||
int fd = open(path, O_RDONLY | O_CLOEXEC);
|
||||
if (fd < 0)
|
||||
return false;
|
||||
|
||||
uint8_t buffer[64 * 1024];
|
||||
bool ok = false;
|
||||
|
||||
if (algo == CHECKSUM_ALGO_MD5) {
|
||||
EVP_MD_CTX* ctx = EVP_MD_CTX_new();
|
||||
if (!ctx) {
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
unsigned int digest_len = 0;
|
||||
if (EVP_DigestInit_ex(ctx, EVP_md5(), NULL) == 1) {
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (EVP_DigestUpdate(ctx, buffer, (size_t)got) != 1) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
if (ok && EVP_DigestFinal_ex(ctx, out, &digest_len) == 1 && digest_len <= out_capacity)
|
||||
*out_len = digest_len;
|
||||
else
|
||||
ok = false;
|
||||
}
|
||||
EVP_MD_CTX_free(ctx);
|
||||
close(fd);
|
||||
return ok;
|
||||
}
|
||||
|
||||
XXH64_state_t xxh64;
|
||||
XXH3_state_t* xxh3 = NULL;
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
XXH64_reset(&xxh64, seed);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3 || algo == CHECKSUM_ALGO_XXH128) {
|
||||
xxh3 = XXH3_createState();
|
||||
if (!xxh3) {
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
if (algo == CHECKSUM_ALGO_XXH3)
|
||||
XXH3_64bits_reset_withSeed(xxh3, seed);
|
||||
else
|
||||
XXH3_128bits_reset_withSeed(xxh3, seed);
|
||||
} else {
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
|
||||
ok = true;
|
||||
ssize_t got;
|
||||
while ((got = read(fd, buffer, sizeof(buffer))) > 0) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64)
|
||||
XXH64_update(&xxh64, buffer, (size_t)got);
|
||||
else if (XXH3_64bits_update(xxh3, buffer, (size_t)got) == XXH_ERROR) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (got < 0)
|
||||
ok = false;
|
||||
|
||||
if (ok) {
|
||||
if (algo == CHECKSUM_ALGO_XXH64) {
|
||||
uint64_t digest = XXH64_digest(&xxh64);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else if (algo == CHECKSUM_ALGO_XXH3) {
|
||||
uint64_t digest = XXH3_64bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
} else {
|
||||
XXH128_hash_t digest = XXH3_128bits_digest(xxh3);
|
||||
memcpy(out, &digest, sizeof(digest));
|
||||
*out_len = sizeof(digest);
|
||||
}
|
||||
}
|
||||
if (xxh3)
|
||||
XXH3_freeState(xxh3);
|
||||
close(fd);
|
||||
return ok;
|
||||
}
|
||||
|
||||
int checksum_algo_from_name(const char* name) {
|
||||
if (!name)
|
||||
return -1;
|
||||
if (strcasecmp(name, "xxh64") == 0 || strcasecmp(name, "xxhash") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH64;
|
||||
if (strcasecmp(name, "xxh3") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH3;
|
||||
if (strcasecmp(name, "xxh128") == 0)
|
||||
return (int)CHECKSUM_ALGO_XXH128;
|
||||
if (strcasecmp(name, "md5") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD5;
|
||||
if (strcasecmp(name, "md4") == 0)
|
||||
return (int)CHECKSUM_ALGO_MD4;
|
||||
if (strcasecmp(name, "sha1") == 0)
|
||||
return (int)CHECKSUM_ALGO_SHA1;
|
||||
if (strcasecmp(name, "none") == 0)
|
||||
return (int)CHECKSUM_ALGO_NONE;
|
||||
return -1;
|
||||
}
|
||||
|
||||
@@ -54,22 +331,56 @@ const char* checksum_algo_name(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
return "xxh64";
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return "xxh3";
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
return "xxh128";
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
return "md5";
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return "md4";
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return "sha1";
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return "none";
|
||||
}
|
||||
return "<unknown>";
|
||||
}
|
||||
|
||||
bool checksum_algo_valid(int algo) {
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5;
|
||||
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5 ||
|
||||
algo == (int)CHECKSUM_ALGO_XXH3 || algo == (int)CHECKSUM_ALGO_XXH128 ||
|
||||
algo == (int)CHECKSUM_ALGO_MD4 || algo == (int)CHECKSUM_ALGO_SHA1 ||
|
||||
algo == (int)CHECKSUM_ALGO_NONE;
|
||||
}
|
||||
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo) {
|
||||
switch (algo) {
|
||||
case CHECKSUM_ALGO_XXH64:
|
||||
case CHECKSUM_ALGO_XXH3:
|
||||
return 8;
|
||||
case CHECKSUM_ALGO_XXH128:
|
||||
case CHECKSUM_ALGO_MD5:
|
||||
case CHECKSUM_ALGO_MD4:
|
||||
return 16;
|
||||
case CHECKSUM_ALGO_SHA1:
|
||||
return 20;
|
||||
case CHECKSUM_ALGO_NONE:
|
||||
return 0;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
ChecksumAlgo checksum_negotiate_default(void) {
|
||||
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
|
||||
* resolves to xxh128. */
|
||||
static const ChecksumAlgo preference[] = {
|
||||
CHECKSUM_ALGO_XXH128, CHECKSUM_ALGO_XXH3, CHECKSUM_ALGO_XXH64, CHECKSUM_ALGO_MD5,
|
||||
CHECKSUM_ALGO_MD4, CHECKSUM_ALGO_SHA1, CHECKSUM_ALGO_NONE,
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(preference) / sizeof(preference[0]); i++) {
|
||||
if (checksum_algo_valid((int)preference[i]))
|
||||
return preference[i];
|
||||
}
|
||||
return CHECKSUM_ALGO_XXH64;
|
||||
}
|
||||
|
||||
+40
-10
@@ -8,18 +8,35 @@
|
||||
/* Whole-file content-digest algorithms selectable with --checksum-choice and
|
||||
* seeded with --checksum-seed. The ids are the values actually placed on the
|
||||
* wire (config frame), so they must be kept stable and validated on receive.
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the default and is byte-for-byte what FastSync
|
||||
* computed before these options existed (xxHash64 with seed 0). */
|
||||
typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
|
||||
* CHECKSUM_ALGO_XXH64 == 0 is the historical FastSync default and its numeric
|
||||
* value is preserved. The full set mirrors the algorithms rsync 3.4.1 can be
|
||||
* built with; every one of them is implemented here. */
|
||||
typedef enum {
|
||||
CHECKSUM_ALGO_XXH64 = 0,
|
||||
CHECKSUM_ALGO_MD5 = 1,
|
||||
CHECKSUM_ALGO_XXH3 = 2,
|
||||
CHECKSUM_ALGO_XXH128 = 3,
|
||||
CHECKSUM_ALGO_MD4 = 4,
|
||||
CHECKSUM_ALGO_SHA1 = 5,
|
||||
CHECKSUM_ALGO_NONE = 6
|
||||
} ChecksumAlgo;
|
||||
|
||||
/* md5 digest is 16 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 16
|
||||
/* FastSync's negotiated default (rsync 3.4.1 auto-negotiates xxh128 first).
|
||||
* The wire default for Config->checksum_algo is this value. */
|
||||
#define CHECKSUM_ALGO_DEFAULT CHECKSUM_ALGO_XXH128
|
||||
|
||||
/* sha1 digest is 20 bytes, the longest supported. */
|
||||
#define CHECKSUM_MAX_DIGEST_LEN 20
|
||||
|
||||
/* Compute the whole-file digest of the first `size` bytes of `data`.
|
||||
*
|
||||
* - CHECKSUM_ALGO_XXH64: xxHash64(data, size, seed) (full 64-bit seed).
|
||||
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP.
|
||||
* md5 has no seed, so `seed` is ignored (documented).
|
||||
* - CHECKSUM_ALGO_XXH3: XXH3_64bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_XXH128: XXH3_128bits_withSeed(data, size, seed).
|
||||
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_MD4: md4(data, size), self-contained RFC 1320 (seed ignored).
|
||||
* - CHECKSUM_ALGO_SHA1: sha1(data, size) via OpenSSL EVP (seed ignored).
|
||||
* - CHECKSUM_ALGO_NONE: no digest; *out_len is 0 and nothing is written.
|
||||
* - `size == 0` hashes the empty input (plus its seed), not a NULL input.
|
||||
*
|
||||
* Writes up to `out_capacity` bytes into `out`, storing the digest length in
|
||||
@@ -28,9 +45,16 @@ typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
|
||||
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Streaming whole-file digest: hash the contents of `path` without holding the
|
||||
* whole file in memory. Same digest/capacity contract as checksum_digest.
|
||||
* Returns false on open/read failure or an undersized buffer. */
|
||||
bool checksum_digest_file(ChecksumAlgo algo, uint64_t seed, const char* path, uint8_t* out,
|
||||
size_t out_capacity, size_t* out_len);
|
||||
|
||||
/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "xxh64" and "xxhash" (both map to CHECKSUM_ALGO_XXH64, rsync's
|
||||
* xxhash spelling) and "md5". Returns -1 for any unsupported name. */
|
||||
* Accepts "xxh64"/"xxhash", "xxh3", "xxh128", "md5", "md4", "sha1", "none".
|
||||
* "auto" is not an algorithm here; the caller resolves it to the negotiated
|
||||
* default. Returns -1 for any unrecognized name. */
|
||||
int checksum_algo_from_name(const char* name);
|
||||
|
||||
/* Canonical name of an algorithm (used in CLI error messages). */
|
||||
@@ -39,7 +63,13 @@ const char* checksum_algo_name(ChecksumAlgo algo);
|
||||
/* True when `algo` is a supported id (used by config receive validation). */
|
||||
bool checksum_algo_valid(int algo);
|
||||
|
||||
/* Digest length in bytes for an algorithm (xxx64 = 8, md5 = 16). */
|
||||
/* Digest length in bytes for an algorithm (xxh64/xxh3 = 8,
|
||||
* md5/md4/xxh128 = 16, sha1 = 20, none = 0). */
|
||||
uint8_t checksum_digest_len(ChecksumAlgo algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list that is
|
||||
* supported on this build (rsync 3.4.1's `--version` order:
|
||||
* xxh128 xxh3 xxh64 md5 md4 sha1 none). Used to resolve "auto". */
|
||||
ChecksumAlgo checksum_negotiate_default(void);
|
||||
|
||||
#endif /* CHECKSUM_H */
|
||||
+170
-77
@@ -1,90 +1,183 @@
|
||||
#include "chmod.h"
|
||||
#include "file.h"
|
||||
#include <stddef.h>
|
||||
#include <string.h>
|
||||
|
||||
static bool parse_clause(mode_t* mode, const char* begin, const char* end) {
|
||||
const char* p = begin;
|
||||
unsigned who = 0;
|
||||
while (p < end && strchr("ugoa", *p)) {
|
||||
if (*p == 'a')
|
||||
who = 7;
|
||||
else
|
||||
who |= *p == 'u' ? 1U : (*p == 'g' ? 2U : 4U);
|
||||
p++;
|
||||
}
|
||||
if (who == 0)
|
||||
who = 7;
|
||||
if (p == end || (*p != '+' && *p != '-' && *p != '='))
|
||||
return false;
|
||||
char operation = *p++;
|
||||
mode_t bits = 0;
|
||||
while (p < end) {
|
||||
mode_t bit;
|
||||
switch (*p++) {
|
||||
case 'r':
|
||||
bit = 4;
|
||||
break;
|
||||
case 'w':
|
||||
bit = 2;
|
||||
break;
|
||||
case 'x':
|
||||
bit = 1;
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
bits |= bit;
|
||||
}
|
||||
for (unsigned class_index = 0; class_index < 3; class_index++) {
|
||||
unsigned class_bit = 1U << class_index;
|
||||
if (!(who & class_bit))
|
||||
continue;
|
||||
mode_t shift = (mode_t)((2U - class_index) * 3U);
|
||||
mode_t mask = (mode_t)(7U << shift);
|
||||
mode_t class_bits = (mode_t)(bits << shift);
|
||||
if (operation == '+')
|
||||
*mode |= class_bits;
|
||||
else if (operation == '-')
|
||||
*mode &= ~class_bits;
|
||||
else
|
||||
*mode = (*mode & ~mask) | class_bits;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
/* rsync's --chmod parser (parse_chmod + tweak_mode). A single clause is
|
||||
* applied as it is completed, so repeated clauses and repeated --chmod options
|
||||
* (joined with commas by the CLI) accumulate exactly like rsync. The D/F
|
||||
* selectors restrict a clause to directories/files; X adds execute only to
|
||||
* directories or files that were already executable. */
|
||||
|
||||
#define CHMOD_BITS 07777
|
||||
#define CHMOD_FLAG_X_KEEP (1U << 0)
|
||||
#define CHMOD_FLAG_DIRS_ONLY (1U << 1)
|
||||
#define CHMOD_FLAG_FILES_ONLY (1U << 2)
|
||||
|
||||
enum chmod_op { CHMOD_OP_ADD = 1, CHMOD_OP_SUB, CHMOD_OP_EQ, CHMOD_OP_SET };
|
||||
enum chmod_state {
|
||||
CHMOD_STATE_ERROR,
|
||||
CHMOD_STATE_1ST_HALF,
|
||||
CHMOD_STATE_2ND_HALF,
|
||||
CHMOD_STATE_OCTAL
|
||||
};
|
||||
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result) {
|
||||
if (!spec || !*spec || !result)
|
||||
return false;
|
||||
bool numeric = true;
|
||||
size_t length = strlen(spec);
|
||||
if (length > 4)
|
||||
numeric = false;
|
||||
for (size_t i = 0; i < length && numeric; i++)
|
||||
numeric = spec[i] >= '0' && spec[i] <= '7';
|
||||
if (numeric) {
|
||||
if (length == 0 || length > 4)
|
||||
return false;
|
||||
mode_t parsed = 0;
|
||||
for (size_t i = 0; i < length; i++)
|
||||
parsed = (mode_t)((parsed << 3) | (spec[i] - '0'));
|
||||
*result = parsed;
|
||||
return true;
|
||||
}
|
||||
|
||||
const mode_t nonperm = mode & ~(mode_t)CHMOD_BITS;
|
||||
const bool initially_executable = (mode & 0111) != 0;
|
||||
mode_t changed = mode;
|
||||
const char* begin = spec;
|
||||
while (*begin) {
|
||||
const char* end = strchr(begin, ',');
|
||||
if (!end)
|
||||
end = begin + strlen(begin);
|
||||
if (!parse_clause(&changed, begin, end))
|
||||
return false;
|
||||
if (*end == '\0')
|
||||
int state = CHMOD_STATE_1ST_HALF;
|
||||
unsigned where = 0;
|
||||
int what = 0, op = 0, topbits = 0, topoct = 0, flags = 0;
|
||||
const char* p = spec;
|
||||
while (state != CHMOD_STATE_ERROR) {
|
||||
if (*p == '\0' || *p == ',') {
|
||||
int bits;
|
||||
if (!op) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
begin = end + 1;
|
||||
if (!*begin)
|
||||
return false;
|
||||
}
|
||||
*result = changed;
|
||||
if (where)
|
||||
bits = (int)(where * (unsigned)what);
|
||||
else {
|
||||
where = 0111;
|
||||
bits = (int)((where * (unsigned)what) & ~(unsigned)file_process_umask());
|
||||
}
|
||||
int mode_and, mode_or;
|
||||
switch (op) {
|
||||
case CHMOD_OP_ADD:
|
||||
mode_and = CHMOD_BITS;
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
case CHMOD_OP_SUB:
|
||||
mode_and = CHMOD_BITS - bits - topoct;
|
||||
mode_or = 0;
|
||||
break;
|
||||
case CHMOD_OP_EQ:
|
||||
mode_and = CHMOD_BITS - (int)(where * 7U) - (topoct ? topbits : 0);
|
||||
mode_or = bits + topoct;
|
||||
break;
|
||||
default:
|
||||
mode_and = 0;
|
||||
mode_or = bits;
|
||||
break;
|
||||
}
|
||||
bool is_dir = S_ISDIR(nonperm);
|
||||
if (!((flags & CHMOD_FLAG_DIRS_ONLY) && !is_dir) &&
|
||||
!((flags & CHMOD_FLAG_FILES_ONLY) && is_dir)) {
|
||||
changed &= (mode_t)mode_and;
|
||||
if ((flags & CHMOD_FLAG_X_KEEP) && !initially_executable && !is_dir)
|
||||
changed |= (mode_t)(mode_or & ~0111);
|
||||
else
|
||||
changed |= (mode_t)mode_or;
|
||||
}
|
||||
if (*p == '\0')
|
||||
break;
|
||||
p++;
|
||||
state = CHMOD_STATE_1ST_HALF;
|
||||
where = 0;
|
||||
what = op = topoct = topbits = flags = 0;
|
||||
continue;
|
||||
}
|
||||
switch (state) {
|
||||
case CHMOD_STATE_1ST_HALF:
|
||||
switch (*p) {
|
||||
case 'D':
|
||||
if (flags & CHMOD_FLAG_FILES_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_DIRS_ONLY;
|
||||
break;
|
||||
case 'F':
|
||||
if (flags & CHMOD_FLAG_DIRS_ONLY) {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
flags |= CHMOD_FLAG_FILES_ONLY;
|
||||
break;
|
||||
case 'u':
|
||||
where |= 0100;
|
||||
topbits |= 04000;
|
||||
break;
|
||||
case 'g':
|
||||
where |= 0010;
|
||||
topbits |= 02000;
|
||||
break;
|
||||
case 'o':
|
||||
where |= 0001;
|
||||
break;
|
||||
case 'a':
|
||||
where |= 0111;
|
||||
break;
|
||||
case '+':
|
||||
op = CHMOD_OP_ADD;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '-':
|
||||
op = CHMOD_OP_SUB;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
case '=':
|
||||
op = CHMOD_OP_EQ;
|
||||
state = CHMOD_STATE_2ND_HALF;
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7' && !where) {
|
||||
op = CHMOD_OP_SET;
|
||||
state = CHMOD_STATE_OCTAL;
|
||||
where = 1;
|
||||
what = *p - '0';
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
case CHMOD_STATE_2ND_HALF:
|
||||
switch (*p) {
|
||||
case 'r':
|
||||
what |= 4;
|
||||
break;
|
||||
case 'w':
|
||||
what |= 2;
|
||||
break;
|
||||
case 'X':
|
||||
flags |= CHMOD_FLAG_X_KEEP;
|
||||
/* fall through */
|
||||
case 'x':
|
||||
what |= 1;
|
||||
break;
|
||||
case 's':
|
||||
if (topbits)
|
||||
topoct |= topbits;
|
||||
else
|
||||
topoct = 04000;
|
||||
break;
|
||||
case 't':
|
||||
topoct |= 01000;
|
||||
break;
|
||||
default:
|
||||
state = CHMOD_STATE_ERROR;
|
||||
break;
|
||||
}
|
||||
break;
|
||||
default:
|
||||
if (*p >= '0' && *p <= '7') {
|
||||
what = what * 8 + (*p - '0');
|
||||
if (what > CHMOD_BITS)
|
||||
state = CHMOD_STATE_ERROR;
|
||||
} else {
|
||||
state = CHMOD_STATE_ERROR;
|
||||
}
|
||||
break;
|
||||
}
|
||||
p++;
|
||||
}
|
||||
if (state == CHMOD_STATE_ERROR)
|
||||
return false;
|
||||
*result = (changed & (mode_t)CHMOD_BITS) | nonperm;
|
||||
return true;
|
||||
}
|
||||
|
||||
+4
-1
@@ -4,7 +4,10 @@
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/* Apply the supported rsync --chmod syntax to a permission mode. */
|
||||
/* Apply rsync's --chmod syntax to a permission mode, including the D/F/X
|
||||
* selectors and the s/t special bits. `mode` should carry the file type bits
|
||||
* (S_IFDIR/S_IFREG) so D/F/X can be evaluated; the type bits are preserved in
|
||||
* `result`. A spec may contain comma-separated clauses, which accumulate. */
|
||||
bool chmod_apply(mode_t mode, const char* spec, mode_t* result);
|
||||
|
||||
#endif
|
||||
|
||||
+100
-96
@@ -1,6 +1,7 @@
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <limits.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -20,6 +21,33 @@
|
||||
#define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024)
|
||||
#define MAX_FILES_PER_CHUNK 65536U
|
||||
|
||||
/* Reserve `charge` against `session`'s connection budget. This mirrors the
|
||||
static protocol_reserve_memory() in protocol.c: the receive-side call sites
|
||||
only have the Data.owner pointer (a ProtocolSession*), and protocol.c is out
|
||||
of scope for this fix, so the same atomic CAS accounting is reproduced here.
|
||||
The matching release always goes through data_destroy()'s Data.owner path. */
|
||||
static bool chunk_session_reserve(ProtocolSession* session, size_t charge) {
|
||||
unsigned long long allocated = atomic_load(&session->total_allocated_bytes);
|
||||
while (true) {
|
||||
if (allocated > MAX_CONNECTION_MEMORY ||
|
||||
(unsigned long long)charge > MAX_CONNECTION_MEMORY - allocated)
|
||||
return false;
|
||||
if (atomic_compare_exchange_weak(&session->total_allocated_bytes, &allocated,
|
||||
allocated + (unsigned long long)charge))
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge) {
|
||||
if (!data || charge == 0 || session == NULL)
|
||||
return true;
|
||||
if (!chunk_session_reserve(session, charge))
|
||||
return false;
|
||||
data->owner = session;
|
||||
data->protocol_charge = charge;
|
||||
return true;
|
||||
}
|
||||
|
||||
Chunk* chunk_create(File** items, int element_count) {
|
||||
if (element_count < 0 || (element_count > 0 && items == NULL))
|
||||
return NULL;
|
||||
@@ -207,17 +235,20 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
return NULL;
|
||||
char* data_pointer = data->data;
|
||||
size_t remaining_size = data->size;
|
||||
/* The element currently being parsed is owned by `files` only after the
|
||||
* array_list_add() at the end of the iteration; until then the error
|
||||
* epilogue destroys it directly. Keeping this one pointer nulled after the
|
||||
* hand-off makes the single cleanup path correct for every failure. */
|
||||
File* file = NULL;
|
||||
|
||||
while (remaining_size > 0) {
|
||||
if ((unsigned int)files->size >= MAX_FILES_PER_CHUNK) {
|
||||
log_message(LOG_LEVEL_ERROR, "Chunk contains too many files");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path length");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
size_t path_len;
|
||||
@@ -227,26 +258,19 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (path_len > SIZE_MAX - 1 || remaining_size < path_len) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
if (path_len == SIZE_MAX) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
char* path = protocol_alloc(path_len + 1);
|
||||
if (path == NULL) {
|
||||
log_perror("Could not allocate memory for file path");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
memcpy(path, data_pointer, path_len);
|
||||
path[path_len] = '\0';
|
||||
if (memchr(path, '\0', path_len) != NULL) {
|
||||
free(path);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
data_pointer += path_len;
|
||||
remaining_size -= path_len;
|
||||
@@ -260,8 +284,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (local_path == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv: received chunk file name cannot be converted to the local charset");
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
path = local_path;
|
||||
path_len = strlen(path);
|
||||
@@ -269,30 +292,23 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (path_len == 0 || has_path_traversal(path)) {
|
||||
free(path);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
File* file = file_create(path);
|
||||
file = file_create(path);
|
||||
free(path);
|
||||
if (file == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (file == NULL)
|
||||
goto error;
|
||||
|
||||
if (remaining_size < sizeof(int)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for entry type");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
int entry_type;
|
||||
memcpy(&entry_type, data_pointer, sizeof(int));
|
||||
if (entry_type != 0 && entry_type != 1 && entry_type != 2 && entry_type != 3) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad entry type");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
file->is_dir = entry_type == 1;
|
||||
file->is_symlink = entry_type == 2;
|
||||
@@ -303,9 +319,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (file->is_special) {
|
||||
if (remaining_size < 2 * (int32_t)sizeof(int32_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for special rdev");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
int32_t special_major, special_minor;
|
||||
memcpy(&special_major, data_pointer, sizeof(special_major));
|
||||
@@ -320,9 +334,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (special_major < 0 || special_minor < 0 || special_major > 0xffff ||
|
||||
special_minor > 0x00ffffff) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: out-of-range special rdev");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
file->rdev_major = special_major;
|
||||
file->rdev_minor = special_minor;
|
||||
@@ -331,37 +343,32 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (use_metadata) {
|
||||
if (remaining_size < sizeof(int)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
// Peek at present flag to determine total size needed before reading
|
||||
/* Peek at the present flag to determine the total record size before
|
||||
decoding. metadata_from_buf() independently bounds-checks every read
|
||||
against remaining_size, so a short body can never over-read. */
|
||||
int present_flag;
|
||||
memcpy(&present_flag, data_pointer, sizeof(int));
|
||||
if ((present_flag != 0 && present_flag != 1) ||
|
||||
(present_flag == 1 && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata body");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
file->metadata = metadata_from_buf(&data_pointer);
|
||||
remaining_size -= sizeof(int);
|
||||
file->metadata = metadata_from_buf((const uint8_t*)data_pointer, remaining_size);
|
||||
size_t metadata_consumed = sizeof(int);
|
||||
if (present_flag == 1) {
|
||||
if (file->metadata == NULL) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
remaining_size -= FILE_METADATA_WIRE_SIZE;
|
||||
if (file->metadata == NULL)
|
||||
goto error;
|
||||
metadata_consumed += FILE_METADATA_WIRE_SIZE;
|
||||
}
|
||||
data_pointer += metadata_consumed;
|
||||
remaining_size -= metadata_consumed;
|
||||
}
|
||||
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for data size");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
size_t file_data_size;
|
||||
@@ -371,34 +378,34 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
|
||||
if (remaining_size < file_data_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for file content");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
// Reject individual file data larger than the maximum allowed size.
|
||||
if (file_data_size > MAX_FILE_DATA_SIZE) {
|
||||
log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size,
|
||||
(unsigned long long)MAX_FILE_DATA_SIZE);
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
|
||||
size_t allocation_size = file_data_size > 0 ? file_data_size : 1;
|
||||
void* file_data = protocol_alloc(allocation_size);
|
||||
if (file_data == NULL) {
|
||||
log_perror("Could not allocate memory for file data");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
memcpy(file_data, data_pointer, file_data_size);
|
||||
Data* replacement = data_create(file_data, file_data_size);
|
||||
if (replacement == NULL) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
if (replacement == NULL)
|
||||
goto error;
|
||||
/* Charge the retained per-file copy to the connection budget (when the
|
||||
inbound chunk carries an owning session) so the queued copies are not
|
||||
held outside MAX_CONNECTION_MEMORY (B6). A NULL owner (e.g. a local
|
||||
batch apply) leaves the copy uncharged. */
|
||||
if (!data_charge_session(replacement, data->owner, allocation_size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for chunk file data");
|
||||
data_destroy(replacement);
|
||||
goto error;
|
||||
}
|
||||
data_destroy(file->data);
|
||||
file->data = replacement;
|
||||
@@ -408,9 +415,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
if (file->is_symlink) {
|
||||
if (remaining_size < sizeof(size_t)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for symlink target");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
size_t target_len;
|
||||
memcpy(&target_len, data_pointer, sizeof(size_t));
|
||||
@@ -418,24 +423,18 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
remaining_size -= sizeof(size_t);
|
||||
if (target_len == 0 || remaining_size < target_len) {
|
||||
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad symlink target");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
char* target = protocol_alloc(target_len + 1);
|
||||
if (!target) {
|
||||
log_perror("Could not allocate memory for symlink target");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
memcpy(target, data_pointer, target_len);
|
||||
target[target_len] = '\0';
|
||||
if (memchr(target, '\0', target_len) != NULL) {
|
||||
free(target);
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
/* The symlink target also rides the wire charset; decode it to the local
|
||||
charset like the path (a target is a path). */
|
||||
@@ -446,9 +445,7 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--iconv: received chunk symlink target cannot be converted to the local "
|
||||
"charset");
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
goto error;
|
||||
}
|
||||
target = local_target;
|
||||
}
|
||||
@@ -457,29 +454,27 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
|
||||
remaining_size -= target_len;
|
||||
}
|
||||
|
||||
if (!array_list_add(files, file)) {
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (!array_list_add(files, file))
|
||||
goto error;
|
||||
file = NULL;
|
||||
}
|
||||
|
||||
File** file_array = (File**)array_list_to_array(files);
|
||||
if (files->size > 0 && file_array == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (files->size > 0 && file_array == NULL)
|
||||
goto error;
|
||||
Chunk* chunk = chunk_create(file_array, files->size);
|
||||
|
||||
free(file_array);
|
||||
if (chunk == NULL) {
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
if (chunk == NULL)
|
||||
goto error;
|
||||
files->item_destroyer = NULL;
|
||||
array_list_delete(files);
|
||||
|
||||
return chunk;
|
||||
|
||||
error:
|
||||
if (file)
|
||||
file_destroy(file);
|
||||
array_list_delete(files);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) {
|
||||
@@ -508,12 +503,21 @@ Chunk* receive_chunk_data(int fd, const Config* config) {
|
||||
}
|
||||
Data* data_to_process = chunk_data;
|
||||
if (config->use_compression) {
|
||||
/* Preserve the inbound session across decompression so the (larger)
|
||||
decompressed chunk is charged to the same connection budget; the
|
||||
compressed buffer's own charge is released by data_destroy below. */
|
||||
ProtocolSession* owner = chunk_data->owner;
|
||||
data_to_process = data_decompress_limited(chunk_data, MAX_CHUNK_SIZE);
|
||||
data_destroy(chunk_data);
|
||||
if (data_to_process == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to decompress chunk");
|
||||
return NULL;
|
||||
}
|
||||
if (!data_charge_session(data_to_process, owner, data_to_process->size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for decompressed chunk");
|
||||
data_destroy(data_to_process);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
// Reject chunks larger than the maximum allowed size to prevent OOM.
|
||||
|
||||
@@ -23,4 +23,15 @@ Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_
|
||||
int compression_threads);
|
||||
Chunk* receive_chunk_data(int fd, const Config* config);
|
||||
|
||||
/* Charge `charge` retained bytes of `data` against `session`'s per-connection
|
||||
* budget (MAX_CONNECTION_MEMORY), mirroring the protocol layer's accounting, and
|
||||
* record them on `data` so data_destroy() returns the charge through the
|
||||
* Data.owner path. Returns false (leaving `data` uncharged) when the ceiling
|
||||
* would be exceeded. A NULL/zero-size charge or a NULL session is a no-op
|
||||
* success. The receive-side decompression and chunk-copy paths know the owning
|
||||
* session only through the Data.owner of the buffer they are processing, so
|
||||
* this is the entry point that lets them participate in the connection budget
|
||||
* without a session handle (B6). */
|
||||
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
|
||||
+555
-103
@@ -2,138 +2,539 @@
|
||||
#include "data.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include <stdlib.h>
|
||||
#include <limits.h>
|
||||
#include <lz4.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <strings.h>
|
||||
#include <threads.h>
|
||||
#include <unistd.h>
|
||||
#include <zlib.h>
|
||||
#include <zstd.h>
|
||||
|
||||
#define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024)
|
||||
#define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */
|
||||
|
||||
static char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv",
|
||||
".zip", ".gz", ".xz", ".zst", NULL};
|
||||
/* rsync 3.4.1's built-in skip-compress suffix list (the `--skip-compress`
|
||||
* defaults, in the man page's order). rsync stores it as space-separated
|
||||
* "*.suffix" globs; FastSync matches the plain suffix after the final dot, so
|
||||
* the leading "*." is omitted here. A user --skip-compress list replaces this
|
||||
* default entirely (matching rsync). */
|
||||
#define DEFAULT_SKIP_COMPRESS_SUFFIXES \
|
||||
"3g2 3gp 7z aac ace apk avi bz2 deb dmg ear f4v flac flv gpg gz iso jar jpeg jpg lrz lz lz4 " \
|
||||
"lzma " \
|
||||
"lzo m1a m1v m2a m2ts m2v m4a m4b m4p m4r m4v mka mkv mov mp1 mp2 mp3 mp4 mpa mpeg mpg mpv mts " \
|
||||
"odb odf odg odi odm odp ods odt oga ogg ogm ogv ogx opus otg oth otp ots ott oxt png qt rar " \
|
||||
"rpm " \
|
||||
"rz rzip spx squashfs sxc sxd sxg sxm sxw sz tbz tbz2 tgz tlz ts txz tzo vob war webm webp xz " \
|
||||
"z " \
|
||||
"zip zst"
|
||||
|
||||
bool compression_should_skip(const char* path) {
|
||||
return compression_should_skip_with_suffixes(path, NULL, -1);
|
||||
/* Self-describing compressed frames: the first byte is the CompressionAlgo id.
|
||||
* zlib/lz4 store the uncompressed size as a little-endian uint32 after the
|
||||
* codec byte so decompression can be exactly pre-sized and bounded. */
|
||||
#define LZ4_SIZE_PREFIX_LEN 4
|
||||
|
||||
static _Atomic int g_compression_algo = COMPRESSION_ALGO_ZSTD;
|
||||
|
||||
/* Case-insensitive match of a bare suffix (no leading dot) against a
|
||||
* space-separated suffix list. */
|
||||
static bool suffix_in_list(const char* name, const char* list) {
|
||||
size_t name_len = strlen(name);
|
||||
while (*list) {
|
||||
while (*list == ' ')
|
||||
list++;
|
||||
const char* start = list;
|
||||
while (*list && *list != ' ')
|
||||
list++;
|
||||
size_t len = (size_t)(list - start);
|
||||
if (len == name_len && strncasecmp(name, start, len) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count) {
|
||||
if (!path)
|
||||
return false;
|
||||
const char* dot = strrchr(path, '.');
|
||||
if (!dot)
|
||||
if (!dot || dot[1] == '\0')
|
||||
return false;
|
||||
if (count < 0) {
|
||||
suffixes = SKIP_COMPRESSION_EXTENSIONS;
|
||||
count = 0;
|
||||
while (SKIP_COMPRESSION_EXTENSIONS[count])
|
||||
count++;
|
||||
}
|
||||
const char* name = dot + 1;
|
||||
/* count < 0 (the user gave no --skip-compress) selects rsync's built-in
|
||||
* default list; a non-negative count is the user's explicit list. */
|
||||
if (count < 0)
|
||||
return suffix_in_list(name, DEFAULT_SKIP_COMPRESS_SUFFIXES);
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (strcasecmp(dot, suffixes[i]) == 0)
|
||||
const char* suffix = suffixes[i];
|
||||
if (suffix[0] == '.')
|
||||
suffix++;
|
||||
if (strcasecmp(name, suffix) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
Data* data_compress(Data* data_to_compress, int compression_level) {
|
||||
return data_compress_with_threads(data_to_compress, compression_level, 0);
|
||||
int compression_algo_from_name(const char* name) {
|
||||
if (!name)
|
||||
return -1;
|
||||
if (strcasecmp(name, "zstd") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZSTD;
|
||||
if (strcasecmp(name, "lz4") == 0)
|
||||
return (int)COMPRESSION_ALGO_LZ4;
|
||||
if (strcasecmp(name, "zlib") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZLIB;
|
||||
if (strcasecmp(name, "zlibx") == 0)
|
||||
return (int)COMPRESSION_ALGO_ZLIBX;
|
||||
if (strcasecmp(name, "none") == 0)
|
||||
return (int)COMPRESSION_ALGO_NONE;
|
||||
return -1;
|
||||
}
|
||||
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
const char* compression_algo_name(CompressionAlgo algo) {
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return "none";
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return "zstd";
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return "lz4";
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
return "zlib";
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return "zlibx";
|
||||
}
|
||||
return "<unknown>";
|
||||
}
|
||||
|
||||
bool compression_algo_valid(int algo) {
|
||||
return algo == (int)COMPRESSION_ALGO_NONE || algo == (int)COMPRESSION_ALGO_ZSTD ||
|
||||
algo == (int)COMPRESSION_ALGO_LZ4 || algo == (int)COMPRESSION_ALGO_ZLIB ||
|
||||
algo == (int)COMPRESSION_ALGO_ZLIBX;
|
||||
}
|
||||
|
||||
bool compression_algo_enabled(CompressionAlgo algo) {
|
||||
return algo != COMPRESSION_ALGO_NONE;
|
||||
}
|
||||
|
||||
CompressionAlgo compression_negotiate_default(void) {
|
||||
/* rsync 3.4.1 default preference order; every entry is compiled in, so this
|
||||
* resolves to zstd. */
|
||||
static const CompressionAlgo preference[] = {
|
||||
COMPRESSION_ALGO_ZSTD, COMPRESSION_ALGO_LZ4, COMPRESSION_ALGO_ZLIBX,
|
||||
COMPRESSION_ALGO_ZLIB, COMPRESSION_ALGO_NONE,
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(preference) / sizeof(preference[0]); i++) {
|
||||
if (compression_algo_valid((int)preference[i]))
|
||||
return preference[i];
|
||||
}
|
||||
return COMPRESSION_ALGO_ZSTD;
|
||||
}
|
||||
|
||||
void compression_set_algo(CompressionAlgo algo) {
|
||||
if (compression_algo_valid((int)algo))
|
||||
atomic_store(&g_compression_algo, (int)algo);
|
||||
}
|
||||
|
||||
CompressionAlgo compression_get_algo(void) {
|
||||
return (CompressionAlgo)atomic_load(&g_compression_algo);
|
||||
}
|
||||
|
||||
/* Per-thread cache of zstd contexts plus the grow-only compression scratch
|
||||
* buffer. zstd contexts are stateful and not safe to share between threads,
|
||||
* so each thread keeps its own (see compression_get_thread_ctx). The cache is
|
||||
* stored in a C11 thread-specific storage slot whose destructor releases the
|
||||
* contexts when the thread exits; this keeps LeakSanitizer clean for the
|
||||
* short-lived sender/receiver/scanner worker threads without every worker
|
||||
* entry point having to remember to call compression_free_thread_contexts().
|
||||
* The main thread's slot is not torn down by tss at process exit, so an atexit
|
||||
* hook releases it (and compression_free_thread_contexts allows eager
|
||||
* release). */
|
||||
typedef struct {
|
||||
ZSTD_CCtx* cctx;
|
||||
ZSTD_DCtx* dctx;
|
||||
void* out_buf; /* reusable ZSTD_compressBound-sized output scratch */
|
||||
size_t out_cap; /* bytes currently allocated for out_buf */
|
||||
int level; /* compression level currently applied to cctx */
|
||||
int workers; /* nbWorkers currently applied to cctx */
|
||||
bool params_set;
|
||||
bool cached; /* false when the TSS slot could not be used: caller owns */
|
||||
} CompressionThreadCtx;
|
||||
|
||||
static once_flag compression_tls_once = ONCE_FLAG_INIT;
|
||||
static tss_t compression_tls_key;
|
||||
static bool compression_tls_ready;
|
||||
|
||||
static void compression_tls_make_key(void);
|
||||
|
||||
static void compression_ctx_free(CompressionThreadCtx* ctx) {
|
||||
if (!ctx)
|
||||
return;
|
||||
if (ctx->cctx)
|
||||
ZSTD_freeCCtx(ctx->cctx);
|
||||
if (ctx->dctx)
|
||||
ZSTD_freeDCtx(ctx->dctx);
|
||||
free(ctx->out_buf);
|
||||
free(ctx);
|
||||
}
|
||||
|
||||
static void compression_tls_destructor(void* value) {
|
||||
compression_ctx_free((CompressionThreadCtx*)value);
|
||||
}
|
||||
|
||||
void compression_free_thread_contexts(void) {
|
||||
call_once(&compression_tls_once, compression_tls_make_key);
|
||||
if (!compression_tls_ready)
|
||||
return;
|
||||
CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key);
|
||||
if (!ctx)
|
||||
return;
|
||||
/* Clear the slot first so the thread-exit destructor cannot free it twice. */
|
||||
tss_set(compression_tls_key, NULL);
|
||||
compression_ctx_free(ctx);
|
||||
}
|
||||
|
||||
static void compression_atexit_cleanup(void) {
|
||||
compression_free_thread_contexts();
|
||||
}
|
||||
|
||||
static void compression_tls_make_key(void) {
|
||||
if (tss_create(&compression_tls_key, compression_tls_destructor) == thrd_success) {
|
||||
compression_tls_ready = true;
|
||||
atexit(compression_atexit_cleanup);
|
||||
}
|
||||
}
|
||||
|
||||
static CompressionThreadCtx* compression_get_thread_ctx(void) {
|
||||
call_once(&compression_tls_once, compression_tls_make_key);
|
||||
if (!compression_tls_ready) {
|
||||
/* Extremely unlikely: fall back to an uncached context the caller frees. */
|
||||
return (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx));
|
||||
}
|
||||
CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key);
|
||||
if (ctx)
|
||||
return ctx;
|
||||
ctx = (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx));
|
||||
if (!ctx)
|
||||
return NULL;
|
||||
ctx->cached = true;
|
||||
if (tss_set(compression_tls_key, ctx) != thrd_success)
|
||||
ctx->cached = false;
|
||||
return ctx;
|
||||
}
|
||||
|
||||
/* Release an uncached context immediately; cached contexts are owned by the
|
||||
* thread's TSS slot and freed on thread exit / compression_free_thread_contexts. */
|
||||
static void compression_ctx_put(CompressionThreadCtx* ctx) {
|
||||
if (ctx && !ctx->cached)
|
||||
compression_ctx_free(ctx);
|
||||
}
|
||||
|
||||
/* Build a frame consisting of a copy of `src` prefixed by `codec`. */
|
||||
static Data* frame_with_codec(const void* src, size_t size, CompressionAlgo codec) {
|
||||
if (size > SIZE_MAX - 1)
|
||||
return NULL;
|
||||
Data* out = data_create_empty(size + 1);
|
||||
if (!out)
|
||||
return NULL;
|
||||
((uint8_t*)out->data)[0] = (uint8_t)codec;
|
||||
if (size > 0)
|
||||
memcpy((uint8_t*)out->data + 1, src, size);
|
||||
out->size = size + 1;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zstd_compress(Data* in, int compression_level, int compression_threads) {
|
||||
size_t dst_size = ZSTD_compressBound(in->size);
|
||||
if (dst_size > SIZE_MAX - 1)
|
||||
return NULL;
|
||||
dst_size += 1; /* codec prefix */
|
||||
|
||||
CompressionThreadCtx* ctx = compression_get_thread_ctx();
|
||||
if (ctx == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD compression context");
|
||||
return NULL;
|
||||
}
|
||||
Data* compressed_data = NULL;
|
||||
|
||||
if (!ctx->cctx) {
|
||||
ctx->cctx = ZSTD_createCCtx();
|
||||
if (!ctx->cctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context");
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->params_set = false;
|
||||
}
|
||||
|
||||
/* Reset only the session: parameters (and any already-allocated zstd worker
|
||||
* pool) stay attached to the context, so compressing the next file does not
|
||||
* rebuild the pool. */
|
||||
ZSTD_CCtx_reset(ctx->cctx, ZSTD_reset_session_only);
|
||||
|
||||
if (!ctx->params_set || ctx->level != compression_level) {
|
||||
size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_compressionLevel, compression_level);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression level: %s", ZSTD_getErrorName(zret));
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->level = compression_level;
|
||||
}
|
||||
|
||||
int available_threads = 0;
|
||||
if (compression_threads > 0) {
|
||||
long online_cpus = sysconf(_SC_NPROCESSORS_ONLN);
|
||||
available_threads = online_cpus > 0 && online_cpus < compression_threads ? (int)online_cpus
|
||||
: compression_threads;
|
||||
}
|
||||
if (!ctx->params_set || ctx->workers != available_threads) {
|
||||
size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_nbWorkers, available_threads);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression threads: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->workers = available_threads;
|
||||
}
|
||||
ctx->params_set = true;
|
||||
|
||||
if (available_threads > 0) {
|
||||
/* Streaming compression needs the source size before threaded mode can end a frame. */
|
||||
size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, in->size);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
goto cleanup;
|
||||
}
|
||||
}
|
||||
|
||||
if (ctx->out_cap < dst_size) {
|
||||
void* grown = protocol_realloc(ctx->out_buf, dst_size);
|
||||
if (grown == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate compression buffer");
|
||||
goto cleanup;
|
||||
}
|
||||
ctx->out_buf = grown;
|
||||
ctx->out_cap = dst_size;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {in->data, in->size, 0};
|
||||
ZSTD_outBuffer output = {(uint8_t*)ctx->out_buf + 1, dst_size - 1, 0};
|
||||
|
||||
size_t ret;
|
||||
do {
|
||||
ret = ZSTD_compressStream2(ctx->cctx, &output, &input, ZSTD_e_end);
|
||||
if (ZSTD_isError(ret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Compression failed: %s", ZSTD_getErrorName(ret));
|
||||
goto cleanup;
|
||||
}
|
||||
} while (ret > 0);
|
||||
|
||||
/* Hand off an exactly-sized copy; the scratch buffer stays cached so the next
|
||||
* call does not reallocate a ZSTD_compressBound-sized block. */
|
||||
compressed_data = data_create_empty(output.pos + 1);
|
||||
if (compressed_data == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate compressed data");
|
||||
goto cleanup;
|
||||
}
|
||||
((uint8_t*)compressed_data->data)[0] = (uint8_t)COMPRESSION_ALGO_ZSTD;
|
||||
if (output.pos > 0)
|
||||
memcpy((uint8_t*)compressed_data->data + 1, (uint8_t*)ctx->out_buf + 1, output.pos);
|
||||
compressed_data->size = output.pos + 1;
|
||||
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu", in->size,
|
||||
compressed_data->size);
|
||||
|
||||
cleanup:
|
||||
compression_ctx_put(ctx);
|
||||
return compressed_data;
|
||||
}
|
||||
|
||||
static Data* lz4_compress(Data* in) {
|
||||
int bound = LZ4_compressBound((int)in->size);
|
||||
if (bound < 0 || in->size > (size_t)INT_MAX)
|
||||
return NULL;
|
||||
Data* out = data_create_empty((size_t)bound + 1 + LZ4_SIZE_PREFIX_LEN);
|
||||
if (!out)
|
||||
return NULL;
|
||||
uint32_t raw_size = (uint32_t)in->size;
|
||||
uint8_t* p = (uint8_t*)out->data;
|
||||
p[0] = (uint8_t)COMPRESSION_ALGO_LZ4;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
p[1 + i] = (uint8_t)((raw_size >> (8 * i)) & 0xff);
|
||||
int written = 0;
|
||||
if (in->size > 0) {
|
||||
written = LZ4_compress_default((const char*)in->data, (char*)p + 1 + LZ4_SIZE_PREFIX_LEN,
|
||||
(int)in->size, bound);
|
||||
if (written <= 0) {
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
out->size = (size_t)written + 1 + LZ4_SIZE_PREFIX_LEN;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zlib_compress(Data* in, CompressionAlgo algo, int compression_level) {
|
||||
int level = compression_level;
|
||||
if (level < 1)
|
||||
level = Z_DEFAULT_COMPRESSION;
|
||||
if (level > 9)
|
||||
level = 9;
|
||||
uLong bound = compressBound((uLong)in->size);
|
||||
if (in->size > (size_t)ULONG_MAX)
|
||||
return NULL;
|
||||
Data* out = data_create_empty((size_t)bound + 1 + LZ4_SIZE_PREFIX_LEN);
|
||||
if (!out)
|
||||
return NULL;
|
||||
uint32_t raw_size = (uint32_t)in->size;
|
||||
uint8_t* p = (uint8_t*)out->data;
|
||||
p[0] = (uint8_t)algo;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
p[1 + i] = (uint8_t)((raw_size >> (8 * i)) & 0xff);
|
||||
uLongf dest_len = bound;
|
||||
int rc = compress2(p + 1 + LZ4_SIZE_PREFIX_LEN, &dest_len, (const Bytef*)in->data,
|
||||
(uLong)in->size, level);
|
||||
if (rc != Z_OK) {
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = (size_t)dest_len + 1 + LZ4_SIZE_PREFIX_LEN;
|
||||
return out;
|
||||
}
|
||||
|
||||
Data* data_compress_codec(Data* data_to_compress, CompressionAlgo algo, int compression_level,
|
||||
int compression_threads) {
|
||||
if (!data_to_compress || (!data_to_compress->data && data_to_compress->size != 0) ||
|
||||
compression_threads < 0 || compression_threads > COMPRESSION_MAX_THREADS)
|
||||
return NULL;
|
||||
if (!compression_algo_valid((int)algo))
|
||||
return NULL;
|
||||
log_message(LOG_LEVEL_DEBUG, "Starting to compress data");
|
||||
size_t dst_size = ZSTD_compressBound(data_to_compress->size);
|
||||
Data* compressed_data = data_create_empty(dst_size);
|
||||
if (compressed_data == NULL)
|
||||
return NULL;
|
||||
|
||||
ZSTD_CCtx* cctx = ZSTD_createCCtx();
|
||||
if (!cctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context");
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
switch (algo) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return frame_with_codec(data_to_compress->data, data_to_compress->size, COMPRESSION_ALGO_NONE);
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return zstd_compress(data_to_compress, compression_level, compression_threads);
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return lz4_compress(data_to_compress);
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return zlib_compress(data_to_compress, algo, compression_level);
|
||||
}
|
||||
|
||||
size_t zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_compressionLevel, compression_level);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression level: %s", ZSTD_getErrorName(zret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
if (compression_threads > 0) {
|
||||
long online_cpus = sysconf(_SC_NPROCESSORS_ONLN);
|
||||
int available_threads = online_cpus > 0 && online_cpus < compression_threads
|
||||
? (int)online_cpus
|
||||
: compression_threads;
|
||||
zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_nbWorkers, available_threads);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression threads: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
}
|
||||
/* Streaming compression needs the source size before threaded mode can end a frame. */
|
||||
zret = ZSTD_CCtx_setPledgedSrcSize(cctx, data_to_compress->size);
|
||||
if (ZSTD_isError(zret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s",
|
||||
ZSTD_getErrorName(zret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0};
|
||||
ZSTD_outBuffer output = {compressed_data->data, dst_size, 0};
|
||||
|
||||
size_t ret;
|
||||
do {
|
||||
ret = ZSTD_compressStream2(cctx, &output, &input, ZSTD_e_end);
|
||||
if (ZSTD_isError(ret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Compression failed: %s", ZSTD_getErrorName(ret));
|
||||
ZSTD_freeCCtx(cctx);
|
||||
data_destroy(compressed_data);
|
||||
return NULL;
|
||||
}
|
||||
} while (ret > 0);
|
||||
|
||||
compressed_data->size = output.pos;
|
||||
ZSTD_freeCCtx(cctx);
|
||||
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu",
|
||||
data_to_compress->size, compressed_data->size);
|
||||
return compressed_data;
|
||||
}
|
||||
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) ||
|
||||
maximum_size == 0)
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads) {
|
||||
return data_compress_codec(data_to_compress, compression_get_algo(), compression_level,
|
||||
compression_threads);
|
||||
}
|
||||
|
||||
Data* data_compress(Data* data_to_compress, int compression_level) {
|
||||
return data_compress_codec(data_to_compress, compression_get_algo(), compression_level, 0);
|
||||
}
|
||||
|
||||
static Data* decompress_none(const Data* compressed_data, size_t maximum_size) {
|
||||
size_t size = compressed_data->size - 1;
|
||||
if (size > maximum_size)
|
||||
return NULL;
|
||||
Data* out = data_create_empty(size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (size > 0)
|
||||
memcpy(out->data, (const uint8_t*)compressed_data->data + 1, size);
|
||||
out->size = size;
|
||||
return out;
|
||||
}
|
||||
|
||||
/* Read the 4-byte little-endian raw size stored after the codec byte. */
|
||||
static bool read_raw_size(const Data* in, uint32_t* raw_size) {
|
||||
if (in->size < 1 + LZ4_SIZE_PREFIX_LEN)
|
||||
return false;
|
||||
const uint8_t* p = (const uint8_t*)in->data;
|
||||
uint32_t v = 0;
|
||||
for (int i = 0; i < LZ4_SIZE_PREFIX_LEN; i++)
|
||||
v |= (uint32_t)p[1 + i] << (8 * i);
|
||||
*raw_size = v;
|
||||
return true;
|
||||
}
|
||||
|
||||
static Data* lz4_decompress(Data* compressed_data, size_t maximum_size, size_t hard_limit) {
|
||||
uint32_t raw_size = 0;
|
||||
if (!read_raw_size(compressed_data, &raw_size))
|
||||
return NULL;
|
||||
if (raw_size > hard_limit || raw_size > maximum_size)
|
||||
return NULL;
|
||||
size_t comp_size = compressed_data->size - 1 - LZ4_SIZE_PREFIX_LEN;
|
||||
Data* out = data_create_empty(raw_size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (raw_size == 0) {
|
||||
out->size = 0;
|
||||
return out;
|
||||
}
|
||||
int rc = LZ4_decompress_safe((const char*)compressed_data->data + 1 + LZ4_SIZE_PREFIX_LEN,
|
||||
(char*)out->data, (int)comp_size, (int)raw_size);
|
||||
if (rc < 0 || (uint32_t)rc != raw_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "LZ4 decompression failed");
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = raw_size;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zlib_decompress(Data* compressed_data, size_t maximum_size, size_t hard_limit) {
|
||||
uint32_t raw_size = 0;
|
||||
if (!read_raw_size(compressed_data, &raw_size))
|
||||
return NULL;
|
||||
if (raw_size > hard_limit || raw_size > maximum_size)
|
||||
return NULL;
|
||||
size_t comp_size = compressed_data->size - 1 - LZ4_SIZE_PREFIX_LEN;
|
||||
Data* out = data_create_empty(raw_size);
|
||||
if (!out)
|
||||
return NULL;
|
||||
if (raw_size == 0) {
|
||||
out->size = 0;
|
||||
return out;
|
||||
}
|
||||
uLongf dest_len = raw_size;
|
||||
int rc =
|
||||
uncompress((Bytef*)out->data, &dest_len,
|
||||
(const Bytef*)compressed_data->data + 1 + LZ4_SIZE_PREFIX_LEN, (uLong)comp_size);
|
||||
if (rc != Z_OK || dest_len != raw_size) {
|
||||
log_message(LOG_LEVEL_ERROR, "zlib decompression failed");
|
||||
data_destroy(out);
|
||||
return NULL;
|
||||
}
|
||||
out->size = raw_size;
|
||||
return out;
|
||||
}
|
||||
|
||||
static Data* zstd_decompress(Data* compressed_data, size_t maximum_size) {
|
||||
/* The zstd frame starts after the codec byte. */
|
||||
const void* frame = (const uint8_t*)compressed_data->data + 1;
|
||||
size_t frame_size = compressed_data->size - 1;
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Start to decompress data");
|
||||
unsigned long long dst_size =
|
||||
ZSTD_getFrameContentSize(compressed_data->data, compressed_data->size);
|
||||
if (ZSTD_isError(dst_size)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: %s",
|
||||
ZSTD_getErrorName(dst_size));
|
||||
unsigned long long dst_size = ZSTD_getFrameContentSize(frame, frame_size);
|
||||
/* ZSTD_isError() is also true for ZSTD_CONTENTSIZE_ERROR and
|
||||
* ZSTD_CONTENTSIZE_UNKNOWN (both are encoded near (size_t)-1), so test the
|
||||
* sentinels explicitly instead of blanket-rejecting every error-ish value:
|
||||
* only CONTENTSIZE_ERROR means an unreadable header, while CONTENTSIZE_UNKNOWN
|
||||
* must reach the estimate fallback below. */
|
||||
if (dst_size == ZSTD_CONTENTSIZE_ERROR) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: invalid zstd frame");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
// ZSTD_CONTENTSIZE_UNKNOWN (~2^64) can cause massive allocation;
|
||||
// fall back to a conservative estimate (3x compressed size) when unknown.
|
||||
if (dst_size == ZSTD_CONTENTSIZE_UNKNOWN) {
|
||||
if (compressed_data->size > ULLONG_MAX / 3)
|
||||
if (frame_size > ULLONG_MAX / 3)
|
||||
return NULL;
|
||||
dst_size = compressed_data->size * 3;
|
||||
dst_size = frame_size * 3;
|
||||
if (dst_size < INITIAL_DECOMPRESS_BUF_SIZE)
|
||||
dst_size = INITIAL_DECOMPRESS_BUF_SIZE;
|
||||
}
|
||||
@@ -144,41 +545,51 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
ZSTD_DCtx* dctx = ZSTD_createDCtx();
|
||||
if (!dctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context");
|
||||
CompressionThreadCtx* ctx = compression_get_thread_ctx();
|
||||
if (ctx == NULL) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD decompression context");
|
||||
return NULL;
|
||||
}
|
||||
Data* uncompressed_data = NULL;
|
||||
|
||||
if (!ctx->dctx) {
|
||||
ctx->dctx = ZSTD_createDCtx();
|
||||
if (!ctx->dctx) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context");
|
||||
goto cleanup;
|
||||
}
|
||||
}
|
||||
/* Reset only the session; decompression parameters are sticky. */
|
||||
ZSTD_DCtx_reset(ctx->dctx, ZSTD_reset_session_only);
|
||||
|
||||
size_t buf_size = (dst_size > 0) ? (size_t)dst_size : INITIAL_DECOMPRESS_BUF_SIZE;
|
||||
if (buf_size > maximum_size)
|
||||
buf_size = maximum_size;
|
||||
Data* uncompressed_data = data_create_empty(buf_size);
|
||||
uncompressed_data = data_create_empty(buf_size);
|
||||
if (!uncompressed_data) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to allocate decompression buffer");
|
||||
ZSTD_freeDCtx(dctx);
|
||||
return NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
|
||||
ZSTD_inBuffer input = {compressed_data->data, compressed_data->size, 0};
|
||||
ZSTD_inBuffer input = {frame, frame_size, 0};
|
||||
ZSTD_outBuffer output = {uncompressed_data->data, buf_size, 0};
|
||||
|
||||
size_t ret;
|
||||
do {
|
||||
ret = ZSTD_decompressStream(dctx, &output, &input);
|
||||
ret = ZSTD_decompressStream(ctx->dctx, &output, &input);
|
||||
if (ZSTD_isError(ret)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Decompression failed: %s", ZSTD_getErrorName(ret));
|
||||
ZSTD_freeDCtx(dctx);
|
||||
data_destroy(uncompressed_data);
|
||||
return NULL;
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
if (ret > 0 && output.pos == output.size) {
|
||||
if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) {
|
||||
log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes",
|
||||
(unsigned long long)MAX_DECOMPRESSED_SIZE);
|
||||
ZSTD_freeDCtx(dctx);
|
||||
data_destroy(uncompressed_data);
|
||||
return NULL;
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
buf_size *= 2;
|
||||
if (buf_size > hard_limit)
|
||||
@@ -186,23 +597,64 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
void* new_data = protocol_realloc(uncompressed_data->data, buf_size);
|
||||
if (!new_data) {
|
||||
log_message(LOG_LEVEL_ERROR, "Failed to grow decompression buffer");
|
||||
ZSTD_freeDCtx(dctx);
|
||||
data_destroy(uncompressed_data);
|
||||
return NULL;
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
uncompressed_data->data = new_data;
|
||||
output.dst = new_data;
|
||||
output.size = buf_size;
|
||||
/* Re-attempt with the larger output buffer; the truncated-frame check
|
||||
* below must not reject a complete frame that merely filled the previous
|
||||
* buffer exactly. */
|
||||
continue;
|
||||
}
|
||||
/* A positive hint with all input consumed means the frame is incomplete: a
|
||||
* truncated stream would otherwise spin here forever (ZSTD_decompressStream
|
||||
* keeps returning the same hint). Fail instead of burning CPU. */
|
||||
if (ret != 0 && input.pos == input.size) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"Truncated zstd frame: input exhausted with %zu bytes still expected", ret);
|
||||
data_destroy(uncompressed_data);
|
||||
uncompressed_data = NULL;
|
||||
goto cleanup;
|
||||
}
|
||||
} while (ret > 0);
|
||||
|
||||
uncompressed_data->size = output.pos;
|
||||
ZSTD_freeDCtx(dctx);
|
||||
|
||||
log_debug_message(LOG_DEBUG_UTIL, "Decompressed data successfully");
|
||||
|
||||
cleanup:
|
||||
compression_ctx_put(ctx);
|
||||
return uncompressed_data;
|
||||
}
|
||||
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
|
||||
if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) ||
|
||||
maximum_size == 0)
|
||||
return NULL;
|
||||
if (compressed_data->size < 1)
|
||||
return NULL;
|
||||
unsigned long long hard_limit =
|
||||
maximum_size < MAX_DECOMPRESSED_SIZE ? maximum_size : MAX_DECOMPRESSED_SIZE;
|
||||
uint8_t codec = ((const uint8_t*)compressed_data->data)[0];
|
||||
if (!compression_algo_valid(codec))
|
||||
return NULL;
|
||||
switch ((CompressionAlgo)codec) {
|
||||
case COMPRESSION_ALGO_NONE:
|
||||
return decompress_none(compressed_data, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_ZSTD:
|
||||
return zstd_decompress(compressed_data, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_LZ4:
|
||||
return lz4_decompress(compressed_data, maximum_size, (size_t)hard_limit);
|
||||
case COMPRESSION_ALGO_ZLIB:
|
||||
case COMPRESSION_ALGO_ZLIBX:
|
||||
return zlib_decompress(compressed_data, maximum_size, (size_t)hard_limit);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
Data* data_decompress(Data* compressed_data) {
|
||||
return data_decompress_limited(compressed_data, MAX_DECOMPRESSED_SIZE);
|
||||
}
|
||||
|
||||
@@ -6,12 +6,67 @@
|
||||
|
||||
#define COMPRESSION_MAX_THREADS 64
|
||||
|
||||
/* Compression algorithms selectable with --compress-choice / -z. The ids are
|
||||
* the values placed on the wire (Config->compression_algo), so they must be
|
||||
* kept stable. NONE is "no compression"; ZSTD is the historical FastSync
|
||||
* default and the negotiated "auto" choice. ZLIBX is rsync's zlib-without-
|
||||
* matched-data variant: FastSync compresses only the delta/token bytes (it does
|
||||
* not put matched file data in the compression stream), so its zlib codec is
|
||||
* already the "x" form and zlib/zlibx share the same implementation, recorded
|
||||
* under distinct ids. */
|
||||
typedef enum {
|
||||
COMPRESSION_ALGO_NONE = 0,
|
||||
COMPRESSION_ALGO_ZSTD = 1,
|
||||
COMPRESSION_ALGO_LZ4 = 2,
|
||||
COMPRESSION_ALGO_ZLIB = 3,
|
||||
COMPRESSION_ALGO_ZLIBX = 4
|
||||
} CompressionAlgo;
|
||||
|
||||
/* Resolve a --compress-choice string (case-insensitive) to an algorithm id.
|
||||
* Accepts "zstd", "lz4", "zlib", "zlibx", "none". "auto" is not an algorithm
|
||||
* here; the caller resolves it to the negotiated default. Returns -1 for any
|
||||
* unrecognized name. */
|
||||
int compression_algo_from_name(const char* name);
|
||||
const char* compression_algo_name(CompressionAlgo algo);
|
||||
bool compression_algo_valid(int algo);
|
||||
|
||||
/* Pick the first algorithm from FastSync's compiled-in preference list
|
||||
* (rsync 3.4.1's `--version` order: zstd lz4 zlibx zlib none). Resolves
|
||||
* "auto". */
|
||||
CompressionAlgo compression_negotiate_default(void);
|
||||
|
||||
/* True when the algorithm actually compresses (i.e. is not NONE). */
|
||||
bool compression_algo_enabled(CompressionAlgo algo);
|
||||
|
||||
/* Select the process-wide codec used by the legacy wrappers below. Each
|
||||
* process serves exactly one transfer config (the server forks per connection,
|
||||
* the client configures itself before spawning transfer threads), so a
|
||||
* process-global default is sufficient and constant for the lifetime of a
|
||||
* transfer. Defaults to ZSTD when never set. Thread-safe. */
|
||||
void compression_set_algo(CompressionAlgo algo);
|
||||
CompressionAlgo compression_get_algo(void);
|
||||
|
||||
/* Codec-aware primitives. The compressed buffer is self-describing: its first
|
||||
* byte is the CompressionAlgo id, so decompression never needs the codec passed
|
||||
* separately (this keeps every existing Decompress call site source-compatible).
|
||||
* `data_compress_codec` returns NULL on invalid input or an unsupported codec. */
|
||||
Data* data_compress_codec(Data* data_to_compress, CompressionAlgo algo, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
|
||||
/* Legacy zstd-default wrappers retained for existing callers/tests. */
|
||||
Data* data_compress(Data* data_to_compress, int compression_level);
|
||||
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
|
||||
int compression_threads);
|
||||
Data* data_decompress(Data* compressed_data);
|
||||
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
|
||||
bool compression_should_skip(const char* path);
|
||||
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count);
|
||||
|
||||
/* Release the calling thread's cached zstd contexts (compressor, decompressor
|
||||
* and scratch buffer). The cache is thread-local and is also released
|
||||
* automatically when a worker thread exits (via a C11 tss destructor) and for
|
||||
* the main thread at process exit; this explicit entry point exists so tests
|
||||
* and long-lived callers can drop the cache deterministically. Safe to call
|
||||
* when no context has been created, and idempotent. */
|
||||
void compression_free_thread_contexts(void);
|
||||
|
||||
#endif
|
||||
|
||||
+707
-672
File diff suppressed because it is too large
Load Diff
+699
-301
File diff suppressed because it is too large
Load Diff
@@ -158,13 +158,13 @@ static int hex_value(char c) {
|
||||
* digits. Such a line is refused loudly (and never accepted) so an operator
|
||||
* cannot keep a replayable bearer digest in place after the protocol bump. */
|
||||
static bool secret_is_legacy_hex(const char* s) {
|
||||
if (!s)
|
||||
if (!s || strlen(s) != 64)
|
||||
return false;
|
||||
for (int i = 0; i < 64; i++) {
|
||||
if (hex_value(s[i]) < 0)
|
||||
return false;
|
||||
}
|
||||
return s[64] == '\0';
|
||||
return true;
|
||||
}
|
||||
|
||||
bool credentials_b64_encode(const uint8_t* in, size_t n, char* out, size_t out_sz) {
|
||||
|
||||
+344
-4
@@ -1,8 +1,13 @@
|
||||
#include "daemon_conf.h"
|
||||
#include "credentials.h"
|
||||
#include "utils.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <ctype.h>
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <netinet/in.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -48,6 +53,191 @@ static bool parse_bool_value(const char* value, bool* out) {
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Parse an IPv4/IPv6 CIDR "addr/prefix" into `bytes`/`*family`. Returns false
|
||||
* for a malformed address, a missing/oversized prefix, or a prefix that does
|
||||
* not fit the address family. */
|
||||
static bool parse_cidr(const char* cidr, int* prefix_out, uint8_t* bytes, int* family_out) {
|
||||
const char* slash = strchr(cidr, '/');
|
||||
if (!slash)
|
||||
return false;
|
||||
size_t addr_len = (size_t)(slash - cidr);
|
||||
if (addr_len == 0 || addr_len >= INET6_ADDRSTRLEN)
|
||||
return false;
|
||||
char addr[INET6_ADDRSTRLEN];
|
||||
memcpy(addr, cidr, addr_len);
|
||||
addr[addr_len] = '\0';
|
||||
char* end = NULL;
|
||||
long prefix = strtol(slash + 1, &end, 10);
|
||||
if (end == slash + 1 || *end != '\0')
|
||||
return false;
|
||||
struct in_addr v4;
|
||||
struct in6_addr v6;
|
||||
if (inet_pton(AF_INET, addr, &v4) == 1) {
|
||||
if (prefix < 0 || prefix > 32)
|
||||
return false;
|
||||
memcpy(bytes, &v4, sizeof(v4));
|
||||
*prefix_out = (int)prefix;
|
||||
*family_out = AF_INET;
|
||||
return true;
|
||||
}
|
||||
if (inet_pton(AF_INET6, addr, &v6) == 1) {
|
||||
if (prefix < 0 || prefix > 128)
|
||||
return false;
|
||||
memcpy(bytes, &v6, sizeof(v6));
|
||||
*prefix_out = (int)prefix;
|
||||
*family_out = AF_INET6;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* A host pattern is valid when it is `*`, a valid IPv4/IPv6 literal, or a valid
|
||||
* CIDR. Peer addresses reaching the matcher are always numeric, so hostname
|
||||
* globs are rejected at parse time: accepting one would create a deny rule that
|
||||
* silently never matches (fail-open). */
|
||||
static bool host_pattern_valid(const char* pattern) {
|
||||
if (!pattern || *pattern == '\0')
|
||||
return false;
|
||||
if (strcmp(pattern, "*") == 0)
|
||||
return true;
|
||||
if (strchr(pattern, '/')) {
|
||||
uint8_t bytes[16];
|
||||
int prefix;
|
||||
int family;
|
||||
return parse_cidr(pattern, &prefix, bytes, &family);
|
||||
}
|
||||
struct in_addr v4;
|
||||
struct in6_addr v6;
|
||||
return inet_pton(AF_INET, pattern, &v4) == 1 || inet_pton(AF_INET6, pattern, &v6) == 1;
|
||||
}
|
||||
|
||||
/* Append every comma- and/or whitespace-separated host pattern in `value` to
|
||||
* the heap-owned list (or replace the list when `replace` is set, which --dparam
|
||||
* uses so an override can narrow access rather than only widen it). Returns
|
||||
* false (err filled) on an invalid pattern or an allocation failure. */
|
||||
static bool store_host_list(char*** list, int* count, const char* value, const char* key,
|
||||
const char* module_name, bool replace, char* err, size_t err_size) {
|
||||
if (replace) {
|
||||
for (int i = 0; i < *count; i++)
|
||||
free((*list)[i]);
|
||||
free(*list);
|
||||
*list = NULL;
|
||||
*count = 0;
|
||||
}
|
||||
char* copy = str_dup(value);
|
||||
if (!copy) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
|
||||
else
|
||||
set_error(err, err_size, "out of memory parsing '%s'", key);
|
||||
return false;
|
||||
}
|
||||
char* save = NULL;
|
||||
int added = 0;
|
||||
for (char* token = strtok_r(copy, ", \t", &save); token; token = strtok_r(NULL, ", \t", &save)) {
|
||||
if (!host_pattern_valid(token)) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': invalid host pattern '%s' in '%s'", module_name,
|
||||
token, key);
|
||||
else
|
||||
set_error(err, err_size, "invalid host pattern '%s' in '%s'", token, key);
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
char** grown = realloc(*list, (size_t)(*count + 1) * sizeof(char*));
|
||||
if (!grown) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
|
||||
else
|
||||
set_error(err, err_size, "out of memory parsing '%s'", key);
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
*list = grown;
|
||||
char* dup = str_dup(token);
|
||||
if (!dup) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
|
||||
else
|
||||
set_error(err, err_size, "out of memory parsing '%s'", key);
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
(*list)[(*count)++] = dup;
|
||||
added++;
|
||||
}
|
||||
free(copy);
|
||||
/* A present key with an empty (or separator-only) value would otherwise
|
||||
* install a zero-length list, i.e. no ACL at all: a strict-parse config must
|
||||
* never silently turn a restrictive directive into "allow everyone". */
|
||||
if (added == 0) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': '%s' must list at least one host pattern", module_name,
|
||||
key);
|
||||
else
|
||||
set_error(err, err_size, "'%s' must list at least one host pattern", key);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse a `max connections` value: a positive integer (0/negative/garbage are
|
||||
* rejected because they would silently disable the cap or admit nothing). */
|
||||
static bool store_max_connections(int* slot, const char* value, const char* module_name, char* err,
|
||||
size_t err_size) {
|
||||
char* end = NULL;
|
||||
errno = 0;
|
||||
long n = strtol(value, &end, 10);
|
||||
if (*value == '\0' || errno != 0 || *end != '\0' || n <= 0 || n > INT_MAX) {
|
||||
if (module_name)
|
||||
set_error(err, err_size,
|
||||
"module '%s': invalid 'max connections' '%s' (must be a positive "
|
||||
"integer)",
|
||||
module_name, value);
|
||||
else
|
||||
set_error(err, err_size, "invalid 'max connections' '%s' (must be a positive integer)",
|
||||
value);
|
||||
return false;
|
||||
}
|
||||
*slot = (int)n;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse a non-negative concurrency cap where 0 means unlimited/disabled
|
||||
* (per-module `max connections`, `max connections per host`,
|
||||
* `auth lockout threshold`). Negative/garbage/oversized values are rejected. */
|
||||
static bool store_optional_cap(int* slot, const char* value, int max_value, const char* key,
|
||||
const char* module_name, char* err, size_t err_size) {
|
||||
char* end = NULL;
|
||||
errno = 0;
|
||||
long n = strtol(value, &end, 10);
|
||||
if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 || n > max_value) {
|
||||
if (module_name)
|
||||
set_error(err, err_size, "module '%s': invalid '%s' '%s' (must be 0-%d)", module_name, key,
|
||||
value, max_value);
|
||||
else
|
||||
set_error(err, err_size, "invalid '%s' '%s' (must be 0-%d)", key, value, max_value);
|
||||
return false;
|
||||
}
|
||||
*slot = (int)n;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Parse an `auth failure delay` value: 0 (disabled) through the configured cap. */
|
||||
static bool store_auth_failure_delay(int* slot, const char* value, char* err, size_t err_size) {
|
||||
char* end = NULL;
|
||||
errno = 0;
|
||||
long n = strtol(value, &end, 10);
|
||||
if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 ||
|
||||
n > DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS) {
|
||||
set_error(err, err_size, "invalid 'auth failure delay' '%s' (must be 0-%d milliseconds)", value,
|
||||
DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS);
|
||||
return false;
|
||||
}
|
||||
*slot = (int)n;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool daemon_module_name_valid(const char* name) {
|
||||
if (!name || *name == '\0')
|
||||
return false;
|
||||
@@ -68,14 +258,28 @@ DaemonConf* daemon_conf_create(void) {
|
||||
if (!conf)
|
||||
return NULL;
|
||||
conf->global.port = DAEMON_CONF_DEFAULT_PORT;
|
||||
conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS;
|
||||
conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS;
|
||||
conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST;
|
||||
conf->global.auth_lockout_threshold = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD;
|
||||
conf->global.auth_lockout_duration_sec = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC;
|
||||
return conf;
|
||||
}
|
||||
|
||||
/* Free a heap-owned pattern list of `count` entries. */
|
||||
static void free_string_list(char** list, int count) {
|
||||
for (int i = 0; i < count; i++)
|
||||
free(list[i]);
|
||||
free(list);
|
||||
}
|
||||
|
||||
void daemon_conf_free(DaemonConf* conf) {
|
||||
if (!conf)
|
||||
return;
|
||||
free(conf->global.motd_file);
|
||||
free(conf->global.address);
|
||||
free_string_list(conf->global.hosts_allow, conf->global.hosts_allow_count);
|
||||
free_string_list(conf->global.hosts_deny, conf->global.hosts_deny_count);
|
||||
for (int i = 0; i < conf->module_count; i++) {
|
||||
DaemonModule* m = &conf->modules[i];
|
||||
free(m->name);
|
||||
@@ -83,6 +287,8 @@ void daemon_conf_free(DaemonConf* conf) {
|
||||
for (int j = 0; j < m->auth_user_count; j++)
|
||||
free(m->auth_users[j]);
|
||||
free(m->auth_users);
|
||||
free_string_list(m->hosts_allow, m->hosts_allow_count);
|
||||
free_string_list(m->hosts_deny, m->hosts_deny_count);
|
||||
}
|
||||
free(conf->modules);
|
||||
free(conf);
|
||||
@@ -122,8 +328,8 @@ static bool store_port(int* slot, const char* value, char* err, size_t err_size)
|
||||
|
||||
/* Apply a global scalar key/value. Keys are case-insensitive. Returns false
|
||||
* (err filled) on an unknown key or an invalid value. */
|
||||
static bool apply_global_key(DaemonConf* conf, char* key, const char* value, char* err,
|
||||
size_t err_size) {
|
||||
static bool apply_global_key(DaemonConf* conf, char* key, const char* value, bool replace_hosts,
|
||||
char* err, size_t err_size) {
|
||||
if (key_equals(key, "port"))
|
||||
return store_port(&conf->global.port, value, err, err_size);
|
||||
if (key_equals(key, "motd file")) {
|
||||
@@ -140,6 +346,28 @@ static bool apply_global_key(DaemonConf* conf, char* key, const char* value, cha
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (key_equals(key, "max connections"))
|
||||
return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size);
|
||||
if (key_equals(key, "max connections per host"))
|
||||
return store_optional_cap(&conf->global.max_connections_per_host, value,
|
||||
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "max connections per host", NULL,
|
||||
err, err_size);
|
||||
if (key_equals(key, "auth failure delay"))
|
||||
return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size);
|
||||
if (key_equals(key, "auth lockout threshold"))
|
||||
return store_optional_cap(&conf->global.auth_lockout_threshold, value,
|
||||
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "auth lockout threshold", NULL,
|
||||
err, err_size);
|
||||
if (key_equals(key, "auth lockout duration"))
|
||||
return store_optional_cap(&conf->global.auth_lockout_duration_sec, value,
|
||||
DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC, "auth lockout duration",
|
||||
NULL, err, err_size);
|
||||
if (key_equals(key, "hosts allow"))
|
||||
return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value,
|
||||
"hosts allow", NULL, replace_hosts, err, err_size);
|
||||
if (key_equals(key, "hosts deny"))
|
||||
return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value,
|
||||
"hosts deny", NULL, replace_hosts, err, err_size);
|
||||
set_error(err, err_size, "unknown global key '%s'", key);
|
||||
return false;
|
||||
}
|
||||
@@ -188,10 +416,17 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char*
|
||||
return false;
|
||||
}
|
||||
char* save = NULL;
|
||||
int added = 0;
|
||||
for (char* token = strtok_r(list, ",", &save); token; token = strtok_r(NULL, ",", &save)) {
|
||||
const char* user = trim_ws(token);
|
||||
if (*user == '\0')
|
||||
continue;
|
||||
if (!credentials_username_valid(user)) {
|
||||
set_error(err, err_size, "module '%s': invalid 'auth users' entry '%s'", module->name,
|
||||
user);
|
||||
free(list);
|
||||
return false;
|
||||
}
|
||||
char** grown =
|
||||
realloc(module->auth_users, (size_t)(module->auth_user_count + 1) * sizeof(char*));
|
||||
if (!grown) {
|
||||
@@ -209,10 +444,27 @@ static bool apply_module_key(DaemonModule* module, char* key, char* value, char*
|
||||
return false;
|
||||
}
|
||||
module->auth_users[module->auth_user_count++] = dup;
|
||||
added++;
|
||||
}
|
||||
free(list);
|
||||
/* An empty/separator-only value must not silently disable authentication:
|
||||
* the key's presence is an explicit request for an allow-list. */
|
||||
if (added == 0) {
|
||||
set_error(err, err_size, "module '%s': 'auth users' must list at least one user",
|
||||
module->name);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (key_equals(key, "max connections"))
|
||||
return store_optional_cap(&module->max_connections, value, DAEMON_CONF_MAX_CONCURRENCY_LIMIT,
|
||||
"max connections", module->name, err, err_size);
|
||||
if (key_equals(key, "hosts allow"))
|
||||
return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow",
|
||||
module->name, false, err, err_size);
|
||||
if (key_equals(key, "hosts deny"))
|
||||
return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny",
|
||||
module->name, false, err, err_size);
|
||||
set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name);
|
||||
return false;
|
||||
}
|
||||
@@ -250,6 +502,11 @@ static int open_module(DaemonConf* conf, int* current_module, const char* name,
|
||||
set_error(err, err_size, "duplicate module '%s'", name);
|
||||
return -1;
|
||||
}
|
||||
if (conf->module_count >= DAEMON_CONF_MAX_MODULES) {
|
||||
set_error(err, err_size, "too many modules (limit %d); module '%s' rejected",
|
||||
DAEMON_CONF_MAX_MODULES, name);
|
||||
return -1;
|
||||
}
|
||||
DaemonModule* grown =
|
||||
realloc(conf->modules, (size_t)(conf->module_count + 1) * sizeof(DaemonModule));
|
||||
if (!grown) {
|
||||
@@ -393,7 +650,7 @@ DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size) {
|
||||
break;
|
||||
}
|
||||
} else {
|
||||
if (!apply_global_key(conf, key, value, err, err_size)) {
|
||||
if (!apply_global_key(conf, key, value, false, err, err_size)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
@@ -448,7 +705,90 @@ int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err
|
||||
set_error(err, err_size, "--dparam '%s' has an empty value", assignment);
|
||||
return -1;
|
||||
}
|
||||
bool ok = apply_global_key(conf, key, value, err, err_size);
|
||||
bool ok = apply_global_key(conf, key, value, true, err, err_size);
|
||||
free(copy);
|
||||
return ok ? 0 : -1;
|
||||
}
|
||||
|
||||
/* Compare the first `prefix` bits of two 16-byte address buffers. */
|
||||
static bool bit_prefix_match(const uint8_t* a, const uint8_t* b, int prefix) {
|
||||
int whole = prefix / 8;
|
||||
if (whole > 0 && memcmp(a, b, (size_t)whole) != 0)
|
||||
return false;
|
||||
int remainder = prefix % 8;
|
||||
if (remainder == 0)
|
||||
return true;
|
||||
uint8_t mask = (uint8_t)(0xffu << (8 - remainder));
|
||||
return (a[whole] & mask) == (b[whole] & mask);
|
||||
}
|
||||
|
||||
/* Case-insensitive glob match used for hostname patterns. Falls back to the
|
||||
* shared case-sensitive matcher when an operand is too long for the stack
|
||||
* buffers. */
|
||||
static bool host_glob_match(const char* pattern, const char* str) {
|
||||
char pbuf[256];
|
||||
char sbuf[256];
|
||||
size_t plen = strlen(pattern);
|
||||
size_t slen = strlen(str);
|
||||
if (plen >= sizeof(pbuf) || slen >= sizeof(sbuf))
|
||||
return glob_match(pattern, str);
|
||||
for (size_t i = 0; i <= plen; i++)
|
||||
pbuf[i] = (char)tolower((unsigned char)pattern[i]);
|
||||
for (size_t i = 0; i <= slen; i++)
|
||||
sbuf[i] = (char)tolower((unsigned char)str[i]);
|
||||
return glob_match(pbuf, sbuf);
|
||||
}
|
||||
|
||||
bool daemon_host_pattern_match(const char* pattern, const char* peer_ip) {
|
||||
if (!pattern || *pattern == '\0' || !peer_ip || *peer_ip == '\0')
|
||||
return false;
|
||||
if (strcmp(pattern, "*") == 0)
|
||||
return true;
|
||||
if (strchr(pattern, '/')) {
|
||||
uint8_t pattern_bytes[16];
|
||||
uint8_t peer_bytes[16];
|
||||
int prefix = 0;
|
||||
int family = AF_UNSPEC;
|
||||
if (!parse_cidr(pattern, &prefix, pattern_bytes, &family))
|
||||
return false;
|
||||
if (inet_pton(family, peer_ip, peer_bytes) != 1)
|
||||
return false;
|
||||
return bit_prefix_match(pattern_bytes, peer_bytes, prefix);
|
||||
}
|
||||
struct in_addr pattern_v4;
|
||||
struct in_addr peer_v4;
|
||||
if (inet_pton(AF_INET, pattern, &pattern_v4) == 1)
|
||||
return inet_pton(AF_INET, peer_ip, &peer_v4) == 1 && pattern_v4.s_addr == peer_v4.s_addr;
|
||||
struct in6_addr pattern_v6;
|
||||
struct in6_addr peer_v6;
|
||||
if (inet_pton(AF_INET6, pattern, &pattern_v6) == 1)
|
||||
return inet_pton(AF_INET6, peer_ip, &peer_v6) == 1 &&
|
||||
memcmp(&pattern_v6, &peer_v6, sizeof(pattern_v6)) == 0;
|
||||
/* Not a literal: a hostname/glob pattern. */
|
||||
return host_glob_match(pattern, peer_ip);
|
||||
}
|
||||
|
||||
bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count,
|
||||
char* const* deny, int deny_count) {
|
||||
if (!peer_ip)
|
||||
return false;
|
||||
for (int i = 0; i < deny_count; i++) {
|
||||
if (daemon_host_pattern_match(deny[i], peer_ip))
|
||||
return false;
|
||||
}
|
||||
if (allow_count > 0) {
|
||||
for (int i = 0; i < allow_count; i++) {
|
||||
if (daemon_host_pattern_match(allow[i], peer_ip))
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny,
|
||||
int deny_count) {
|
||||
(void)allow;
|
||||
(void)deny;
|
||||
return allow_count > 0 || deny_count > 0;
|
||||
}
|
||||
|
||||
@@ -52,6 +52,15 @@ typedef struct DaemonModule {
|
||||
activities. Without it the daemon refuses all of them. */
|
||||
char** auth_users; /* `auth users = a,b`; Wave B credential list */
|
||||
int auth_user_count;
|
||||
/* `max connections = N` (optional per-module cap). 0 means unlimited. The
|
||||
* per-connection child records the selected module in the shared registry
|
||||
* (daemon_limits.c) once the config frame names it, so the cap is enforced
|
||||
* across all forked children; the parent reclaims the slot on SIGCHLD. */
|
||||
int max_connections;
|
||||
char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
char** hosts_deny; /* `hosts deny = a,b`; host access deny patterns */
|
||||
int hosts_deny_count;
|
||||
} DaemonModule;
|
||||
|
||||
/* Global (pre-module) scalar keys. `motd file` is parsed and stored but has
|
||||
@@ -60,6 +69,25 @@ typedef struct DaemonConfGlobals {
|
||||
int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */
|
||||
char* motd_file; /* `motd file`, may be NULL */
|
||||
char* address; /* `address` (optional bind address), may be NULL */
|
||||
int max_connections; /* `max connections`, default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */
|
||||
int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */
|
||||
int max_connections_per_host; /* `max connections per host`, concurrent cap per
|
||||
source IP; default
|
||||
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST (0 =
|
||||
unlimited) */
|
||||
int auth_lockout_threshold; /* `auth lockout threshold`, failed attempts from
|
||||
one source before lockout; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD (0
|
||||
disables) */
|
||||
int auth_lockout_duration_sec; /* `auth lockout duration`, seconds; default
|
||||
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC
|
||||
(0 disables) */
|
||||
char** hosts_allow; /* `hosts allow`; global host access allow patterns */
|
||||
int hosts_allow_count;
|
||||
char** hosts_deny; /* `hosts deny`; global host access deny patterns */
|
||||
int hosts_deny_count;
|
||||
} DaemonConfGlobals;
|
||||
|
||||
typedef struct DaemonConf {
|
||||
@@ -69,6 +97,31 @@ typedef struct DaemonConf {
|
||||
} DaemonConf;
|
||||
|
||||
#define DAEMON_CONF_DEFAULT_PORT 873
|
||||
/* Default global connection cap when `max connections` is absent. Matches the
|
||||
* historical hardcoded listener value. */
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100
|
||||
/* Default `auth failure delay` in milliseconds (0 disables the throttle). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500
|
||||
/* Default `max connections per host` (0 = unlimited). */
|
||||
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST 0
|
||||
/* Default cross-process auth lockout: 10 failed attempts from one source lock
|
||||
* it out for 300 s (0 disables either knob). */
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD 10
|
||||
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC 300
|
||||
/* Upper bound on a `max connections per host` or `auth lockout threshold`
|
||||
* value, so a typo cannot size the shared registry absurdly. */
|
||||
#define DAEMON_CONF_MAX_CONCURRENCY_LIMIT 1000000
|
||||
/* Upper bound on `auth lockout duration` (7 days). */
|
||||
#define DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC 604800
|
||||
/* Largest accepted `auth failure delay`, so a typo cannot pin a connection
|
||||
* child in nanosleep for an absurd time. */
|
||||
/* Bounded well below the socket I/O timeout so a failed-auth child cannot hold
|
||||
* a connection slot for long enough to amplify connection-cap exhaustion. */
|
||||
#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000
|
||||
/* Upper bound on the number of [module] sections, so the shared registry's
|
||||
* per-module counter array stays fixed-size. The parser rejects the next
|
||||
* section past this bound. */
|
||||
#define DAEMON_CONF_MAX_MODULES 256
|
||||
/* Longest accepted config line (excluding the trailing newline). Longer lines
|
||||
* are rejected rather than buffered unboundedly. */
|
||||
#define DAEMON_CONF_MAX_LINE 4096
|
||||
@@ -99,9 +152,31 @@ const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char*
|
||||
bool daemon_module_name_valid(const char* name);
|
||||
|
||||
/* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and
|
||||
* apply it to the global scalars only. Keys are case-insensitive and limited
|
||||
* to the global scalar keys defined by the grammar (port, motd file, address).
|
||||
* apply it to the global keys only. Keys are case-insensitive and limited to
|
||||
* the global keys defined by the grammar (port, motd file, address,
|
||||
* max connections, max connections per host, auth failure delay,
|
||||
* auth lockout threshold, auth lockout duration, hosts allow, hosts deny).
|
||||
* Returns 0 on success, -1 on error (err filled). */
|
||||
int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size);
|
||||
|
||||
/* Host access-control matching (pure; no I/O). `daemon_host_pattern_match`
|
||||
* matches one configured pattern against a numeric peer IP string. Supported
|
||||
* patterns: `*` (match anything), an IPv4/IPv6 literal, an IPv4/IPv6 CIDR
|
||||
* (`10.0.0.0/8`, `2001:db8::/32`), or a glob (`*.example.com`) evaluated with
|
||||
* the same matcher as file globs; a glob only matches a peer string of the
|
||||
* same shape, so a numeric peer never matches a hostname glob. */
|
||||
bool daemon_host_pattern_match(const char* pattern, const char* peer_ip);
|
||||
|
||||
/* rsync-like combined decision over a deny list and an allow list: a matching
|
||||
* deny rejects (deny takes precedence); otherwise, when any allow entries
|
||||
* exist, a peer that matches none is rejected; with no allow entries every
|
||||
* peer not denied is accepted. An empty/unset pair returns true. */
|
||||
bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count,
|
||||
char* const* deny, int deny_count);
|
||||
|
||||
/* True when at least one allow or deny pattern is configured (i.e. an
|
||||
* unprovable peer must fail closed rather than being treated as unrestricted). */
|
||||
bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny,
|
||||
int deny_count);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -0,0 +1,494 @@
|
||||
#include "daemon_limits.h"
|
||||
#include "daemon_conf.h"
|
||||
#include "log.h"
|
||||
#include <arpa/inet.h>
|
||||
#include <netinet/in.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdint.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/mman.h>
|
||||
#include <time.h>
|
||||
|
||||
/* The two module-count bounds must agree: the daemon config parser never
|
||||
* produces more than DAEMON_CONF_MAX_MODULES modules, so the shared registry's
|
||||
* per-module counter array is sized from the same bound. */
|
||||
_Static_assert(DAEMON_LIMITS_MAX_MODULES == DAEMON_CONF_MAX_MODULES,
|
||||
"daemon_limits module bound must match daemon_conf");
|
||||
|
||||
/* Slot lifecycle states (stored in slot_state). */
|
||||
enum {
|
||||
SLOT_FREE = 0,
|
||||
SLOT_CLAIMED = 1,
|
||||
SLOT_REGISTERED = 2,
|
||||
};
|
||||
|
||||
/* The registry header lives at the base of the shared mapping; the pointer
|
||||
* fields point at the arrays carved out of the same mapping. Absolute pointers
|
||||
* remain valid in a forked child because fork() clones the address space and
|
||||
* mapping, so parent and child observe the same virtual addresses. */
|
||||
struct DaemonLimitRegistry {
|
||||
int max_slots;
|
||||
int module_count;
|
||||
int host_slots; /* power of two; 1 when no per-source tracking is needed */
|
||||
int per_host_cap;
|
||||
int lockout_threshold;
|
||||
int lockout_duration_sec;
|
||||
size_t map_size;
|
||||
_Atomic long long host_full_warn; /* last "table full" warning epoch */
|
||||
_Atomic int* slot_state;
|
||||
_Atomic int* slot_pid;
|
||||
_Atomic int* slot_module;
|
||||
_Atomic int* slot_host; /* per-source table bucket, or -1 */
|
||||
_Atomic int* module_active;
|
||||
_Atomic uint64_t* host_key; /* 0 == empty bucket */
|
||||
_Atomic int* host_active;
|
||||
_Atomic int* host_fail;
|
||||
_Atomic long long* host_until; /* epoch seconds the lockout expires */
|
||||
_Atomic long long* host_last_use; /* epoch seconds the bucket was last touched */
|
||||
};
|
||||
|
||||
static size_t round_up(size_t n, size_t align) {
|
||||
return (n + align - 1) & ~(align - 1);
|
||||
}
|
||||
|
||||
static size_t next_pow2(size_t n) {
|
||||
size_t p = 1;
|
||||
while (p < n)
|
||||
p <<= 1;
|
||||
return p;
|
||||
}
|
||||
|
||||
/* Parse a numeric IPv4/IPv6 peer string into family + raw bytes. */
|
||||
static bool parse_peer_ip(const char* peer_ip, int* family, unsigned char* bytes) {
|
||||
if (!peer_ip || *peer_ip == '\0')
|
||||
return false;
|
||||
struct in_addr v4;
|
||||
if (inet_pton(AF_INET, peer_ip, &v4) == 1) {
|
||||
memcpy(bytes, &v4, sizeof(v4));
|
||||
*family = AF_INET;
|
||||
return true;
|
||||
}
|
||||
struct in6_addr v6;
|
||||
if (inet_pton(AF_INET6, peer_ip, &v6) == 1) {
|
||||
memcpy(bytes, &v6, sizeof(v6));
|
||||
*family = AF_INET6;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) {
|
||||
if (ok)
|
||||
*ok = false;
|
||||
unsigned char bytes[16];
|
||||
int family = AF_UNSPEC;
|
||||
if (!parse_peer_ip(peer_ip, &family, bytes))
|
||||
return 0;
|
||||
uint64_t hash = 14695981039346656037ULL ^ (uint64_t)(uint32_t)family;
|
||||
size_t length = family == AF_INET ? 4 : 16;
|
||||
for (size_t i = 0; i < length; i++) {
|
||||
hash ^= bytes[i];
|
||||
hash *= 1099511628211ULL;
|
||||
}
|
||||
if (hash == 0)
|
||||
hash = 0x9e3779b97f4a7c15ULL;
|
||||
if (ok)
|
||||
*ok = true;
|
||||
return hash;
|
||||
}
|
||||
|
||||
/* True when the registry must maintain per-source buckets: either the per-host
|
||||
* cap is configured, or the auth lockout is (threshold AND duration > 0). A
|
||||
* lockout threshold without a duration is a no-op, so it must not size or intern
|
||||
* the table. create(), register() and the lockout paths all agree on this. */
|
||||
static bool registry_tracks_hosts(const DaemonLimitRegistry* registry) {
|
||||
return registry->per_host_cap > 0 ||
|
||||
(registry->lockout_threshold > 0 && registry->lockout_duration_sec > 0);
|
||||
}
|
||||
|
||||
/* Find the bucket holding `peer_ip`, or -1 when it has no entry. Finding a
|
||||
* bucket refreshes its last-use time so the eviction policy sees it as live. */
|
||||
static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], (long long)time(NULL),
|
||||
memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0)
|
||||
return -1; /* no tombstones: an empty bucket ends the probe chain */
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* A bucket with no live connection may be repurposed: immediately when its
|
||||
* lockout deadline has already passed (the review's "expired" case), or after an
|
||||
* idle window when it holds no pending lockout. A bucket with a future lockout
|
||||
* deadline is retained so the lockout actually lasts its configured duration. */
|
||||
static bool host_bucket_reclaimable(DaemonLimitRegistry* registry, size_t idx, long long now) {
|
||||
if (atomic_load_explicit(®istry->host_active[idx], memory_order_relaxed) != 0)
|
||||
return false;
|
||||
long long until = atomic_load_explicit(®istry->host_until[idx], memory_order_relaxed);
|
||||
if (until != 0)
|
||||
return until <= now;
|
||||
long long last_use = atomic_load_explicit(®istry->host_last_use[idx], memory_order_relaxed);
|
||||
/* A bucket whose key is published but whose last_use has not yet been stamped
|
||||
* (last_use == 0) must be treated as live: reclaiming it here would steal a
|
||||
* bucket a racing child just claimed. The claim path also stamps last_use
|
||||
* before publishing the key, so this window cannot persist. */
|
||||
return last_use != 0 && now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC;
|
||||
}
|
||||
|
||||
/* Emit at most one "per-source table full" warning per
|
||||
* DAEMON_LIMITS_HOST_FULL_WARN_SEC across all forked children. Called from a
|
||||
* normal (non-signal) child path, so logging is safe here. */
|
||||
static void host_warn_table_full(DaemonLimitRegistry* registry, long long now) {
|
||||
long long last = atomic_load_explicit(®istry->host_full_warn, memory_order_relaxed);
|
||||
if (last != 0 && now - last < DAEMON_LIMITS_HOST_FULL_WARN_SEC)
|
||||
return;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_full_warn, &last, now,
|
||||
memory_order_relaxed, memory_order_relaxed)) {
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"daemon: per-source registry is full (%d slots) and no bucket can be reclaimed; "
|
||||
"'max connections per host' and the auth lockout are temporarily not enforced for "
|
||||
"new sources (the per-module cap and host ACLs still apply)",
|
||||
registry->host_slots);
|
||||
}
|
||||
}
|
||||
|
||||
/* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two
|
||||
* forked children racing on the same source converge on one bucket.
|
||||
*
|
||||
* When the probe finds no empty bucket it reclaims, via a key CAS, the first
|
||||
* bucket that is reclaimable (expired lockout or idle, and no active
|
||||
* connection) and resets its counters. This bounds the table's lifetime so it
|
||||
* cannot fill permanently and stay fail-open. Returns -1 only when the address
|
||||
* is unparseable or the table is genuinely full of live/locked buckets
|
||||
* (callers fail open: the global/module caps and ACLs still apply). */
|
||||
static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
bool ok = false;
|
||||
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
|
||||
if (!ok)
|
||||
return -1;
|
||||
long long now = (long long)time(NULL);
|
||||
size_t mask = (size_t)registry->host_slots - 1;
|
||||
size_t start = (size_t)(key & mask);
|
||||
/* A couple of passes bound the work: the first normally claims/seeds a bucket;
|
||||
* a lost eviction CAS retries once against the freshly observed table. */
|
||||
for (int pass = 0; pass < 2; pass++) {
|
||||
int evict = -1;
|
||||
uint64_t evict_key = 0;
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t idx = (start + i) & mask;
|
||||
uint64_t current = atomic_load_explicit(®istry->host_key[idx], memory_order_acquire);
|
||||
if (current == key) {
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
return (int)idx;
|
||||
}
|
||||
if (current == 0) {
|
||||
/* Stamp last_use *before* publishing the key so a reclaimer racing the
|
||||
* claim can never observe a claimed bucket with last_use == 0 and
|
||||
* evict it. A pre-stamp is harmless if the CAS loses: the bucket is
|
||||
* either still empty (never inspected for reclaim) or has just been
|
||||
* taken by another source that wants a fresh timestamp anyway. */
|
||||
atomic_store_explicit(®istry->host_last_use[idx], now, memory_order_relaxed);
|
||||
uint64_t expected = 0;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[idx], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
return (int)idx;
|
||||
}
|
||||
if (atomic_load_explicit(®istry->host_key[idx], memory_order_acquire) == key) {
|
||||
return (int)idx;
|
||||
}
|
||||
continue; /* another child won this empty bucket; keep probing */
|
||||
}
|
||||
if (evict < 0 && host_bucket_reclaimable(registry, idx, now)) {
|
||||
evict = (int)idx;
|
||||
evict_key = current;
|
||||
}
|
||||
}
|
||||
if (evict >= 0) {
|
||||
/* Refresh the timestamp before the key changes hands so the reused bucket
|
||||
* is not seen as immediately idle by a racing reclaimer. */
|
||||
atomic_store_explicit(®istry->host_last_use[evict], now, memory_order_relaxed);
|
||||
uint64_t expected = evict_key;
|
||||
if (atomic_compare_exchange_strong_explicit(®istry->host_key[evict], &expected, key,
|
||||
memory_order_acq_rel, memory_order_acquire)) {
|
||||
/* The bucket now belongs to the new source; clear the evicted source's
|
||||
* stale lockout/failure state. */
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_fail[evict], 0, memory_order_relaxed);
|
||||
atomic_store_explicit(®istry->host_until[evict], 0, memory_order_relaxed);
|
||||
/* Two children can race to intern the same brand-new key into different
|
||||
* eviction targets, leaving the table with duplicate buckets for `key`.
|
||||
* Re-scan for the first (canonical) bucket holding `key`; when it
|
||||
* precedes `evict`, drop our duplicate's occupancy and hand back the
|
||||
* canonical bucket so per-source counts are not orphaned on the
|
||||
* duplicate. The duplicate keeps its key, so no tombstone hole is
|
||||
* created and probe chains stay intact; it ages out normally. */
|
||||
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
|
||||
size_t candidate = (start + i) & mask;
|
||||
uint64_t found =
|
||||
atomic_load_explicit(®istry->host_key[candidate], memory_order_acquire);
|
||||
if (found == key) {
|
||||
if (candidate != (size_t)evict) {
|
||||
atomic_store_explicit(®istry->host_active[evict], 0, memory_order_relaxed);
|
||||
return (int)candidate;
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (found == 0)
|
||||
break; /* the key is present at `evict`, so this cannot happen first */
|
||||
}
|
||||
return evict;
|
||||
}
|
||||
continue; /* lost the race; re-probe with fresh observations */
|
||||
}
|
||||
break; /* no free and no reclaimable bucket: genuinely full */
|
||||
}
|
||||
host_warn_table_full(registry, now);
|
||||
return -1;
|
||||
}
|
||||
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec) {
|
||||
if (max_slots < DAEMON_LIMITS_MIN_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MIN_SLOTS;
|
||||
if (max_slots > DAEMON_LIMITS_MAX_SLOTS)
|
||||
max_slots = DAEMON_LIMITS_MAX_SLOTS;
|
||||
if (module_count < 1)
|
||||
module_count = 1;
|
||||
if (module_count > DAEMON_LIMITS_MAX_MODULES)
|
||||
module_count = DAEMON_LIMITS_MAX_MODULES;
|
||||
if (per_host_cap < 0)
|
||||
per_host_cap = 0;
|
||||
if (lockout_threshold < 0)
|
||||
lockout_threshold = 0;
|
||||
if (lockout_duration_sec < 0)
|
||||
lockout_duration_sec = 0;
|
||||
|
||||
bool need_hosts = per_host_cap > 0 || (lockout_threshold > 0 && lockout_duration_sec > 0);
|
||||
int host_slots = 1;
|
||||
if (need_hosts) {
|
||||
size_t want = (size_t)max_slots * 4;
|
||||
if (want < 64)
|
||||
want = 64;
|
||||
if (want > DAEMON_LIMITS_MAX_HOST_SLOTS)
|
||||
want = DAEMON_LIMITS_MAX_HOST_SLOTS;
|
||||
host_slots = (int)next_pow2(want);
|
||||
}
|
||||
|
||||
size_t header = round_up(sizeof(DaemonLimitRegistry), 16);
|
||||
size_t slot_bytes =
|
||||
round_up((size_t)max_slots * sizeof(_Atomic int), 16) * 4; /* state,pid,module,host */
|
||||
size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16);
|
||||
size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16);
|
||||
size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2;
|
||||
size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16) * 2;
|
||||
size_t total =
|
||||
header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16;
|
||||
|
||||
void* map = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0);
|
||||
if (map == MAP_FAILED)
|
||||
return NULL;
|
||||
memset(map, 0, total);
|
||||
|
||||
DaemonLimitRegistry* registry = (DaemonLimitRegistry*)map;
|
||||
registry->max_slots = max_slots;
|
||||
registry->module_count = module_count;
|
||||
registry->host_slots = host_slots;
|
||||
registry->per_host_cap = per_host_cap;
|
||||
registry->lockout_threshold = lockout_threshold;
|
||||
registry->lockout_duration_sec = lockout_duration_sec;
|
||||
registry->map_size = total;
|
||||
|
||||
unsigned char* cursor = (unsigned char*)map + header;
|
||||
registry->slot_state = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_pid = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_module = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->slot_host = (atomic_int*)cursor;
|
||||
cursor += (size_t)max_slots * sizeof(_Atomic int);
|
||||
registry->module_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)module_count * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_key = (_Atomic uint64_t*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic uint64_t);
|
||||
registry->host_active = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
registry->host_fail = (atomic_int*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic int);
|
||||
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
|
||||
registry->host_until = (atomic_llong*)cursor;
|
||||
cursor += (size_t)host_slots * sizeof(_Atomic long long);
|
||||
registry->host_last_use = (atomic_llong*)cursor;
|
||||
|
||||
for (int i = 0; i < max_slots; i++) {
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
}
|
||||
return registry;
|
||||
}
|
||||
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
munmap(registry, registry->map_size);
|
||||
}
|
||||
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
int expected = SLOT_FREE;
|
||||
if (atomic_compare_exchange_strong(®istry->slot_state[i], &expected, SLOT_CLAIMED)) {
|
||||
atomic_store(®istry->slot_pid[i], 0);
|
||||
atomic_store(®istry->slot_module[i], -1);
|
||||
atomic_store(®istry->slot_host[i], -1);
|
||||
return i;
|
||||
}
|
||||
}
|
||||
return DAEMON_LIMITS_NO_SLOT;
|
||||
}
|
||||
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_store(®istry->slot_pid[slot], (int)pid);
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return;
|
||||
atomic_exchange_explicit(®istry->slot_state[slot], SLOT_FREE, memory_order_acq_rel);
|
||||
atomic_store_explicit(®istry->slot_pid[slot], 0, memory_order_relaxed);
|
||||
/* The module/host occupancy arrays are derived from the slot table; do not
|
||||
* decrement here or a SIGKILL between a child's increment and its REGISTERED
|
||||
* publish would leak a count. Callers that need the derived counts call
|
||||
* daemon_limits_recompute. */
|
||||
}
|
||||
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) {
|
||||
if (!registry || pid <= 0)
|
||||
return;
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load(®istry->slot_state[i]) == SLOT_FREE)
|
||||
continue;
|
||||
if (atomic_load(®istry->slot_pid[i]) == (int)pid) {
|
||||
daemon_limits_reclaim_slot(registry, i);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry) {
|
||||
if (!registry)
|
||||
return;
|
||||
/* Zero the derived arrays, then re-derive solely from the REGISTERED slots.
|
||||
* A child that was SIGKILLed after incrementing a counter but before
|
||||
* publishing REGISTERED is not counted, and its leaked increment is erased by
|
||||
* the zeroing, so the leak cannot persist. */
|
||||
for (int m = 0; m < registry->module_count; m++)
|
||||
atomic_store_explicit(®istry->module_active[m], 0, memory_order_relaxed);
|
||||
for (int h = 0; h < registry->host_slots; h++)
|
||||
atomic_store_explicit(®istry->host_active[h], 0, memory_order_relaxed);
|
||||
for (int i = 0; i < registry->max_slots; i++) {
|
||||
if (atomic_load_explicit(®istry->slot_state[i], memory_order_acquire) != SLOT_REGISTERED)
|
||||
continue;
|
||||
int module = atomic_load_explicit(®istry->slot_module[i], memory_order_relaxed);
|
||||
if (module >= 0 && module < registry->module_count)
|
||||
atomic_fetch_add_explicit(®istry->module_active[module], 1, memory_order_relaxed);
|
||||
int host = atomic_load_explicit(®istry->slot_host[i], memory_order_relaxed);
|
||||
if (host >= 0 && host < registry->host_slots)
|
||||
atomic_fetch_add_explicit(®istry->host_active[host], 1, memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap) {
|
||||
if (!registry || slot < 0 || slot >= registry->max_slots)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (module_index < 0 || module_index >= registry->module_count)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
if (atomic_load_explicit(®istry->slot_state[slot], memory_order_acquire) != SLOT_CLAIMED)
|
||||
return DAEMON_LIMIT_UNAVAILABLE;
|
||||
|
||||
int host = -1;
|
||||
if (registry_tracks_hosts(registry))
|
||||
host = host_intern(registry, peer_ip);
|
||||
|
||||
int module_count = atomic_fetch_add(®istry->module_active[module_index], 1) + 1;
|
||||
if (module_cap > 0 && module_count > module_cap) {
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_MODULE_FULL;
|
||||
}
|
||||
if (host >= 0) {
|
||||
int host_count = atomic_fetch_add(®istry->host_active[host], 1) + 1;
|
||||
if (registry->per_host_cap > 0 && host_count > registry->per_host_cap) {
|
||||
atomic_fetch_sub(®istry->host_active[host], 1);
|
||||
atomic_fetch_sub(®istry->module_active[module_index], 1);
|
||||
return DAEMON_LIMIT_HOST_FULL;
|
||||
}
|
||||
}
|
||||
atomic_store(®istry->slot_module[slot], module_index);
|
||||
atomic_store(®istry->slot_host[slot], host);
|
||||
atomic_store_explicit(®istry->slot_state[slot], SLOT_REGISTERED, memory_order_release);
|
||||
return DAEMON_LIMIT_OK;
|
||||
}
|
||||
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return false;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return false;
|
||||
long long until = atomic_load(®istry->host_until[bucket]);
|
||||
long long now = (long long)time(NULL);
|
||||
if (until > now) {
|
||||
if (seconds_remaining)
|
||||
*seconds_remaining = (int)(until - now);
|
||||
return true;
|
||||
}
|
||||
if (until != 0) {
|
||||
/* The previous lockout has expired: clear the stale counter so the source
|
||||
* gets a fresh allowance. */
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
|
||||
return;
|
||||
int bucket = host_intern(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
int failures = atomic_fetch_add(®istry->host_fail[bucket], 1) + 1;
|
||||
if (failures >= registry->lockout_threshold) {
|
||||
long long now = (long long)time(NULL);
|
||||
atomic_store(®istry->host_until[bucket], now + (long long)registry->lockout_duration_sec);
|
||||
}
|
||||
}
|
||||
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip) {
|
||||
if (!registry)
|
||||
return;
|
||||
int bucket = host_lookup(registry, peer_ip);
|
||||
if (bucket < 0)
|
||||
return;
|
||||
atomic_store(®istry->host_fail[bucket], 0);
|
||||
atomic_store(®istry->host_until[bucket], 0);
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
#ifndef DAEMON_LIMITS_H
|
||||
#define DAEMON_LIMITS_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
|
||||
/* Cross-process daemon connection registry.
|
||||
*
|
||||
* The daemon listener forks ONE child per accepted connection, so any
|
||||
* per-module / per-source accounting must live in state shared across the
|
||||
* forked children. This module owns a fixed-size registry carved out of an
|
||||
* anonymous shared mapping (mmap(MAP_SHARED | MAP_ANONYMOUS)) created by the
|
||||
* accept-loop PARENT before it forks; every child inherits the mapping (and the
|
||||
* pointer to it) across fork().
|
||||
*
|
||||
* Rules:
|
||||
* - ONLY C11 atomics (atomic_*); never mtx_t/pthread locks, which can deadlock
|
||||
* in a forked child if another thread held them at fork time.
|
||||
* - No heap allocation after fork: the mapping is fixed-size and all access is
|
||||
* atomic load/store/CAS over preallocated arrays.
|
||||
*
|
||||
* Slot lifecycle (the parent reclaims even when a child is SIGKILLed):
|
||||
* FREE --(parent claim_slot)--> CLAIMED
|
||||
* CLAIMED --(child register)--> REGISTERED
|
||||
* any --(parent reclaim)--> FREE
|
||||
* The child records its module index and per-source bucket into the slot before
|
||||
* publishing REGISTERED; the parent's SIGCHLD handler matches the reaped pid to
|
||||
* the slot and, when REGISTERED, decrements the module/per-source counters.
|
||||
* A child killed before registering holds no counts, so reclaiming a CLAIMED
|
||||
* slot only frees the slot.
|
||||
*
|
||||
* Per-source identity is the normalized numeric peer IP (IPv4-mapped IPv6 is
|
||||
* already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an
|
||||
* open-addressed, linear-probing table keyed by a 64-bit hash. The same table
|
||||
* also carries the cross-process auth-failure counter and lockout deadline.
|
||||
*
|
||||
* Per-source table lifetime: a bucket's key is never cleared back to empty (that
|
||||
* would break every later probe chain that passed through it). Instead the
|
||||
* table has a bounded-lifetime eviction policy: when no empty bucket exists, the
|
||||
* first bucket that is reclaimable -- no active connection AND (its lockout
|
||||
* deadline has passed OR it has been idle for
|
||||
* DAEMON_LIMITS_HOST_EVICT_IDLE_SEC) -- is atomically repurposed for the new
|
||||
* source via a CAS of its key, and its counters are reset. The table therefore
|
||||
* cannot fill permanently, and a full table degrades to fail-open for the
|
||||
* per-source cap/lockout of new sources (the per-module cap and host ACLs still
|
||||
* apply) instead of staying fail-open forever. A rate-limited warning is logged
|
||||
* on the fail-open path. The eviction race with a concurrent
|
||||
* registration/reclaim on the same bucket is benign: it can at worst lose one
|
||||
* source's counter (fail-open), never corrupt memory or the module caps.
|
||||
*/
|
||||
|
||||
typedef struct DaemonLimitRegistry DaemonLimitRegistry;
|
||||
|
||||
/* Result of a per-connection admission check. */
|
||||
typedef enum {
|
||||
DAEMON_LIMIT_OK = 0, /* admitted; slot is now REGISTERED */
|
||||
DAEMON_LIMIT_MODULE_FULL, /* module's `max connections` cap reached */
|
||||
DAEMON_LIMIT_HOST_FULL, /* global `max connections per host` cap reached */
|
||||
DAEMON_LIMIT_UNAVAILABLE, /* registry/slot unusable (caller fails open) */
|
||||
} DaemonLimitResult;
|
||||
|
||||
/* Bounds for registry sizing. A slot is one concurrently live child. */
|
||||
#define DAEMON_LIMITS_MIN_SLOTS 16
|
||||
#define DAEMON_LIMITS_MAX_SLOTS 65536
|
||||
#define DAEMON_LIMITS_MAX_HOST_SLOTS 65536
|
||||
#define DAEMON_LIMITS_NO_SLOT (-1)
|
||||
/* Upper bound on `module_count`, matching daemon_conf.h's DAEMON_CONF_MAX_MODULES
|
||||
* (asserted in daemon_limits.c) so a caller can never size the per-module counter
|
||||
* array larger than the config parser can produce. */
|
||||
#define DAEMON_LIMITS_MAX_MODULES 256
|
||||
|
||||
/* Per-source table lifetime: a bucket with no active connection and no pending
|
||||
* lockout is reclaimable once it has been idle this long, so a flood of distinct
|
||||
* sources cannot pin the table full forever. A bucket whose lockout deadline
|
||||
* has passed is reclaimable immediately (independent of this idle window). */
|
||||
#define DAEMON_LIMITS_HOST_EVICT_IDLE_SEC 300
|
||||
/* Minimum spacing between "per-source table is full" warnings, so a table-full
|
||||
* attack cannot flood the log. */
|
||||
#define DAEMON_LIMITS_HOST_FULL_WARN_SEC 60
|
||||
|
||||
/* Create the shared registry in the calling (parent) process. `max_slots` is
|
||||
* the number of concurrently live children to track (clamped to
|
||||
* [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the
|
||||
* number of daemon modules (clamped to
|
||||
* [1, DAEMON_LIMITS_MAX_MODULES]); `per_host_cap` and the lockout pair come
|
||||
* from the daemon config (0 disables). Returns NULL on failure (e.g. mmap
|
||||
* allocation); callers must degrade gracefully (global cap + ACLs still
|
||||
* apply). */
|
||||
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
|
||||
int lockout_threshold, int lockout_duration_sec);
|
||||
|
||||
/* Unmap the registry. Only the creating process may call this. */
|
||||
void daemon_limits_destroy(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Parent side: reserve a slot for the next fork. Returns the slot index or
|
||||
* DAEMON_LIMITS_NO_SLOT when every slot is in use. */
|
||||
int daemon_limits_claim_slot(DaemonLimitRegistry* registry);
|
||||
/* Parent side: record the forked child's pid in a claimed slot. */
|
||||
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid);
|
||||
/* Parent side: release a slot. The slot becomes FREE; the module/per-source
|
||||
* occupancy arrays are DERIVED state and are only refreshed by
|
||||
* daemon_limits_recompute, which callers must invoke afterwards when they rely
|
||||
* on the derived counts (the SIGCHLD handler batches one recompute for the whole
|
||||
* reap). Idempotent. */
|
||||
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot);
|
||||
/* Parent SIGCHLD side: release the slot owned by `pid` (no-op when not found).
|
||||
* Like reclaim_slot this does not touch the derived occupancy arrays; call
|
||||
* daemon_limits_recompute after a batch of releases. */
|
||||
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid);
|
||||
|
||||
/* Parent side (async-signal-safe; atomics only, no malloc/log): rebuild
|
||||
* module_active[] / host_active[] from scratch by scanning the REGISTERED slots.
|
||||
* The slot table is the single source of truth, so this self-heals any
|
||||
* count leaked by a child that was SIGKILLed mid-registration (it zeroes the
|
||||
* arrays and re-derives them). Bounded by max_slots + host_slots. A
|
||||
* registration racing this call can be transiently undercounted until the next
|
||||
* recompute, which can only relax a cap briefly -- never corrupt memory. */
|
||||
void daemon_limits_recompute(DaemonLimitRegistry* registry);
|
||||
|
||||
/* Child side: admit the connection for `module_index` from `peer_ip`. Always
|
||||
* tracks the module/per-source occupancy (so the parent's reclaim is
|
||||
* symmetric); when `module_cap` > 0 it additionally enforces the per-module
|
||||
* cap. A NULL/empty or non-numeric `peer_ip` skips the per-source track (the
|
||||
* callers use that to exempt a trusted loopback peer from the per-host cap; the
|
||||
* per-module cap still applies). Returns DAEMON_LIMIT_OK and publishes the
|
||||
* slot, or a refusal reason. */
|
||||
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
|
||||
const char* peer_ip, int module_cap);
|
||||
|
||||
/* Child side: true when `peer_ip` is currently locked out after too many failed
|
||||
* authentications. `seconds_remaining` may be NULL. */
|
||||
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
|
||||
int* seconds_remaining);
|
||||
/* Child side: count one failed authentication for `peer_ip`; once the threshold
|
||||
* is reached the source is locked out for the configured duration. */
|
||||
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
/* Child side: clear the failure counter/lockout for a source that authenticated
|
||||
* successfully (no-op when the source has no table entry). */
|
||||
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip);
|
||||
|
||||
/* Pure helper: 64-bit FNV-1a hash of a numeric peer IP plus its family, used to
|
||||
* index the per-source table. *ok is set false (and 0 returned) for a NULL or
|
||||
* non-numeric address. Exposed for unit testing. */
|
||||
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok);
|
||||
|
||||
#endif
|
||||
+7
-1
@@ -23,6 +23,7 @@ Data* data_create_reserve(size_t size) {
|
||||
d->data = NULL;
|
||||
d->size = size;
|
||||
d->protocol_charge = 0;
|
||||
d->owner = NULL;
|
||||
return d;
|
||||
}
|
||||
|
||||
@@ -36,14 +37,19 @@ Data* data_create(void* data, size_t data_size) {
|
||||
new_data->data = data;
|
||||
new_data->size = data_size;
|
||||
new_data->protocol_charge = 0;
|
||||
new_data->owner = NULL;
|
||||
return new_data;
|
||||
}
|
||||
|
||||
void data_destroy(Data* data) {
|
||||
if (data == NULL)
|
||||
return;
|
||||
if (data->protocol_charge != 0)
|
||||
if (data->protocol_charge != 0) {
|
||||
if (data->owner != NULL)
|
||||
protocol_release_memory_for_session(data->owner, data->protocol_charge);
|
||||
else
|
||||
protocol_release_memory(data->protocol_charge);
|
||||
}
|
||||
free(data->data);
|
||||
free(data);
|
||||
}
|
||||
|
||||
@@ -3,11 +3,25 @@
|
||||
|
||||
#include <stdlib.h>
|
||||
|
||||
/* Forward declaration for the connection budget a received Data is charged
|
||||
* against; defined in protocol.h (which includes this header). */
|
||||
typedef struct ProtocolSession ProtocolSession;
|
||||
|
||||
typedef struct {
|
||||
void* data;
|
||||
size_t size;
|
||||
/* Non-zero only for a buffer charged to the protocol connection budget. */
|
||||
size_t protocol_charge;
|
||||
/* Session whose budget `protocol_charge` was reserved from. When non-NULL,
|
||||
* the charge is returned to this session directly, regardless of which
|
||||
* session (if any) is bound to the destroying thread. owner is not
|
||||
* guaranteed to be set whenever protocol_charge is non-zero: it is NULL for
|
||||
* uncharged Data and for Data that has no recorded owner, in which case any
|
||||
* charge falls back to the session bound at destroy time.
|
||||
*
|
||||
* Lifetime contract: a Data with a non-NULL owner must not outlive that
|
||||
* ProtocolSession -- data_destroy dereferences owner to return the charge. */
|
||||
ProtocolSession* owner;
|
||||
} Data;
|
||||
|
||||
Data* data_create_empty(size_t data_size);
|
||||
@@ -15,5 +29,9 @@ Data* data_create_reserve(size_t size);
|
||||
Data* data_create(void* data, size_t data_size);
|
||||
void data_destroy(Data* data);
|
||||
void protocol_release_memory(size_t charge);
|
||||
/* Release `charge` against `session` directly instead of the thread-local bound
|
||||
* session. Used by data_destroy to honor Data.owner; `session` must outlive
|
||||
* the Data whose charge is being returned. A NULL session is a no-op. */
|
||||
void protocol_release_memory_for_session(ProtocolSession* session, size_t charge);
|
||||
|
||||
#endif
|
||||
|
||||
@@ -264,6 +264,20 @@ static bool delay_publish_entry(DelayUpdatesContext* context, const Config* conf
|
||||
const StagedFileEntry* entry) {
|
||||
if (!delay_publish_backup(context, config, entry))
|
||||
return false;
|
||||
/* --force: an incoming regular file/symlink may replace a destination
|
||||
DIRECTORY (possibly non-empty). The immediate-install path handles this in
|
||||
file_receive; a --delay-updates run stages elsewhere and only discovers the
|
||||
blocking directory here, so clear it before the rename (rsync's
|
||||
"could not make way for new regular file" without --force). */
|
||||
if (config && config->force_delete && file_directory_exists_secure(entry->final_path)) {
|
||||
if (!file_remove_tree_secure(entry->final_path)) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
log_message(LOG_LEVEL_ERROR, "could not remove destination directory blocking '%s': %s",
|
||||
escaped ? escaped : "<allocation failed>", strerror(errno));
|
||||
free(escaped);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
if (!file_rename_secure(entry->staged_path, entry->final_path)) {
|
||||
if (errno == EXDEV) {
|
||||
char* escaped = output_escape(entry->final_path, false);
|
||||
|
||||
@@ -18,8 +18,9 @@ typedef struct {
|
||||
/* Receiver-side --delay-updates staging registry. All successfully written
|
||||
files land under a private staging directory inside the receive root and are
|
||||
atomically renamed into their final destination only at the very end of the
|
||||
transfer. A single PipelineContextReceiver has exactly one writer thread,
|
||||
but the registry is still mutex-protected so the same object can be safely
|
||||
transfer. A single receiver pipeline (see src/server/receiver_pipeline.h)
|
||||
has exactly one writer thread, but the registry is still mutex-protected so
|
||||
the same object can be safely
|
||||
shared with the publish/cleanup phase that runs after the threads join. */
|
||||
typedef struct DelayUpdatesContext {
|
||||
char* root_directory; /* receive root the staging dir lives under */
|
||||
|
||||
@@ -0,0 +1,893 @@
|
||||
#include "delete_plan.h"
|
||||
|
||||
#include "charset.h"
|
||||
#include "delay_updates.h"
|
||||
#include "file.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <dirent.h>
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <limits.h>
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/stat.h>
|
||||
#include <unistd.h>
|
||||
|
||||
/* Mirrors MAX_SERVER_DELETE_COUNT in file_receive.c: the server's hard bound on
|
||||
* the number of entries one deletion commit may remove. A client
|
||||
* --max-delete=NUM smaller than this replaces it for the run. */
|
||||
#define DELETE_PLAN_SERVER_LIMIT 100000U
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* Sender: plan builder */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
typedef struct PlanNode {
|
||||
char* dir;
|
||||
ArrayList* files; /* basenames kept directly in dir */
|
||||
ArrayList* dirs; /* basenames of kept child directories */
|
||||
bool sent;
|
||||
struct PlanNode* hash_next;
|
||||
} PlanNode;
|
||||
|
||||
struct DeletePlanSender {
|
||||
PlanNode** buckets;
|
||||
size_t capacity;
|
||||
size_t count;
|
||||
bool config_sent;
|
||||
bool all_synced;
|
||||
const ArrayList* synced_dirs;
|
||||
/* Owned by the caller's synced_dirs list; non-NULL only for a general -R
|
||||
transfer, where it is the destination prefix the delete walk is confined
|
||||
to. NULL means the whole receive root (or a --files-from scope). */
|
||||
const char* walk_root;
|
||||
const ArrayList* protected_prefixes;
|
||||
const ArrayList* size_skipped;
|
||||
const ArrayList* missing_args;
|
||||
size_t entries;
|
||||
/* Transmitted FILE entries only. The caller's "empty scan" safety guard keys
|
||||
off this (an I/O error that hid every file must refuse to delete even when
|
||||
some directories were traversed), so directory keep entries do not count. */
|
||||
size_t file_entries;
|
||||
};
|
||||
|
||||
static size_t plan_hash(const char* key) {
|
||||
size_t h = 5381;
|
||||
for (const unsigned char* p = (const unsigned char*)key; *p; p++)
|
||||
h = ((h << 5) + h) + *p;
|
||||
return h;
|
||||
}
|
||||
|
||||
static bool list_contains_str(const ArrayList* list, const char* value) {
|
||||
if (!list)
|
||||
return false;
|
||||
for (int i = 0; i < list->size; i++) {
|
||||
if (strcmp((const char*)list->items[i], value) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool list_add_str_unique(ArrayList* list, const char* value) {
|
||||
if (!list || !value)
|
||||
return false;
|
||||
if (list_contains_str(list, value))
|
||||
return true;
|
||||
char* copy = str_dup(value);
|
||||
if (!copy)
|
||||
return false;
|
||||
if (!array_list_add(list, copy)) {
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
DeletePlanSender* delete_plan_sender_create(void) {
|
||||
DeletePlanSender* sender = calloc(1, sizeof(DeletePlanSender));
|
||||
if (!sender)
|
||||
return NULL;
|
||||
sender->capacity = 64;
|
||||
sender->buckets = calloc(sender->capacity, sizeof(PlanNode*));
|
||||
if (!sender->buckets) {
|
||||
free(sender);
|
||||
return NULL;
|
||||
}
|
||||
sender->all_synced = true;
|
||||
return sender;
|
||||
}
|
||||
|
||||
static void plan_node_destroy(PlanNode* node) {
|
||||
if (!node)
|
||||
return;
|
||||
free(node->dir);
|
||||
array_list_delete(node->files);
|
||||
array_list_delete(node->dirs);
|
||||
free(node);
|
||||
}
|
||||
|
||||
void delete_plan_sender_destroy(DeletePlanSender* sender) {
|
||||
if (!sender)
|
||||
return;
|
||||
for (size_t i = 0; i < sender->capacity; i++) {
|
||||
PlanNode* node = sender->buckets[i];
|
||||
while (node) {
|
||||
PlanNode* next = node->hash_next;
|
||||
plan_node_destroy(node);
|
||||
node = next;
|
||||
}
|
||||
}
|
||||
free(sender->buckets);
|
||||
free(sender);
|
||||
}
|
||||
|
||||
static PlanNode* plan_find(const DeletePlanSender* sender, const char* dir) {
|
||||
size_t index = plan_hash(dir) & (sender->capacity - 1);
|
||||
for (PlanNode* node = sender->buckets[index]; node; node = node->hash_next) {
|
||||
if (strcmp(node->dir, dir) == 0)
|
||||
return node;
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
static bool plan_grow(DeletePlanSender* sender) {
|
||||
size_t new_capacity = sender->capacity * 2;
|
||||
PlanNode** buckets = calloc(new_capacity, sizeof(PlanNode*));
|
||||
if (!buckets)
|
||||
return false;
|
||||
for (size_t i = 0; i < sender->capacity; i++) {
|
||||
PlanNode* node = sender->buckets[i];
|
||||
while (node) {
|
||||
PlanNode* next = node->hash_next;
|
||||
size_t index = plan_hash(node->dir) & (new_capacity - 1);
|
||||
node->hash_next = buckets[index];
|
||||
buckets[index] = node;
|
||||
node = next;
|
||||
}
|
||||
}
|
||||
free(sender->buckets);
|
||||
sender->buckets = buckets;
|
||||
sender->capacity = new_capacity;
|
||||
return true;
|
||||
}
|
||||
|
||||
static PlanNode* plan_ensure(DeletePlanSender* sender, const char* dir) {
|
||||
PlanNode* node = plan_find(sender, dir);
|
||||
if (node)
|
||||
return node;
|
||||
if (sender->count + 1 > sender->capacity * 3 / 4 && !plan_grow(sender))
|
||||
return NULL;
|
||||
node = calloc(1, sizeof(PlanNode));
|
||||
if (!node)
|
||||
return NULL;
|
||||
node->dir = str_dup(dir);
|
||||
node->files = array_list_create(free);
|
||||
node->dirs = array_list_create(free);
|
||||
if (!node->dir || !node->files || !node->dirs) {
|
||||
plan_node_destroy(node);
|
||||
return NULL;
|
||||
}
|
||||
size_t index = plan_hash(dir) & (sender->capacity - 1);
|
||||
node->hash_next = sender->buckets[index];
|
||||
sender->buckets[index] = node;
|
||||
sender->count++;
|
||||
return node;
|
||||
}
|
||||
|
||||
static char* path_parent_dir(const char* path) {
|
||||
const char* slash = strrchr(path, '/');
|
||||
if (!slash)
|
||||
return str_dup(".");
|
||||
if (slash == path)
|
||||
return str_dup(".");
|
||||
size_t len = (size_t)(slash - path);
|
||||
char* parent = malloc(len + 1);
|
||||
if (!parent)
|
||||
return NULL;
|
||||
memcpy(parent, path, len);
|
||||
parent[len] = '\0';
|
||||
return parent;
|
||||
}
|
||||
|
||||
static char* path_base_name(const char* path) {
|
||||
const char* slash = strrchr(path, '/');
|
||||
return str_dup(slash ? slash + 1 : path);
|
||||
}
|
||||
|
||||
/* Copy `path`, stripping a leading '/' and any trailing '/'. */
|
||||
static char* plan_clean_path(const char* path) {
|
||||
while (*path == '/')
|
||||
path++;
|
||||
size_t len = strlen(path);
|
||||
while (len > 0 && path[len - 1] == '/')
|
||||
len--;
|
||||
char* clean = malloc(len + 1);
|
||||
if (!clean)
|
||||
return NULL;
|
||||
memcpy(clean, path, len);
|
||||
clean[len] = '\0';
|
||||
return clean;
|
||||
}
|
||||
|
||||
static bool plan_ensure_ancestors(DeletePlanSender* sender, const char* dir) {
|
||||
char* current = str_dup(dir);
|
||||
if (!current)
|
||||
return false;
|
||||
bool ok = true;
|
||||
while (strcmp(current, ".") != 0) {
|
||||
char* parent = path_parent_dir(current);
|
||||
char* base = path_base_name(current);
|
||||
PlanNode* parent_node = parent ? plan_ensure(sender, parent) : NULL;
|
||||
if (!parent || !base || !parent_node || !list_add_str_unique(parent_node->dirs, base)) {
|
||||
ok = false;
|
||||
free(parent);
|
||||
free(base);
|
||||
break;
|
||||
}
|
||||
free(base);
|
||||
free(current);
|
||||
current = parent;
|
||||
}
|
||||
free(current);
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool delete_plan_sender_add(DeletePlanSender* sender, const char* path, bool is_dir) {
|
||||
if (!sender || !path)
|
||||
return false;
|
||||
char* clean = plan_clean_path(path);
|
||||
if (!clean)
|
||||
return false;
|
||||
if (*clean == '\0') {
|
||||
free(clean);
|
||||
return true;
|
||||
}
|
||||
char* parent = path_parent_dir(clean);
|
||||
char* base = path_base_name(clean);
|
||||
PlanNode* parent_node = parent ? plan_ensure(sender, parent) : NULL;
|
||||
bool ok = parent && base && parent_node;
|
||||
if (ok) {
|
||||
if (is_dir) {
|
||||
ok = list_add_str_unique(parent_node->dirs, base) && plan_ensure(sender, clean) != NULL;
|
||||
} else {
|
||||
ok = list_add_str_unique(parent_node->files, base);
|
||||
}
|
||||
}
|
||||
if (ok)
|
||||
ok = plan_ensure_ancestors(sender, parent);
|
||||
if (ok) {
|
||||
sender->entries++;
|
||||
if (!is_dir)
|
||||
sender->file_entries++;
|
||||
}
|
||||
free(clean);
|
||||
free(parent);
|
||||
free(base);
|
||||
return ok;
|
||||
}
|
||||
|
||||
void delete_plan_sender_finalize(DeletePlanSender* sender, const ArrayList* synced_dirs,
|
||||
const char* walk_root) {
|
||||
if (!sender)
|
||||
return;
|
||||
sender->synced_dirs = synced_dirs;
|
||||
sender->all_synced = synced_dirs == NULL && walk_root == NULL;
|
||||
sender->walk_root = walk_root;
|
||||
}
|
||||
|
||||
bool delete_plan_sender_empty(const DeletePlanSender* sender) {
|
||||
return !sender || sender->file_entries == 0;
|
||||
}
|
||||
|
||||
void delete_plan_sender_set_config(DeletePlanSender* sender, const ArrayList* protected_prefixes,
|
||||
const ArrayList* size_skipped, const ArrayList* missing_args) {
|
||||
if (!sender)
|
||||
return;
|
||||
sender->protected_prefixes = protected_prefixes;
|
||||
sender->size_skipped = size_skipped;
|
||||
sender->missing_args = missing_args;
|
||||
}
|
||||
|
||||
/* True when `dir` is `root` itself or a descendant of it (path-component
|
||||
* aware, so "foo" does not match "foobar"). */
|
||||
static bool path_at_or_under(const char* dir, const char* root) {
|
||||
if (!dir || !root)
|
||||
return false;
|
||||
size_t n = strlen(root);
|
||||
return strncmp(dir, root, n) == 0 && (dir[n] == '\0' || dir[n] == '/');
|
||||
}
|
||||
|
||||
static bool plan_is_allowed(const DeletePlanSender* sender, const char* dir) {
|
||||
if (sender->all_synced)
|
||||
return true;
|
||||
if (sender->walk_root)
|
||||
return path_at_or_under(dir, sender->walk_root);
|
||||
return list_contains_str(sender->synced_dirs, dir);
|
||||
}
|
||||
|
||||
static int send_str_section(int fd, const ArrayList* list) {
|
||||
int count = list ? list->size : 0;
|
||||
if (!send_int(fd, count))
|
||||
return -1;
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (!send_wire_str(fd, (const char*)list->items[i]))
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int send_plan_node(int fd, DeletePlanSender* sender, PlanNode* node) {
|
||||
if (!send_status(fd, STATUS_DELETE_PLAN))
|
||||
return -1;
|
||||
if (!send_int(fd, sender->config_sent ? 0 : 1))
|
||||
return -1;
|
||||
if (!sender->config_sent) {
|
||||
if (send_str_section(fd, sender->protected_prefixes) != 0 ||
|
||||
send_str_section(fd, sender->size_skipped) != 0 ||
|
||||
send_str_section(fd, sender->missing_args) != 0)
|
||||
return -1;
|
||||
sender->config_sent = true;
|
||||
}
|
||||
if (!send_wire_str(fd, node->dir))
|
||||
return -1;
|
||||
if (send_str_section(fd, node->dirs) != 0 || send_str_section(fd, node->files) != 0)
|
||||
return -1;
|
||||
node->sent = true;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int send_prefix_plan(int fd, DeletePlanSender* sender, const char* dir) {
|
||||
PlanNode* node = plan_find(sender, dir);
|
||||
if (!node || node->sent)
|
||||
return 0;
|
||||
if (!plan_is_allowed(sender, dir))
|
||||
return 0;
|
||||
return send_plan_node(fd, sender, node);
|
||||
}
|
||||
|
||||
int delete_plan_send_root(int fd, DeletePlanSender* sender) {
|
||||
if (!sender)
|
||||
return -1;
|
||||
const char* root = sender->walk_root ? sender->walk_root : ".";
|
||||
if (!plan_ensure(sender, root))
|
||||
return -1;
|
||||
return send_prefix_plan(fd, sender, root);
|
||||
}
|
||||
|
||||
int delete_plan_send_for_path(int fd, DeletePlanSender* sender, const char* path, bool is_dir) {
|
||||
if (!sender || !path)
|
||||
return -1;
|
||||
char* clean = plan_clean_path(path);
|
||||
if (!clean)
|
||||
return -1;
|
||||
/* The walk root (the -R prefix, or ".") is sent up front by
|
||||
delete_plan_send_root(); never emit the receive-root plan for a scoped -R
|
||||
run, whose "." keep list would delete the prefix's siblings. */
|
||||
int rc = sender->walk_root ? 0 : send_prefix_plan(fd, sender, ".");
|
||||
if (rc == 0 && *clean != '\0') {
|
||||
size_t len = strlen(clean);
|
||||
size_t end = len;
|
||||
if (!is_dir) {
|
||||
const char* slash = strrchr(clean, '/');
|
||||
end = slash ? (size_t)(slash - clean) : 0;
|
||||
}
|
||||
for (size_t i = 1; i <= end && rc == 0; i++) {
|
||||
if (i == end || clean[i] == '/') {
|
||||
char* prefix = malloc(i + 1);
|
||||
if (!prefix) {
|
||||
rc = -1;
|
||||
break;
|
||||
}
|
||||
memcpy(prefix, clean, i);
|
||||
prefix[i] = '\0';
|
||||
rc = send_prefix_plan(fd, sender, prefix);
|
||||
free(prefix);
|
||||
}
|
||||
}
|
||||
}
|
||||
free(clean);
|
||||
return rc;
|
||||
}
|
||||
|
||||
int delete_plan_send_remaining(int fd, DeletePlanSender* sender, const ArrayList* dirs) {
|
||||
if (!sender || !dirs)
|
||||
return 0;
|
||||
for (int i = 0; i < dirs->size; i++) {
|
||||
const char* dir = (const char*)dirs->items[i];
|
||||
if (delete_plan_send_for_path(fd, sender, dir, true) != 0)
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* Receiver: delete session */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
struct DeletePlanSession {
|
||||
bool defer;
|
||||
bool dry_run;
|
||||
size_t max_delete;
|
||||
size_t deleted;
|
||||
size_t skipped;
|
||||
bool limit_hit;
|
||||
bool limit_logged;
|
||||
bool config_seen;
|
||||
bool missing_applied;
|
||||
ArrayList* protected_prefixes;
|
||||
ArrayList* size_skipped;
|
||||
ArrayList* missing;
|
||||
ArrayList* deferred;
|
||||
};
|
||||
|
||||
DeletePlanSession* delete_plan_session_create(const Config* config) {
|
||||
if (!config)
|
||||
return NULL;
|
||||
DeletePlanSession* session = calloc(1, sizeof(DeletePlanSession));
|
||||
if (!session)
|
||||
return NULL;
|
||||
session->defer = config->delete_delay;
|
||||
session->dry_run = config->dry_run;
|
||||
bool user_limited =
|
||||
config->max_delete >= 0 && (size_t)config->max_delete < DELETE_PLAN_SERVER_LIMIT;
|
||||
session->max_delete =
|
||||
user_limited ? (size_t)config->max_delete : (size_t)DELETE_PLAN_SERVER_LIMIT;
|
||||
session->protected_prefixes = array_list_create(free);
|
||||
session->size_skipped = array_list_create(free);
|
||||
session->missing = array_list_create(free);
|
||||
session->deferred = array_list_create(free);
|
||||
if (!session->protected_prefixes || !session->size_skipped || !session->missing ||
|
||||
!session->deferred) {
|
||||
delete_plan_session_destroy(session);
|
||||
return NULL;
|
||||
}
|
||||
return session;
|
||||
}
|
||||
|
||||
void delete_plan_session_destroy(DeletePlanSession* session) {
|
||||
if (!session)
|
||||
return;
|
||||
array_list_delete(session->protected_prefixes);
|
||||
array_list_delete(session->size_skipped);
|
||||
array_list_delete(session->missing);
|
||||
array_list_delete(session->deferred);
|
||||
free(session);
|
||||
}
|
||||
|
||||
bool delete_plan_session_limit_reached(const DeletePlanSession* session) {
|
||||
return session && session->limit_hit;
|
||||
}
|
||||
|
||||
size_t delete_plan_session_deleted(const DeletePlanSession* session) {
|
||||
return session ? session->deleted : 0;
|
||||
}
|
||||
|
||||
/* True for a destination-relative path section entry (non-empty, relative,
|
||||
* traversal-free). */
|
||||
static bool valid_rel_path(const char* value) {
|
||||
return value && value[0] != '\0' && value[0] != '/' && !has_path_traversal(value);
|
||||
}
|
||||
|
||||
/* True for a single child name (non-empty, no slash, not "."/".."). */
|
||||
static bool valid_name(const char* value) {
|
||||
return value && value[0] != '\0' && strcmp(value, ".") != 0 && strcmp(value, "..") != 0 &&
|
||||
strchr(value, '/') == NULL;
|
||||
}
|
||||
|
||||
/* Read one count-prefixed section. `bytes` is the running per-frame budget,
|
||||
* shared across every section of the frame so a hostile peer cannot retain more
|
||||
* than MAX_MANIFEST_BYTES from one STATUS_DELETE_PLAN frame. */
|
||||
static bool read_section(int fd, ArrayList* list, bool rel_path, size_t* bytes) {
|
||||
int count;
|
||||
if (!receive_int(fd, &count) || count < 0 || count > MAX_MANIFEST_ENTRIES)
|
||||
return false;
|
||||
for (int i = 0; i < count; i++) {
|
||||
char* value = receive_wire_str(fd);
|
||||
bool ok = value && (rel_path ? valid_rel_path(value) : valid_name(value));
|
||||
if (ok) {
|
||||
size_t entry_size = strlen(value) + sizeof(char*) + 16;
|
||||
if (entry_size > MAX_MANIFEST_BYTES - *bytes) {
|
||||
ok = false;
|
||||
} else {
|
||||
*bytes += entry_size;
|
||||
ok = array_list_add(list, value);
|
||||
}
|
||||
}
|
||||
if (!ok) {
|
||||
free(value);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static int open_plan_dir(const Config* config, const char* dir) {
|
||||
char* full = (strcmp(dir, ".") == 0) ? str_dup(config->receive_root_directory)
|
||||
: path_cat(config->receive_root_directory, dir);
|
||||
if (!full)
|
||||
return -1;
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
int fd = -1;
|
||||
if (root_fd >= 0) {
|
||||
if (utils_get_authorized_root_path())
|
||||
fd = utils_open_authorized_destination(full);
|
||||
else if (strcmp(dir, ".") == 0)
|
||||
fd = dup(root_fd);
|
||||
} else {
|
||||
fd = open(full, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
}
|
||||
free(full);
|
||||
return fd;
|
||||
}
|
||||
|
||||
typedef struct PlanSkips {
|
||||
DeleteSkipEntry* entries;
|
||||
int count;
|
||||
} PlanSkips;
|
||||
|
||||
static bool build_plan_skips(const Config* config, const DeletePlanSession* session,
|
||||
PlanSkips* out) {
|
||||
out->entries = NULL;
|
||||
out->count = 0;
|
||||
int count = (config->delay_updates ? 1 : 0) + config->basis_count +
|
||||
session->protected_prefixes->size + session->size_skipped->size;
|
||||
if (count == 0)
|
||||
return true;
|
||||
out->entries = calloc((size_t)count, sizeof(DeleteSkipEntry));
|
||||
if (!out->entries)
|
||||
return false;
|
||||
int idx = 0;
|
||||
if (config->delay_updates) {
|
||||
out->entries[idx].prefix = DELAY_UPDATES_STAGING_DIR;
|
||||
out->entries[idx].top_level_only = true;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < config->basis_count; i++) {
|
||||
out->entries[idx].prefix = config->basis_dirs[i].path;
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < session->protected_prefixes->size; i++) {
|
||||
out->entries[idx].prefix = (const char*)session->protected_prefixes->items[i];
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
for (int i = 0; i < session->size_skipped->size; i++) {
|
||||
out->entries[idx].prefix = (const char*)session->size_skipped->items[i];
|
||||
out->entries[idx].top_level_only = false;
|
||||
idx++;
|
||||
}
|
||||
out->count = idx;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool budget_available(const DeletePlanSession* session) {
|
||||
return session->deleted < session->max_delete;
|
||||
}
|
||||
|
||||
static void note_skipped(DeletePlanSession* session) {
|
||||
session->limit_hit = true;
|
||||
session->skipped++;
|
||||
}
|
||||
|
||||
static void log_deleted(const char* rel) {
|
||||
char* escaped = output_escape(rel, log_get_8_bit_output());
|
||||
fprintf(stderr, " Deleted: %s\n", escaped ? escaped : "<allocation failed>");
|
||||
free(escaped);
|
||||
}
|
||||
|
||||
/* Append a snapshot path for --delete-delay. */
|
||||
static bool defer_add(DeletePlanSession* session, const char* rel) {
|
||||
char* copy = str_dup(rel);
|
||||
if (!copy)
|
||||
return false;
|
||||
if (!array_list_add(session->deferred, copy)) {
|
||||
free(copy);
|
||||
return false;
|
||||
}
|
||||
session->deleted++;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Process the direct children of one directory. `keep_dirs`/`keep_files`
|
||||
* (basenames) are the source entries that must be kept; NULL means every child
|
||||
* is an extra (the forced path used inside a removed extra directory tree).
|
||||
* `survives` reports that at least one child remains (kept, protected, or
|
||||
* skipped by the budget). `force_now` removes even in --delete-delay mode
|
||||
* (type conflicts must clear before the incoming data). */
|
||||
static bool process_children(int dirfd, const char* dir_rel, const ArrayList* keep_dirs,
|
||||
const ArrayList* keep_files, bool at_root, bool force_now,
|
||||
const PlanSkips* skips, DeletePlanSession* session, bool* survives);
|
||||
|
||||
static bool process_extra_dir(int dirfd, const char* name, const char* child_rel, bool force_now,
|
||||
const PlanSkips* skips, DeletePlanSession* session, bool* removed) {
|
||||
*removed = false;
|
||||
int childfd = openat(dirfd, name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (childfd < 0) {
|
||||
if (errno == ENOENT) {
|
||||
*removed = true;
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
bool survives = false;
|
||||
bool ok =
|
||||
process_children(childfd, child_rel, NULL, NULL, false, force_now, skips, session, &survives);
|
||||
close(childfd);
|
||||
if (!ok)
|
||||
return false;
|
||||
if (survives)
|
||||
return true;
|
||||
if (!budget_available(session)) {
|
||||
note_skipped(session);
|
||||
return true;
|
||||
}
|
||||
if (session->defer && !force_now) {
|
||||
if (!defer_add(session, child_rel))
|
||||
return false;
|
||||
*removed = true;
|
||||
return true;
|
||||
}
|
||||
if (unlinkat(dirfd, name, AT_REMOVEDIR) == 0) {
|
||||
session->deleted++;
|
||||
log_deleted(child_rel);
|
||||
*removed = true;
|
||||
return true;
|
||||
}
|
||||
if (errno == ENOENT) {
|
||||
*removed = true;
|
||||
return true;
|
||||
}
|
||||
/* ENOTEMPTY/EEXIST: a protected entry the walker leaves behind survived, so
|
||||
the directory stays; any other errno is a genuine failure. */
|
||||
return errno == ENOTEMPTY || errno == EEXIST;
|
||||
}
|
||||
|
||||
static bool process_extra_file(int dirfd, const char* name, const char* child_rel, bool force_now,
|
||||
DeletePlanSession* session) {
|
||||
if (!budget_available(session)) {
|
||||
note_skipped(session);
|
||||
return true;
|
||||
}
|
||||
if (session->defer && !force_now) {
|
||||
return defer_add(session, child_rel);
|
||||
}
|
||||
if (unlinkat(dirfd, name, 0) == 0) {
|
||||
session->deleted++;
|
||||
log_deleted(child_rel);
|
||||
} else if (errno != ENOENT) {
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool process_children(int dirfd, const char* dir_rel, const ArrayList* keep_dirs,
|
||||
const ArrayList* keep_files, bool at_root, bool force_now,
|
||||
const PlanSkips* skips, DeletePlanSession* session, bool* survives) {
|
||||
*survives = false;
|
||||
int scanfd = openat(dirfd, ".", O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (scanfd < 0)
|
||||
return false;
|
||||
DIR* dir = fdopendir(scanfd);
|
||||
if (!dir) {
|
||||
close(scanfd);
|
||||
return false;
|
||||
}
|
||||
bool operation_ok = true;
|
||||
bool local_survives = false;
|
||||
const struct dirent* entry;
|
||||
while ((entry = readdir(dir)) != NULL) {
|
||||
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
|
||||
continue;
|
||||
char* child_rel =
|
||||
(strcmp(dir_rel, ".") == 0) ? str_dup(entry->d_name) : path_cat(dir_rel, entry->d_name);
|
||||
if (!child_rel) {
|
||||
operation_ok = false;
|
||||
continue;
|
||||
}
|
||||
if (path_under_skip_prefix(child_rel, at_root, skips->entries, skips->count)) {
|
||||
local_survives = true;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
if (errno != ENOENT)
|
||||
operation_ok = false;
|
||||
free(child_rel);
|
||||
continue;
|
||||
}
|
||||
bool is_dir = S_ISDIR(st.st_mode);
|
||||
bool in_keep_dirs = is_dir && list_contains_str(keep_dirs, entry->d_name);
|
||||
bool in_keep_files = !is_dir && list_contains_str(keep_files, entry->d_name);
|
||||
if (in_keep_dirs) {
|
||||
local_survives = true;
|
||||
} else if (keep_dirs && !is_dir && list_contains_str(keep_dirs, entry->d_name)) {
|
||||
/* Destination file blocks a source directory: clear it now, whatever the
|
||||
delete timing, so the directory can be created. */
|
||||
if (!process_extra_file(dirfd, entry->d_name, child_rel, true, session))
|
||||
operation_ok = false;
|
||||
} else if (in_keep_files) {
|
||||
local_survives = true;
|
||||
} else if (keep_files && is_dir && list_contains_str(keep_files, entry->d_name)) {
|
||||
/* Destination directory blocks a source file: remove it now. */
|
||||
bool removed = false;
|
||||
if (!process_extra_dir(dirfd, entry->d_name, child_rel, true, skips, session, &removed))
|
||||
operation_ok = false;
|
||||
else if (!removed)
|
||||
local_survives = true;
|
||||
} else if (is_dir) {
|
||||
bool removed = false;
|
||||
if (!process_extra_dir(dirfd, entry->d_name, child_rel, force_now, skips, session, &removed))
|
||||
operation_ok = false;
|
||||
else if (!removed)
|
||||
local_survives = true;
|
||||
} else {
|
||||
if (!process_extra_file(dirfd, entry->d_name, child_rel, force_now, session))
|
||||
operation_ok = false;
|
||||
}
|
||||
free(child_rel);
|
||||
}
|
||||
closedir(dir);
|
||||
*survives = local_survives;
|
||||
return operation_ok;
|
||||
}
|
||||
|
||||
static bool apply_plan_dir(DeletePlanSession* session, const Config* config, const char* dir,
|
||||
const ArrayList* dirs, const ArrayList* files) {
|
||||
int dirfd = open_plan_dir(config, dir);
|
||||
if (dirfd < 0) {
|
||||
/* An absent destination directory has nothing to delete. */
|
||||
return errno == ENOENT || errno == ENOTDIR;
|
||||
}
|
||||
PlanSkips skips;
|
||||
if (!build_plan_skips(config, session, &skips)) {
|
||||
close(dirfd);
|
||||
return false;
|
||||
}
|
||||
bool survives = false;
|
||||
bool ok = process_children(dirfd, dir, dirs, files, strcmp(dir, ".") == 0, false, &skips, session,
|
||||
&survives);
|
||||
free(skips.entries);
|
||||
close(dirfd);
|
||||
if (!ok)
|
||||
log_message(LOG_LEVEL_ERROR, "deletion failed while removing extraneous files");
|
||||
return ok;
|
||||
}
|
||||
|
||||
static bool apply_missing(DeletePlanSession* session, const Config* config) {
|
||||
if (session->missing_applied)
|
||||
return true;
|
||||
session->missing_applied = true;
|
||||
/* The server clears delete_missing_args when its --allow-delete policy is
|
||||
off; never honor the client's exact-path requests then. */
|
||||
if (!config->delete_missing_args || session->missing->size == 0)
|
||||
return true;
|
||||
DeleteManifest manifest = {
|
||||
.keeps = NULL, .protected = NULL, .missing = session->missing, .dirs = NULL};
|
||||
size_t remaining = budget_available(session) ? session->max_delete - session->deleted : 0;
|
||||
size_t deleted = 0;
|
||||
size_t skipped = 0;
|
||||
bool limit = false;
|
||||
bool ok = manifest_delete_missing_args_limited(config, &manifest, remaining, &deleted, &skipped,
|
||||
&limit);
|
||||
session->deleted += deleted;
|
||||
session->skipped += skipped;
|
||||
if (limit)
|
||||
session->limit_hit = true;
|
||||
return ok;
|
||||
}
|
||||
|
||||
int delete_plan_session_receive(DeletePlanSession* session, const Config* config, int fd) {
|
||||
if (!session || !config) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return -1;
|
||||
}
|
||||
int has_config;
|
||||
if (!receive_int(fd, &has_config) || (has_config != 0 && has_config != 1)) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return -1;
|
||||
}
|
||||
size_t bytes = 0;
|
||||
if (has_config) {
|
||||
if (session->config_seen || !read_section(fd, session->protected_prefixes, true, &bytes) ||
|
||||
!read_section(fd, session->size_skipped, true, &bytes) ||
|
||||
!read_section(fd, session->missing, true, &bytes)) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return -1;
|
||||
}
|
||||
session->config_seen = true;
|
||||
}
|
||||
char* dir = receive_wire_str(fd);
|
||||
ArrayList* dirs = array_list_create(free);
|
||||
ArrayList* files = array_list_create(free);
|
||||
bool parsed = dir && (strcmp(dir, ".") == 0 || valid_rel_path(dir)) && dirs && files &&
|
||||
read_section(fd, dirs, false, &bytes) && read_section(fd, files, false, &bytes);
|
||||
if (!parsed) {
|
||||
free(dir);
|
||||
array_list_delete(dirs);
|
||||
array_list_delete(files);
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return -1;
|
||||
}
|
||||
bool enabled = config->use_delete || config->delete_missing_args;
|
||||
bool ok = true;
|
||||
if (!session->dry_run && enabled) {
|
||||
if (!session->defer && !apply_missing(session, config))
|
||||
ok = false;
|
||||
if (ok && !apply_plan_dir(session, config, dir, dirs, files))
|
||||
ok = false;
|
||||
}
|
||||
free(dir);
|
||||
array_list_delete(dirs);
|
||||
array_list_delete(files);
|
||||
if (!ok) {
|
||||
send_status(fd, STATUS_ERROR);
|
||||
return -1;
|
||||
}
|
||||
if (session->limit_hit && !session->limit_logged) {
|
||||
session->limit_logged = true;
|
||||
log_message(LOG_LEVEL_WARNING, "Deletions stopped due to the delete limit (%zu skipped)",
|
||||
session->skipped);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Apply one snapshotted --delete-delay path (post-order: children precede their
|
||||
* parent directory). */
|
||||
static bool apply_deferred_path(DeletePlanSession* session, const Config* config, const char* rel) {
|
||||
(void)session;
|
||||
char* full = path_cat(config->receive_root_directory, rel);
|
||||
if (!full)
|
||||
return false;
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(full, &leaf, false);
|
||||
free(full);
|
||||
if (parent_fd < 0) {
|
||||
free(leaf);
|
||||
return errno == ENOENT || errno == ENOTDIR;
|
||||
}
|
||||
struct stat st;
|
||||
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0) {
|
||||
bool absent = errno == ENOENT;
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
return absent;
|
||||
}
|
||||
int rc;
|
||||
if (S_ISDIR(st.st_mode))
|
||||
rc = unlinkat(parent_fd, leaf, AT_REMOVEDIR);
|
||||
else
|
||||
rc = unlinkat(parent_fd, leaf, 0);
|
||||
bool ok = rc == 0 || errno == ENOENT || errno == ENOTEMPTY || errno == EEXIST;
|
||||
if (rc == 0)
|
||||
log_deleted(rel);
|
||||
close(parent_fd);
|
||||
free(leaf);
|
||||
return ok;
|
||||
}
|
||||
|
||||
DeleteCommitResult delete_plan_session_commit(DeletePlanSession* session, const Config* config) {
|
||||
if (!session || !config)
|
||||
return DELETE_COMMIT_ERROR;
|
||||
/* Central no-mutation guard (mirrors manifest_delete_all): a dry-run never
|
||||
deletes. The receive path already skips plan application, but a hostile or
|
||||
buggy peer could still reach the commit, so treat it as a no-op. */
|
||||
if (session->dry_run)
|
||||
return DELETE_COMMIT_OK;
|
||||
bool ok = true;
|
||||
if (session->defer) {
|
||||
for (int i = 0; i < session->deferred->size && ok; i++)
|
||||
ok = apply_deferred_path(session, config, (const char*)session->deferred->items[i]);
|
||||
}
|
||||
if (ok)
|
||||
ok = apply_missing(session, config);
|
||||
if (!ok)
|
||||
return DELETE_COMMIT_ERROR;
|
||||
if (session->limit_hit)
|
||||
return DELETE_COMMIT_LIMIT_REACHED;
|
||||
return DELETE_COMMIT_OK;
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
#ifndef DELETE_PLAN_H
|
||||
#define DELETE_PLAN_H
|
||||
|
||||
#include "array_list.h"
|
||||
#include "config.h"
|
||||
#include "file_receive.h"
|
||||
#include "protocol.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
/* Per-directory delete plans (protocol 2.24.0).
|
||||
*
|
||||
* rsync's --delete-during removes a directory's extras while the generator
|
||||
* processes that directory, and --delete-delay records the deletion list during
|
||||
* the scan but applies it only after a fully-successful transfer. FastSync has
|
||||
* no per-directory generator pass; instead the sender streams one plan per
|
||||
* source directory, in directory order, and the receiver applies it when it
|
||||
* arrives (during) or snapshots its extras and commits them at the end (delay).
|
||||
*
|
||||
* The sender side builds a plan set from the path-only pre-scan (it needs every
|
||||
* directory's complete direct-child list before the first data byte of that
|
||||
* directory). The receiver side is a session that carries the global protected
|
||||
* prefixes (filter-excluded and size-skipped source mirrors), the
|
||||
* --delete-missing-args exact deletions, the shared --max-delete budget and,
|
||||
* for --delete-delay, the snapshotted extras. */
|
||||
|
||||
/* ---- Sender: plan builder ---- */
|
||||
|
||||
typedef struct DeletePlanSender DeletePlanSender;
|
||||
|
||||
DeletePlanSender* delete_plan_sender_create(void);
|
||||
void delete_plan_sender_destroy(DeletePlanSender* sender);
|
||||
/* Record one transmitted entry. `path` is the destination-relative wire path;
|
||||
* is_dir marks an explicit directory entry (--dirs, a -x mount point). */
|
||||
bool delete_plan_sender_add(DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Drop plans for directories outside `synced_dirs` (the --files-from
|
||||
* synchronization scope; pass NULL when a full recursive transfer synchronized
|
||||
* every directory). The receive root is the "." sentinel.
|
||||
*
|
||||
* `walk_root` scopes a general -R transfer: when non-NULL it is the
|
||||
* reconstructed destination prefix the run actually transferred, and only the
|
||||
* plan for that prefix (and directories below it) is ever transmitted, so the
|
||||
* prefix's parent-directory siblings are never walked. Pass NULL for a plain
|
||||
* recursive transfer and for --files-from. */
|
||||
void delete_plan_sender_finalize(DeletePlanSender* sender, const ArrayList* synced_dirs,
|
||||
const char* walk_root);
|
||||
/* True when no transmitted FILE entry was recorded (an ambiguous empty scan).
|
||||
Directory keep entries do not count, so an I/O error that hid every file
|
||||
still refuses to delete. */
|
||||
bool delete_plan_sender_empty(const DeletePlanSender* sender);
|
||||
/* Attach the global config sections advertised on the first plan frame. */
|
||||
void delete_plan_sender_set_config(DeletePlanSender* sender, const ArrayList* protected_prefixes,
|
||||
const ArrayList* size_skipped, const ArrayList* missing_args);
|
||||
/* Send the root plan (even before any data, so root extras are handled like
|
||||
* rsync's first generator directory). Returns -1 on I/O error. */
|
||||
int delete_plan_send_root(int fd, DeletePlanSender* sender);
|
||||
/* Send the plans for every ancestor of `path` (root-first) and, when is_dir,
|
||||
* for `path` itself; already-sent plans are skipped. */
|
||||
int delete_plan_send_for_path(int fd, DeletePlanSender* sender, const char* path, bool is_dir);
|
||||
/* Send the plan for every directory in `dirs` that has not been transmitted
|
||||
* yet. Called after the data stream so an empty source directory's plan still
|
||||
* clears its destination extras even though no file frame triggered it. */
|
||||
int delete_plan_send_remaining(int fd, DeletePlanSender* sender, const ArrayList* dirs);
|
||||
|
||||
/* ---- Receiver: delete session ---- */
|
||||
|
||||
typedef struct DeletePlanSession DeletePlanSession;
|
||||
|
||||
DeletePlanSession* delete_plan_session_create(const Config* config);
|
||||
void delete_plan_session_destroy(DeletePlanSession* session);
|
||||
/* Read one STATUS_DELETE_PLAN frame (the leading status already consumed) and
|
||||
* act on it. Returns 0 on success (including a dry-run/disabled no-op) and -1
|
||||
* after signalling STATUS_ERROR on a malformed frame or a deletion failure. */
|
||||
int delete_plan_session_receive(DeletePlanSession* session, const Config* config, int fd);
|
||||
/* Apply the deferred snapshot (--delete-delay) and the missing-args deletions.
|
||||
* Safe to call once; returns the commit outcome. */
|
||||
DeleteCommitResult delete_plan_session_commit(DeletePlanSession* session, const Config* config);
|
||||
/* True once the shared --max-delete budget stopped part of a deletion. */
|
||||
bool delete_plan_session_limit_reached(const DeletePlanSession* session);
|
||||
/* Number of destination entries the session's plans removed (or, for
|
||||
--delete-delay, snapshotted for removal), for the end-of-transfer stats. */
|
||||
size_t delete_plan_session_deleted(const DeletePlanSession* session);
|
||||
|
||||
#endif
|
||||
+332
-144
@@ -6,6 +6,7 @@
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <limits.h>
|
||||
#include <pthread.h>
|
||||
#include <stdatomic.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
@@ -39,19 +40,27 @@ static bool write_all(int fd, const void* data, unsigned long long size) {
|
||||
}
|
||||
|
||||
/* Preallocate `size` bytes on `fd` before any data is written (--preallocate).
|
||||
* posix_fallocate reserves real disk blocks, so an out-of-space condition
|
||||
* fallocate(2) reserves real disk blocks, so an out-of-space condition
|
||||
* (ENOSPC/EDQUOT) surfaces up front instead of partway through a transfer;
|
||||
* unavoidable fragmentation of a streamed file is also reduced. Some
|
||||
* filesystems (e.g. tmpfs, ZFS) do not support it and return EOPNOTSUPP/ENOSYS,
|
||||
* where we fall back to ftruncate, which still extends the logical size so the
|
||||
* fail-fast/contiguity intent degrades gracefully but never fails. Genuine
|
||||
* allocation failures are propagated as the error code (caller fails the write).
|
||||
* posix_fallocate leaves the fd's file offset unchanged, so the subsequent
|
||||
* write_all at offset 0 is unaffected. Returns 0 on success (including the
|
||||
* fallback) or a nonzero error code. */
|
||||
* unavoidable fragmentation of a streamed file is also reduced. rsync favors
|
||||
* the syscall over glibc posix_fallocate (whose emulation can be subtly
|
||||
* different), so try fallocate(2) first and only fall back to posix_fallocate,
|
||||
* then to ftruncate on filesystems (e.g. tmpfs, ZFS) that support neither. The
|
||||
* logical size is always extended, so the fail-fast/contiguity intent degrades
|
||||
* gracefully but never fails on an unsupported filesystem; genuine allocation
|
||||
* failures are propagated as the error code (caller fails the write). Neither
|
||||
* leaves the fd's file offset guaranteed, so the caller seeks back to 0 before
|
||||
* writing. Returns 0 on success (including the fallback) or a nonzero error
|
||||
* code. */
|
||||
static int preallocate_fd(int fd, unsigned long long size) {
|
||||
if (size == 0)
|
||||
return 0;
|
||||
#ifdef __linux__
|
||||
if (fallocate(fd, 0, 0, (off_t)size) == 0)
|
||||
return 0;
|
||||
if (errno != EOPNOTSUPP && errno != ENOSYS && errno != EINVAL)
|
||||
return errno;
|
||||
#endif
|
||||
int rc = posix_fallocate(fd, 0, (off_t)size);
|
||||
if (rc == EOPNOTSUPP || rc == ENOSYS) {
|
||||
if (ftruncate(fd, (off_t)size) == 0)
|
||||
@@ -72,6 +81,58 @@ static unsigned long long next_temp_sequence(void) {
|
||||
return atomic_fetch_add_explicit(&sequence, 1, memory_order_relaxed);
|
||||
}
|
||||
|
||||
/* Process-wide umask, captured exactly once. Reading the umask requires a
|
||||
* get+set round trip (umask(0); umask(old)); doing that per write would be racy
|
||||
* in the multithreaded receiver, so the value is captured at process startup by
|
||||
* file_umask_capture() (called at the top of main(), before any threads exist).
|
||||
* The pthread_once fallback keeps a caller that never called the capture (e.g. a
|
||||
* unit test) correct. */
|
||||
static unsigned g_process_umask;
|
||||
static atomic_bool g_process_umask_captured;
|
||||
static pthread_once_t g_process_umask_once = PTHREAD_ONCE_INIT;
|
||||
|
||||
static void file_capture_umask_now(void) {
|
||||
mode_t mask = umask(0);
|
||||
umask(mask);
|
||||
g_process_umask = (unsigned)mask;
|
||||
atomic_store_explicit(&g_process_umask_captured, true, memory_order_release);
|
||||
}
|
||||
|
||||
static void file_capture_umask_once(void) {
|
||||
if (atomic_load_explicit(&g_process_umask_captured, memory_order_acquire))
|
||||
return;
|
||||
file_capture_umask_now();
|
||||
}
|
||||
|
||||
/* Re-captures the umask. Must only be called while the process is still
|
||||
* single-threaded (startup, or the daemon's post-fork setup after umask(0)),
|
||||
* so a later re-capture can refresh the cached value before any receiver
|
||||
* thread exists. */
|
||||
void file_umask_capture(void) {
|
||||
file_capture_umask_now();
|
||||
}
|
||||
|
||||
unsigned file_process_umask(void) {
|
||||
if (!atomic_load_explicit(&g_process_umask_captured, memory_order_acquire))
|
||||
pthread_once(&g_process_umask_once, file_capture_umask_once);
|
||||
return g_process_umask;
|
||||
}
|
||||
|
||||
/* Base mode applied when the policy does not take the source mode wholesale
|
||||
* (i.e. --perms is off). A pre-existing destination keeps its own mode; a
|
||||
* brand-new file is created like rsync: source_mode & 0777 & ~umask (special
|
||||
* bits are not part of a mode-preserving transfer without -p). Only when no
|
||||
* metadata is available at all does the historical fixed 0644 default apply.
|
||||
* The -E rule (and no-op for a plain -t) is layered on top of this base. */
|
||||
static mode_t file_mode_base(const FileMetadata* metadata, bool existing_known,
|
||||
mode_t existing_mode) {
|
||||
if (existing_known)
|
||||
return existing_mode;
|
||||
if (metadata)
|
||||
return metadata->mode & 0777 & ~(mode_t)file_process_umask();
|
||||
return S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH;
|
||||
}
|
||||
|
||||
bool file_checksum(File* file, ChecksumAlgo algo, uint64_t seed, uint8_t* out, size_t out_capacity,
|
||||
size_t* out_len) {
|
||||
if (!file || !out || !out_len || !file->data)
|
||||
@@ -124,6 +185,8 @@ File* file_create(const char* path) {
|
||||
file->rdev_major = 0;
|
||||
file->rdev_minor = 0;
|
||||
file->xattrs = NULL;
|
||||
file->dest_state = (OutputDestState){0};
|
||||
file->matched_bytes = 0;
|
||||
return file;
|
||||
}
|
||||
|
||||
@@ -284,28 +347,6 @@ size_t file_content_to_buffer(File* file) {
|
||||
|
||||
/* ---- Secure filesystem primitives ---- */
|
||||
|
||||
static int authorized_root_fd = -1;
|
||||
static char* authorized_root_path;
|
||||
|
||||
static bool path_is_within_root(const char* root, const char* path) {
|
||||
size_t root_len = strlen(root);
|
||||
return strncmp(root, path, root_len) == 0 && (path[root_len] == '\0' || path[root_len] == '/');
|
||||
}
|
||||
|
||||
bool file_set_authorized_root(int fd, const char* canonical_path) {
|
||||
char* path_copy = canonical_path ? str_dup(canonical_path) : NULL;
|
||||
if (canonical_path && !path_copy) {
|
||||
authorized_root_fd = -1;
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = NULL;
|
||||
return false;
|
||||
}
|
||||
authorized_root_fd = fd;
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = path_copy;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool file_path_exists_secure(const char* path) {
|
||||
if (!path)
|
||||
return false;
|
||||
@@ -362,10 +403,6 @@ void file_set_keep_dirlinks(bool enable) {
|
||||
file_keep_dirlinks = enable;
|
||||
}
|
||||
|
||||
bool file_get_keep_dirlinks(void) {
|
||||
return file_keep_dirlinks;
|
||||
}
|
||||
|
||||
/* --trust-sender (Phase 5) receiver process-wide policy: when set, the receiver
|
||||
* trusts the sender's file list and skips its own redundant up-front re-
|
||||
* validation (empty/".." path rejection, escaping-symlink-target containment).
|
||||
@@ -383,10 +420,69 @@ bool file_get_trust_sender(void) {
|
||||
return file_trust_sender;
|
||||
}
|
||||
|
||||
/* True when `target` is a lexical symlink target that can never escape the
|
||||
* receive root once created beneath it: relative (not absolute) and containing
|
||||
* no ".." path component. Used by --munge-links' sender-side containment: an
|
||||
* escaping target is never transmitted (the entry is skipped/contained). */
|
||||
/* rsync 3.4.1 unsafe_symlink(): true when `target` (the link's destination
|
||||
* string) points outside the transfer tree rooted at the symlink's own
|
||||
* location. `link_path` is the symlink's path relative to the top of the
|
||||
* transfer (including its name). This is a purely lexical test matching
|
||||
* rsync's util1.c: absolute/empty targets are always unsafe; leading "../"
|
||||
* components are counted against the symlink's own directory depth; a ".."
|
||||
* that would climb above the transfer root is unsafe. rsync 3.4.1 additionally
|
||||
* rejects any INTERNAL "/../" component and a trailing "/..". */
|
||||
bool file_symlink_unsafe(const char* target, const char* link_path) {
|
||||
if (!target || target[0] == '\0' || target[0] == '/')
|
||||
return true;
|
||||
const char* rest = target;
|
||||
while (strncmp(rest, "../", 3) == 0) {
|
||||
rest += 3;
|
||||
while (*rest == '/')
|
||||
rest++;
|
||||
}
|
||||
if (strstr(rest, "/../") != NULL)
|
||||
return true;
|
||||
size_t target_len = strlen(target);
|
||||
if (target_len > 3 && strcmp(&target[target_len - 3], "/..") == 0)
|
||||
return true;
|
||||
|
||||
int depth = 0;
|
||||
const char* name;
|
||||
const char* slash;
|
||||
const char* src = link_path ? link_path : "";
|
||||
for (name = src; (slash = strchr(name, '/')) != NULL; name = slash + 1) {
|
||||
if (*name == '.' && (name[1] == '/' || (name[1] == '.' && name[2] == '/'))) {
|
||||
if (name[1] == '.')
|
||||
depth = 0;
|
||||
} else {
|
||||
depth++;
|
||||
}
|
||||
while (slash[1] == '/')
|
||||
slash++;
|
||||
}
|
||||
if (*name == '.' && name[1] == '.' && name[2] == '\0')
|
||||
depth = 0;
|
||||
|
||||
for (name = target; (slash = strchr(name, '/')) != NULL; name = slash + 1) {
|
||||
if (*name == '.' && (name[1] == '/' || (name[1] == '.' && name[2] == '/'))) {
|
||||
if (name[1] == '.') {
|
||||
if (--depth < 0)
|
||||
return true;
|
||||
}
|
||||
} else {
|
||||
depth++;
|
||||
}
|
||||
while (slash[1] == '/')
|
||||
slash++;
|
||||
}
|
||||
if (*name == '.' && name[1] == '.' && name[2] == '\0')
|
||||
depth--;
|
||||
return depth < 0;
|
||||
}
|
||||
|
||||
/* Strict lexical helper: true when `target` is relative (not absolute) and
|
||||
* contains no ".." component at all, so it can never escape the directory it
|
||||
* is created in. This is stricter than rsync's unsafe_symlink() (which allows
|
||||
* an in-tree ".."); the scanner/receiver use file_symlink_unsafe()/--safe-links
|
||||
* for rsync parity, and this helper is retained for callers that want the
|
||||
* ".."-free guarantee. */
|
||||
bool file_symlink_target_contained(const char* target) {
|
||||
if (!target || target[0] == '\0' || target[0] == '/')
|
||||
return false;
|
||||
@@ -417,8 +513,9 @@ bool file_symlink_unmunge(char* target) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Owned copy of `target` prefixed with SYMLINK_MUNGE_PREFIX (the sender-side
|
||||
* --munge-links rewriting). Returns NULL on allocation failure. */
|
||||
/* Owned copy of `target` prefixed with SYMLINK_MUNGE_PREFIX (the receiver-side
|
||||
* --munge-links rewriting, matching rsync's receiver). Returns NULL on
|
||||
* allocation failure. */
|
||||
char* file_symlink_munge(const char* target) {
|
||||
if (!target)
|
||||
return NULL;
|
||||
@@ -439,23 +536,14 @@ char* file_symlink_munge(const char* target) {
|
||||
* the target is ever followed. The final component is never dereferenced: an
|
||||
* existing non-directory entry at `path` is unlinked by name before the link is
|
||||
* placed; an existing directory there is left untouched (returns false, so a
|
||||
* caller can treat it as a collision). As a receiver-side trust-boundary
|
||||
* invariant, `target` must be file_symlink_target_contained() (relative and
|
||||
* ".."-free): an absolute or escaping target is rejected outright (returns
|
||||
* false) so a malicious sender can never materialize a symlink that points
|
||||
* outside the receive root. */
|
||||
* caller can treat it as a collision). The link VALUE `target` is copied
|
||||
* verbatim, matching rsync -l (which stores absolute and ".."-bearing targets
|
||||
* as-is); target policy is the caller's job -- the scanner applies
|
||||
* --safe-links/--copy-unsafe-links, and the receiver applies --munge-links.
|
||||
* The PLACEMENT path is always confined below the authorized root. */
|
||||
bool file_symlink_at_secure(const char* path, const char* target) {
|
||||
/* The link itself (`path`) is always kept below the authorized root. The
|
||||
TARGET may point anywhere: normally only a contained (relative, ".."-free)
|
||||
target is permitted so a malicious sender can never plant a symlink that
|
||||
later dereferences outside the root. Under --trust-sender that target
|
||||
containment check is relaxed (the receiver trusts the sender and copies the
|
||||
link verbatim, matching rsync -l), but path/leaf confinement is never
|
||||
disabled, so the link still cannot be placed outside the tree. */
|
||||
if (!path || !target || has_path_traversal(path))
|
||||
return false;
|
||||
if (!file_trust_sender && !file_symlink_target_contained(target))
|
||||
return false;
|
||||
char* leaf = NULL;
|
||||
int parent_fd = file_open_secure_parent(path, &leaf, true);
|
||||
if (parent_fd < 0)
|
||||
@@ -492,7 +580,10 @@ static int open_dir_beneath_root(const char* resolved, const char* root) {
|
||||
rel++;
|
||||
if (*rel == '\0')
|
||||
return -1;
|
||||
int fd = dup(authorized_root_fd);
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
if (root_fd < 0)
|
||||
return -1;
|
||||
int fd = dup(root_fd);
|
||||
if (fd < 0)
|
||||
return -1;
|
||||
char* copy = str_dup(rel);
|
||||
@@ -534,20 +625,21 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs)
|
||||
return -1;
|
||||
}
|
||||
int fd;
|
||||
if (authorized_root_fd >= 0) {
|
||||
if (!authorized_root_path || path[0] != '/' ||
|
||||
!path_is_within_root(authorized_root_path, path)) {
|
||||
int root_fd = utils_get_authorized_root_fd();
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
if (root_fd >= 0) {
|
||||
if (!root_path || path[0] != '/' || !path_is_within_root(root_path, path)) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
fd = dup(authorized_root_fd);
|
||||
fd = dup(root_fd);
|
||||
if (fd < 0) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
size_t root_len = strlen(authorized_root_path);
|
||||
size_t root_len = strlen(root_path);
|
||||
char* relative = str_dup(path + root_len);
|
||||
if (!relative) {
|
||||
free(copy);
|
||||
@@ -580,7 +672,7 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs)
|
||||
if (strcmp(component, ".") != 0) {
|
||||
int next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (next < 0 && create_dirs && errno == ENOENT) {
|
||||
bool created = mkdirat(fd, component, 0755) == 0;
|
||||
bool created = mkdirat(fd, component, (mode_t)(0777 & ~(mode_t)file_process_umask())) == 0;
|
||||
if (created || errno == EEXIST) {
|
||||
/* P7 Wave E: --copy-as owns EVERY entry, including the intermediate
|
||||
directories this walk creates implicitly. Its target ids are a
|
||||
@@ -611,15 +703,14 @@ int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs)
|
||||
O_NOFOLLOW walk. Only honoured when the symlink resolves to a
|
||||
directory that stays beneath the authorized root, so a malicious link
|
||||
can never redirect the write outside it. */
|
||||
if (next < 0 && file_keep_dirlinks && authorized_root_path != NULL &&
|
||||
if (next < 0 && file_keep_dirlinks && root_path != NULL &&
|
||||
(errno == ELOOP || errno == ENOTDIR || errno == EACCES)) {
|
||||
struct stat lst;
|
||||
if (fstatat(fd, component, &lst, AT_SYMLINK_NOFOLLOW) == 0 && S_ISLNK(lst.st_mode)) {
|
||||
char candidate[PATH_MAX];
|
||||
char root[PATH_MAX];
|
||||
if (realpath(authorized_root_path, root) &&
|
||||
snprintf(candidate, sizeof(candidate), "%s%s/%s", root, rel_buf, component) <
|
||||
(int)sizeof(candidate)) {
|
||||
if (realpath(root_path, root) && snprintf(candidate, sizeof(candidate), "%s%s/%s", root,
|
||||
rel_buf, component) < (int)sizeof(candidate)) {
|
||||
char resolved[PATH_MAX];
|
||||
if (realpath(candidate, resolved) && strcmp(resolved, root) != 0 &&
|
||||
strncmp(root, resolved, strlen(root)) == 0 &&
|
||||
@@ -694,8 +785,9 @@ bool file_ensure_directory_secure(const char* path) {
|
||||
return false;
|
||||
/* The authorized root is already an open directory, and the filesystem root
|
||||
is always present: there is no final component left to create for them. */
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
bool root_is_open =
|
||||
authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0;
|
||||
utils_get_authorized_root_fd() >= 0 && root_path && strcmp(norm, root_path) == 0;
|
||||
if (root_is_open || strcmp(norm, "/") == 0) {
|
||||
free(norm);
|
||||
return true;
|
||||
@@ -709,7 +801,7 @@ bool file_ensure_directory_secure(const char* path) {
|
||||
int dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
bool created = false;
|
||||
if (dir_fd < 0 && errno == ENOENT) {
|
||||
if (mkdirat(parent_fd, leaf, 0755) == 0) {
|
||||
if (mkdirat(parent_fd, leaf, (mode_t)(0777 & ~(mode_t)file_process_umask())) == 0) {
|
||||
created = true;
|
||||
dir_fd = openat(parent_fd, leaf, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
} else if (errno == EEXIST) {
|
||||
@@ -742,8 +834,9 @@ bool file_directory_exists_secure(const char* path) {
|
||||
char* norm = normalize_directory_path(path);
|
||||
if (!norm)
|
||||
return false;
|
||||
const char* root_path = utils_get_authorized_root_path();
|
||||
bool root_is_open =
|
||||
authorized_root_fd >= 0 && authorized_root_path && strcmp(norm, authorized_root_path) == 0;
|
||||
utils_get_authorized_root_fd() >= 0 && root_path && strcmp(norm, root_path) == 0;
|
||||
if (root_is_open || strcmp(norm, "/") == 0) {
|
||||
free(norm);
|
||||
return true;
|
||||
@@ -876,29 +969,48 @@ int file_open_private_dir(const char* dir_path) {
|
||||
return fd;
|
||||
}
|
||||
|
||||
/* Open a --temp-dir scratch directory exactly as rsync does: the directory must
|
||||
* already exist and is used as given (an absolute path is used verbatim, a
|
||||
* relative one was already resolved against the destination root by the
|
||||
* caller). Unlike file_open_private_dir this neither creates it nor confines
|
||||
* it below the receive root, because rsync accepts any temp dir -- including
|
||||
* one outside the destination tree or on another filesystem. Returns an
|
||||
* O_DIRECTORY|O_CLOEXEC fd, or -1 on error. */
|
||||
int file_open_temp_dir(const char* dir_path) {
|
||||
if (!dir_path)
|
||||
return -1;
|
||||
return open(dir_path, O_RDONLY | O_DIRECTORY | O_CLOEXEC);
|
||||
}
|
||||
|
||||
/* After the content and mode/times are restored on the just-written file, apply
|
||||
* the per-file xattrs (-X/-A) and, for --fake-super, park the source's
|
||||
* uid/gid/mode/mtime in the reserved xattr. All fd-relative (confined to the
|
||||
* destination file) and best-effort: a per-attribute or privilege failure is
|
||||
* logged and skipped, never fatal. */
|
||||
static void restore_extra_fd(int fd, const FileMetadata* metadata, const FileXattrList* xattrs,
|
||||
bool fake_super) {
|
||||
bool fake_super, FileAttrPolicy policy) {
|
||||
xattr_apply_fd(fd, xattrs);
|
||||
if (fake_super && metadata) {
|
||||
fake_super_store_fd(fd, (uint32_t)metadata->uid, (uint32_t)metadata->gid,
|
||||
(uint32_t)metadata->mode, metadata->mtime_sec, metadata->mtime_nsec);
|
||||
/* Replay: re-apply the recorded uid/gid/mode/mtime fd-relative so a save
|
||||
under --fake-super restores the attrs (when privileged) instead of only
|
||||
recording them. Best-effort; fake_super_restore_fd silently skips a
|
||||
non-root fchown EPERM/EACCES and never fatal. */
|
||||
fake_super_restore_fd(fd);
|
||||
/* Record the ownership that WOULD have been applied: when an explicit
|
||||
ownership request (--chown/--usermap/--groupmap/--copy-as or -o/-g) is
|
||||
active, the resolved mapping; otherwise the source's own id. The real
|
||||
chown is suppressed (identity_apply_ownership early-returns under
|
||||
--fake-super) so recording never defeats the flag. Mode/mtime are still
|
||||
replayed (policy-gated) so unprivileged --fake-super keeps working. */
|
||||
uint32_t store_uid;
|
||||
uint32_t store_gid;
|
||||
identity_resolve_storage_ids((int32_t)metadata->uid, (int32_t)metadata->gid, &store_uid,
|
||||
&store_gid);
|
||||
fake_super_store_fd(fd, store_uid, store_gid, (uint32_t)metadata->mode, metadata->mtime_sec,
|
||||
metadata->mtime_nsec);
|
||||
fake_super_restore_fd(fd, policy);
|
||||
}
|
||||
}
|
||||
|
||||
static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool update, bool no_replace,
|
||||
FileAttrPolicy policy, bool update, bool no_replace,
|
||||
bool use_fsync, const char* temp_dir,
|
||||
const FileXattrList* xattrs, bool fake_super,
|
||||
bool keep_partial) {
|
||||
@@ -908,27 +1020,65 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
return false;
|
||||
int fd = -1;
|
||||
bool ok = false;
|
||||
/* Set when a --temp-dir install fails with EXDEV: rsync then falls back to a
|
||||
* non-atomic write directly in the destination directory (see the tail of
|
||||
* this function). */
|
||||
bool cross_device_fallback = false;
|
||||
/* The base mode applied when --perms is off (neither the source mode nor an
|
||||
* exec-only change is taken wholesale): a pre-existing destination keeps its
|
||||
* own mode (special bits dropped), while a brand-new file uses
|
||||
* source&~umask when metadata is available (see file_mode_base) or 0644 when
|
||||
* there is none. Captured from the destination probe before the write. */
|
||||
mode_t existing_mode = S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH;
|
||||
bool existing_mode_known = false;
|
||||
if (inplace) {
|
||||
/* --inplace writes directly into the destination; a scratch --temp-dir
|
||||
does not apply and must never redirect these writes. */
|
||||
fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_CLOEXEC | O_NOFOLLOW, 0644);
|
||||
/* Type gate BEFORE opening: an existing destination entry that is not a
|
||||
regular file (FIFO, socket, char/block device, directory) must never be
|
||||
opened for writing. Opening a FIFO would block the receive thread
|
||||
forever and writing into a device would bypass the --write-devices /
|
||||
super-mode gate (a client-controlled device write). fstatat with
|
||||
AT_SYMLINK_NOFOLLOW does not follow a symlink and does not block. */
|
||||
struct stat pre_stat;
|
||||
if (fstatat(dirfd, leaf, &pre_stat, AT_SYMLINK_NOFOLLOW) == 0) {
|
||||
if (!S_ISREG(pre_stat.st_mode)) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
/* Capture the old destination mode before the overwrite so a no--p/-E
|
||||
* write can restore it (the write itself may clear setuid/setgid). */
|
||||
existing_mode = pre_stat.st_mode & 0777;
|
||||
existing_mode_known = true;
|
||||
}
|
||||
/* O_NONBLOCK: a no-op for a regular file, but a raced-in FIFO cannot block
|
||||
the open before the post-open S_ISREG re-check rejects it. */
|
||||
fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_CLOEXEC | O_NOFOLLOW | O_NONBLOCK, 0644);
|
||||
if (fd >= 0) {
|
||||
struct stat destination_stat;
|
||||
/* Re-check the opened descriptor: a concurrent replacement between the
|
||||
fstatat probe and the open (or a device/FIFO raced in) must never be
|
||||
written through. */
|
||||
if (fstat(fd, &destination_stat) != 0 || !S_ISREG(destination_stat.st_mode)) {
|
||||
close(fd);
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
bool newer = false;
|
||||
if (update && metadata && fstat(fd, &destination_stat) == 0 &&
|
||||
S_ISREG(destination_stat.st_mode)) {
|
||||
newer = stat_is_newer(&destination_stat, metadata);
|
||||
if (update && metadata && stat_is_newer(&destination_stat, metadata)) {
|
||||
newer = true;
|
||||
}
|
||||
if (newer) {
|
||||
ok = true;
|
||||
} else {
|
||||
/* Preallocate the expected payload size before writing so an
|
||||
out-of-space condition fails cleanly up front (--preallocate).
|
||||
--sparse takes precedence: posix_fallocate would allocate every
|
||||
block, defeating the holes the sparse writer would create, so the
|
||||
two never combine here (the ftruncate presize below stays). */
|
||||
rsync lets --preallocate win over --sparse (the reserved blocks
|
||||
survive the sparse writer's seeks), so both flags can be active. */
|
||||
int prealloc_rc = 0;
|
||||
if (preallocate && !sparse && data_size > 0) {
|
||||
if (preallocate && data_size > 0) {
|
||||
prealloc_rc = preallocate_fd(fd, data_size);
|
||||
if (prealloc_rc != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
@@ -952,15 +1102,25 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
/* Normalize the mode: apply the metadata-derived safe mode when the
|
||||
sender supplied metadata (setuid/setgid/sticky are never honored);
|
||||
otherwise fall back to a safe default so dangerous bits on an
|
||||
existing destination cannot survive an overwrite. */
|
||||
existing destination cannot survive an overwrite. When the policy
|
||||
requests neither -p nor -E the source mode is deliberately ignored
|
||||
and the pre-existing destination mode (or 0644 for a new file) is
|
||||
restored instead. The exec-bits-only -E change is likewise applied
|
||||
on top of that destination-derived base, not the scratch file's
|
||||
0600. */
|
||||
if (ok) {
|
||||
if (metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
else if (fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0)
|
||||
if (metadata) {
|
||||
if (!policy.perms &&
|
||||
fchmod(fd, file_mode_base(metadata, existing_mode_known, existing_mode)) != 0)
|
||||
ok = false;
|
||||
if (ok)
|
||||
ok = file_restore_metadata_fd(fd, metadata, policy);
|
||||
} else if (fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0) {
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
if (ok)
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super);
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super, policy);
|
||||
if (ok && use_fsync)
|
||||
ok = fsync(fd) == 0;
|
||||
}
|
||||
@@ -973,25 +1133,31 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
install failure (partial data may exist, --partial may retain it) from a
|
||||
pre-write validation failure (nothing to retain). */
|
||||
bool write_attempted = false;
|
||||
if (update && metadata) {
|
||||
/* This check protects the normal atomic path as far as possible. A
|
||||
concurrent replacement can still occur before the final rename. */
|
||||
/* Probe the destination ONCE up front: it both drives the --update check
|
||||
and records the pre-existing mode the no--p/-E fallback preserves. */
|
||||
struct stat destination_stat;
|
||||
if (fstatat(dirfd, leaf, &destination_stat, AT_SYMLINK_NOFOLLOW) == 0 &&
|
||||
S_ISREG(destination_stat.st_mode) && stat_is_newer(&destination_stat, metadata)) {
|
||||
bool destination_is_regular =
|
||||
fstatat(dirfd, leaf, &destination_stat, AT_SYMLINK_NOFOLLOW) == 0 &&
|
||||
S_ISREG(destination_stat.st_mode);
|
||||
if (destination_is_regular) {
|
||||
existing_mode = destination_stat.st_mode & 0777;
|
||||
existing_mode_known = true;
|
||||
}
|
||||
if (update && metadata && destination_is_regular &&
|
||||
stat_is_newer(&destination_stat, metadata)) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
/* Scratch directory for the temporary working copy. When NULL the temp
|
||||
file is created in the destination directory, exactly as historically. */
|
||||
int scratch_dirfd = -1;
|
||||
if (temp_dir) {
|
||||
scratch_dirfd = file_open_private_dir(temp_dir);
|
||||
scratch_dirfd = file_open_temp_dir(temp_dir);
|
||||
if (scratch_dirfd < 0) {
|
||||
int saved_errno = errno;
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s",
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--temp-dir '%s' could not be opened (rsync requires it to already exist): %s",
|
||||
temp_dir, strerror(saved_errno));
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
@@ -1038,7 +1204,7 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
if (fd < 0)
|
||||
continue; /* EEXIST (or a transient open error): try a fresh name. */
|
||||
int prealloc_rc = 0;
|
||||
if (preallocate && !sparse && data_size > 0) {
|
||||
if (preallocate && data_size > 0) {
|
||||
prealloc_rc = preallocate_fd(fd, data_size);
|
||||
if (prealloc_rc != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
@@ -1060,10 +1226,19 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
? file_store_write_sparse(fd, (const unsigned char*)data, data_size)
|
||||
: write_all(fd, data, data_size);
|
||||
}
|
||||
if (ok && metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
if (ok) {
|
||||
if (metadata) {
|
||||
if (!policy.perms &&
|
||||
fchmod(fd, file_mode_base(metadata, existing_mode_known, existing_mode)) != 0)
|
||||
ok = false;
|
||||
if (ok)
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super);
|
||||
ok = file_restore_metadata_fd(fd, metadata, policy);
|
||||
} else if (fchmod(fd, S_IRUSR | S_IWUSR | S_IRGRP | S_IROTH) != 0) {
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
if (ok)
|
||||
restore_extra_fd(fd, metadata, xattrs, fake_super, policy);
|
||||
if (ok && use_fsync)
|
||||
ok = fsync(fd) == 0;
|
||||
}
|
||||
@@ -1080,17 +1255,16 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
errno != ENOENT)
|
||||
ok = false;
|
||||
} else {
|
||||
/* Cross-device (or otherwise impossible) link: rsync falls back to
|
||||
writing the file directly in the destination directory. Record
|
||||
it and retry below with no scratch dir. */
|
||||
if (scratch_dirfd >= 0 && errno == EXDEV)
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"temp dir is on a different filesystem than the destination; cannot "
|
||||
"link file into place (EXDEV); no fallback copy is attempted");
|
||||
cross_device_fallback = true;
|
||||
ok = false;
|
||||
}
|
||||
} else if (renameat(scratch_dirfd >= 0 ? scratch_dirfd : dirfd, tmp, dirfd, leaf) != 0) {
|
||||
if (scratch_dirfd >= 0 && errno == EXDEV)
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"temp dir is on a different filesystem than the destination; cannot "
|
||||
"atomically install file (EXDEV); no fallback copy is attempted");
|
||||
cross_device_fallback = true;
|
||||
ok = false;
|
||||
}
|
||||
}
|
||||
@@ -1123,43 +1297,49 @@ static bool file_to_disk_secure_impl(const char* path, const void* data,
|
||||
close(fd);
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
if (cross_device_fallback) {
|
||||
/* rsync semantics: a --temp-dir on another filesystem must not abort the
|
||||
write. Retry once with no scratch dir so the file is written and
|
||||
installed non-atomically in the destination directory. */
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"temp dir is on a different filesystem than the destination; falling back to a "
|
||||
"non-atomic copy into the destination directory");
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
policy, update, no_replace, use_fsync, NULL, xattrs, fake_super,
|
||||
keep_partial);
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, const char* temp_dir) {
|
||||
FileAttrPolicy policy, const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, false, false, false, temp_dir, NULL,
|
||||
false, false);
|
||||
policy, false, false, false, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, true, false, false, temp_dir, NULL, false,
|
||||
false);
|
||||
policy, true, false, false, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
const char* temp_dir) {
|
||||
FileAttrPolicy policy, bool use_fsync, const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, false, false, use_fsync, temp_dir, NULL,
|
||||
false, false);
|
||||
policy, false, false, use_fsync, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
unsigned long long data_size, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, false, sparse, preallocate, metadata,
|
||||
preserve_executability, false, true, false, temp_dir, NULL, false,
|
||||
false);
|
||||
policy, false, true, false, temp_dir, NULL, false, false);
|
||||
}
|
||||
|
||||
/* Receiver write-path variant that also applies the per-file xattrs (-X/-A)
|
||||
@@ -1169,13 +1349,12 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
* failed write's temp. See file_to_disk_secure_impl for the semantics. */
|
||||
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool update, bool no_replace, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super, bool keep_partial,
|
||||
const char* temp_dir) {
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
|
||||
bool fake_super, bool keep_partial, const char* temp_dir) {
|
||||
return file_to_disk_secure_impl(path, data, data_size, inplace, sparse, preallocate, metadata,
|
||||
preserve_executability, update, no_replace, use_fsync, temp_dir,
|
||||
xattrs, fake_super, keep_partial);
|
||||
policy, update, no_replace, use_fsync, temp_dir, xattrs,
|
||||
fake_super, keep_partial);
|
||||
}
|
||||
|
||||
/* Atomic --link-dest install. The destination is replaced (via a temporary
|
||||
@@ -1196,7 +1375,7 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
|
||||
static bool file_to_disk_secure_link_impl(const char* path, const char* basis_path,
|
||||
const void* data, unsigned long long data_size,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
FileAttrPolicy policy, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir) {
|
||||
if (!path || !basis_path)
|
||||
@@ -1208,11 +1387,12 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
|
||||
int scratch_dirfd = -1;
|
||||
if (temp_dir) {
|
||||
scratch_dirfd = file_open_private_dir(temp_dir);
|
||||
scratch_dirfd = file_open_temp_dir(temp_dir);
|
||||
if (scratch_dirfd < 0) {
|
||||
int saved_errno = errno;
|
||||
log_message(LOG_LEVEL_ERROR, "could not open --temp-dir scratch directory '%s': %s", temp_dir,
|
||||
strerror(saved_errno));
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--temp-dir '%s' could not be opened (rsync requires it to already exist): %s",
|
||||
temp_dir, strerror(saved_errno));
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
@@ -1247,10 +1427,18 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
if (linked) {
|
||||
int target_dirfd = scratch_dirfd >= 0 ? scratch_dirfd : dirfd;
|
||||
if (use_fsync) {
|
||||
int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (tfd < 0 || fsync(tfd) != 0) {
|
||||
/* O_NONBLOCK: the freshly linked temp is normally the basis's regular
|
||||
file, but a raced-in FIFO at the name must not block this reopen
|
||||
forever. With O_NONBLOCK such an open fails with ENXIO instead of
|
||||
blocking, which is treated as a benign fsync-skip (the link itself
|
||||
is still installed); any other open/fsync failure falls back to the
|
||||
byte-copy path as before. */
|
||||
int tfd = openat(target_dirfd, tmp, O_RDONLY | O_NOFOLLOW | O_CLOEXEC | O_NONBLOCK);
|
||||
if (tfd < 0) {
|
||||
if (errno != ENXIO)
|
||||
linked = false;
|
||||
} else if (fsync(tfd) != 0) {
|
||||
linked = false;
|
||||
if (tfd >= 0)
|
||||
close(tfd);
|
||||
} else {
|
||||
close(tfd);
|
||||
@@ -1277,8 +1465,8 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
/* The basis file could not be linked in (missing, cross-device, refused
|
||||
by the filesystem). Write a byte-identical local copy instead. */
|
||||
return file_to_disk_secure_attrs(path, data, data_size, false, false, preallocate, metadata,
|
||||
preserve_executability, false, false, use_fsync, xattrs,
|
||||
fake_super, false, temp_dir);
|
||||
policy, false, false, use_fsync, xattrs, fake_super, false,
|
||||
temp_dir);
|
||||
}
|
||||
|
||||
if (scratch_dirfd >= 0)
|
||||
@@ -1290,25 +1478,25 @@ static bool file_to_disk_secure_link_impl(const char* path, const char* basis_pa
|
||||
|
||||
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool use_fsync, const char* temp_dir) {
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata,
|
||||
preserve_executability, use_fsync, NULL, false, temp_dir);
|
||||
policy, use_fsync, NULL, false, temp_dir);
|
||||
}
|
||||
|
||||
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir) {
|
||||
return file_to_disk_secure_link_impl(path, basis_path, data, data_size, preallocate, metadata,
|
||||
preserve_executability, use_fsync, xattrs, fake_super,
|
||||
temp_dir);
|
||||
policy, use_fsync, xattrs, fake_super, temp_dir);
|
||||
}
|
||||
|
||||
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse) {
|
||||
if (!path || (!data && data_size != 0) || has_path_traversal(path))
|
||||
return false;
|
||||
return file_to_disk_secure(path, data, data_size, inplace, sparse, false, NULL, false, NULL);
|
||||
FileAttrPolicy policy = {false, false, false, false};
|
||||
return file_to_disk_secure(path, data, data_size, inplace, sparse, false, NULL, policy, NULL);
|
||||
}
|
||||
|
||||
+50
-32
@@ -28,17 +28,36 @@ void file_metadata_destroy(void* metadata);
|
||||
/* --open-noatime process-wide sender policy; see file.c. */
|
||||
void file_set_open_noatime(bool enable);
|
||||
bool file_get_open_noatime(void);
|
||||
/* Capture the process umask ONCE, before any threads are created. Call this at
|
||||
* the very top of main() in both entry points so the cached value is read while
|
||||
* the process is still single-threaded: reading the umask needs a get+set round
|
||||
* trip (umask(0); umask(old)), which would race against receiver threads
|
||||
* creating files if it happened during the first write. Idempotent and safe to
|
||||
* call more than once. */
|
||||
void file_umask_capture(void);
|
||||
/* Process-wide umask, captured once (thread-safe). Used to derive the mode of
|
||||
* a brand-new destination like rsync: source_mode & 0777 & ~umask. Falls back
|
||||
* to file_umask_capture() (behind pthread_once) if capture was never called. */
|
||||
unsigned file_process_umask(void);
|
||||
/* Open `path` read-only for transfer, honouring --open-noatime when set. */
|
||||
int file_open_for_read(const char* path);
|
||||
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse);
|
||||
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave). --munge-links
|
||||
* sender-side marker: every transmitted symlink target is prefixed with this
|
||||
* while the flag is on; the receiver strips it to restore the real target. */
|
||||
#define SYMLINK_MUNGE_PREFIX "#SYMLINK/"
|
||||
/* Symlink trust-boundary helpers (Phase 4, symlink wave; rsync parity).
|
||||
* --munge-links is a RECEIVER-side rewrite: rsync prefixes every stored symlink
|
||||
* target with this marker, making the link unusable while the referenced
|
||||
* directory does not exist. A SENDER receiving a munged source strips it back
|
||||
* off before transmitting (so a munged tree round-trips through the receiver's
|
||||
* re-munging). */
|
||||
#define SYMLINK_MUNGE_PREFIX "/rsyncd-munged/"
|
||||
|
||||
char* file_symlink_munge(const char* target);
|
||||
/* rsync 3.4.1 unsafe_symlink(): true when `target` escapes the transfer tree
|
||||
* rooted at `link_path` (the symlink's transfer-relative path incl. its name).
|
||||
* Absolute/empty targets and targets climbing above the transfer root (via
|
||||
* "..") are unsafe, as are internal "/../" components and trailing "/..". */
|
||||
bool file_symlink_unsafe(const char* target, const char* link_path);
|
||||
/* True when a lexical target is relative and contains no ".." component, so it
|
||||
* can never escape the receive root once created beneath it. */
|
||||
bool file_symlink_target_contained(const char* target);
|
||||
@@ -46,13 +65,13 @@ bool file_symlink_target_contained(const char* target);
|
||||
* returns true when a marker was removed. */
|
||||
bool file_symlink_unmunge(char* target);
|
||||
/* Create a symlink at `path` -> `target`, confined below the authorized root
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). Returns
|
||||
* false when a directory already occupies `path`. */
|
||||
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). The link
|
||||
* value is copied verbatim (rsync -l); only the placement path is confined.
|
||||
* Returns false when a directory already occupies `path`. */
|
||||
bool file_symlink_at_secure(const char* path, const char* target);
|
||||
/* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing
|
||||
* symlink-to-directory to be followed as a directory. */
|
||||
void file_set_keep_dirlinks(bool enable);
|
||||
bool file_get_keep_dirlinks(void);
|
||||
|
||||
/* --trust-sender receiver process-wide policy (Phase 5). When set, the
|
||||
* receiver trusts that the sender already produced a clean file list and skips
|
||||
@@ -63,9 +82,6 @@ bool file_get_keep_dirlinks(void);
|
||||
void file_set_trust_sender(bool enable);
|
||||
bool file_get_trust_sender(void);
|
||||
|
||||
/* A configured fd without a canonical identity deliberately rejects paths. */
|
||||
bool file_set_authorized_root(int fd, const char* canonical_path);
|
||||
|
||||
/* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */
|
||||
bool file_path_exists_secure(const char* path);
|
||||
bool file_stat_secure(const char* path, struct stat* st);
|
||||
@@ -79,37 +95,40 @@ bool file_rename_secure(const char* old_path, const char* new_path);
|
||||
regular file. See the .c for the exact success semantics. */
|
||||
bool file_remove_tree_secure(const char* path);
|
||||
/* Open a private 0700 directory (creating it on demand) that must live below
|
||||
the authorized root. Used for the --temp-dir scratch directory and the
|
||||
--delay-updates staging directory. */
|
||||
the authorized root. Used for the --delay-updates staging directory. */
|
||||
int file_open_private_dir(const char* dir_path);
|
||||
|
||||
/* Open an existing --temp-dir scratch directory as-is (absolute or relative;
|
||||
no creation, no root confinement), matching rsync's --temp-dir handling. */
|
||||
int file_open_temp_dir(const char* dir_path);
|
||||
|
||||
/* The file_to_disk_secure* variants write a temporary copy in the destination
|
||||
directory and atomically rename it over `path`. temp_dir is an absolute,
|
||||
root-confined scratch directory (already validated by the caller): when it
|
||||
is non-NULL the temporary copy is instead created there (with a name unique
|
||||
across the whole scratch directory) and atomically renamed into the
|
||||
destination directory once fully written and fsynced. A rename across
|
||||
filesystems (EXDEV) fails the write with an error; the file is never
|
||||
silently copied into place. Pass NULL for the historical same-directory
|
||||
behavior. --inplace writes never use temp_dir. */
|
||||
directory and atomically rename it over `path`. temp_dir is a scratch
|
||||
directory (an absolute path, or one the caller already resolved against the
|
||||
destination root): when it is non-NULL the temporary copy is instead created
|
||||
there (with a name unique across the whole scratch directory) and atomically
|
||||
renamed into the destination directory once fully written and fsynced. When
|
||||
that rename/link fails with EXDEV (the scratch dir is on another filesystem)
|
||||
the write falls back to a non-atomic copy directly in the destination
|
||||
directory, matching rsync. Pass NULL for the same-directory behavior.
|
||||
--inplace writes never use temp_dir. */
|
||||
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, const char* temp_dir);
|
||||
FileAttrPolicy policy, const char* temp_dir);
|
||||
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
|
||||
unsigned long long data_size, bool inplace, bool sparse,
|
||||
bool preallocate, const FileMetadata* metadata,
|
||||
bool preserve_executability, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
FileAttrPolicy policy, bool use_fsync, const char* temp_dir);
|
||||
/* With update enabled, an existing newer destination is left untouched. The
|
||||
check is descriptor-based for inplace writes; atomic replacement still has
|
||||
an unavoidable final rename race without filesystem locking. */
|
||||
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
unsigned long long data_size, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
const char* temp_dir);
|
||||
/* Receiver write-path variant that also applies per-file xattrs (-X/-A) and the
|
||||
* --fake-super stat xattr fd-relative before the final rename. `update` /
|
||||
@@ -117,10 +136,9 @@ bool file_to_disk_secure_no_replace(const char* path, const void* data,
|
||||
* enables --partial best-effort retention of a failed write's temp. */
|
||||
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool update, bool no_replace, bool use_fsync,
|
||||
const FileXattrList* xattrs, bool fake_super, bool keep_partial,
|
||||
const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool update,
|
||||
bool no_replace, bool use_fsync, const FileXattrList* xattrs,
|
||||
bool fake_super, bool keep_partial, const char* temp_dir);
|
||||
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
|
||||
(via a temp name + rename); fall back to a byte-identical local copy from
|
||||
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
|
||||
@@ -129,15 +147,15 @@ bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long
|
||||
never re-allocated). */
|
||||
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
bool use_fsync, const char* temp_dir);
|
||||
const FileMetadata* metadata, FileAttrPolicy policy, bool use_fsync,
|
||||
const char* temp_dir);
|
||||
/* Like file_to_disk_secure_link, but the byte-copy fallback also applies the
|
||||
* per-file xattrs (-X/-A) and --fake-super stat xattr (fd-relative). On a
|
||||
* successful hard link no attributes are applied (the shared inode already
|
||||
* carries the basis's). */
|
||||
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
|
||||
unsigned long long data_size, bool preallocate,
|
||||
const FileMetadata* metadata, bool preserve_executability,
|
||||
const FileMetadata* metadata, FileAttrPolicy policy,
|
||||
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
|
||||
const char* temp_dir);
|
||||
|
||||
|
||||
@@ -0,0 +1,38 @@
|
||||
#ifndef FILE_ATTR_H
|
||||
#define FILE_ATTR_H
|
||||
|
||||
#include "config.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
|
||||
/*
|
||||
* Per-attribute receiver policy for applying a transmitted FileMetadata. This
|
||||
* is the split-out replacement for the former single use_metadata bundle: each
|
||||
* flag is applied independently, matching rsync's -p/-t/-o/-g/-E/-U semantics.
|
||||
* `use_metadata` remains the transport/presence gate (whether the metadata frame
|
||||
* travelled at all); this struct decides which attributes are ACTUALLY applied.
|
||||
*
|
||||
* It lives in its own header (rather than metadata.h) because xattr.h's
|
||||
* fake_super_restore_fd() takes one and metadata.h <-> file_types.h form an
|
||||
* include cycle that must not be entered from xattr.h.
|
||||
*
|
||||
* The mode leg is: perms wins over executability; an exec-bits-only change is
|
||||
* made only when perms is off; when neither is set the receiver deliberately
|
||||
* sets no source mode. file.c then substitutes the pre-existing destination
|
||||
* mode for a brand-new destination with metadata it uses the sanitized
|
||||
* source-mode-&-umask base (S_IWGRP|S_IWOTH cleared), and the fixed 0644
|
||||
* default only when no metadata is available at all, so a no--p overwrite
|
||||
* does not lose the destination's perms.
|
||||
*/
|
||||
typedef struct FileAttrPolicy {
|
||||
bool perms; /* config->preserve_perms: apply the source mode bits */
|
||||
bool times; /* config->preserve_times: apply the source mtime */
|
||||
bool atimes; /* config->preserve_atimes (-U): apply the source atime */
|
||||
bool executability; /* config->use_executability (-E): exec-bits-only mode */
|
||||
} FileAttrPolicy;
|
||||
|
||||
/* Build the per-attribute policy from a connection's Config. A NULL config
|
||||
* yields the all-off policy (no attribute application). */
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config);
|
||||
|
||||
#endif
|
||||
+99
-22
@@ -2,6 +2,7 @@
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -22,6 +23,8 @@ static void string_list_destroy(StringList* list) {
|
||||
|
||||
static bool string_list_add(StringList* list, const char* text) {
|
||||
if (list->count == list->capacity) {
|
||||
if (list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_cap = list->capacity > 0 ? list->capacity * 2 : 16;
|
||||
char** grown = realloc(list->items, (size_t)new_cap * sizeof(char*));
|
||||
if (!grown)
|
||||
@@ -51,11 +54,18 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings,
|
||||
if (len == 0)
|
||||
return 0;
|
||||
if (raw[0] == '/') {
|
||||
snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", (int)len, raw);
|
||||
int print_len = len > (size_t)INT_MAX ? INT_MAX : (int)len;
|
||||
snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", print_len, raw);
|
||||
return -1;
|
||||
}
|
||||
/* Reject NUL bytes inside a token defensively. In NUL-delimited mode the
|
||||
* delimiter itself is the final byte and is expected; in line mode any NUL is
|
||||
* embedded garbage (strlen-based parsing would otherwise silently truncate). */
|
||||
size_t scan_len = strip_line_endings ? len : len - 1;
|
||||
if (memchr(raw, '\0', scan_len)) {
|
||||
snprintf(err, err_size, "entry contains an embedded NUL byte");
|
||||
return -1;
|
||||
}
|
||||
/* Reject NUL bytes inside a token defensively (NUL-delimited mode splits on
|
||||
* them, so this only guards against embedded garbage). */
|
||||
char* dup = malloc(len + 1);
|
||||
if (!dup) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
@@ -102,8 +112,27 @@ static int normalize_entry(const char* raw, size_t len, bool strip_line_endings,
|
||||
return result;
|
||||
}
|
||||
|
||||
/* Build the membership index over the exact entries only. `file_list_affects`
|
||||
combines the exact/descendant lookups with a walk of the query's own ancestor
|
||||
prefixes, so no ancestor prefix is ever materialized as a copy and the index
|
||||
stays O(entry count) memory regardless of path depth. An empty entry (the
|
||||
source root) sets whole_tree and short-circuits every query. */
|
||||
static bool file_list_index_build(FileListSet* set, char* err, size_t err_size) {
|
||||
if (!path_index_build(&set->index, (const char* const*)set->entries, (size_t)set->count)) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < set->count; i++) {
|
||||
if (set->entries[i][0] == '\0') {
|
||||
set->whole_tree = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_size) {
|
||||
FileListSet* set = malloc(sizeof(FileListSet));
|
||||
FileListSet* set = calloc(1, sizeof(FileListSet));
|
||||
if (!set) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
@@ -112,6 +141,10 @@ static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_si
|
||||
set->entries = raw->items;
|
||||
raw->items = NULL;
|
||||
raw->count = 0;
|
||||
if (!file_list_index_build(set, err, err_size)) {
|
||||
file_list_destroy(set);
|
||||
return NULL;
|
||||
}
|
||||
return set;
|
||||
}
|
||||
|
||||
@@ -133,10 +166,20 @@ FileListSet* file_list_load(const char* path, bool null_separated, char* err, si
|
||||
StringList raw = {0};
|
||||
char* line = NULL;
|
||||
size_t line_cap = 0;
|
||||
ssize_t n;
|
||||
bool ok = true;
|
||||
char delim = null_separated ? '\0' : '\n';
|
||||
while (ok && (n = getdelim(&line, &line_cap, delim, fp)) != -1) {
|
||||
while (ok) {
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, delim, UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
if (errno == EFBIG)
|
||||
snprintf(err, err_size, "entry in file list exceeds %d bytes", (int)UTILS_MAX_LINE_LEN);
|
||||
else
|
||||
snprintf(err, err_size, "error reading file list: %s", strerror(errno));
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (n == 0)
|
||||
break;
|
||||
int r = normalize_entry(line, (size_t)n, !null_separated, &raw, err, err_size);
|
||||
if (r < 0) {
|
||||
ok = false;
|
||||
@@ -158,34 +201,68 @@ FileListSet* file_list_load(const char* path, bool null_separated, char* err, si
|
||||
void file_list_destroy(FileListSet* set) {
|
||||
if (!set)
|
||||
return;
|
||||
path_index_free(&set->index);
|
||||
for (int i = 0; i < set->count; i++)
|
||||
free(set->entries[i]);
|
||||
free(set->entries);
|
||||
free(set);
|
||||
}
|
||||
|
||||
static bool path_has_prefix(const char* path, const char* prefix) {
|
||||
size_t plen = strlen(prefix);
|
||||
if (strncmp(path, prefix, plen) != 0)
|
||||
return false;
|
||||
return path[plen] == '/' || path[plen] == '\0';
|
||||
}
|
||||
|
||||
bool file_list_affects(const FileListSet* set, const char* rel) {
|
||||
if (!set)
|
||||
return true;
|
||||
if (!rel)
|
||||
return false;
|
||||
for (int i = 0; i < set->count; i++) {
|
||||
const char* entry = set->entries[i];
|
||||
if (entry[0] == '\0')
|
||||
if (set->whole_tree)
|
||||
return true; /* whole tree listed */
|
||||
if (strcmp(rel, entry) == 0)
|
||||
return true; /* the entry itself is listed */
|
||||
if (path_has_prefix(rel, entry))
|
||||
return true; /* rel lives under a listed directory */
|
||||
if (path_has_prefix(entry, rel))
|
||||
return true; /* rel is an ancestor directory of a listed entry */
|
||||
/* An exact entry match means `rel` itself is listed. */
|
||||
if (path_index_contains(&set->index, rel))
|
||||
return true;
|
||||
/* Otherwise `rel` is affected when a listed entry is an ancestor directory of
|
||||
it; walk rel's own directory prefixes (which preserve path-boundary
|
||||
semantics) and test each for an exact entry. No prefixes are stored. */
|
||||
size_t len = strlen(rel);
|
||||
while (len > 0) {
|
||||
const char* slash = NULL;
|
||||
for (size_t i = len; i-- > 0;) {
|
||||
if (rel[i] == '/') {
|
||||
slash = rel + i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!slash)
|
||||
break;
|
||||
len = (size_t)(slash - rel);
|
||||
if (path_index_contains_n(&set->index, rel, len))
|
||||
return true;
|
||||
}
|
||||
/* Finally `rel` is affected when it is an ancestor directory of a listed
|
||||
entry (binary search for the first entry at or after `rel` + '/'). */
|
||||
return path_index_has_descendant(&set->index, rel);
|
||||
}
|
||||
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel) {
|
||||
if (!set || set->whole_tree)
|
||||
return true;
|
||||
if (!rel || rel[0] == '\0')
|
||||
return false;
|
||||
/* `rel` itself is listed, or one of its ancestor prefixes is an exact listed
|
||||
directory (a listed prefix of a directory path is necessarily a
|
||||
directory). */
|
||||
size_t len = strlen(rel);
|
||||
while (len > 0) {
|
||||
const char* slash = NULL;
|
||||
for (size_t i = len; i-- > 0;) {
|
||||
if (rel[i] == '/') {
|
||||
slash = rel + i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!slash)
|
||||
break;
|
||||
len = (size_t)(slash - rel);
|
||||
if (path_index_contains_n(&set->index, rel, len))
|
||||
return true;
|
||||
}
|
||||
return path_index_contains(&set->index, rel);
|
||||
}
|
||||
|
||||
+20
-2
@@ -1,6 +1,7 @@
|
||||
#ifndef FILE_LIST_H
|
||||
#define FILE_LIST_H
|
||||
|
||||
#include "utils.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
|
||||
@@ -12,11 +13,18 @@
|
||||
* of "." means the whole tree, absolute entries and ".." traversal are
|
||||
* rejected at parse time. The set is immutable and shared read-only across
|
||||
* scanner worker threads.
|
||||
*/
|
||||
|
||||
*
|
||||
* Membership is answered from `index`, built once at load time over the exact
|
||||
* entries only: `index.exact` matches a listed path, the sorted view detects an
|
||||
* ancestor directory of a listed entry, and `rel`'s own directory prefixes are
|
||||
* matched against the exact set while descending. No ancestor prefix is stored
|
||||
* as a separate string, so the index is O(entry count) memory however deep the
|
||||
* paths are, and each query is O(path length) comparisons. */
|
||||
typedef struct {
|
||||
char** entries; /* normalized rel paths; "" means the whole tree */
|
||||
int count;
|
||||
PathIndex index;
|
||||
bool whole_tree; /* an entry of "" lists the source root */
|
||||
} FileListSet;
|
||||
|
||||
/* Load and validate a --files-from file. When `null_separated` (-0/--from0)
|
||||
@@ -32,4 +40,14 @@ void file_list_destroy(FileListSet* set);
|
||||
* this returns true, files are transferred only when it returns true. */
|
||||
bool file_list_affects(const FileListSet* set, const char* rel);
|
||||
|
||||
/* True when the DIRECTORY `rel` (path relative to the source root) is inside a
|
||||
* listed directory subtree: `rel` itself is a listed entry, or one of `rel`'s
|
||||
* ancestor directory prefixes is an exact listed entry. Unlike
|
||||
* file_list_affects this does NOT treat an ancestor of a listed entry as
|
||||
* affected, so an implied parent directory of a listed file is not synchronized
|
||||
* (rsync deletes nothing in it). With no set or a whole-tree set every
|
||||
* directory is in scope. This is the delete-walker's "synchronized directory"
|
||||
* predicate. */
|
||||
bool file_list_dir_in_scope(const FileListSet* set, const char* rel);
|
||||
|
||||
#endif
|
||||
|
||||
+1326
-595
File diff suppressed because it is too large
Load Diff
+88
-14
@@ -7,6 +7,15 @@
|
||||
|
||||
/* Server-side file receive/save path. */
|
||||
|
||||
/* Cumulative caps for the deferred directory-time accumulator. The sender may
|
||||
* legitimately split a large tree across repeated STATUS_DIR_TIMES frames, so a
|
||||
* per-frame bound is not enough: the receiver must bound the TOTAL it retains
|
||||
* against a hostile sender. Mirror the delete-manifest limits
|
||||
* (MAX_MANIFEST_ENTRIES / MAX_MANIFEST_BYTES): the entry count bounds the
|
||||
* metadata array and the byte budget bounds the concatenated path strings. */
|
||||
#define MAX_DIR_TIME_ENTRIES (1024 * 1024)
|
||||
#define MAX_DIR_TIME_BYTES (16ULL * 1024 * 1024)
|
||||
|
||||
File* file_receive(const Config* config, int file_descriptor);
|
||||
File* file_receive_directory(int file_descriptor, const Config* config);
|
||||
File* file_receive_dir_time(int file_descriptor, const Config* config);
|
||||
@@ -15,6 +24,13 @@ File* file_receive_symlink(int file_descriptor, const Config* config);
|
||||
File* file_receive_special(int file_descriptor);
|
||||
bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode);
|
||||
File* receive_incremental_check(int fd, const Config* config, bool* skipped);
|
||||
/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set
|
||||
* true only on the server-contacting --dry-run path when the file is not up to
|
||||
* date: the receiver has already sent STATUS_DRY_RUN_TRANSFER and returns NULL
|
||||
* without storing anything. On that path `*skipped` is true for an up-to-date
|
||||
* (STATUS_OK) file and both flags are false for a genuine error. */
|
||||
File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped,
|
||||
bool* would_transfer);
|
||||
|
||||
/* P7 Wave D directory-time accumulator. The receiver collects the metadata of
|
||||
* every directory it creates/receives (STATUS_MKDIR with metadata and/or the
|
||||
@@ -26,21 +42,37 @@ File* receive_incremental_check(int fd, const Config* config, bool* skipped);
|
||||
typedef struct {
|
||||
char** paths; /* owned, destination-relative wire paths */
|
||||
FileMetadata* entries; /* owned, parallel to paths */
|
||||
FileXattrList** xattrs; /* owned, parallel to paths; NULL when none */
|
||||
size_t count;
|
||||
size_t capacity;
|
||||
size_t bytes; /* cumulative strlen of every retained path */
|
||||
} DirTimeList;
|
||||
|
||||
/* Capture gate shared by the sender-side and receiver-side sinks: directory
|
||||
* metadata is accumulated only when a directory attribute is requested
|
||||
* (-p/--perms for directory modes, or -t/--times for directory mtimes with
|
||||
* -O/--omit-dir-times not suppressing them) and metadata rides the wire. Kept
|
||||
* here, next to the accumulator it guards, so both call sites express the same
|
||||
* condition. */
|
||||
bool dir_metadata_should_capture(const Config* config);
|
||||
|
||||
void dir_time_list_init(DirTimeList* list);
|
||||
void dir_time_list_free(DirTimeList* list);
|
||||
/* Deep-copy one directory's path + metadata into the list. Returns false on
|
||||
* allocation failure (the caller fails the transfer). */
|
||||
bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata);
|
||||
/* Apply every accumulated directory's mtime (and atime when captured) beneath
|
||||
* `root_directory`, confined fd-relative. Best-effort per entry: an absent
|
||||
* directory (an empty/pruned source dir that was deliberately not created) or a
|
||||
* non-directory at the path is skipped QUIETLY, an unreachable one with a
|
||||
* warning, and never fatal. */
|
||||
void dir_time_list_apply(const DirTimeList* list, const char* root_directory);
|
||||
/* Deep-copy one directory's path + metadata (and, when non-NULL, its captured
|
||||
* xattr/ACL block) into the list. Returns false on allocation failure OR when
|
||||
* the cumulative entry/byte caps would be exceeded (the caller fails the
|
||||
* transfer). */
|
||||
bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata,
|
||||
const FileXattrList* xattrs);
|
||||
/* Apply every accumulated directory's metadata beneath `root_directory`,
|
||||
* confined fd-relative: ownership through the negotiated identity policy,
|
||||
* times (mtime, plus atime when -U captured one under -t), the mode (through
|
||||
* --chmod when configured, under -p), and the captured xattrs/ACLs (under
|
||||
* -X/-A). Best-effort per entry: an absent directory (an empty/pruned source
|
||||
* dir that was deliberately not created) or a non-directory at the path is
|
||||
* skipped QUIETLY, an unreachable one with a warning, and never fatal. */
|
||||
void dir_metadata_list_apply(const DirTimeList* list, const char* root_directory,
|
||||
const Config* config);
|
||||
|
||||
/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative
|
||||
paths the sender transferred/keeps) plus `protected`, destination-relative
|
||||
@@ -56,11 +88,18 @@ typedef struct DeleteManifest {
|
||||
ArrayList* keeps;
|
||||
ArrayList* protected;
|
||||
ArrayList* missing;
|
||||
/* Destination-relative paths of the directories the sender synchronized for
|
||||
this run. The extras walker only removes entries directly inside one of
|
||||
these (the receive root is the "." sentinel); `--files-from` runs therefore
|
||||
leave untransmitted directories and the unlisted parts of listed ones
|
||||
alone, matching rsync's "delete only in synchronized directories". */
|
||||
ArrayList* dirs;
|
||||
} DeleteManifest;
|
||||
|
||||
void delete_manifest_free(DeleteManifest* manifest);
|
||||
/* Read a delete-manifest frame: keep count + keeps, then protected count +
|
||||
protected prefixes, then missing count + missing paths (self-delimiting; the
|
||||
/* Read a delete-manifest frame (protocol 2.23.0): keep count + keeps, then
|
||||
protected count + protected prefixes, then missing count + missing paths,
|
||||
then synchronized-directory count + directory paths (self-delimiting; the
|
||||
leading STATUS_MANIFEST code has been consumed). Returns an owned
|
||||
DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */
|
||||
DeleteManifest* receive_manifest_entries(int fd);
|
||||
@@ -79,11 +118,46 @@ bool manifest_delete_extras(const Config* config, DeleteManifest* manifest);
|
||||
confinement or I/O error (the run then fails); tolerated per-path cases are
|
||||
reported and skipped. */
|
||||
bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest);
|
||||
/* Budgeted form of manifest_delete_missing_args for the per-directory delete
|
||||
session: each removed mirror draws from `max_delete` (SIZE_MAX = unlimited)
|
||||
and the tallies are accumulated into `*deleted`/`*skipped`. `*limit_hit` is set
|
||||
when the budget stopped the pass with entries left over. Returns false only
|
||||
on a genuine deletion error. */
|
||||
bool manifest_delete_missing_args_limited(const Config* config, DeleteManifest* manifest,
|
||||
size_t max_delete, size_t* deleted, size_t* skipped,
|
||||
bool* limit_hit);
|
||||
/* Outcome of committing a delete manifest. LIMIT_REACHED reports rsync's
|
||||
partial --max-delete result: the budget allowed some deletions and the rest
|
||||
were skipped (the run still stores all file data but the client exits 25). */
|
||||
typedef enum {
|
||||
DELETE_COMMIT_OK = 0,
|
||||
DELETE_COMMIT_LIMIT_REACHED,
|
||||
DELETE_COMMIT_ERROR
|
||||
} DeleteCommitResult;
|
||||
|
||||
/* Run every deletion family the manifest carries: the --delete-missing-args
|
||||
exact-path deletions first (user requests are not blocked by exclusion
|
||||
protection), then the ordinary extras walk when --delete is active. Returns
|
||||
true when nothing to do or everything committed. */
|
||||
bool manifest_delete_all(const Config* config, DeleteManifest* manifest);
|
||||
protection), then the ordinary extras walk when --delete is active. Both
|
||||
share one --max-delete budget. Returns DELETE_COMMIT_OK when nothing was to
|
||||
do or everything committed, DELETE_COMMIT_LIMIT_REACHED when the budget
|
||||
stopped part of the work, or DELETE_COMMIT_ERROR on a genuine failure. */
|
||||
DeleteCommitResult manifest_delete_all(const Config* config, DeleteManifest* manifest);
|
||||
/* Like manifest_delete_all, but reports how many destination entries the commit
|
||||
removed (for the end-of-transfer wire stats). `deleted` may be NULL. */
|
||||
DeleteCommitResult manifest_delete_all_counted(const Config* config, DeleteManifest* manifest,
|
||||
size_t* deleted);
|
||||
|
||||
/* -n/--dry-run --delete would-delete reporting: walk the destination exactly as
|
||||
the delete pass would and append (strdup'd) destination-relative paths that
|
||||
WOULD be removed to `out`, without touching disk. Uses the same staging-dir,
|
||||
basis-dir and protected-prefix skips as the real commit. Returns true on a
|
||||
clean walk; `*count_out` receives the number of paths appended. */
|
||||
bool manifest_would_delete_list(const Config* config, DeleteManifest* manifest, ArrayList* out,
|
||||
size_t* count_out);
|
||||
/* Convert one basis-directory path to the receive-root-relative protection
|
||||
prefix the delete walker uses (NULL when it lies outside the root). Exposed
|
||||
for unit tests of the root-of-"/" and normalization edge cases. */
|
||||
char* file_receive_basis_delete_relative(const Config* config, const char* path);
|
||||
|
||||
/* Outcome of a single file_save_to_disk operation. The receiver needs to
|
||||
distinguish "written" from "skipped" so --remove-source-files can be told
|
||||
|
||||
+11
-2
@@ -143,10 +143,17 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
|
||||
}
|
||||
|
||||
off_t offset = 0;
|
||||
/* A non-positive --timeout disables the deadline: poll blocks until the
|
||||
* socket is writable (rsync's --timeout=0 default). */
|
||||
int io_timeout_sec = protocol_get_io_timeout_sec();
|
||||
struct timespec deadline;
|
||||
if (io_timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += 60;
|
||||
deadline.tv_sec += io_timeout_sec;
|
||||
}
|
||||
while ((unsigned long long)offset < file_size) {
|
||||
int timeout = -1;
|
||||
if (io_timeout_sec > 0) {
|
||||
struct timespec now;
|
||||
clock_gettime(CLOCK_MONOTONIC, &now);
|
||||
long long remaining = (long long)(deadline.tv_sec - now.tv_sec) * 1000LL +
|
||||
@@ -155,8 +162,9 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
timeout = remaining > INT_MAX ? INT_MAX : (int)remaining;
|
||||
}
|
||||
struct pollfd pfd = {.fd = file_descriptor, .events = POLLOUT};
|
||||
int timeout = remaining > INT_MAX ? INT_MAX : (int)remaining;
|
||||
int polled = poll(&pfd, 1, timeout);
|
||||
if (polled <= 0 || (pfd.revents & (POLLERR | POLLHUP | POLLNVAL))) {
|
||||
close(fd);
|
||||
@@ -174,6 +182,7 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
|
||||
close(fd);
|
||||
return false;
|
||||
}
|
||||
protocol_note_bytes_written((unsigned long long)sent);
|
||||
}
|
||||
|
||||
close(fd);
|
||||
|
||||
@@ -1,129 +1,7 @@
|
||||
#include <errno.h>
|
||||
#include <fcntl.h>
|
||||
#include <libgen.h>
|
||||
#include <limits.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#include "file_store.h"
|
||||
#include "metadata.h"
|
||||
#include "utils.h"
|
||||
|
||||
static int authorized_root_fd = -1;
|
||||
static char* authorized_root_path;
|
||||
|
||||
static bool path_is_within_root(const char* root, const char* path) {
|
||||
size_t root_length = strlen(root);
|
||||
return strncmp(root, path, root_length) == 0 &&
|
||||
(path[root_length] == '\0' || path[root_length] == '/');
|
||||
}
|
||||
|
||||
bool file_store_set_authorized_root(int fd, const char* canonical_path) {
|
||||
char* new_path = canonical_path ? str_dup(canonical_path) : NULL;
|
||||
if (canonical_path && !new_path) {
|
||||
authorized_root_fd = -1;
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = NULL;
|
||||
return false;
|
||||
}
|
||||
free(authorized_root_path);
|
||||
authorized_root_path = new_path;
|
||||
authorized_root_fd = fd;
|
||||
return true;
|
||||
}
|
||||
|
||||
int file_store_open_secure_parent(const char* path, char** leaf_out) {
|
||||
char* copy = str_dup(path);
|
||||
if (!copy)
|
||||
return -1;
|
||||
char* parent = dirname(copy);
|
||||
const char* slash = strrchr(path, '/');
|
||||
char* leaf = str_dup(slash ? slash + 1 : path);
|
||||
if (!leaf) {
|
||||
free(copy);
|
||||
return -1;
|
||||
}
|
||||
int fd;
|
||||
if (authorized_root_fd >= 0) {
|
||||
if (!authorized_root_path || path[0] != '/' ||
|
||||
!path_is_within_root(authorized_root_path, path)) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
fd = dup(authorized_root_fd);
|
||||
if (fd < 0) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
size_t root_length = strlen(authorized_root_path);
|
||||
char* relative = str_dup(path + root_length);
|
||||
if (!relative) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
close(fd);
|
||||
return -1;
|
||||
}
|
||||
free(copy);
|
||||
copy = relative;
|
||||
parent = dirname(copy);
|
||||
} else {
|
||||
fd = (parent[0] == '/') ? open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC)
|
||||
: open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
|
||||
}
|
||||
if (fd < 0) {
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
char* save = NULL;
|
||||
char* component = strtok_r(parent, "/", &save);
|
||||
while (component) {
|
||||
if (strcmp(component, "..") == 0) {
|
||||
close(fd);
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
if (strcmp(component, ".") != 0) {
|
||||
int next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
if (next < 0 && errno == ENOENT) {
|
||||
if (mkdirat(fd, component, 0755) == 0 || errno == EEXIST)
|
||||
next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
|
||||
}
|
||||
if (next < 0) {
|
||||
close(fd);
|
||||
free(copy);
|
||||
free(leaf);
|
||||
return -1;
|
||||
}
|
||||
close(fd);
|
||||
fd = next;
|
||||
}
|
||||
component = strtok_r(NULL, "/", &save);
|
||||
}
|
||||
free(copy);
|
||||
*leaf_out = leaf;
|
||||
return fd;
|
||||
}
|
||||
|
||||
bool file_store_rename_secure(const char* old_path, const char* new_path) {
|
||||
char *old_leaf = NULL, *new_leaf = NULL;
|
||||
int old_parent = file_store_open_secure_parent(old_path, &old_leaf);
|
||||
int new_parent = file_store_open_secure_parent(new_path, &new_leaf);
|
||||
bool ok = old_parent >= 0 && new_parent >= 0 &&
|
||||
renameat(old_parent, old_leaf, new_parent, new_leaf) == 0;
|
||||
if (old_parent >= 0)
|
||||
close(old_parent);
|
||||
if (new_parent >= 0)
|
||||
close(new_parent);
|
||||
free(old_leaf);
|
||||
free(new_leaf);
|
||||
return ok;
|
||||
}
|
||||
|
||||
static bool write_all(int fd, const void* data, unsigned long long size) {
|
||||
const unsigned char* p = data;
|
||||
@@ -176,67 +54,3 @@ bool file_store_write_sparse(int fd, const unsigned char* data, unsigned long lo
|
||||
}
|
||||
return ftruncate(fd, (off_t)size) == 0;
|
||||
}
|
||||
|
||||
bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, const FileMetadata* metadata,
|
||||
bool preserve_executability) {
|
||||
char* leaf = NULL;
|
||||
int dirfd = file_store_open_secure_parent(path, &leaf);
|
||||
if (dirfd < 0)
|
||||
return false;
|
||||
int fd = -1;
|
||||
bool ok = false;
|
||||
if (inplace) {
|
||||
fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC | O_NOFOLLOW, 0644);
|
||||
if (fd >= 0) {
|
||||
if (sparse && data_size > 0) {
|
||||
if (ftruncate(fd, (off_t)data_size) == 0)
|
||||
ok = file_store_write_sparse(fd, data, data_size);
|
||||
} else {
|
||||
ok = write_all(fd, data, data_size);
|
||||
}
|
||||
if (ok && metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
}
|
||||
} else {
|
||||
int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%u", leaf, (long)getpid(), 99U);
|
||||
if (tmp_size < 0) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
char* tmp = malloc((size_t)tmp_size + 1);
|
||||
if (!tmp) {
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return false;
|
||||
}
|
||||
for (unsigned int i = 0; i < 100 && !ok; ++i) {
|
||||
snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i);
|
||||
fd = openat(dirfd, tmp, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | O_NOFOLLOW, 0600);
|
||||
if (fd < 0)
|
||||
continue;
|
||||
if (sparse && data_size > 0)
|
||||
ok = ftruncate(fd, (off_t)data_size) == 0;
|
||||
if (ok || (!sparse || data_size == 0))
|
||||
ok = (sparse && data_size > 0)
|
||||
? file_store_write_sparse(fd, (const unsigned char*)data, data_size)
|
||||
: write_all(fd, data, data_size);
|
||||
if (ok && metadata)
|
||||
ok = file_restore_metadata_fd(fd, metadata, preserve_executability);
|
||||
if (close(fd) != 0)
|
||||
ok = false;
|
||||
fd = -1;
|
||||
if (ok && renameat(dirfd, tmp, dirfd, leaf) != 0)
|
||||
ok = false;
|
||||
if (!ok)
|
||||
unlinkat(dirfd, tmp, 0);
|
||||
}
|
||||
free(tmp);
|
||||
}
|
||||
if (fd >= 0)
|
||||
close(fd);
|
||||
close(dirfd);
|
||||
free(leaf);
|
||||
return ok;
|
||||
}
|
||||
|
||||
@@ -1,15 +1,8 @@
|
||||
#ifndef FILE_STORE_H
|
||||
#define FILE_STORE_H
|
||||
|
||||
#include "file.h"
|
||||
#include <stdbool.h>
|
||||
|
||||
bool file_store_set_authorized_root(int fd, const char* canonical_path);
|
||||
int file_store_open_secure_parent(const char* path, char** leaf_out);
|
||||
bool file_store_rename_secure(const char* old_path, const char* new_path);
|
||||
bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size,
|
||||
bool inplace, bool sparse, const FileMetadata* metadata,
|
||||
bool preserve_executability);
|
||||
/* Sparse-aware write (--sparse/-S): every all-zero run of at least
|
||||
* SPARSE_HOLE_MIN bytes is skipped with lseek(SEEK_CUR) so it becomes a real
|
||||
* hole; every other byte is written. The caller pre-sizes the file with
|
||||
|
||||
@@ -2,6 +2,7 @@
|
||||
#define FILE_TYPES_H
|
||||
|
||||
#include "data.h"
|
||||
#include "format.h"
|
||||
#include "xattr.h"
|
||||
#include <stdbool.h>
|
||||
#include <sys/stat.h>
|
||||
@@ -86,6 +87,16 @@ typedef struct {
|
||||
* Receiver: parsed off the wire, attached here, and applied fd-relative on
|
||||
* the written file. NULL/0 == the file carries no xattrs. */
|
||||
FileXattrList* xattrs;
|
||||
/* Sender-side output-parity state (never serialized): the receiver-reported
|
||||
* pre-transfer destination snapshot for this entry, filled by the per-file
|
||||
* STATUS_CHECK exchange when report_dest_info is set. `known` is false when
|
||||
* no report was requested/received, in which case -i/--out-format treats the
|
||||
* entry conservatively as newly created. */
|
||||
OutputDestState dest_state;
|
||||
/* Receiver-only wire-stats tally: the number of bytes reconstructed from the
|
||||
* basis file (matched delta blocks) for this entry. 0 when the file was sent
|
||||
* whole. Accumulated into ReceiverStats.matched_data by the receiver sink. */
|
||||
unsigned long long matched_bytes;
|
||||
} File;
|
||||
|
||||
/* The path that should be sent on the wire and used for the receiver-side
|
||||
|
||||
+616
-225
@@ -1,162 +1,27 @@
|
||||
#include "filter.h"
|
||||
#include "log.h"
|
||||
#include "utils.h"
|
||||
#include <ctype.h>
|
||||
#include <errno.h>
|
||||
#include <limits.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
/* ---- Single rule parsing ---- */
|
||||
|
||||
static bool rule_text_is_unsupported_word(const char* p, size_t len) {
|
||||
static const char* const words[] = {"merge", "dir-merge", "hide", "show",
|
||||
"protect", "risk", "clear"};
|
||||
for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) {
|
||||
size_t wl = strlen(words[i]);
|
||||
if (len == wl && strncmp(p, words[i], wl) == 0)
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
/* Write a diagnostic message into the caller's optional buffer. A NULL `err`
|
||||
* (or a zero size) is a no-op, so a caller that only needs the boolean status
|
||||
* may pass NULL without the snprintf-on-NULL undefined behaviour. */
|
||||
static void filter_set_error(char* err, size_t err_size, const char* fmt, ...) {
|
||||
if (!err || err_size == 0)
|
||||
return;
|
||||
va_list ap;
|
||||
va_start(ap, fmt);
|
||||
vsnprintf(err, err_size, fmt, ap);
|
||||
va_end(ap);
|
||||
}
|
||||
|
||||
/* rsync include/exclude rule modifiers we do NOT implement. A rule whose +/- is
|
||||
* immediately followed by one of these is rejected instead of being silently
|
||||
* parsed as a literal pattern. */
|
||||
static bool is_unsupported_rule_modifier(char c) {
|
||||
return c == '!' || c == 'C' || c == 's' || c == 'r' || c == 'p' || c == 'x';
|
||||
}
|
||||
|
||||
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (!line)
|
||||
return NULL;
|
||||
char* text = str_dup(line);
|
||||
if (!text) {
|
||||
if (err)
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
size_t len = strlen(text);
|
||||
while (len > 0 && (text[len - 1] == '\n' || text[len - 1] == '\r'))
|
||||
text[--len] = '\0';
|
||||
|
||||
const char* p = text;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
if (*p == '\0') {
|
||||
snprintf(err, err_size, "empty filter rule");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
FilterAction action = FILTER_ACTION_EXCLUDE;
|
||||
if (*p == '+' || *p == '-') {
|
||||
action = *p == '+' ? FILTER_ACTION_INCLUDE : FILTER_ACTION_EXCLUDE;
|
||||
p++;
|
||||
/* rsync attaches rule modifiers directly to the +/- (e.g. "-s foo"). Only
|
||||
* the '/' anchor modifier is supported; anything else is a clear error
|
||||
* rather than a silently-ignored literal. */
|
||||
if (*p != ' ' && *p != '\t' && *p != '\0' && is_unsupported_rule_modifier(*p)) {
|
||||
snprintf(err, err_size,
|
||||
"filter rule modifier '%c' is not supported (only the '/' anchor after +/- "
|
||||
"is implemented; put a space between +/- and the pattern)",
|
||||
*p);
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
} else {
|
||||
/* ':' (dir-merge) and '.' (merge) are rsync filter-rule shorthands. At the
|
||||
* start of a rule they mean "merge this file", so reject them instead of
|
||||
* silently turning them into inert exclude patterns. */
|
||||
if (*p == ':' || *p == '.' || *p == '!') {
|
||||
snprintf(err, err_size,
|
||||
"filter rule starting with '%c' is not supported (merge/dir-merge/list-clear "
|
||||
"shorthands are not implemented; use +/- include/exclude rules)",
|
||||
*p);
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
const char* sp = p;
|
||||
while (*sp != '\0' && *sp != ' ' && *sp != '\t')
|
||||
sp++;
|
||||
size_t word_len = (size_t)(sp - p);
|
||||
if (rule_text_is_unsupported_word(p, word_len)) {
|
||||
snprintf(err, err_size,
|
||||
"'%.*s' filter directives are not supported (only +/- include/exclude rules "
|
||||
"with an optional '/' anchor and trailing '/' dir marker)",
|
||||
(int)word_len, p);
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
if (word_len == strlen("include") && strncmp(p, "include", word_len) == 0) {
|
||||
action = FILTER_ACTION_INCLUDE;
|
||||
p = sp;
|
||||
} else if (word_len == strlen("exclude") && strncmp(p, "exclude", word_len) == 0) {
|
||||
action = FILTER_ACTION_EXCLUDE;
|
||||
p = sp;
|
||||
}
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
}
|
||||
|
||||
if (*p == '\0') {
|
||||
snprintf(err, err_size, "filter rule has no pattern");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* A pattern beginning with '/' is anchored (either as "-/foo" or "- /foo"). */
|
||||
bool anchored = false;
|
||||
if (*p == '/') {
|
||||
anchored = true;
|
||||
p++;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
}
|
||||
if (*p == '\0') {
|
||||
snprintf(err, err_size, "filter rule has no pattern after '/' anchor");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Pattern runs to the end of the rule; a single trailing '/' marks dir-only. */
|
||||
size_t pat_len = strlen(p);
|
||||
bool dir_only = false;
|
||||
if (pat_len > 1 && p[pat_len - 1] == '/') {
|
||||
dir_only = true;
|
||||
pat_len--;
|
||||
} else if (pat_len == 1 && p[0] == '/') {
|
||||
/* "//" anchored with nothing after: meaningless. */
|
||||
snprintf(err, err_size, "filter rule has no pattern");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
rule->pattern = malloc(pat_len + 1);
|
||||
if (!rule->pattern) {
|
||||
free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
free(text);
|
||||
return NULL;
|
||||
}
|
||||
memcpy(rule->pattern, p, pat_len);
|
||||
rule->pattern[pat_len] = '\0';
|
||||
rule->action = action;
|
||||
rule->anchored = anchored;
|
||||
rule->dir_only = dir_only;
|
||||
rule->owner = NULL;
|
||||
free(text);
|
||||
return rule;
|
||||
}
|
||||
/* ---- Ordered rule lists ---- */
|
||||
|
||||
void filter_rule_free(FilterRule* rule) {
|
||||
if (!rule)
|
||||
@@ -166,8 +31,6 @@ void filter_rule_free(FilterRule* rule) {
|
||||
free(rule);
|
||||
}
|
||||
|
||||
/* ---- Ordered rule lists ---- */
|
||||
|
||||
FilterRuleList* filter_rule_list_create(void) {
|
||||
return calloc(1, sizeof(FilterRuleList));
|
||||
}
|
||||
@@ -176,6 +39,8 @@ bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) {
|
||||
if (!list || !rule)
|
||||
return false;
|
||||
if (list->count == list->capacity) {
|
||||
if (list->capacity > INT_MAX / 2)
|
||||
return false;
|
||||
int new_cap = list->capacity > 0 ? list->capacity * 2 : 8;
|
||||
FilterRule** grown = realloc(list->items, (size_t)new_cap * sizeof(FilterRule*));
|
||||
if (!grown)
|
||||
@@ -187,28 +52,42 @@ bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err,
|
||||
size_t err_size) {
|
||||
FilterRule* rule = filter_rule_parse(line, err, err_size);
|
||||
if (!rule)
|
||||
return false;
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
void filter_rule_list_free(FilterRuleList* list) {
|
||||
if (!list)
|
||||
return;
|
||||
for (int i = 0; i < list->count; i++)
|
||||
filter_rule_free(list->items[i]);
|
||||
for (int i = 0; i < list->dir_merge_count; i++)
|
||||
free(list->dir_merge_names[i]);
|
||||
free(list->dir_merge_names);
|
||||
free(list->items);
|
||||
free(list);
|
||||
}
|
||||
|
||||
/* Register a per-directory merge-file basename (for "dir-merge NAME"/": NAME"
|
||||
* and -F's .rsync-filter). Duplicate names are ignored. */
|
||||
bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name) {
|
||||
if (!list || !name || name[0] == '\0')
|
||||
return false;
|
||||
for (int i = 0; i < list->dir_merge_count; i++) {
|
||||
if (strcmp(list->dir_merge_names[i], name) == 0)
|
||||
return true;
|
||||
}
|
||||
if (list->dir_merge_count == list->dir_merge_capacity) {
|
||||
int new_cap = list->dir_merge_capacity > 0 ? list->dir_merge_capacity * 2 : 4;
|
||||
char** grown = realloc(list->dir_merge_names, (size_t)new_cap * sizeof(char*));
|
||||
if (!grown)
|
||||
return false;
|
||||
list->dir_merge_names = grown;
|
||||
list->dir_merge_capacity = new_cap;
|
||||
}
|
||||
char* dup = str_dup(name);
|
||||
if (!dup)
|
||||
return false;
|
||||
list->dir_merge_names[list->dir_merge_count++] = dup;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool set_rule_owner(FilterRule* rule, const char* owner) {
|
||||
char* dup = str_dup(owner ? owner : "");
|
||||
if (!dup)
|
||||
@@ -218,7 +97,315 @@ static bool set_rule_owner(FilterRule* rule, const char* owner) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* ---- CVS default excludes (-C) ---- */
|
||||
/* ---- Rule parsing ---- */
|
||||
|
||||
/* A short rule prefix is a single character; a long rule name is alphabetic
|
||||
* (with '-'). `is_short` distinguishes the modifier-attachment rules. */
|
||||
typedef enum {
|
||||
RULE_KIND_EXCLUDE,
|
||||
RULE_KIND_INCLUDE,
|
||||
RULE_KIND_HIDE,
|
||||
RULE_KIND_SHOW,
|
||||
RULE_KIND_PROTECT,
|
||||
RULE_KIND_RISK,
|
||||
RULE_KIND_MERGE,
|
||||
RULE_KIND_DIR_MERGE,
|
||||
RULE_KIND_CLEAR,
|
||||
RULE_KIND_UNKNOWN,
|
||||
} RuleKind;
|
||||
|
||||
static bool short_rule_char(char c, RuleKind* kind) {
|
||||
switch (c) {
|
||||
case '-':
|
||||
*kind = RULE_KIND_EXCLUDE;
|
||||
return true;
|
||||
case '+':
|
||||
*kind = RULE_KIND_INCLUDE;
|
||||
return true;
|
||||
case 'H':
|
||||
*kind = RULE_KIND_HIDE;
|
||||
return true;
|
||||
case 'S':
|
||||
*kind = RULE_KIND_SHOW;
|
||||
return true;
|
||||
case 'P':
|
||||
*kind = RULE_KIND_PROTECT;
|
||||
return true;
|
||||
case 'R':
|
||||
*kind = RULE_KIND_RISK;
|
||||
return true;
|
||||
case '.':
|
||||
*kind = RULE_KIND_MERGE;
|
||||
return true;
|
||||
case ':':
|
||||
*kind = RULE_KIND_DIR_MERGE;
|
||||
return true;
|
||||
case '!':
|
||||
*kind = RULE_KIND_CLEAR;
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
static bool long_rule_name(const char* name, size_t len, RuleKind* kind) {
|
||||
struct {
|
||||
const char* word;
|
||||
RuleKind kind;
|
||||
} table[] = {
|
||||
{"exclude", RULE_KIND_EXCLUDE}, {"include", RULE_KIND_INCLUDE},
|
||||
{"hide", RULE_KIND_HIDE}, {"show", RULE_KIND_SHOW},
|
||||
{"protect", RULE_KIND_PROTECT}, {"risk", RULE_KIND_RISK},
|
||||
{"merge", RULE_KIND_MERGE}, {"dir-merge", RULE_KIND_DIR_MERGE},
|
||||
{"clear", RULE_KIND_CLEAR},
|
||||
};
|
||||
for (size_t i = 0; i < sizeof(table) / sizeof(table[0]); i++) {
|
||||
if (strlen(table[i].word) == len && strncmp(name, table[i].word, len) == 0) {
|
||||
*kind = table[i].kind;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool is_modifier_char(char c) {
|
||||
return c == 's' || c == 'r' || c == 'p' || c == 'x' || c == '/' || c == '!' || c == 'C';
|
||||
}
|
||||
|
||||
/* Parse "RULE[,MODIFIERS] [PATTERN]". On success `kind`, `sides`,
|
||||
* `sides_explicit`, `negate`, `anchored_mod`, `perishable`, `xattr`,
|
||||
* `cvs_inject` and the pattern span (`pat_start`/`pat_len`, possibly 0 for
|
||||
* merge/clear) are filled. Returns true on success. */
|
||||
static bool parse_rule_syntax(const char* text, RuleKind* kind, unsigned* sides,
|
||||
bool* sides_explicit, bool* negate, bool* anchored_mod,
|
||||
bool* perishable, bool* xattr, bool* cvs_inject,
|
||||
const char** pat_start, size_t* pat_len) {
|
||||
const char* p = text;
|
||||
*sides = FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER;
|
||||
*sides_explicit = false;
|
||||
*negate = false;
|
||||
*anchored_mod = false;
|
||||
*perishable = false;
|
||||
*xattr = false;
|
||||
*cvs_inject = false;
|
||||
*pat_start = NULL;
|
||||
*pat_len = 0;
|
||||
|
||||
bool is_short = false;
|
||||
if (short_rule_char(*p, kind)) {
|
||||
is_short = true;
|
||||
p++;
|
||||
} else {
|
||||
const char* name_start = p;
|
||||
while (isalpha((unsigned char)*p) || *p == '-')
|
||||
p++;
|
||||
size_t name_len = (size_t)(p - name_start);
|
||||
if (name_len == 0 || !long_rule_name(name_start, name_len, kind))
|
||||
return false;
|
||||
/* A long name must be followed by a separator, a comma or the end. */
|
||||
if (*p != '\0' && *p != ',' && *p != ' ' && *p != '_')
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Modifiers: long names require a comma; short names may attach directly.
|
||||
Only commit a modifier run that terminates at a separator or the end, so a
|
||||
pattern such as "*.tmp" written as "-*.tmp" is not mistaken for modifiers. */
|
||||
const char* mod_start = p;
|
||||
const char* mod_end = p;
|
||||
if (*p == ',') {
|
||||
p++;
|
||||
mod_start = p;
|
||||
while (is_modifier_char(*p))
|
||||
p++;
|
||||
mod_end = p;
|
||||
} else if (is_short) {
|
||||
const char* scan = p;
|
||||
while (is_modifier_char(*scan))
|
||||
scan++;
|
||||
if (*scan == '\0' || *scan == ' ' || *scan == '_') {
|
||||
mod_start = p;
|
||||
mod_end = scan;
|
||||
p = scan;
|
||||
}
|
||||
}
|
||||
for (const char* m = mod_start; m < mod_end; m++) {
|
||||
switch (*m) {
|
||||
case 's':
|
||||
*sides = FILTER_SIDE_SENDER;
|
||||
*sides_explicit = true;
|
||||
break;
|
||||
case 'r':
|
||||
*sides = FILTER_SIDE_RECEIVER;
|
||||
*sides_explicit = true;
|
||||
break;
|
||||
case '!':
|
||||
*negate = true;
|
||||
break;
|
||||
case '/':
|
||||
*anchored_mod = true;
|
||||
break;
|
||||
case 'p':
|
||||
*perishable = true;
|
||||
break;
|
||||
case 'x':
|
||||
*xattr = true;
|
||||
break;
|
||||
case 'C':
|
||||
*cvs_inject = true;
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
/* A single space or underscore separates the rule/modifiers from the
|
||||
pattern; further spaces/underscores belong to the pattern. */
|
||||
const char* pat = p;
|
||||
if (*pat == ' ' || *pat == '_')
|
||||
pat++;
|
||||
/* Trim a trailing newline/CR (the caller may pass a raw file line). */
|
||||
*pat_start = pat;
|
||||
*pat_len = strlen(pat);
|
||||
while (*pat_len > 0 && (pat[*pat_len - 1] == '\n' || pat[*pat_len - 1] == '\r'))
|
||||
(*pat_len)--;
|
||||
return true;
|
||||
}
|
||||
|
||||
FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err,
|
||||
size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (!line)
|
||||
return NULL;
|
||||
const char* p = line;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
if (*p == '\0' || *p == '\n' || *p == '\r') {
|
||||
filter_set_error(err, err_size, "empty filter rule");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
RuleKind kind = RULE_KIND_UNKNOWN;
|
||||
unsigned sides;
|
||||
bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject;
|
||||
const char* pat;
|
||||
size_t pat_len;
|
||||
if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable,
|
||||
&xattr, &cvs_inject, &pat, &pat_len)) {
|
||||
filter_set_error(err, err_size, "unrecognized filter rule syntax");
|
||||
return NULL;
|
||||
}
|
||||
if (cvs_inject) {
|
||||
/* The C modifier expands to the CVS defaults in place; the rule itself
|
||||
carries no pattern and is handled by the caller. */
|
||||
filter_set_error(err, err_size, "the C modifier is handled by the rule-list parser");
|
||||
return NULL;
|
||||
}
|
||||
if (xattr) {
|
||||
filter_set_error(err, err_size, "xattr-name filter rules (the x modifier) are not supported");
|
||||
return NULL;
|
||||
}
|
||||
if (kind == RULE_KIND_MERGE || kind == RULE_KIND_DIR_MERGE) {
|
||||
filter_set_error(err, err_size, "merge/dir-merge rules are handled by the rule-list parser");
|
||||
return NULL;
|
||||
}
|
||||
if (kind == RULE_KIND_CLEAR) {
|
||||
if (pat_len != 0) {
|
||||
filter_set_error(err, err_size, "clear takes no pattern");
|
||||
return NULL;
|
||||
}
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
rule->action = FILTER_ACTION_NONE; /* clear marker: no pattern */
|
||||
rule->sides = 0;
|
||||
return rule;
|
||||
}
|
||||
|
||||
FilterAction action;
|
||||
switch (kind) {
|
||||
case RULE_KIND_INCLUDE:
|
||||
case RULE_KIND_SHOW:
|
||||
case RULE_KIND_RISK:
|
||||
action = FILTER_ACTION_INCLUDE;
|
||||
break;
|
||||
default:
|
||||
action = FILTER_ACTION_EXCLUDE;
|
||||
break;
|
||||
}
|
||||
if (kind == RULE_KIND_HIDE)
|
||||
sides = FILTER_SIDE_SENDER;
|
||||
else if (kind == RULE_KIND_SHOW)
|
||||
sides = FILTER_SIDE_SENDER;
|
||||
else if (kind == RULE_KIND_PROTECT)
|
||||
sides = FILTER_SIDE_RECEIVER;
|
||||
else if (kind == RULE_KIND_RISK)
|
||||
sides = FILTER_SIDE_RECEIVER;
|
||||
if (kind == RULE_KIND_HIDE || kind == RULE_KIND_SHOW || kind == RULE_KIND_PROTECT ||
|
||||
kind == RULE_KIND_RISK)
|
||||
sides_explicit = true;
|
||||
/* --delete-excluded turns an unqualified (no explicit s/r) rule into a
|
||||
sender-side-only rule, so it no longer protects the receiver. */
|
||||
if (opts && opts->delete_excluded && !sides_explicit)
|
||||
sides = FILTER_SIDE_SENDER;
|
||||
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "filter rule has no pattern");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
bool anchored = anchored_mod;
|
||||
const char* pat_begin = pat;
|
||||
if (*pat_begin == '/') {
|
||||
anchored = true;
|
||||
pat_begin++;
|
||||
/* Drop the spaces that could follow the anchor in the "-/ foo" form. */
|
||||
while (*pat_begin == ' ' || *pat_begin == '\t')
|
||||
pat_begin++;
|
||||
pat_len = strlen(pat_begin);
|
||||
while (pat_len > 0 && (pat_begin[pat_len - 1] == '\n' || pat_begin[pat_len - 1] == '\r'))
|
||||
pat_len--;
|
||||
}
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "filter rule has no pattern after '/' anchor");
|
||||
return NULL;
|
||||
}
|
||||
bool dir_only = false;
|
||||
if (pat_len > 1 && pat_begin[pat_len - 1] == '/') {
|
||||
dir_only = true;
|
||||
pat_len--;
|
||||
}
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "filter rule has no pattern");
|
||||
return NULL;
|
||||
}
|
||||
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
rule->pattern = malloc(pat_len + 1);
|
||||
if (!rule->pattern) {
|
||||
free(rule);
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
memcpy(rule->pattern, pat_begin, pat_len);
|
||||
rule->pattern[pat_len] = '\0';
|
||||
rule->action = action;
|
||||
rule->sides = sides;
|
||||
rule->anchored = anchored;
|
||||
rule->dir_only = dir_only;
|
||||
rule->negate = negate;
|
||||
rule->perishable = perishable;
|
||||
(void)xattr; /* xattr-name rules never match file/dir names; accepted/ignored */
|
||||
return rule;
|
||||
}
|
||||
|
||||
/* ---- CVS default excludes (-C and the C modifier) ---- */
|
||||
|
||||
typedef struct {
|
||||
const char* pattern;
|
||||
@@ -237,12 +424,13 @@ static const CvsDefaultRule CVS_DEFAULTS[] = {
|
||||
{".svn/", true}, {".git/", true}, {".hg/", true}, {".bzr/", true},
|
||||
};
|
||||
|
||||
static bool cvs_rule_list_append(FilterRuleList* list) {
|
||||
static bool filter_list_append_cvs(FilterRuleList* list, unsigned sides) {
|
||||
for (size_t i = 0; i < sizeof(CVS_DEFAULTS) / sizeof(CVS_DEFAULTS[0]); i++) {
|
||||
FilterRule* rule = calloc(1, sizeof(FilterRule));
|
||||
if (!rule)
|
||||
return false;
|
||||
rule->action = FILTER_ACTION_EXCLUDE;
|
||||
rule->sides = sides;
|
||||
rule->dir_only = CVS_DEFAULTS[i].dir_only;
|
||||
size_t plen = strlen(CVS_DEFAULTS[i].pattern);
|
||||
if (rule->dir_only && plen > 0 && CVS_DEFAULTS[i].pattern[plen - 1] == '/')
|
||||
@@ -266,98 +454,261 @@ static bool cvs_rule_list_append(FilterRuleList* list) {
|
||||
return true;
|
||||
}
|
||||
|
||||
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
|
||||
#define FILTER_MAX_MERGE_DEPTH 16
|
||||
|
||||
static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* base_dir,
|
||||
int depth, char* err, size_t err_size);
|
||||
|
||||
/* Read a merge file and splice its rules into `list`. A relative path is
|
||||
* resolved below `base_dir` when given, else used as-is (rsync resolves a
|
||||
* command-line merge file relative to the current directory). */
|
||||
static bool filter_list_merge_file(FilterRuleList* list, const char* name,
|
||||
const FilterParseOptions* opts, const char* base_dir, int depth,
|
||||
char* err, size_t err_size) {
|
||||
if (name[0] == '\0') {
|
||||
filter_set_error(err, err_size, "merge requires a filename");
|
||||
return false;
|
||||
}
|
||||
char* path =
|
||||
(base_dir && base_dir[0] && name[0] != '/') ? path_cat(base_dir, name) : str_dup(name);
|
||||
if (!path) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
FILE* fp = fopen(path, "r");
|
||||
if (!fp) {
|
||||
filter_set_error(err, err_size, "could not read merge file '%s': %s", path, strerror(errno));
|
||||
free(path);
|
||||
return false;
|
||||
}
|
||||
char* line = NULL;
|
||||
size_t cap = 0;
|
||||
bool ok = true;
|
||||
while (true) {
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &cap, '\n', UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
filter_set_error(err, err_size, "error reading merge file '%s'", path);
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (n == 0)
|
||||
break;
|
||||
const char* lp = line;
|
||||
while (*lp == ' ' || *lp == '\t')
|
||||
lp++;
|
||||
if (*lp == '\0' || *lp == '\n' || *lp == '\r' || *lp == '#')
|
||||
continue;
|
||||
if (!filter_list_parse_append_depth(list, lp, opts, base_dir, depth + 1, err, err_size)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
free(line);
|
||||
fclose(fp);
|
||||
free(path);
|
||||
return ok;
|
||||
}
|
||||
|
||||
/* Parse one line and append/merge it into `list`. Handles clear, merge and
|
||||
* dir-merge at the list level. */
|
||||
static bool filter_list_parse_append_depth(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* base_dir,
|
||||
int depth, char* err, size_t err_size) {
|
||||
if (depth > FILTER_MAX_MERGE_DEPTH) {
|
||||
filter_set_error(err, err_size, "merge files nested too deeply");
|
||||
return false;
|
||||
}
|
||||
const char* p = line;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
if (*p == '\0' || *p == '\n' || *p == '\r')
|
||||
return true;
|
||||
|
||||
RuleKind kind = RULE_KIND_UNKNOWN;
|
||||
unsigned sides;
|
||||
bool sides_explicit, negate, anchored_mod, perishable, xattr, cvs_inject;
|
||||
const char* pat;
|
||||
size_t pat_len;
|
||||
if (!parse_rule_syntax(p, &kind, &sides, &sides_explicit, &negate, &anchored_mod, &perishable,
|
||||
&xattr, &cvs_inject, &pat, &pat_len)) {
|
||||
filter_set_error(err, err_size, "unrecognized filter rule syntax: %s", p);
|
||||
return false;
|
||||
}
|
||||
(void)sides_explicit;
|
||||
(void)negate;
|
||||
(void)anchored_mod;
|
||||
(void)perishable;
|
||||
(void)xattr;
|
||||
|
||||
if (cvs_inject) {
|
||||
/* "C" injects the CVS defaults in place; no pattern is expected. */
|
||||
return filter_list_append_cvs(list, sides);
|
||||
}
|
||||
if (kind == RULE_KIND_CLEAR) {
|
||||
if (pat_len != 0) {
|
||||
filter_set_error(err, err_size, "clear takes no pattern");
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < list->count; i++)
|
||||
filter_rule_free(list->items[i]);
|
||||
list->count = 0;
|
||||
return true;
|
||||
}
|
||||
if (kind == RULE_KIND_MERGE) {
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "merge requires a filename");
|
||||
return false;
|
||||
}
|
||||
char* name = malloc(pat_len + 1);
|
||||
if (!name) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
memcpy(name, pat, pat_len);
|
||||
name[pat_len] = '\0';
|
||||
bool ok = filter_list_merge_file(list, name, opts, base_dir, depth, err, err_size);
|
||||
free(name);
|
||||
return ok;
|
||||
}
|
||||
if (kind == RULE_KIND_DIR_MERGE) {
|
||||
if (pat_len == 0) {
|
||||
filter_set_error(err, err_size, "dir-merge requires a filename");
|
||||
return false;
|
||||
}
|
||||
char* name = malloc(pat_len + 1);
|
||||
if (!name) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
memcpy(name, pat, pat_len);
|
||||
name[pat_len] = '\0';
|
||||
bool ok = filter_rule_list_add_dir_merge(list, name);
|
||||
free(name);
|
||||
if (!ok) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
FilterRule* rule = filter_rule_parse(p, opts, err, err_size);
|
||||
if (!rule)
|
||||
return false;
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* merge_base_dir,
|
||||
char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (!list)
|
||||
return false;
|
||||
return filter_list_parse_append_depth(list, line, opts, merge_base_dir, 0, err, err_size);
|
||||
}
|
||||
|
||||
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
|
||||
bool delete_excluded, char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
FilterRuleList* list = filter_rule_list_create();
|
||||
if (!list) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
FilterParseOptions opts = {.delete_excluded = delete_excluded, .cvs_exclude = cvs_exclude};
|
||||
for (int i = 0; i < rule_count; i++) {
|
||||
if (!rule_texts || !rule_texts[i])
|
||||
continue;
|
||||
FilterRule* rule = filter_rule_parse(rule_texts[i], err, err_size);
|
||||
if (!rule) {
|
||||
if (!filter_rule_list_parse_append(list, rule_texts[i], &opts, NULL, err, err_size)) {
|
||||
filter_rule_list_free(list);
|
||||
return NULL;
|
||||
}
|
||||
if (!set_rule_owner(rule, "")) {
|
||||
filter_rule_free(rule);
|
||||
filter_rule_list_free(list);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
if (cvs_exclude && !filter_list_append_cvs(list, FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER)) {
|
||||
filter_rule_list_free(list);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
}
|
||||
if (cvs_exclude && !cvs_rule_list_append(list)) {
|
||||
filter_rule_list_free(list);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
return list;
|
||||
}
|
||||
|
||||
/* ---- Per-directory .rsync-filter files ---- */
|
||||
/* ---- Per-directory merge files ---- */
|
||||
|
||||
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
|
||||
/* Undo the rules and dir-merge registrations that one merge file appended,
|
||||
* leaving the caller's earlier content intact. A "clear" rule inside the file
|
||||
* frees every rule, including the caller's; clamp to the surviving count so
|
||||
* those already-freed rules are never resurrected and freed a second time. */
|
||||
static void filter_file_rollback(FilterRuleList* list, int rules_before, int dir_merges_before) {
|
||||
int first = rules_before < list->count ? rules_before : list->count;
|
||||
for (int i = first; i < list->count; i++)
|
||||
filter_rule_free(list->items[i]);
|
||||
list->count = first;
|
||||
for (int i = dir_merges_before; i < list->dir_merge_count; i++)
|
||||
free(list->dir_merge_names[i]);
|
||||
list->dir_merge_count = dir_merges_before;
|
||||
}
|
||||
|
||||
bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts, bool* exists,
|
||||
char* err, size_t err_size) {
|
||||
if (err && err_size > 0)
|
||||
err[0] = '\0';
|
||||
if (exists)
|
||||
*exists = false;
|
||||
char* filter_path = path_cat(dir_path, ".rsync-filter");
|
||||
if (!list)
|
||||
return false;
|
||||
char* filter_path = path_cat(dir_path, name);
|
||||
if (!filter_path) {
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return false;
|
||||
}
|
||||
FILE* fp = fopen(filter_path, "r");
|
||||
free(filter_path);
|
||||
if (!fp) {
|
||||
if (errno == ENOENT || errno == ENOTDIR)
|
||||
return filter_rule_list_create();
|
||||
log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s", dir_path,
|
||||
strerror(errno));
|
||||
return filter_rule_list_create();
|
||||
return true;
|
||||
char* escaped_dir = output_escape(dir_path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Could not read %s in %s: %s", name,
|
||||
escaped_dir ? escaped_dir : "<allocation failed>", strerror(errno));
|
||||
free(escaped_dir);
|
||||
return true;
|
||||
}
|
||||
if (exists)
|
||||
*exists = true;
|
||||
FilterRuleList* list = filter_rule_list_create();
|
||||
if (!list) {
|
||||
fclose(fp);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
int rules_before = list->count;
|
||||
int dir_merges_before = list->dir_merge_count;
|
||||
char* line = NULL;
|
||||
size_t line_cap = 0;
|
||||
ssize_t n;
|
||||
bool ok = true;
|
||||
while ((n = getline(&line, &line_cap, fp)) != -1) {
|
||||
while (true) {
|
||||
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, '\n', UTILS_MAX_LINE_LEN);
|
||||
if (n < 0) {
|
||||
if (errno == EFBIG) {
|
||||
filter_set_error(err, err_size, "line in %s exceeds %d bytes", name,
|
||||
(int)UTILS_MAX_LINE_LEN);
|
||||
} else {
|
||||
filter_set_error(err, err_size, "error reading %s: %s", name, strerror(errno));
|
||||
}
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (n == 0)
|
||||
break;
|
||||
const char* p = line;
|
||||
while (*p == ' ' || *p == '\t')
|
||||
p++;
|
||||
if (*p == '\0' || *p == '\n' || *p == '\r' || *p == '#')
|
||||
continue;
|
||||
FilterRule* rule = filter_rule_parse(p, err, err_size);
|
||||
if (!rule) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (!set_rule_owner(rule, owner_rel)) {
|
||||
filter_rule_free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
if (!filter_rule_list_add(list, rule)) {
|
||||
filter_rule_free(rule);
|
||||
snprintf(err, err_size, "memory allocation failed");
|
||||
/* Merge files inside a per-directory file resolve relative to that
|
||||
directory. */
|
||||
if (!filter_list_parse_append_depth(list, p, opts, dir_path, 0, err, err_size)) {
|
||||
ok = false;
|
||||
break;
|
||||
}
|
||||
@@ -365,12 +716,40 @@ FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bo
|
||||
free(line);
|
||||
fclose(fp);
|
||||
if (!ok) {
|
||||
filter_file_rollback(list, rules_before, dir_merges_before);
|
||||
return false;
|
||||
}
|
||||
for (int i = rules_before; i < list->count; i++) {
|
||||
if (!set_rule_owner(list->items[i], owner_rel)) {
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
filter_file_rollback(list, rules_before, dir_merges_before);
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
FilterRuleList* filter_file_read_named(const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts,
|
||||
bool* exists, char* err, size_t err_size) {
|
||||
FilterRuleList* list = filter_rule_list_create();
|
||||
if (!list) {
|
||||
if (err && err_size > 0)
|
||||
filter_set_error(err, err_size, "memory allocation failed");
|
||||
return NULL;
|
||||
}
|
||||
if (!filter_file_append(list, dir_path, name, owner_rel, opts, exists, err, err_size)) {
|
||||
filter_rule_list_free(list);
|
||||
return NULL;
|
||||
}
|
||||
return list;
|
||||
}
|
||||
|
||||
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
|
||||
char* err, size_t err_size) {
|
||||
return filter_file_read_named(dir_path, ".rsync-filter", owner_rel, NULL, exists, err, err_size);
|
||||
}
|
||||
|
||||
/* ---- Rule matching ---- */
|
||||
|
||||
/* Match a pattern that contains '/' (non-anchored) against the end of the
|
||||
@@ -386,10 +765,10 @@ static bool glob_suffix_match(const char* pattern, const char* str) {
|
||||
}
|
||||
|
||||
static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, const char* leaf,
|
||||
bool is_dir) {
|
||||
bool is_dir, unsigned side) {
|
||||
if (!rule || !rule->pattern)
|
||||
return FILTER_ACTION_NONE;
|
||||
if (rule->dir_only && !is_dir)
|
||||
if (!(rule->sides & side))
|
||||
return FILTER_ACTION_NONE;
|
||||
/* A rule applies only to entries below its owner directory. */
|
||||
const char* rel2 = rel_path;
|
||||
@@ -404,24 +783,36 @@ static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, c
|
||||
if (rel2[0] == '\0')
|
||||
return FILTER_ACTION_NONE;
|
||||
bool matched;
|
||||
if (rule->anchored) {
|
||||
if (rule->dir_only && !is_dir)
|
||||
matched = false;
|
||||
else if (rule->anchored)
|
||||
matched = glob_match(rule->pattern, rel2);
|
||||
} else if (strchr(rule->pattern, '/') != NULL) {
|
||||
else if (strchr(rule->pattern, '/') != NULL)
|
||||
matched = glob_suffix_match(rule->pattern, rel2);
|
||||
} else {
|
||||
else
|
||||
matched = glob_match(rule->pattern, leaf);
|
||||
}
|
||||
return matched ? rule->action : FILTER_ACTION_NONE;
|
||||
if (rule->negate)
|
||||
matched = !matched;
|
||||
if (!matched)
|
||||
return FILTER_ACTION_NONE;
|
||||
if (side == FILTER_SIDE_RECEIVER)
|
||||
return rule->action == FILTER_ACTION_EXCLUDE ? FILTER_ACTION_PROTECT : FILTER_ACTION_RISK;
|
||||
return rule->action;
|
||||
}
|
||||
|
||||
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
|
||||
bool is_dir) {
|
||||
FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path,
|
||||
const char* leaf, bool is_dir, unsigned side) {
|
||||
if (!list)
|
||||
return FILTER_ACTION_NONE;
|
||||
for (int i = 0; i < list->count; i++) {
|
||||
FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir);
|
||||
FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir, side);
|
||||
if (action != FILTER_ACTION_NONE)
|
||||
return action;
|
||||
}
|
||||
return FILTER_ACTION_NONE;
|
||||
}
|
||||
|
||||
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
|
||||
bool is_dir) {
|
||||
return filter_rules_apply_side(list, rel_path, leaf, is_dir, FILTER_SIDE_SENDER);
|
||||
}
|
||||
|
||||
+88
-34
@@ -4,38 +4,51 @@
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
|
||||
/* rsync-style filter rule engine (client-side file selection).
|
||||
/* rsync-style filter rule engine (client-side file selection and the
|
||||
* receiver-side protection set it feeds).
|
||||
*
|
||||
* Supported rule syntax (documented subset):
|
||||
* [+|-] [anchored '/' prefix] pattern [trailing '/' for dir-only]
|
||||
*
|
||||
* "+ PATTERN" include rule (first match wins)
|
||||
* "- PATTERN" exclude rule
|
||||
* "PATTERN" implicit exclude rule (rsync default)
|
||||
* "include PATTERN" / "exclude PATTERN" word forms
|
||||
* leading '/' after the +/- anchors the pattern to its owner directory
|
||||
* (the transfer root for command-line/-C rules, the directory that
|
||||
* contains a .rsync-filter file for per-directory rules)
|
||||
* a trailing '/' makes the rule match directories only
|
||||
*
|
||||
* Rejected explicitly (no silent no-ops): the rsync merge/dir-merge/list-clear
|
||||
* shorthands written as a rule that starts with ':' or '.' or '!', the
|
||||
* merge/dir-merge/hide/show/protect/risk/clear words, and every include/exclude
|
||||
* rule modifier other than '/' (! C s r p x). The pattern must be separated
|
||||
* from +/- by a space (or a single '/' anchor), exactly like rsync's
|
||||
* "-s foo"/"-p ..." modifier syntax is refused.
|
||||
* Rule syntax (see the rsync man page FILTER RULES section):
|
||||
* RULE [PATTERN_OR_FILENAME]
|
||||
* RULE,MODIFIERS [PATTERN_OR_FILENAME]
|
||||
* Short RULE names may attach MODIFIERS directly ("-sr foo"); the long name
|
||||
* form requires the comma. The pattern/filename is separated from the rule by
|
||||
* one space or underscore. Rule names:
|
||||
* exclude/- exclude (by default both sender-hide and receiver-protect)
|
||||
* include/+ include (by default both sender-show and receiver-risk)
|
||||
* hide/H sender-only exclude
|
||||
* show/S sender-only include
|
||||
* protect/P receiver-only exclude (protect from deletion)
|
||||
* risk/R receiver-only include (allow deletion)
|
||||
* merge/. read a client-side merge file for more rules
|
||||
* dir-merge/: per-directory merge file (registered for the scanner)
|
||||
* clear/! clear the current rule list (takes no argument)
|
||||
* Modifiers: '/' absolute anchor, '!' negate match, 'C' inject CVS defaults,
|
||||
* 's' sender side, 'r' receiver side, 'p' perishable, 'x' xattr name rule.
|
||||
* A trailing '/' makes a pattern match directories only. A leading '/' anchors
|
||||
* the pattern to its owner directory.
|
||||
*/
|
||||
|
||||
typedef enum {
|
||||
FILTER_ACTION_NONE = 0, /* no rule matched */
|
||||
FILTER_ACTION_EXCLUDE = -1,
|
||||
FILTER_ACTION_INCLUDE = 1
|
||||
FILTER_ACTION_INCLUDE = 1,
|
||||
/* Receiver-side-only verdicts: the entry is transferred but its destination
|
||||
* mirror is protected from --delete (PROTECT) or explicitly left at risk
|
||||
* (RISK). */
|
||||
FILTER_ACTION_PROTECT = 2,
|
||||
FILTER_ACTION_RISK = 3,
|
||||
} FilterAction;
|
||||
|
||||
#define FILTER_SIDE_SENDER 1u
|
||||
#define FILTER_SIDE_RECEIVER 2u
|
||||
|
||||
typedef struct {
|
||||
FilterAction action;
|
||||
FilterAction action; /* EXCLUDE or INCLUDE (the base pattern action) */
|
||||
unsigned sides; /* FILTER_SIDE_SENDER | FILTER_SIDE_RECEIVER */
|
||||
bool anchored; /* pattern anchored to the rule's owner directory */
|
||||
bool dir_only; /* pattern had a trailing '/': matches directories only */
|
||||
bool negate; /* '!' modifier: match succeeds when the pattern does not */
|
||||
bool perishable; /* 'p' modifier (ignored in deleted directories) */
|
||||
char* owner; /* owning directory rel path ("" == transfer root) */
|
||||
char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */
|
||||
} FilterRule;
|
||||
@@ -44,19 +57,41 @@ typedef struct {
|
||||
FilterRule** items; /* owned array of rule pointers */
|
||||
int count;
|
||||
int capacity;
|
||||
/* Per-directory merge-file basenames registered by "dir-merge NAME"/": NAME"
|
||||
* or by -F (.rsync-filter). Owned strings; the scanner reads each name in
|
||||
* every directory it traverses. */
|
||||
char** dir_merge_names;
|
||||
int dir_merge_count;
|
||||
int dir_merge_capacity;
|
||||
} FilterRuleList;
|
||||
|
||||
/* Context needed while parsing a rule list (merge files, --delete-excluded). */
|
||||
typedef struct {
|
||||
bool delete_excluded; /* --delete-excluded: default sides become sender-only */
|
||||
bool cvs_exclude; /* -C: expand the CVS default excludes */
|
||||
} FilterParseOptions;
|
||||
|
||||
/* Parse a single filter-rule line (no trailing newline required). Returns an
|
||||
* owned rule, or NULL on unsupported/invalid syntax with a message in `err`. */
|
||||
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size);
|
||||
* owned rule, or NULL on unsupported/invalid syntax with a message in `err`.
|
||||
* `opts` may be NULL (no merge expansion / no delete-excluded). */
|
||||
FilterRule* filter_rule_parse(const char* line, const FilterParseOptions* opts, char* err,
|
||||
size_t err_size);
|
||||
void filter_rule_free(FilterRule* rule);
|
||||
|
||||
FilterRuleList* filter_rule_list_create(void);
|
||||
/* Append a fully-parsed rule (takes ownership). Returns false on OOM. */
|
||||
bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule);
|
||||
/* Parse `line` and append it. Returns false and fills `err` on bad syntax. */
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err,
|
||||
size_t err_size);
|
||||
/* Register a per-directory merge-file basename (idempotent). Returns false on
|
||||
* OOM. Used by the scanner to read custom "dir-merge" files. */
|
||||
bool filter_rule_list_add_dir_merge(FilterRuleList* list, const char* name);
|
||||
/* Parse `line` and append it. Handles "clear"/"!" (resets the list), "merge
|
||||
* FILE"/". FILE" (splices the file's rules) and "dir-merge NAME"/": NAME"
|
||||
* (registers a per-directory filename). Returns false and fills `err` on bad
|
||||
* syntax or an unreadable merge file. `merge_base_dir` resolves a relative
|
||||
* merge-file path (NULL means the process working directory). */
|
||||
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line,
|
||||
const FilterParseOptions* opts, const char* merge_base_dir,
|
||||
char* err, size_t err_size);
|
||||
void filter_rule_list_free(FilterRuleList* list);
|
||||
|
||||
/* Build the command-line filter set: `rule_texts` (--filter=RULE in the order
|
||||
@@ -64,19 +99,38 @@ void filter_rule_list_free(FilterRuleList* list);
|
||||
* cvs_exclude is true. All rules are owned by "" (the transfer root).
|
||||
* Returns NULL on unsupported rule text (message in `err`). */
|
||||
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
|
||||
bool delete_excluded, char* err, size_t err_size);
|
||||
|
||||
/* Read "<dir_path>/<name>" and return its rules, each owned by `owner_rel`. A
|
||||
* missing file yields an empty list with *exists=false; an unreadable file is
|
||||
* treated as missing. Returns NULL only on parse or allocation failure
|
||||
* (message in `err`). `opts` may be NULL. */
|
||||
FilterRuleList* filter_file_read_named(const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts,
|
||||
bool* exists, char* err, size_t err_size);
|
||||
|
||||
/* Append the rules of "<dir_path>/<name>" into an existing list (each owned by
|
||||
* `owner_rel`). A missing file yields *exists=false and no error. Returns
|
||||
* false only on parse/allocation failure (message in `err`). */
|
||||
bool filter_file_append(FilterRuleList* list, const char* dir_path, const char* name,
|
||||
const char* owner_rel, const FilterParseOptions* opts, bool* exists,
|
||||
char* err, size_t err_size);
|
||||
|
||||
/* Read "<dir_path>/.rsync-filter" and return its rules, each owned by
|
||||
* `owner_rel`. A missing file yields an empty list with *exists=false; an
|
||||
* unreadable file is treated as missing. Returns NULL only on parse or
|
||||
* allocation failure (message in `err`). */
|
||||
/* filter_file_read_named with the default ".rsync-filter" name. */
|
||||
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
|
||||
char* err, size_t err_size);
|
||||
|
||||
/* Evaluate an entry against one ordered rule list. Returns FILTER_ACTION_NONE
|
||||
* when no rule matched, otherwise the first matching rule's action.
|
||||
* `rel_path` is the entry's path relative to the transfer root ("" == root),
|
||||
* `leaf` its final name, `is_dir` whether it is a directory. */
|
||||
/* Evaluate an entry against one ordered rule list for one side. Returns
|
||||
* FILTER_ACTION_NONE when no rule matched, otherwise the first matching rule's
|
||||
* action (for the receiver side an EXCLUDE is reported as
|
||||
* FILTER_ACTION_PROTECT and an INCLUDE as FILTER_ACTION_RISK). `rel_path` is
|
||||
* the entry's path relative to the transfer root ("" == root), `leaf` its final
|
||||
* name, `is_dir` whether it is a directory. */
|
||||
FilterAction filter_rules_apply_side(const FilterRuleList* list, const char* rel_path,
|
||||
const char* leaf, bool is_dir, unsigned side);
|
||||
|
||||
/* Sender-side convenience wrapper (kept for callers/tests that only need the
|
||||
* transfer decision). */
|
||||
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
|
||||
bool is_dir);
|
||||
|
||||
|
||||
@@ -0,0 +1,129 @@
|
||||
#include "format.h"
|
||||
#include "protocol.h"
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
|
||||
bool format_human_size_decimal(unsigned long long bytes, char* buffer, size_t buffer_size) {
|
||||
if (!buffer || buffer_size == 0)
|
||||
return false;
|
||||
if (bytes < 1000ULL) {
|
||||
int written = snprintf(buffer, buffer_size, "%llu", bytes);
|
||||
return written >= 0 && (size_t)written < buffer_size;
|
||||
}
|
||||
static const char units[] = "KMGTPE";
|
||||
double value = (double)bytes;
|
||||
size_t divisions = 0;
|
||||
while (value >= 1000.0 && divisions < sizeof(units) - 1) {
|
||||
value /= 1000.0;
|
||||
divisions++;
|
||||
}
|
||||
int written = snprintf(buffer, buffer_size, "%.2f%c", value, units[divisions - 1]);
|
||||
return written >= 0 && (size_t)written < buffer_size;
|
||||
}
|
||||
|
||||
bool format_big_num(unsigned long long value, bool human_readable, char* buffer,
|
||||
size_t buffer_size) {
|
||||
if (human_readable)
|
||||
return format_human_size_decimal(value, buffer, buffer_size);
|
||||
char digits[32];
|
||||
int written = snprintf(digits, sizeof(digits), "%llu", value);
|
||||
if (written < 0 || (size_t)written >= sizeof(digits))
|
||||
return false;
|
||||
size_t len = (size_t)written;
|
||||
size_t separators = len > 1 ? (len - 1) / 3 : 0;
|
||||
size_t total = len + separators;
|
||||
if (total + 1 > buffer_size)
|
||||
return false;
|
||||
size_t out = total;
|
||||
buffer[out] = '\0';
|
||||
size_t digits_since_sep = 0;
|
||||
for (size_t i = len; i > 0; i--) {
|
||||
buffer[--out] = digits[i - 1];
|
||||
digits_since_sep++;
|
||||
if (digits_since_sep == 3 && i > 1) {
|
||||
buffer[--out] = ',';
|
||||
digits_since_sep = 0;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
bool format_rsync_datetime(time_t when, bool dash, char* buffer, size_t buffer_size) {
|
||||
if (!buffer || buffer_size == 0)
|
||||
return false;
|
||||
struct tm broken_down;
|
||||
if (localtime_r(&when, &broken_down) == NULL)
|
||||
return false;
|
||||
const char* format = dash ? "%Y/%m/%d-%H:%M:%S" : "%Y/%m/%d %H:%M:%S";
|
||||
return strftime(buffer, buffer_size, format, &broken_down) != 0;
|
||||
}
|
||||
|
||||
bool format_dest_state_send(int fd, const OutputDestState* state) {
|
||||
if (!state)
|
||||
return false;
|
||||
int32_t has_old = state->existed ? 1 : 0;
|
||||
uint64_t size = (uint64_t)state->size;
|
||||
int64_t mtime = (int64_t)state->mtime_sec;
|
||||
int64_t mtime_nsec = state->mtime_nsec;
|
||||
uint32_t mode = state->mode;
|
||||
int32_t uid = state->uid;
|
||||
int32_t gid = state->gid;
|
||||
return send_n_data(fd, &has_old, sizeof(has_old)) && send_n_data(fd, &size, sizeof(size)) &&
|
||||
send_n_data(fd, &mtime, sizeof(mtime)) &&
|
||||
send_n_data(fd, &mtime_nsec, sizeof(mtime_nsec)) && send_n_data(fd, &mode, sizeof(mode)) &&
|
||||
send_n_data(fd, &uid, sizeof(uid)) && send_n_data(fd, &gid, sizeof(gid));
|
||||
}
|
||||
|
||||
bool format_dest_state_receive(int fd, OutputDestState* state) {
|
||||
if (!state)
|
||||
return false;
|
||||
int32_t has_old = 0;
|
||||
uint64_t size = 0;
|
||||
int64_t mtime = 0;
|
||||
int64_t mtime_nsec = 0;
|
||||
uint32_t mode = 0;
|
||||
int32_t uid = 0;
|
||||
int32_t gid = 0;
|
||||
if (!receive_n_data(fd, &has_old, sizeof(has_old)) || !receive_n_data(fd, &size, sizeof(size)) ||
|
||||
!receive_n_data(fd, &mtime, sizeof(mtime)) ||
|
||||
!receive_n_data(fd, &mtime_nsec, sizeof(mtime_nsec)) ||
|
||||
!receive_n_data(fd, &mode, sizeof(mode)) || !receive_n_data(fd, &uid, sizeof(uid)) ||
|
||||
!receive_n_data(fd, &gid, sizeof(gid)))
|
||||
return false;
|
||||
memset(state, 0, sizeof(*state));
|
||||
state->known = true;
|
||||
state->existed = has_old != 0;
|
||||
state->size = size;
|
||||
state->mtime_sec = mtime;
|
||||
state->mtime_nsec = mtime_nsec;
|
||||
state->mode = mode;
|
||||
state->uid = uid;
|
||||
state->gid = gid;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool format_stats_send(int fd, const ReceiverStats* stats) {
|
||||
if (!stats)
|
||||
return false;
|
||||
unsigned long long matched = stats->matched_data;
|
||||
unsigned long long deleted = stats->deleted_files;
|
||||
unsigned long long would = stats->would_delete_count;
|
||||
return send_n_data(fd, &matched, sizeof(matched)) && send_n_data(fd, &deleted, sizeof(deleted)) &&
|
||||
send_n_data(fd, &would, sizeof(would));
|
||||
}
|
||||
|
||||
bool format_stats_receive(int fd, ReceiverStats* stats) {
|
||||
if (!stats)
|
||||
return false;
|
||||
unsigned long long matched = 0;
|
||||
unsigned long long deleted = 0;
|
||||
unsigned long long would = 0;
|
||||
if (!receive_n_data(fd, &matched, sizeof(matched)) ||
|
||||
!receive_n_data(fd, &deleted, sizeof(deleted)) || !receive_n_data(fd, &would, sizeof(would)))
|
||||
return false;
|
||||
memset(stats, 0, sizeof(*stats));
|
||||
stats->matched_data = matched;
|
||||
stats->deleted_files = deleted;
|
||||
stats->would_delete_count = would;
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
#ifndef FORMAT_H
|
||||
#define FORMAT_H
|
||||
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <time.h>
|
||||
|
||||
/* Low-level output-formatting primitives shared by the change-event model
|
||||
* (change_list.c) and the transfer driver (client_send.c).
|
||||
*
|
||||
* The functions here are pure/string-level except for the STATUS_DEST_INFO
|
||||
* codec, which lets the receiver report the pre-transfer destination entry so
|
||||
* the sender can render rsync-accurate --itemize-changes / --out-format
|
||||
* columns (see protocol.h). */
|
||||
|
||||
/* Pre-transfer destination snapshot, reported by the receiver when the wire
|
||||
* config carries report_dest_info. `known` distinguishes "no report was
|
||||
* requested/received" from "the destination did not exist" (`existed == false`
|
||||
* with `known == true`). */
|
||||
typedef struct {
|
||||
bool known;
|
||||
bool existed;
|
||||
unsigned long long size;
|
||||
long long mtime_sec;
|
||||
long long mtime_nsec;
|
||||
uint32_t mode;
|
||||
int32_t uid;
|
||||
int32_t gid;
|
||||
} OutputDestState;
|
||||
|
||||
/* rsync's -h/--human-readable size (decimal, base 1000): integers below 1000
|
||||
* print verbatim; larger values use the largest unit that keeps the value
|
||||
* below 1000 (K/M/G/T/P/E) with exactly two decimals, so 1500000 -> "1.50M"
|
||||
* and 999999 -> "1000.00K" (matching rsync's human_num). Returns false when
|
||||
* the buffer is too small (nothing is written). */
|
||||
bool format_human_size_decimal(unsigned long long bytes, char* buffer, size_t buffer_size);
|
||||
|
||||
/* rsync's general number formatting (big_num). When `human_readable` is true
|
||||
* this is format_human_size_decimal; otherwise the integer is rendered with a
|
||||
* ',' thousands separator every three digits (rsync's separator in the C
|
||||
* locale). Returns false on an undersized buffer. */
|
||||
bool format_big_num(unsigned long long value, bool human_readable, char* buffer,
|
||||
size_t buffer_size);
|
||||
|
||||
/* rsync's %M/%t timestamp. When `dash` is true the separator between the date
|
||||
* and the time is '-' (the %M form: "YYYY/MM/DD-HH:MM:SS"); otherwise it is a
|
||||
* space (the %t form: "YYYY/MM/DD HH:MM:SS"). Local time. Returns false on a
|
||||
* bad time or an undersized buffer. */
|
||||
bool format_rsync_datetime(time_t when, bool dash, char* buffer, size_t buffer_size);
|
||||
|
||||
/* Fixed-width STATUS_DEST_INFO record codec (int32 has_old, uint64 size,
|
||||
* int64 mtime, int64 mtime_nsec, uint32 mode, int32 uid, int32 gid). The
|
||||
* status frame itself is sent/received by the caller. Returns false on I/O
|
||||
* failure. */
|
||||
bool format_dest_state_send(int fd, const OutputDestState* state);
|
||||
bool format_dest_state_receive(int fd, OutputDestState* state);
|
||||
|
||||
/* End-of-transfer receiver counters reported through STATUS_STATS (protocol
|
||||
* 2.25.0) when the wire config carries report_stats. `would_delete_count` is
|
||||
* the number of destination-relative paths the receiver would have deleted in a
|
||||
* -n/--dry-run --delete run; that many wire strings immediately follow the
|
||||
* fixed record (sent/read by the caller). */
|
||||
typedef struct {
|
||||
unsigned long long matched_data;
|
||||
unsigned long long deleted_files;
|
||||
unsigned long long would_delete_count;
|
||||
} ReceiverStats;
|
||||
|
||||
/* Fixed-width STATUS_STATS counter record. The status frame and the optional
|
||||
* would-delete path list are sent/received by the caller. Returns false on I/O
|
||||
* failure. */
|
||||
bool format_stats_send(int fd, const ReceiverStats* stats);
|
||||
bool format_stats_receive(int fd, ReceiverStats* stats);
|
||||
|
||||
#endif
|
||||
+471
-128
@@ -30,20 +30,40 @@ typedef struct {
|
||||
/* --super / --no-super tri-state (SUPER_MODE_AUTO when unset). Snapshotted
|
||||
* per connection so privilege_super_permitted() can gate super-user
|
||||
* activities without a Config argument. */
|
||||
int super_mode;
|
||||
SuperMode super_mode;
|
||||
/* --copy-as=USER[:GROUP]: snapshotted so the ownership resolver can force the
|
||||
* target ids without a Config argument. */
|
||||
bool copy_as_set;
|
||||
int32_t copy_as_uid;
|
||||
int32_t copy_as_gid;
|
||||
/* -o/--owner and -g/--group: preserve the source owner/group through the
|
||||
* normal name/identity resolution path. Split out of the former
|
||||
* use_metadata bundle; unlike --numeric-ids/--chown/--usermap/--groupmap/-a
|
||||
* these are a preserve-source request, not an arbitrary client-chosen owner,
|
||||
* so they are tracked separately from the explicit ownership gate. */
|
||||
bool preserve_owner;
|
||||
bool preserve_group;
|
||||
/* --fake-super: when active the receiver must only RECORD the (resolved)
|
||||
* ownership in the reserved xattr, never perform a real chown. Snapshotted
|
||||
* so the fd-relative ownership helpers can suppress the chown without a
|
||||
* Config argument. */
|
||||
bool fake_super;
|
||||
bool set;
|
||||
} IdentityActive;
|
||||
|
||||
static IdentityActive g_identity;
|
||||
|
||||
static void identity_active_reset(void) {
|
||||
if (g_identity.usermap) {
|
||||
for (int i = 0; i < g_identity.usermap_count; i++)
|
||||
free(g_identity.usermap[i].to_name);
|
||||
free(g_identity.usermap);
|
||||
}
|
||||
if (g_identity.groupmap) {
|
||||
for (int i = 0; i < g_identity.groupmap_count; i++)
|
||||
free(g_identity.groupmap[i].to_name);
|
||||
free(g_identity.groupmap);
|
||||
}
|
||||
g_identity.usermap = NULL;
|
||||
g_identity.groupmap = NULL;
|
||||
g_identity.usermap_count = 0;
|
||||
@@ -57,6 +77,9 @@ static void identity_active_reset(void) {
|
||||
g_identity.copy_as_set = false;
|
||||
g_identity.copy_as_uid = 0;
|
||||
g_identity.copy_as_gid = 0;
|
||||
g_identity.preserve_owner = false;
|
||||
g_identity.preserve_group = false;
|
||||
g_identity.fake_super = false;
|
||||
g_identity.set = false;
|
||||
}
|
||||
|
||||
@@ -77,32 +100,57 @@ bool identity_set_active(const Config* config) {
|
||||
g_identity.copy_as_set = config->copy_as_set;
|
||||
g_identity.copy_as_uid = config->copy_as_uid;
|
||||
g_identity.copy_as_gid = config->copy_as_gid;
|
||||
g_identity.preserve_owner = config->preserve_owner;
|
||||
g_identity.preserve_group = config->preserve_group;
|
||||
g_identity.fake_super = config->fake_super;
|
||||
if (config->usermap_count > 0) {
|
||||
g_identity.usermap = calloc((size_t)config->usermap_count, sizeof(IdentityMap));
|
||||
if (!g_identity.usermap)
|
||||
goto alloc_failed;
|
||||
memcpy(g_identity.usermap, config->usermap,
|
||||
(size_t)config->usermap_count * sizeof(IdentityMap));
|
||||
for (int i = 0; i < config->usermap_count; i++) {
|
||||
g_identity.usermap[i] = config->usermap[i];
|
||||
g_identity.usermap[i].to_name =
|
||||
config->usermap[i].to_name ? str_dup(config->usermap[i].to_name) : NULL;
|
||||
if (config->usermap[i].to_name && !g_identity.usermap[i].to_name) {
|
||||
g_identity.usermap_count = i; /* free only the entries already duplicated */
|
||||
goto alloc_failed;
|
||||
}
|
||||
}
|
||||
g_identity.usermap_count = config->usermap_count;
|
||||
}
|
||||
if (config->groupmap_count > 0) {
|
||||
g_identity.groupmap = calloc((size_t)config->groupmap_count, sizeof(IdentityMap));
|
||||
if (!g_identity.groupmap)
|
||||
goto alloc_failed;
|
||||
memcpy(g_identity.groupmap, config->groupmap,
|
||||
(size_t)config->groupmap_count * sizeof(IdentityMap));
|
||||
for (int i = 0; i < config->groupmap_count; i++) {
|
||||
g_identity.groupmap[i] = config->groupmap[i];
|
||||
g_identity.groupmap[i].to_name =
|
||||
config->groupmap[i].to_name ? str_dup(config->groupmap[i].to_name) : NULL;
|
||||
if (config->groupmap[i].to_name && !g_identity.groupmap[i].to_name) {
|
||||
g_identity.groupmap_count = i;
|
||||
goto alloc_failed;
|
||||
}
|
||||
}
|
||||
g_identity.groupmap_count = config->groupmap_count;
|
||||
}
|
||||
g_identity.set = true;
|
||||
/* A root receiver would honor any client-supplied ownership request (a
|
||||
--usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids).
|
||||
Surface that prominently; a privileged daemon applying arbitrary client
|
||||
ownership is a deliberate, opt-in choice the operator should be aware of. */
|
||||
if (geteuid() == 0)
|
||||
--usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids)
|
||||
ONLY when super-user activities are permitted. --no-super (or a daemon
|
||||
veto that forced SUPER_MODE_OFF) forbids the chown even for root, so do
|
||||
not claim the ownership will be honored in that case. */
|
||||
if (geteuid() == 0) {
|
||||
if (privilege_super_mode_permitted(g_identity.super_mode))
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"identity mapping active and running as root: client-supplied "
|
||||
"ownership (usermap/groupmap/chown/numeric-ids) will be honored; "
|
||||
"run the daemon as an unprivileged user unless intended");
|
||||
else
|
||||
log_message(LOG_LEVEL_WARNING,
|
||||
"identity mapping active and running as root, but super-user activities are "
|
||||
"disabled (--no-super): requested ownership will NOT be applied; run the "
|
||||
"daemon as an unprivileged user unless intended");
|
||||
}
|
||||
/* --super explicitly requests super-user activities, but FastSync never
|
||||
elevates privileges: when the receiver is not already root the kernel will
|
||||
refuse those confined attempts and each is skipped per entry. Warn exactly
|
||||
@@ -128,7 +176,7 @@ bool privilege_super_permitted(void) {
|
||||
return privilege_super_mode_permitted(g_identity.super_mode);
|
||||
}
|
||||
|
||||
bool privilege_super_mode_permitted(int mode) {
|
||||
bool privilege_super_mode_permitted(SuperMode mode) {
|
||||
/* AUTO and ON both attempt the confined operation; OFF forbids it even for a
|
||||
* root receiver. AUTO is the historical FastSync behavior (always attempt
|
||||
* and let the kernel refuse an unprivileged call, which the caller skips), so
|
||||
@@ -138,25 +186,55 @@ bool privilege_super_mode_permitted(int mode) {
|
||||
}
|
||||
|
||||
bool identity_active_enabled(void) {
|
||||
/* numeric_ids is included: this set only gates identity_apply_ownership,
|
||||
which runs only when metadata is present (a -M/--preserve transfer). A
|
||||
standalone --numeric-ids (no ownership-affecting flag) carries no
|
||||
metadata, never reaches identity_apply_ownership, and therefore correctly
|
||||
stays inert; combined with -M it activates raw-id application. --super /
|
||||
--no-super does NOT enable ownership: it only permits or forbids the
|
||||
already-requested super-user activities, so a --super with no explicit
|
||||
identity flag must never silently apply client-chosen ownership. */
|
||||
/* --numeric-ids is deliberately NOT included: it is a mapping MODIFIER (use
|
||||
* the transmitted numeric id raw instead of a name lookup), not a request to
|
||||
* change ownership. rsync's --numeric-ids on its own never chowns anything;
|
||||
* it only changes how an already-requested -o/-g/map resolves. Ownership is
|
||||
* activated only by an explicit request: --chown/--usermap/--groupmap/
|
||||
* --copy-as or a preserve-source -o/--owner / -g/--group. --super/--no-super
|
||||
* likewise does NOT enable ownership: it only permits or forbids the
|
||||
* already-requested super-user activities. */
|
||||
return g_identity.set &&
|
||||
(g_identity.numeric_ids || g_identity.chown_uid_set || g_identity.chown_gid_set ||
|
||||
g_identity.usermap_count > 0 || g_identity.groupmap_count > 0 || g_identity.copy_as_set);
|
||||
(g_identity.chown_uid_set || g_identity.chown_gid_set || g_identity.usermap_count > 0 ||
|
||||
g_identity.groupmap_count > 0 || g_identity.copy_as_set || g_identity.preserve_owner ||
|
||||
g_identity.preserve_group);
|
||||
}
|
||||
|
||||
bool identity_owner_requested(void) {
|
||||
return g_identity.set && (g_identity.copy_as_set || g_identity.chown_uid_set ||
|
||||
g_identity.preserve_owner || g_identity.usermap_count > 0);
|
||||
}
|
||||
|
||||
bool identity_group_requested(void) {
|
||||
return g_identity.set && (g_identity.copy_as_set || g_identity.chown_gid_set ||
|
||||
g_identity.preserve_group || g_identity.groupmap_count > 0);
|
||||
}
|
||||
|
||||
bool identity_ownership_requested(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
/* Every value that makes the receiver act on a client-chosen owner, plus an
|
||||
* explicit --super (super-user device-node activities). Pure config, so the
|
||||
* daemon gate can evaluate it before identity_set_active(). */
|
||||
/* General-awareness predicate: every value that makes the receiver act on a
|
||||
* client-chosen owner, plus an explicit --super (super-user device-node
|
||||
* activities) and the preserve-source -o/-g requests. Pure config, so callers
|
||||
* can evaluate it before identity_set_active(). The daemon module gate uses
|
||||
* the narrower identity_explicit_ownership_requested() below, which treats a
|
||||
* plain -o/-g/-a as a preserve-source request rather than arbitrary
|
||||
* client-chosen ownership. */
|
||||
return config->numeric_ids || config->chown_uid_set || config->chown_gid_set ||
|
||||
config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set ||
|
||||
config->preserve_owner || config->preserve_group || config->fake_super ||
|
||||
config->super_mode == SUPER_MODE_ON;
|
||||
}
|
||||
|
||||
bool identity_explicit_ownership_requested(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
/* The narrow set the daemon gate refuses for a non-opted module: a request
|
||||
* that lets the CLIENT choose an arbitrary owner/group (rather than preserve
|
||||
* the source's own). Deliberately EXCLUDES preserve_owner/preserve_group so a
|
||||
* plain -a/-o/-g push is not refused; for those the gate instead forces
|
||||
* super-user ownership activity off (no chown happens) unless the module has
|
||||
* `client owner = yes`. */
|
||||
return config->numeric_ids || config->chown_uid_set || config->chown_gid_set ||
|
||||
config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set ||
|
||||
config->fake_super || config->super_mode == SUPER_MODE_ON;
|
||||
@@ -177,6 +255,29 @@ bool identity_copy_as_refused(const Config* config) {
|
||||
return geteuid() != 0 || config->super_mode == SUPER_MODE_OFF;
|
||||
}
|
||||
|
||||
/* Validate one received FROM:TO map rule. `from` is a single id, the LOW end
|
||||
* of an inclusive range, IDENTITY_MATCH_ANY, or IDENTITY_MATCH_UNNAMED; a
|
||||
* sentinel FROM must carry the same value in from_hi. `to` is a non-negative
|
||||
* id, IDENTITY_CURRENT, or ignored when a bounded receiver-resolved `to_name`
|
||||
* is present. */
|
||||
static bool identity_wire_map_valid(const IdentityMap* map) {
|
||||
if (!map)
|
||||
return false;
|
||||
if (map->from < IDENTITY_MATCH_UNNAMED)
|
||||
return false;
|
||||
if (map->from < 0) {
|
||||
if (map->from_hi != map->from)
|
||||
return false;
|
||||
} else if (map->from_hi < map->from) {
|
||||
return false;
|
||||
}
|
||||
if (map->to < IDENTITY_CURRENT)
|
||||
return false;
|
||||
if (map->to_name && strlen(map->to_name) > 255)
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool identity_wire_valid(const Config* config) {
|
||||
if (!config)
|
||||
return false;
|
||||
@@ -188,11 +289,11 @@ bool identity_wire_valid(const Config* config) {
|
||||
if (config->chown_gid_set && config->chown_gid < IDENTITY_MATCH_ANY)
|
||||
return false;
|
||||
for (int i = 0; i < config->usermap_count; i++) {
|
||||
if (config->usermap[i].from < IDENTITY_MATCH_ANY || config->usermap[i].to < IDENTITY_CURRENT)
|
||||
if (!identity_wire_map_valid(&config->usermap[i]))
|
||||
return false;
|
||||
}
|
||||
for (int i = 0; i < config->groupmap_count; i++) {
|
||||
if (config->groupmap[i].from < IDENTITY_MATCH_ANY || config->groupmap[i].to < IDENTITY_CURRENT)
|
||||
if (!identity_wire_map_valid(&config->groupmap[i]))
|
||||
return false;
|
||||
}
|
||||
/* Defense-in-depth: a --copy-as block must never carry a negative (sentinel)
|
||||
@@ -250,15 +351,137 @@ static int identity_resolve_token(const char* token, bool is_group, int32_t* out
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int identity_append_rule(IdentityMap** map, int* count, int32_t from, int32_t to) {
|
||||
static bool identity_all_digits(const char* token) {
|
||||
if (!token || *token == '\0')
|
||||
return false;
|
||||
for (const char* p = token; *p; p++)
|
||||
if (*p < '0' || *p > '9')
|
||||
return false;
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool identity_token_has_glob(const char* token) {
|
||||
return token && (strchr(token, '*') || strchr(token, '?') || strchr(token, '['));
|
||||
}
|
||||
|
||||
/* Parse a --usermap/--groupmap FROM token into a matcher (from/from_hi). rsync
|
||||
* accepts a name, a numeric id, an inclusive LOW-HIGH range, '*' (any id), or an
|
||||
* empty token (ids with no name on the sender). Returns 0 on success, -1 on a
|
||||
* malformed token or an unresolvable sender-side name. */
|
||||
static int identity_parse_from(const char* token, bool is_group, int32_t* out_from,
|
||||
int32_t* out_hi) {
|
||||
if (token[0] == '\0') {
|
||||
*out_from = IDENTITY_MATCH_UNNAMED;
|
||||
*out_hi = IDENTITY_MATCH_UNNAMED;
|
||||
return 0;
|
||||
}
|
||||
if (strcmp(token, "*") == 0) {
|
||||
*out_from = IDENTITY_MATCH_ANY;
|
||||
*out_hi = IDENTITY_MATCH_ANY;
|
||||
return 0;
|
||||
}
|
||||
const char* num = token[0] == '@' ? token + 1 : token;
|
||||
if (identity_all_digits(num)) {
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_from = id;
|
||||
*out_hi = id;
|
||||
return 0;
|
||||
}
|
||||
/* An inclusive LOW-HIGH numeric range. */
|
||||
const char* dash = strchr(num, '-');
|
||||
if (dash && dash != num && dash[1] != '\0' && strchr(dash + 1, '-') == NULL) {
|
||||
size_t lo_len = (size_t)(dash - num);
|
||||
size_t hi_len = strlen(dash + 1);
|
||||
char low[16];
|
||||
char high[16];
|
||||
if (lo_len < sizeof(low) && hi_len < sizeof(high)) {
|
||||
memcpy(low, num, lo_len);
|
||||
low[lo_len] = '\0';
|
||||
memcpy(high, dash + 1, hi_len);
|
||||
high[hi_len] = '\0';
|
||||
if (identity_all_digits(low) && identity_all_digits(high)) {
|
||||
char* endptr = NULL;
|
||||
errno = 0;
|
||||
long lo = strtol(low, &endptr, 10);
|
||||
if (errno != 0 || !endptr || *endptr != '\0')
|
||||
return -1;
|
||||
errno = 0;
|
||||
long hi = strtol(high, &endptr, 10);
|
||||
if (errno != 0 || !endptr || *endptr != '\0' || hi < lo || hi > INT32_MAX)
|
||||
return -1;
|
||||
*out_from = (int32_t)lo;
|
||||
*out_hi = (int32_t)hi;
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
/* Not a numeric LOW-HIGH range: fall through and treat as a name (a
|
||||
* hyphenated account name like "wayne-smith" must still resolve). */
|
||||
}
|
||||
/* A sender-side name. A wildcard other than the bare '*' is matched by rsync
|
||||
* against the sender's names; because FastSync transmits numeric ids only, the
|
||||
* receiver cannot evaluate it, so reject rather than silently mis-match. */
|
||||
if (identity_token_has_glob(token)) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%smap FROM '%s': name wildcards other than '*' are not supported "
|
||||
"(FastSync transmits numeric ids, so sender names are unavailable on the "
|
||||
"receiver)",
|
||||
is_group ? "--group" : "--user", token);
|
||||
return -1;
|
||||
}
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_from = id;
|
||||
*out_hi = id;
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Parse a --usermap/--groupmap TO token. '*', a bare numeric id, or an @N id is
|
||||
* stored numerically; every other non-empty token is a NAME resolved on the
|
||||
* RECEIVER at apply time (rsync resolves TO names against the receiving side).
|
||||
* Returns 0 on success, -1 on an empty/malformed token. */
|
||||
static int identity_parse_to(const char* token, bool is_group, int32_t* out_to, char** out_name) {
|
||||
if (token[0] == '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "%smap TO value is missing", is_group ? "--group" : "--user");
|
||||
return -1;
|
||||
}
|
||||
if (strcmp(token, "*") == 0) {
|
||||
*out_to = IDENTITY_CURRENT;
|
||||
*out_name = NULL;
|
||||
return 0;
|
||||
}
|
||||
const char* num = token[0] == '@' ? token + 1 : token;
|
||||
if (identity_all_digits(num)) {
|
||||
int32_t id;
|
||||
if (identity_resolve_token(token, is_group, &id) != 0)
|
||||
return -1;
|
||||
*out_to = id;
|
||||
*out_name = NULL;
|
||||
return 0;
|
||||
}
|
||||
if (identity_token_has_glob(token)) {
|
||||
log_message(LOG_LEVEL_ERROR, "%smap TO '%s' may not contain a wildcard",
|
||||
is_group ? "--group" : "--user", token);
|
||||
return -1;
|
||||
}
|
||||
char* name = str_dup(token);
|
||||
if (!name)
|
||||
return -1;
|
||||
*out_to = 0;
|
||||
*out_name = name;
|
||||
return 0;
|
||||
}
|
||||
|
||||
static int identity_append_rule(IdentityMap** map, int* count, const IdentityMap* rule) {
|
||||
if (*count >= MAX_IDENTITY_MAP)
|
||||
return -1;
|
||||
IdentityMap* grown = realloc(*map, (size_t)(*count + 1) * sizeof(IdentityMap));
|
||||
if (!grown)
|
||||
return -1;
|
||||
*map = grown;
|
||||
(*map)[*count].from = from;
|
||||
(*map)[*count].to = to;
|
||||
(*map)[*count] = *rule;
|
||||
(*count)++;
|
||||
return 0;
|
||||
}
|
||||
@@ -275,7 +498,7 @@ int identity_parse_map(Config* config, const char* value, bool is_group) {
|
||||
char* saveptr = NULL;
|
||||
for (char* rule = strtok_r(list, ",", &saveptr); rule; rule = strtok_r(NULL, ",", &saveptr)) {
|
||||
char* colon = strchr(rule, ':');
|
||||
if (!colon || colon == rule) {
|
||||
if (!colon) {
|
||||
/* Log before freeing: `rule` points into the str_dup'd list. */
|
||||
log_message(LOG_LEVEL_ERROR, "%s rules must be FROM:TO (got '%s')", optname, rule);
|
||||
free(list);
|
||||
@@ -284,25 +507,25 @@ int identity_parse_map(Config* config, const char* value, bool is_group) {
|
||||
*colon = '\0';
|
||||
char* from_token = rule;
|
||||
char* to_token = colon + 1;
|
||||
if (*to_token == '\0') {
|
||||
IdentityMap parsed;
|
||||
memset(&parsed, 0, sizeof(parsed));
|
||||
if (identity_parse_from(from_token, is_group, &parsed.from, &parsed.from_hi) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%s could not resolve FROM '%s' in '%s' (a name must exist on the "
|
||||
"source; use @N for a numeric id)",
|
||||
optname, from_token, value);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR, "%s rule 'FROM:' is missing the TO value (got '%s')", optname,
|
||||
value);
|
||||
return -1;
|
||||
}
|
||||
int32_t from_id, to_id;
|
||||
if (identity_resolve_token(from_token, is_group, &from_id) != 0 ||
|
||||
identity_resolve_token(to_token, is_group, &to_id) != 0) {
|
||||
if (identity_parse_to(to_token, is_group, &parsed.to, &parsed.to_name) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "%s could not parse TO '%s' in '%s'", optname, to_token, value);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"%s could not resolve '%s' (name must exist on the source; use "
|
||||
"@N for a numeric id)",
|
||||
optname, value);
|
||||
return -1;
|
||||
}
|
||||
if (identity_append_rule(is_group ? &config->groupmap : &config->usermap,
|
||||
is_group ? &config->groupmap_count : &config->usermap_count, from_id,
|
||||
to_id) != 0) {
|
||||
is_group ? &config->groupmap_count : &config->usermap_count,
|
||||
&parsed) != 0) {
|
||||
free(parsed.to_name);
|
||||
free(list);
|
||||
log_message(LOG_LEVEL_ERROR, "%s has too many rules (max %d)", optname, MAX_IDENTITY_MAP);
|
||||
return -1;
|
||||
@@ -366,6 +589,67 @@ static int identity_split_chown(const char* value, char** puser, char** pgroup)
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* --chown is rsync's shorthand for "--usermap=*:USER --groupmap=*:GROUP", so a
|
||||
* name TO value must be resolved on the RECEIVER, not on the sender. Append the
|
||||
* equivalent map rule (FROM matches every id). The numeric/'*' forms are stored
|
||||
* numerically exactly as rsync's id_parse/user_to_uid would. Returns 0 on
|
||||
* success, -1 on a malformed numeric token or allocation failure. */
|
||||
static int identity_append_chown_rule(Config* config, bool is_group, const char* token) {
|
||||
IdentityMap rule;
|
||||
memset(&rule, 0, sizeof(rule));
|
||||
rule.from = IDENTITY_MATCH_ANY;
|
||||
rule.from_hi = IDENTITY_MATCH_ANY;
|
||||
if (strcmp(token, "*") == 0) {
|
||||
rule.to = IDENTITY_CURRENT;
|
||||
} else if (identity_all_digits(token[0] == '@' ? token + 1 : token)) {
|
||||
if (identity_resolve_token(token, is_group, &rule.to) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown numeric id is out of range: %s", token);
|
||||
return -1;
|
||||
}
|
||||
} else {
|
||||
rule.to = 0;
|
||||
rule.to_name = str_dup(token);
|
||||
if (!rule.to_name)
|
||||
return -1;
|
||||
}
|
||||
if (identity_append_rule(is_group ? &config->groupmap : &config->usermap,
|
||||
is_group ? &config->groupmap_count : &config->usermap_count,
|
||||
&rule) != 0) {
|
||||
free(rule.to_name);
|
||||
log_message(LOG_LEVEL_ERROR, "--chown has too many rules (max %d)", MAX_IDENTITY_MAP);
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Resolve/record one --chown side. The source-side numeric value is kept in
|
||||
* chown_uid/chown_gid purely as a fallback (the appended map rule resolves the
|
||||
* name on the receiver and wins); a name that does not exist on the sender is
|
||||
* accepted and left to receiver-side resolution, matching rsync. */
|
||||
static int identity_parse_chown_side(Config* config, bool is_group, const char* token) {
|
||||
if (identity_append_chown_rule(config, is_group, token) != 0)
|
||||
return -1;
|
||||
bool numeric = identity_all_digits(token[0] == '@' ? token + 1 : token);
|
||||
int32_t resolved;
|
||||
if (identity_resolve_token(token, is_group, &resolved) == 0) {
|
||||
if (is_group) {
|
||||
config->chown_gid = resolved;
|
||||
config->chown_gid_set = true;
|
||||
} else {
|
||||
config->chown_uid = resolved;
|
||||
config->chown_uid_set = true;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
if (numeric) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown could not resolve numeric id '%s'", token);
|
||||
return -1;
|
||||
}
|
||||
/* Unknown sender-side name: rsync accepts it and resolves it (or warns) on
|
||||
* the receiver; do the same instead of failing the whole run. */
|
||||
return 0;
|
||||
}
|
||||
|
||||
int identity_parse_chown(Config* config, const char* value) {
|
||||
if (!config || !value || *value == '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown requires a value (USER:GROUP, USER, or :GROUP)");
|
||||
@@ -404,33 +688,19 @@ int identity_parse_chown(Config* config, const char* value) {
|
||||
if (*user == '\0') {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown requires a user or group (got '%s')", value);
|
||||
ret = -1;
|
||||
} else if (identity_resolve_token(user, false, &config->chown_uid) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR,
|
||||
"--chown could not resolve user '%s' (use a name that exists "
|
||||
"on the source, '*', or @N)",
|
||||
value);
|
||||
} else if (identity_parse_chown_side(config, false, user) != 0) {
|
||||
ret = -1;
|
||||
} else {
|
||||
config->chown_uid_set = true;
|
||||
}
|
||||
} else {
|
||||
/* --chown=USER:GROUP, --chown=:GROUP, --chown=USER: */
|
||||
if (*user != '\0') {
|
||||
if (identity_resolve_token(user, false, &config->chown_uid) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown could not resolve user '%s'", value);
|
||||
if (*user != '\0' && identity_parse_chown_side(config, false, user) != 0) {
|
||||
ret = -1;
|
||||
goto done;
|
||||
}
|
||||
config->chown_uid_set = true;
|
||||
}
|
||||
if (*group != '\0') {
|
||||
if (identity_resolve_token(group, true, &config->chown_gid) != 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown could not resolve group '%s'", value);
|
||||
if (*group != '\0' && identity_parse_chown_side(config, true, group) != 0) {
|
||||
ret = -1;
|
||||
goto done;
|
||||
}
|
||||
config->chown_gid_set = true;
|
||||
}
|
||||
if (!*user && !*group) {
|
||||
log_message(LOG_LEVEL_ERROR, "--chown must set a user, a group, or both (got '%s')", value);
|
||||
ret = -1;
|
||||
@@ -567,9 +837,6 @@ int identity_parse_copy_as(Config* config, const char* value) {
|
||||
config->copy_as_set = true;
|
||||
config->copy_as_uid = uid;
|
||||
config->copy_as_gid = gid;
|
||||
/* Ownership application needs the metadata path (the source uid/gid must be
|
||||
* transmitted); imply it exactly like --chown/--usermap/--groupmap. */
|
||||
config->use_metadata = true;
|
||||
ret = 0;
|
||||
|
||||
done:
|
||||
@@ -580,34 +847,120 @@ done:
|
||||
|
||||
/* ---- Receiver-side ownership application ---- */
|
||||
|
||||
static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id,
|
||||
/* True when a map rule's FROM matcher accepts `id`. A sentinel FROM never
|
||||
* carries a range. IDENTITY_MATCH_UNNAMED mirrors rsync's empty FROM: it
|
||||
* matches only ids that have no name in the account database (rsync matches the
|
||||
* sender's names; FastSync transmits numeric ids only, so it approximates this
|
||||
* with the receiver's database -- documented in RSYNC_COMPAT.md). */
|
||||
static bool identity_map_from_matches(const IdentityMap* map, int32_t id, bool is_group) {
|
||||
if (map->from == IDENTITY_MATCH_ANY)
|
||||
return true;
|
||||
if (map->from == IDENTITY_MATCH_UNNAMED)
|
||||
return is_group ? (getgrgid((gid_t)id) == NULL) : (getpwuid((uid_t)id) == NULL);
|
||||
return id >= map->from && id <= map->from_hi;
|
||||
}
|
||||
|
||||
/* First matching rule wins. A rule whose TO is a receiver-side name resolves it
|
||||
* against the receiver's account database here; an unresolvable TO name is
|
||||
* skipped with a warning and the next rule is considered (rsync prints "Unknown
|
||||
* --usermap name on receiver" and leaves the id unmapped rather than aborting). */
|
||||
static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id, bool is_group,
|
||||
int32_t* out_to) {
|
||||
for (int i = 0; i < count; i++) {
|
||||
if (map[i].from == IDENTITY_MATCH_ANY || map[i].from == source_id) {
|
||||
if (!identity_map_from_matches(&map[i], source_id, is_group))
|
||||
continue;
|
||||
if (map[i].to_name) {
|
||||
if (is_group) {
|
||||
struct group* gr = getgrnam(map[i].to_name);
|
||||
if (!gr) {
|
||||
log_message(LOG_LEVEL_WARNING, "Unknown --groupmap name on receiver: %s", map[i].to_name);
|
||||
continue;
|
||||
}
|
||||
*out_to = (int32_t)gr->gr_gid;
|
||||
} else {
|
||||
struct passwd* pw = getpwnam(map[i].to_name);
|
||||
if (!pw) {
|
||||
log_message(LOG_LEVEL_WARNING, "Unknown --usermap name on receiver: %s", map[i].to_name);
|
||||
continue;
|
||||
}
|
||||
*out_to = (int32_t)pw->pw_uid;
|
||||
}
|
||||
} else {
|
||||
*out_to = map[i].to;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/* Resolve the owner side from the negotiated policy. Sets *out and returns
|
||||
* true when an owner-affecting request is active (a usermap, --chown USER, or
|
||||
* -o/--owner); returns false (leaving *out untouched) when the owner side is
|
||||
* not requested, so callers can pass (uid_t)-1 to fchown and leave it as-is.
|
||||
* --numeric-ids only changes the RESOLUTION (raw id instead of a name lookup);
|
||||
* it never makes the side requested. */
|
||||
static bool identity_resolve_owner(int32_t source_uid, uid_t* out) {
|
||||
if (!(g_identity.chown_uid_set || g_identity.preserve_owner || g_identity.usermap_count > 0))
|
||||
return false;
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, false,
|
||||
&target)) {
|
||||
*out = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target;
|
||||
} else if (g_identity.chown_uid_set) {
|
||||
*out = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
*out = (uid_t)source_uid;
|
||||
} else {
|
||||
/* Best-effort name mapping against the receiver's own database. When the
|
||||
* transmitted (numeric) id has no name here, fall back to the raw numeric id
|
||||
* so -o still preserves the source owner. */
|
||||
struct passwd* pw = getpwuid((uid_t)source_uid);
|
||||
if (pw) {
|
||||
const struct passwd* mapped = getpwnam(pw->pw_name);
|
||||
*out = mapped ? mapped->pw_uid : (uid_t)source_uid;
|
||||
} else {
|
||||
*out = (uid_t)source_uid;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Group-side counterpart of identity_resolve_owner(). */
|
||||
static bool identity_resolve_group(int32_t source_gid, gid_t* out) {
|
||||
if (!(g_identity.chown_gid_set || g_identity.preserve_group || g_identity.groupmap_count > 0))
|
||||
return false;
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, true,
|
||||
&target)) {
|
||||
*out = target == IDENTITY_CURRENT ? getegid() : (gid_t)target;
|
||||
} else if (g_identity.chown_gid_set) {
|
||||
*out = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
*out = (gid_t)source_gid;
|
||||
} else {
|
||||
struct group* gr = getgrgid((gid_t)source_gid);
|
||||
if (gr) {
|
||||
const struct group* mapped = getgrnam(gr->gr_name);
|
||||
*out = mapped ? mapped->gr_gid : (gid_t)source_gid;
|
||||
} else {
|
||||
*out = (gid_t)source_gid;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Resolve the target ownership from the negotiated policy against the entry's
|
||||
* current stat. Shared by the fd (regular file) and no-follow (symlink) apply
|
||||
* paths. Returns false when no side is to be changed. */
|
||||
static bool identity_resolve_targets(const struct stat* st, int32_t source_uid, int32_t source_gid,
|
||||
uid_t* out_uid, gid_t* out_gid) {
|
||||
bool set_uid = false;
|
||||
bool set_gid = false;
|
||||
uid_t uid = 0;
|
||||
gid_t gid = 0;
|
||||
|
||||
/* --copy-as (P7 Wave E) has the highest priority: it forces BOTH the owner
|
||||
* and group of every written entry to the requested ids, beating usermap /
|
||||
* groupmap / --chown / --numeric-ids and the best-effort name lookup. Only
|
||||
* skip when the entry already carries exactly those ids. */
|
||||
if (g_identity.copy_as_set) {
|
||||
uid = (uid_t)g_identity.copy_as_uid;
|
||||
gid = (gid_t)g_identity.copy_as_gid;
|
||||
uid_t uid = (uid_t)g_identity.copy_as_uid;
|
||||
gid_t gid = (gid_t)g_identity.copy_as_gid;
|
||||
if (st->st_uid == uid && st->st_gid == gid)
|
||||
return false;
|
||||
*out_uid = uid;
|
||||
@@ -615,67 +968,52 @@ static bool identity_resolve_targets(const struct stat* st, int32_t source_uid,
|
||||
return true;
|
||||
}
|
||||
|
||||
int32_t target;
|
||||
if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, &target)) {
|
||||
uid = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target;
|
||||
set_uid = true;
|
||||
} else if (g_identity.chown_uid_set) {
|
||||
uid = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid;
|
||||
set_uid = true;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
uid = (uid_t)source_uid;
|
||||
set_uid = true;
|
||||
} else {
|
||||
/* Best-effort name mapping against the receiver's own database: if the
|
||||
* transmitted (numeric) id resolves to a name present on this machine,
|
||||
* re-resolve it. On a shared-account host this is the identity operation;
|
||||
* when the id has no name here, the user side is left alone. */
|
||||
struct passwd* pw = getpwuid((uid_t)source_uid);
|
||||
if (pw) {
|
||||
const struct passwd* mapped = getpwnam(pw->pw_name);
|
||||
if (mapped) {
|
||||
uid = mapped->pw_uid;
|
||||
set_uid = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, &target)) {
|
||||
gid = target == IDENTITY_CURRENT ? getegid() : (gid_t)target;
|
||||
set_gid = true;
|
||||
} else if (g_identity.chown_gid_set) {
|
||||
gid = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid;
|
||||
set_gid = true;
|
||||
} else if (g_identity.numeric_ids) {
|
||||
gid = (gid_t)source_gid;
|
||||
set_gid = true;
|
||||
} else {
|
||||
struct group* gr = getgrgid((gid_t)source_gid);
|
||||
if (gr) {
|
||||
const struct group* mapped = getgrnam(gr->gr_name);
|
||||
if (mapped) {
|
||||
gid = mapped->gr_gid;
|
||||
set_gid = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (!set_uid && !set_gid)
|
||||
/* Each side is resolved independently: -o/-g and the explicit identity flags
|
||||
* request the owner/group respectively, and a side that is NOT requested must
|
||||
* be left exactly as it is (`-1` to fchown on that side). This is what lets
|
||||
* plain -g change only the group, or -o only the owner. */
|
||||
uid_t uid = (uid_t)-1;
|
||||
gid_t gid = (gid_t)-1;
|
||||
bool owner_requested = identity_resolve_owner(source_uid, &uid);
|
||||
bool group_requested = identity_resolve_group(source_gid, &gid);
|
||||
if (!owner_requested && !group_requested)
|
||||
return false;
|
||||
/* An unset side keeps the file's current id so the other side can change. */
|
||||
if (!set_uid)
|
||||
uid = st->st_uid;
|
||||
if (!set_gid)
|
||||
gid = st->st_gid;
|
||||
/* Only change ownership when the target differs (avoid needless syscalls and
|
||||
* any chance of clearing setuid/setgid on an already-correct entry). */
|
||||
if (st->st_uid == uid && st->st_gid == gid)
|
||||
|
||||
/* Only change ownership when a requested side actually differs (avoid
|
||||
* needless syscalls and any chance of clearing setuid/setgid on an
|
||||
* already-correct entry). */
|
||||
bool changed = (owner_requested && uid != st->st_uid) || (group_requested && gid != st->st_gid);
|
||||
if (!changed)
|
||||
return false;
|
||||
*out_uid = uid;
|
||||
*out_gid = gid;
|
||||
return true;
|
||||
}
|
||||
|
||||
/* --fake-super storage resolution: the receiver records the ownership it WOULD
|
||||
* have applied. A requested side uses the resolved mapping (--copy-as /
|
||||
* usermap / --chown / -o/-g, with --numeric-ids as the raw-id modifier); a side
|
||||
* that was not requested keeps the source's own id, so a plain --fake-super run
|
||||
* records the source owner untouched. */
|
||||
void identity_resolve_storage_ids(int32_t source_uid, int32_t source_gid, uint32_t* out_uid,
|
||||
uint32_t* out_gid) {
|
||||
if (g_identity.copy_as_set) {
|
||||
*out_uid = (uint32_t)g_identity.copy_as_uid;
|
||||
*out_gid = (uint32_t)g_identity.copy_as_gid;
|
||||
return;
|
||||
}
|
||||
uid_t uid = (uid_t)source_uid;
|
||||
gid_t gid = (gid_t)source_gid;
|
||||
uid_t resolved_uid;
|
||||
gid_t resolved_gid;
|
||||
if (identity_resolve_owner(source_uid, &resolved_uid))
|
||||
uid = resolved_uid;
|
||||
if (identity_resolve_group(source_gid, &resolved_gid))
|
||||
gid = resolved_gid;
|
||||
*out_uid = (uint32_t)uid;
|
||||
*out_gid = (uint32_t)gid;
|
||||
}
|
||||
|
||||
static void identity_log_chown_failure(const char* what, uid_t uid, gid_t gid) {
|
||||
/* EPERM/EACCES are expected when the receiver is not privileged (e.g. the CI
|
||||
* `nobody` user): warn and continue, never abort the transfer. Any other
|
||||
@@ -710,8 +1048,12 @@ bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) {
|
||||
/* Ownership application is OFF unless the client requested an identity flag.
|
||||
* This is the controlled gate: a default (or plain -M) transfer never changes
|
||||
* ownership, byte-for-byte preserving FastSync's existing behavior. --no-super
|
||||
* additionally forbids it even when the receiver is root. */
|
||||
if (!identity_active_enabled() || !privilege_super_permitted() || fd < 0)
|
||||
* additionally forbids it even when the receiver is root. --fake-super never
|
||||
* performs a REAL chown: that would defeat the point of the flag (record the
|
||||
* source ownership on an unprivileged receiver for a later privileged
|
||||
* restore); the resolved ownership is stored in the reserved xattr instead by
|
||||
* fake_super_store_fd(). */
|
||||
if (!identity_active_enabled() || g_identity.fake_super || !privilege_super_permitted() || fd < 0)
|
||||
return true;
|
||||
struct stat st;
|
||||
if (fstat(fd, &st) != 0)
|
||||
@@ -731,7 +1073,8 @@ bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) {
|
||||
|
||||
bool identity_apply_ownership_link(int parent_fd, const char* leaf, int32_t source_uid,
|
||||
int32_t source_gid) {
|
||||
if (!identity_active_enabled() || !privilege_super_permitted() || parent_fd < 0 || !leaf)
|
||||
if (!identity_active_enabled() || g_identity.fake_super || !privilege_super_permitted() ||
|
||||
parent_fd < 0 || !leaf)
|
||||
return true;
|
||||
struct stat st;
|
||||
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0)
|
||||
|
||||
+36
-7
@@ -80,16 +80,37 @@ void identity_clear_active(void);
|
||||
* snapshot. Ownership stays OFF ("do not apply") for every transfer that
|
||||
* requests none of them, preserving FastSync's existing behavior. --super /
|
||||
* --no-super alone does NOT enable ownership; an explicit identity flag
|
||||
* (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) is required. */
|
||||
* (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) or a
|
||||
* preserve-source -o/--owner / -g/--group request is required. */
|
||||
bool identity_active_enabled(void);
|
||||
|
||||
/* Pure, config-only predicate: true when the client requested ANY
|
||||
* client-chosen ownership or super-user activity (--numeric-ids, --chown,
|
||||
* --usermap/--groupmap, --copy-as, --fake-super, or an explicit --super). Used
|
||||
* by the daemon module gate to decide whether a module's per-module opt-in is
|
||||
* required; it never reads the per-connection snapshot. */
|
||||
/* Per-side predicates over the ACTIVE per-connection snapshot (call
|
||||
* identity_set_active() first). They mirror the owner_requested /
|
||||
* group_requested conditions inside identity_resolve_targets() exactly, so
|
||||
* callers that must apply only one side (e.g. the --fake-super owner replay)
|
||||
* can pass (uid_t)-1 / (gid_t)-1 for the side that was NOT requested and leave
|
||||
* it untouched. The owner side is requested by --copy-as, --chown USER,
|
||||
* --numeric-ids, -o/--owner, or a non-empty --usermap; the group side by
|
||||
* --copy-as, --chown :GROUP, --numeric-ids, -g/--group, or a non-empty
|
||||
* --groupmap. */
|
||||
bool identity_owner_requested(void);
|
||||
bool identity_group_requested(void);
|
||||
|
||||
/* Pure, config-only predicate: true when the client requested ANY client-chosen
|
||||
* ownership or super-user activity (--numeric-ids, --chown, --usermap/--groupmap,
|
||||
* --copy-as, --fake-super, an explicit --super, or a preserve-source -o/-g).
|
||||
* General awareness only; the daemon module gate uses the narrower
|
||||
* identity_explicit_ownership_requested() below. Never reads the snapshot. */
|
||||
bool identity_ownership_requested(const Config* config);
|
||||
|
||||
/* Pure, config-only predicate for the narrow set that lets the CLIENT choose an
|
||||
* arbitrary owner/group: --numeric-ids, --chown, --usermap/--groupmap,
|
||||
* --copy-as, --fake-super, or an explicit --super. Deliberately EXCLUDES a
|
||||
* plain -o/--owner / -g/--group (or -a) preserve-source request, which the
|
||||
* daemon gate handles by forcing super-user ownership activity off rather than
|
||||
* refusing the whole transfer. Never reads the snapshot. */
|
||||
bool identity_explicit_ownership_requested(const Config* config);
|
||||
|
||||
/* Apply the negotiated ownership to an already-written file descriptor.
|
||||
* source_uid/source_gid are the transmitted numeric ids. Resolution order:
|
||||
* --copy-as (highest priority, forces both ids), then a matching
|
||||
@@ -106,6 +127,14 @@ bool identity_ownership_requested(const Config* config);
|
||||
* is active returns true. */
|
||||
bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid);
|
||||
|
||||
/* Resolve the ownership that --fake-super should RECORD in the reserved xattr
|
||||
* (rather than chown for real). A requested side (--copy-as / usermap /
|
||||
* --chown / -o / -g, with --numeric-ids as the raw-id modifier) yields the
|
||||
* resolved target; a side that was not requested keeps the transmitted source
|
||||
* id. Must be called after identity_set_active(). */
|
||||
void identity_resolve_storage_ids(int32_t source_uid, int32_t source_gid, uint32_t* out_uid,
|
||||
uint32_t* out_gid);
|
||||
|
||||
/* P7 Wave D: the no-follow (symlink) counterpart. Resolves the same
|
||||
* usermap/groupmap/chown/numeric-ids/copy-as policy but applies it with
|
||||
* fchownat(..., AT_SYMLINK_NOFOLLOW) so a symlink's own ownership is changed
|
||||
@@ -128,6 +157,6 @@ bool identity_wire_valid(const Config* config);
|
||||
* best-effort behavior where an unprivileged attempt is refused by the kernel
|
||||
* and skipped. Neither EVER elevates privileges. */
|
||||
bool privilege_super_permitted(void);
|
||||
bool privilege_super_mode_permitted(int mode);
|
||||
bool privilege_super_mode_permitted(SuperMode mode);
|
||||
|
||||
#endif
|
||||
+72
-26
@@ -3,7 +3,9 @@
|
||||
#include <stdbool.h>
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <threads.h>
|
||||
#include <time.h>
|
||||
|
||||
static const char* log_level_strings[] = {"DEBUG", "INFO", "WARN", "ERROR"};
|
||||
@@ -15,6 +17,18 @@ static FILE* log_fp = NULL;
|
||||
static _Thread_local bool eight_bit_output;
|
||||
static LogStderrMode stderr_mode = LOG_STDERR_ERRORS;
|
||||
|
||||
/* Serializes access to log_fp and makes each emitted line atomic: the
|
||||
* timestamp prefix, formatted body, and trailing newline are written as one
|
||||
* critical section so concurrent threads cannot interleave partial lines.
|
||||
* Initialized lazily (matching the protocol.c bw_mutex idiom) because logging
|
||||
* can happen before main() installs any synchronization. */
|
||||
static mtx_t log_mutex;
|
||||
static once_flag log_mutex_once = ONCE_FLAG_INIT;
|
||||
|
||||
static void log_mutex_init(void) {
|
||||
mtx_init(&log_mutex, mtx_plain);
|
||||
}
|
||||
|
||||
void set_log_level(LogLevel level) {
|
||||
current_log_level = level;
|
||||
}
|
||||
@@ -41,7 +55,10 @@ uint32_t get_log_info_flags(void) {
|
||||
}
|
||||
|
||||
void log_set_file(FILE* fp) {
|
||||
call_once(&log_mutex_once, log_mutex_init);
|
||||
mtx_lock(&log_mutex);
|
||||
log_fp = fp;
|
||||
mtx_unlock(&log_mutex);
|
||||
}
|
||||
|
||||
void log_set_8_bit_output(bool enabled) {
|
||||
@@ -60,13 +77,48 @@ LogStderrMode log_get_stderr_mode(void) {
|
||||
return stderr_mode;
|
||||
}
|
||||
|
||||
static inline void write_message(FILE* dest_io, LogLevel log_level, struct tm t, const char* format,
|
||||
/* Format one complete log line (timestamp prefix + body + newline) into a
|
||||
* freshly allocated buffer. This is pure CPU/malloc work and must happen
|
||||
* OUTSIDE the log mutex: the mutex only guards the log_fp pointer, so a
|
||||
* stalled stderr/stdout pipe cannot block every logging thread. Returns NULL
|
||||
* on allocation/formatting failure. */
|
||||
static char* format_log_line(LogLevel log_level, const struct tm* t, const char* format,
|
||||
va_list args) {
|
||||
fprintf(dest_io, "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t.tm_year + 1900, t.tm_mon + 1,
|
||||
t.tm_mday, t.tm_hour, t.tm_min, t.tm_sec, log_level_strings[log_level]);
|
||||
char prefix[64];
|
||||
int prefix_len = snprintf(
|
||||
prefix, sizeof(prefix), "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t->tm_year + 1900,
|
||||
t->tm_mon + 1, t->tm_mday, t->tm_hour, t->tm_min, t->tm_sec, log_level_strings[log_level]);
|
||||
if (prefix_len < 0 || prefix_len >= (int)sizeof(prefix))
|
||||
return NULL;
|
||||
va_list copy;
|
||||
va_copy(copy, args);
|
||||
int body_len = vsnprintf(NULL, 0, format, copy);
|
||||
va_end(copy);
|
||||
if (body_len < 0)
|
||||
return NULL;
|
||||
size_t total = (size_t)prefix_len + (size_t)body_len;
|
||||
char* line = malloc(total + 2); /* body bytes + '\n' + NUL */
|
||||
if (!line)
|
||||
return NULL;
|
||||
memcpy(line, prefix, (size_t)prefix_len);
|
||||
vsnprintf(line + prefix_len, (size_t)body_len + 1, format, args);
|
||||
line[total] = '\n';
|
||||
line[total + 1] = '\0';
|
||||
return line;
|
||||
}
|
||||
|
||||
vfprintf(dest_io, format, args);
|
||||
fprintf(dest_io, "\n");
|
||||
/* Write an already-formatted line to the console and, if configured, the log
|
||||
* file. Only the log_fp pointer is read under the mutex (so log_set_file /
|
||||
* config_delete cannot free it while it is in use); the single console fputs
|
||||
* runs unlocked but is internally atomic per stdio stream. */
|
||||
static void emit_log_line(FILE* console, const char* line) {
|
||||
fputs(line, console);
|
||||
call_once(&log_mutex_once, log_mutex_init);
|
||||
mtx_lock(&log_mutex);
|
||||
FILE* file = log_fp;
|
||||
if (file)
|
||||
fputs(line, file);
|
||||
mtx_unlock(&log_mutex);
|
||||
}
|
||||
|
||||
void log_message(LogLevel log_level, const char* format, ...) {
|
||||
@@ -86,14 +138,12 @@ void log_message(LogLevel log_level, const char* format, ...) {
|
||||
|
||||
va_list args;
|
||||
va_start(args, format);
|
||||
write_message(dest_io, log_level, t, format, args);
|
||||
char* line = format_log_line(log_level, &t, format, args);
|
||||
va_end(args);
|
||||
|
||||
if (log_fp) {
|
||||
va_start(args, format);
|
||||
write_message(log_fp, log_level, t, format, args);
|
||||
va_end(args);
|
||||
}
|
||||
if (!line)
|
||||
return;
|
||||
emit_log_line(dest_io, line);
|
||||
free(line);
|
||||
}
|
||||
|
||||
void log_debug_message(LogDebugFlag flag, const char* format, ...) {
|
||||
@@ -107,14 +157,12 @@ void log_debug_message(LogDebugFlag flag, const char* format, ...) {
|
||||
|
||||
va_list args;
|
||||
va_start(args, format);
|
||||
write_message(stdout, LOG_LEVEL_DEBUG, t, format, args);
|
||||
char* line = format_log_line(LOG_LEVEL_DEBUG, &t, format, args);
|
||||
va_end(args);
|
||||
|
||||
if (log_fp) {
|
||||
va_start(args, format);
|
||||
write_message(log_fp, LOG_LEVEL_DEBUG, t, format, args);
|
||||
va_end(args);
|
||||
}
|
||||
if (!line)
|
||||
return;
|
||||
emit_log_line(stdout, line);
|
||||
free(line);
|
||||
}
|
||||
|
||||
void log_info_message(LogInfoFlag flag, const char* format, ...) {
|
||||
@@ -129,14 +177,12 @@ void log_info_message(LogInfoFlag flag, const char* format, ...) {
|
||||
|
||||
va_list args;
|
||||
va_start(args, format);
|
||||
write_message(stdout, LOG_LEVEL_INFO, t, format, args);
|
||||
char* line = format_log_line(LOG_LEVEL_INFO, &t, format, args);
|
||||
va_end(args);
|
||||
|
||||
if (log_fp) {
|
||||
va_start(args, format);
|
||||
write_message(log_fp, LOG_LEVEL_INFO, t, format, args);
|
||||
va_end(args);
|
||||
}
|
||||
if (!line)
|
||||
return;
|
||||
emit_log_line(stdout, line);
|
||||
free(line);
|
||||
}
|
||||
|
||||
void log_perror(const char* context) {
|
||||
|
||||
+149
-194
@@ -90,63 +90,65 @@ void metadata_to_buf(char** buf, const FileMetadata* m) {
|
||||
*buf += sizeof(crtime_nsec);
|
||||
}
|
||||
|
||||
FileMetadata* metadata_from_buf(char** buf) {
|
||||
FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len) {
|
||||
if (buf == NULL || len < sizeof(int32_t))
|
||||
return NULL;
|
||||
int32_t present;
|
||||
memcpy(&present, *buf, sizeof(present));
|
||||
*buf += sizeof(present);
|
||||
if (present != 0 && present != 1)
|
||||
memcpy(&present, buf, sizeof(present));
|
||||
if (present != 1)
|
||||
return NULL;
|
||||
if (!present)
|
||||
if (len < sizeof(int32_t) + FILE_METADATA_WIRE_SIZE)
|
||||
return NULL;
|
||||
const uint8_t* cursor = buf + sizeof(int32_t);
|
||||
FileMetadata* m = protocol_alloc(sizeof(FileMetadata));
|
||||
if (m == NULL)
|
||||
return NULL;
|
||||
int32_t mode;
|
||||
memcpy(&mode, *buf, sizeof(mode));
|
||||
*buf += sizeof(mode);
|
||||
memcpy(&mode, cursor, sizeof(mode));
|
||||
cursor += sizeof(mode);
|
||||
m->mode = (mode_t)mode;
|
||||
int32_t uid;
|
||||
memcpy(&uid, *buf, sizeof(uid));
|
||||
*buf += sizeof(uid);
|
||||
memcpy(&uid, cursor, sizeof(uid));
|
||||
cursor += sizeof(uid);
|
||||
m->uid = (uid_t)uid;
|
||||
int32_t gid;
|
||||
memcpy(&gid, *buf, sizeof(gid));
|
||||
*buf += sizeof(gid);
|
||||
memcpy(&gid, cursor, sizeof(gid));
|
||||
cursor += sizeof(gid);
|
||||
m->gid = (gid_t)gid;
|
||||
int64_t mtime_sec;
|
||||
memcpy(&mtime_sec, *buf, sizeof(mtime_sec));
|
||||
*buf += sizeof(mtime_sec);
|
||||
memcpy(&mtime_sec, cursor, sizeof(mtime_sec));
|
||||
cursor += sizeof(mtime_sec);
|
||||
m->mtime_sec = (time_t)mtime_sec;
|
||||
int64_t mtime_nsec;
|
||||
memcpy(&mtime_nsec, *buf, sizeof(mtime_nsec));
|
||||
*buf += sizeof(mtime_nsec);
|
||||
memcpy(&mtime_nsec, cursor, sizeof(mtime_nsec));
|
||||
cursor += sizeof(mtime_nsec);
|
||||
m->mtime_nsec = (long)mtime_nsec;
|
||||
int32_t atime_valid;
|
||||
memcpy(&atime_valid, *buf, sizeof(atime_valid));
|
||||
*buf += sizeof(atime_valid);
|
||||
memcpy(&atime_valid, cursor, sizeof(atime_valid));
|
||||
cursor += sizeof(atime_valid);
|
||||
int64_t atime_sec;
|
||||
memcpy(&atime_sec, *buf, sizeof(atime_sec));
|
||||
*buf += sizeof(atime_sec);
|
||||
memcpy(&atime_sec, cursor, sizeof(atime_sec));
|
||||
cursor += sizeof(atime_sec);
|
||||
int64_t atime_nsec;
|
||||
memcpy(&atime_nsec, *buf, sizeof(atime_nsec));
|
||||
*buf += sizeof(atime_nsec);
|
||||
memcpy(&atime_nsec, cursor, sizeof(atime_nsec));
|
||||
cursor += sizeof(atime_nsec);
|
||||
int32_t crtime_valid;
|
||||
memcpy(&crtime_valid, *buf, sizeof(crtime_valid));
|
||||
*buf += sizeof(crtime_valid);
|
||||
memcpy(&crtime_valid, cursor, sizeof(crtime_valid));
|
||||
cursor += sizeof(crtime_valid);
|
||||
int64_t crtime_sec;
|
||||
memcpy(&crtime_sec, *buf, sizeof(crtime_sec));
|
||||
*buf += sizeof(crtime_sec);
|
||||
memcpy(&crtime_sec, cursor, sizeof(crtime_sec));
|
||||
cursor += sizeof(crtime_sec);
|
||||
int64_t crtime_nsec;
|
||||
memcpy(&crtime_nsec, *buf, sizeof(crtime_nsec));
|
||||
*buf += sizeof(crtime_nsec);
|
||||
memcpy(&crtime_nsec, cursor, sizeof(crtime_nsec));
|
||||
cursor += sizeof(crtime_nsec);
|
||||
m->atime_valid = atime_valid != 0;
|
||||
m->atime_sec = (time_t)atime_sec;
|
||||
m->atime_nsec = (long)atime_nsec;
|
||||
m->crtime_valid = crtime_valid != 0;
|
||||
m->crtime_sec = (time_t)crtime_sec;
|
||||
m->crtime_nsec = (long)crtime_nsec;
|
||||
if (present != 1 || mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 ||
|
||||
gid < 0 || atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 ||
|
||||
if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 ||
|
||||
atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 ||
|
||||
(atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) ||
|
||||
(crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) {
|
||||
free(m);
|
||||
@@ -157,33 +159,17 @@ FileMetadata* metadata_from_buf(char** buf) {
|
||||
|
||||
bool metadata_send(int file_descriptor, const FileMetadata* m) {
|
||||
if (m == NULL) {
|
||||
int32_t zero = 0;
|
||||
return send_n_data(file_descriptor, &zero, sizeof(zero));
|
||||
int32_t absent = 0;
|
||||
return send_n_data(file_descriptor, &absent, sizeof(absent));
|
||||
}
|
||||
int32_t present = 1;
|
||||
int32_t mode = (int32_t)m->mode;
|
||||
int32_t uid = (int32_t)m->uid;
|
||||
int32_t gid = (int32_t)m->gid;
|
||||
int64_t mtime_sec = (int64_t)m->mtime_sec;
|
||||
int64_t mtime_nsec = (int64_t)m->mtime_nsec;
|
||||
int32_t atime_valid = m->atime_valid ? 1 : 0;
|
||||
int64_t atime_sec = (int64_t)m->atime_sec;
|
||||
int64_t atime_nsec = (int64_t)m->atime_nsec;
|
||||
int32_t crtime_valid = m->crtime_valid ? 1 : 0;
|
||||
int64_t crtime_sec = (int64_t)m->crtime_sec;
|
||||
int64_t crtime_nsec = (int64_t)m->crtime_nsec;
|
||||
return send_n_data(file_descriptor, &present, sizeof(present)) &&
|
||||
send_n_data(file_descriptor, &mode, sizeof(mode)) &&
|
||||
send_n_data(file_descriptor, &uid, sizeof(uid)) &&
|
||||
send_n_data(file_descriptor, &gid, sizeof(gid)) &&
|
||||
send_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec)) &&
|
||||
send_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec)) &&
|
||||
send_n_data(file_descriptor, &atime_valid, sizeof(atime_valid)) &&
|
||||
send_n_data(file_descriptor, &atime_sec, sizeof(atime_sec)) &&
|
||||
send_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec)) &&
|
||||
send_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid)) &&
|
||||
send_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec)) &&
|
||||
send_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec));
|
||||
/* One packed frame (protocol 2.20.0): the int32 present flag followed by the
|
||||
fixed FILE_METADATA_WIRE_SIZE-byte field record. metadata_to_buf() emits
|
||||
exactly that layout (present + fields), so build it once and write the
|
||||
whole record in a single call instead of one frame per field. */
|
||||
char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE];
|
||||
char* cursor = packed;
|
||||
metadata_to_buf(&cursor, m);
|
||||
return send_n_data(file_descriptor, packed, sizeof(packed));
|
||||
}
|
||||
|
||||
FileMetadata* metadata_receive(int file_descriptor, int* ok) {
|
||||
@@ -203,130 +189,78 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) {
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
FileMetadata* m = protocol_alloc(sizeof(FileMetadata));
|
||||
/* Rebuild the packed record metadata_from_buf() expects: the present flag we
|
||||
just read, followed by exactly FILE_METADATA_WIRE_SIZE field bytes. */
|
||||
char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE];
|
||||
memcpy(packed, &present, sizeof(present));
|
||||
if (!receive_n_data(file_descriptor, packed + sizeof(present), FILE_METADATA_WIRE_SIZE)) {
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
FileMetadata* m = metadata_from_buf((const uint8_t*)packed, sizeof(packed));
|
||||
if (m == NULL) {
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int32_t mode;
|
||||
if (!receive_n_data(file_descriptor, &mode, sizeof(mode))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->mode = (mode_t)mode;
|
||||
int32_t uid;
|
||||
if (!receive_n_data(file_descriptor, &uid, sizeof(uid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->uid = (uid_t)uid;
|
||||
int32_t gid;
|
||||
if (!receive_n_data(file_descriptor, &gid, sizeof(gid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->gid = (gid_t)gid;
|
||||
int64_t mtime_sec;
|
||||
if (!receive_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->mtime_sec = (time_t)mtime_sec;
|
||||
int64_t mtime_nsec;
|
||||
if (!receive_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->mtime_nsec = (long)mtime_nsec;
|
||||
int32_t atime_valid;
|
||||
if (!receive_n_data(file_descriptor, &atime_valid, sizeof(atime_valid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t atime_sec;
|
||||
if (!receive_n_data(file_descriptor, &atime_sec, sizeof(atime_sec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t atime_nsec;
|
||||
if (!receive_n_data(file_descriptor, &atime_nsec, sizeof(atime_nsec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int32_t crtime_valid;
|
||||
if (!receive_n_data(file_descriptor, &crtime_valid, sizeof(crtime_valid))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t crtime_sec;
|
||||
if (!receive_n_data(file_descriptor, &crtime_sec, sizeof(crtime_sec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
int64_t crtime_nsec;
|
||||
if (!receive_n_data(file_descriptor, &crtime_nsec, sizeof(crtime_nsec))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
m->atime_valid = atime_valid != 0;
|
||||
m->atime_sec = (time_t)atime_sec;
|
||||
m->atime_nsec = (long)atime_nsec;
|
||||
m->crtime_valid = crtime_valid != 0;
|
||||
m->crtime_sec = (time_t)crtime_sec;
|
||||
m->crtime_nsec = (long)crtime_nsec;
|
||||
if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 ||
|
||||
atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 ||
|
||||
(atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) ||
|
||||
(crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) {
|
||||
free(m);
|
||||
if (ok)
|
||||
*ok = 0;
|
||||
return NULL;
|
||||
}
|
||||
if (ok)
|
||||
*ok = 1;
|
||||
return m;
|
||||
}
|
||||
|
||||
static mode_t metadata_mode(const FileMetadata* metadata, mode_t current_mode,
|
||||
bool preserve_executability) {
|
||||
bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrPolicy policy,
|
||||
mode_t* out_mode) {
|
||||
const mode_t execute_bits = S_IXUSR | S_IXGRP | S_IXOTH;
|
||||
if (preserve_executability)
|
||||
return (current_mode & 0777 & ~execute_bits) | (metadata->mode & execute_bits);
|
||||
return metadata->mode & 0777 & ~(S_IWGRP | S_IWOTH);
|
||||
if (policy.perms) {
|
||||
/* rsync --perms copies the source's permission and special bits exactly,
|
||||
* including group/other write and setuid/setgid/sticky. The kernel may
|
||||
* still clear setgid when the receiver is not in the file's group; the
|
||||
* caller logs a failed chmod rather than silently masking the bits here. */
|
||||
*out_mode = source_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
return true;
|
||||
}
|
||||
if (policy.executability) {
|
||||
/* -E/--executability (rsync 3.4 rule): do NOT copy the source's execute
|
||||
* bits per class. If the source is executable at all, derive the execute
|
||||
* bits from the DESTINATION's own read bits (so a class that can read may
|
||||
* execute); otherwise clear every execute bit. This runs on the
|
||||
* destination-derived base (pre-existing dest mode, or source&~umask for a
|
||||
* new file), and leaves the special bits untouched. --perms wins when both
|
||||
* are set (handled above). */
|
||||
mode_t base = current_mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
if (source_mode & 0111)
|
||||
*out_mode = base | ((base & 0444) >> 2);
|
||||
else
|
||||
*out_mode = base & ~execute_bits;
|
||||
return true;
|
||||
}
|
||||
/* Neither requested: no source mode is applied at all. */
|
||||
return false;
|
||||
}
|
||||
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool preserve_executability) {
|
||||
FileAttrPolicy file_attr_policy_from_config(const Config* config) {
|
||||
FileAttrPolicy policy = {false, false, false, false};
|
||||
if (config) {
|
||||
policy.perms = config->preserve_perms;
|
||||
policy.times = config->preserve_times;
|
||||
policy.atimes = config->preserve_atimes;
|
||||
policy.executability = config->use_executability;
|
||||
}
|
||||
return policy;
|
||||
}
|
||||
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata, FileAttrPolicy policy) {
|
||||
if (metadata == NULL)
|
||||
return;
|
||||
bool apply_mode = false;
|
||||
mode_t safe_mode = 0;
|
||||
if (policy.perms || policy.executability) {
|
||||
struct stat current;
|
||||
mode_t current_mode = stat(path, ¤t) == 0 ? current.st_mode : 0;
|
||||
mode_t safe_mode = metadata_mode(metadata, current_mode, preserve_executability);
|
||||
if (chmod(path, safe_mode) != 0) {
|
||||
apply_mode = metadata_mode_for_policy(metadata->mode, current_mode, policy, &safe_mode);
|
||||
}
|
||||
if (apply_mode && chmod(path, safe_mode) != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Failed to chmod %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
|
||||
@@ -334,31 +268,34 @@ void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
}
|
||||
/* Never apply client-supplied ownership. The descriptor API below is the
|
||||
receiver write path; retain this legacy API only for compatibility. */
|
||||
struct timespec times[2];
|
||||
times[0].tv_sec = 0;
|
||||
times[0].tv_nsec = UTIME_OMIT;
|
||||
if (policy.times || (policy.atimes && metadata->atime_valid)) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
if (metadata->atime_valid) {
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
if (metadata->crtime_valid) {
|
||||
log_message(LOG_LEVEL_DEBUG,
|
||||
"crtime (birth time) %lld.%09ld transmitted for %s but not applied: no portable "
|
||||
"setter exists",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec, path);
|
||||
}
|
||||
if (utimensat(AT_FDCWD, path, times, 0) != 0) {
|
||||
char* escaped_path = output_escape(path, log_get_8_bit_output());
|
||||
log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s",
|
||||
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
|
||||
free(escaped_path);
|
||||
}
|
||||
}
|
||||
if (metadata->crtime_valid) {
|
||||
log_message(LOG_LEVEL_DEBUG,
|
||||
"crtime (birth time) %lld.%09ld transmitted for %s but not applied: no portable "
|
||||
"setter exists",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec, path);
|
||||
}
|
||||
}
|
||||
|
||||
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool omit_link_times) {
|
||||
FileAttrPolicy policy, bool omit_link_times) {
|
||||
if (path == NULL || metadata == NULL)
|
||||
return !identity_copy_as_active();
|
||||
char* leaf = NULL;
|
||||
@@ -371,18 +308,25 @@ bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadat
|
||||
best-effort. */
|
||||
bool owned = identity_apply_ownership_link(parent_fd, leaf, (int32_t)metadata->uid,
|
||||
(int32_t)metadata->gid);
|
||||
/* Symlink mode: not settable on Linux (fchmodat AT_SYMLINK_NOFOLLOW returns
|
||||
EOPNOTSUPP/ENOTSUP); attempt it for platforms that support it and quietly
|
||||
ignore the unsupported case so the transfer never fails over it. */
|
||||
mode_t link_mode = metadata->mode & 0777;
|
||||
/* Symlink mode: only when -p is in effect. It is not settable on Linux
|
||||
(fchmodat AT_SYMLINK_NOFOLLOW returns EOPNOTSUPP/ENOTSUP); attempt it for
|
||||
platforms that support it and quietly ignore the unsupported case so the
|
||||
transfer never fails over it. */
|
||||
if (policy.perms) {
|
||||
mode_t link_mode = metadata->mode & (mode_t)(S_ISUID | S_ISGID | S_ISVTX | 0777);
|
||||
if (fchmodat(parent_fd, leaf, link_mode, AT_SYMLINK_NOFOLLOW) != 0 && errno != EOPNOTSUPP &&
|
||||
errno != ENOTSUP && errno != ENOSYS) {
|
||||
log_message(LOG_LEVEL_DEBUG, "Could not set symlink mode on %s: %s", path, strerror(errno));
|
||||
}
|
||||
if (!omit_link_times) {
|
||||
}
|
||||
if (!omit_link_times && (policy.times || (policy.atimes && metadata->atime_valid))) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
|
||||
if (metadata->atime_valid) {
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
@@ -398,19 +342,13 @@ bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadat
|
||||
return owned;
|
||||
}
|
||||
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability) {
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, FileAttrPolicy policy) {
|
||||
if (fd < 0 || metadata == NULL)
|
||||
return metadata == NULL;
|
||||
bool ok = true;
|
||||
struct stat current;
|
||||
if (fstat(fd, ¤t) != 0)
|
||||
return false;
|
||||
mode_t safe_mode = metadata_mode(metadata, current.st_mode, preserve_executability);
|
||||
if (fchmod(fd, safe_mode) != 0)
|
||||
ok = false;
|
||||
/* Client uid/gid values are deliberately not authoritative UNLESS the client
|
||||
explicitly opted in with an identity flag (--numeric-ids / --usermap /
|
||||
--groupmap / --chown). identity_apply_ownership is the controlled,
|
||||
--groupmap / --chown / -o/-g). identity_apply_ownership is the controlled,
|
||||
privilege-gated path: it consults the negotiated policy, resolves the
|
||||
target ids, and applies them via an fd-relative fchown() that is confined
|
||||
to the just-written file (EPERM/EACCES are logged, never fatal) -- EXCEPT
|
||||
@@ -418,14 +356,19 @@ bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserv
|
||||
marks this entry as failed instead of reporting a wrong-owner write as
|
||||
success. With no identity flag set it is a no-op, so a default or plain -M
|
||||
transfer keeps FastSync's existing behavior of never applying client
|
||||
ownership. */
|
||||
ownership. Ownership runs BEFORE the mode because a chown clears
|
||||
setuid/setgid; rsync likewise chowns first and then restores the source
|
||||
mode (including its special bits). */
|
||||
if (!identity_apply_ownership(fd, (int32_t)metadata->uid, (int32_t)metadata->gid))
|
||||
ok = false;
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
|
||||
if (metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
if (policy.perms || policy.executability) {
|
||||
struct stat current;
|
||||
if (fstat(fd, ¤t) != 0)
|
||||
return false;
|
||||
mode_t safe_mode = 0;
|
||||
bool apply_mode = metadata_mode_for_policy(metadata->mode, current.st_mode, policy, &safe_mode);
|
||||
if (apply_mode && fchmod(fd, safe_mode) != 0)
|
||||
ok = false;
|
||||
}
|
||||
/* --crtimes captures and transmits the source birth time, but there is no
|
||||
* portable way to set a birth time (utimensat can only set atime/mtime), so
|
||||
@@ -437,7 +380,19 @@ bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserv
|
||||
"crtime (birth time) %lld.%09ld transmitted but not applied: no portable setter",
|
||||
(long long)metadata->crtime_sec, metadata->crtime_nsec);
|
||||
}
|
||||
if (policy.times || (policy.atimes && metadata->atime_valid)) {
|
||||
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
|
||||
{.tv_sec = 0, .tv_nsec = UTIME_OMIT}};
|
||||
if (policy.times) {
|
||||
times[1].tv_sec = metadata->mtime_sec;
|
||||
times[1].tv_nsec = metadata->mtime_nsec;
|
||||
}
|
||||
if (policy.atimes && metadata->atime_valid) {
|
||||
times[0].tv_sec = metadata->atime_sec;
|
||||
times[0].tv_nsec = metadata->atime_nsec;
|
||||
}
|
||||
if (futimens(fd, times) != 0)
|
||||
ok = false;
|
||||
}
|
||||
return ok;
|
||||
}
|
||||
|
||||
+35
-9
@@ -2,7 +2,9 @@
|
||||
#define METADATA_H
|
||||
|
||||
#include "file.h"
|
||||
#include "file_attr.h"
|
||||
#include <stdbool.h>
|
||||
#include <stddef.h>
|
||||
#include <stdint.h>
|
||||
#include <sys/stat.h>
|
||||
#include <time.h>
|
||||
@@ -29,26 +31,50 @@
|
||||
|
||||
/* Size of metadata fields on wire, excluding the int32_t `present` field that
|
||||
* is always sent first. The total wire size for present metadata is
|
||||
* sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms). */
|
||||
* sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms).
|
||||
*
|
||||
* metadata_send()/metadata_receive() (protocol 2.20.0) frame the metadata as a
|
||||
* single packed record: one int32 present flag (0 = absent) followed, when
|
||||
* present, by exactly FILE_METADATA_WIRE_SIZE bytes of field data. This is the
|
||||
* same present+fields byte layout metadata_to_buf()/metadata_from_buf() use, so
|
||||
* the wire metadata is now one frame instead of one frame per field. */
|
||||
#define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 5 + sizeof(int64_t) * 6)
|
||||
|
||||
void metadata_to_buf(char** buf, const FileMetadata* m);
|
||||
FileMetadata* metadata_from_buf(char** buf);
|
||||
/* Decode one packed metadata record (an int32 present flag followed, when
|
||||
* present, by FILE_METADATA_WIRE_SIZE field bytes) from `buf`, which has `len`
|
||||
* readable bytes. Every read is bounds-checked against `len`, so the function
|
||||
* can never over-read the caller's buffer: a too-short record, an absent
|
||||
* (present == 0) record and a malformed record all return NULL. A successful
|
||||
* decode returns a heap-allocated FileMetadata owned by the caller. */
|
||||
FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len);
|
||||
bool metadata_send(int file_descriptor, const FileMetadata* m);
|
||||
FileMetadata* metadata_receive(int file_descriptor, int* ok);
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool preserve_executability);
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability);
|
||||
void file_restore_metadata(const char* path, const FileMetadata* metadata, FileAttrPolicy policy);
|
||||
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, FileAttrPolicy policy);
|
||||
|
||||
/* Shared mode-policy helper: the single source of truth for the receiver's
|
||||
* mode rule. Given a source mode and the destination's CURRENT mode, returns
|
||||
* true and stores the exact mode to apply in *out_mode when `policy` requests
|
||||
* a change, or false when it requests neither --perms nor --executability (the
|
||||
* caller then leaves the destination mode alone). --perms wins over -E; the
|
||||
* -E rule derives exec bits from the destination's read bits (rsync 3.4);
|
||||
* group/other write is never granted from a client-supplied mode. Shared by
|
||||
* file_restore_metadata_fd() and the --fake-super replay so the two cannot
|
||||
* diverge. */
|
||||
bool metadata_mode_for_policy(mode_t source_mode, mode_t current_mode, FileAttrPolicy policy,
|
||||
mode_t* out_mode);
|
||||
/* P7 Wave D: apply a SYMLINK's own metadata using no-follow primitives only
|
||||
* (utimensat/lchown/fchmodat with AT_SYMLINK_NOFOLLOW), confined fd-relative
|
||||
* under the authorized root. `omit_link_times` (-J/--omit-link-times)
|
||||
* suppresses the timestamps; the link's mode/ownership are still attempted
|
||||
* (ownership stays gated by the identity policy and by default is not applied).
|
||||
* under the authorized root. The link's mode is applied only when policy.perms;
|
||||
* policy.times (further suppressed by `omit_link_times` for -J) applies the
|
||||
* mtime with policy.atimes controlling the atime slot; ownership stays gated by
|
||||
* the identity policy and by default is not applied.
|
||||
* A null metadata or an unfollowable parent is a harmless no-op. Returns false
|
||||
* only when a REQUIRED --copy-as ownership application failed, so the caller can
|
||||
* report the entry as failed instead of claiming a wrong-owner success. */
|
||||
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata,
|
||||
bool omit_link_times);
|
||||
FileAttrPolicy policy, bool omit_link_times);
|
||||
|
||||
/* Compare timestamps using rsync's whole-second modification window. */
|
||||
bool metadata_mtime_matches(time_t left_sec, long left_nsec, time_t right_sec, long right_nsec,
|
||||
|
||||
+98
-247
@@ -1,15 +1,16 @@
|
||||
#include "multiprocessing.h"
|
||||
#include "receiver.h"
|
||||
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "data.h"
|
||||
#include "file.h"
|
||||
#include "file_receive.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "utils.h"
|
||||
#include <stdint.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
@@ -25,12 +26,18 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que
|
||||
context->queue_loader = queue_loader;
|
||||
context->scanner_done = false;
|
||||
context->loader_done = false;
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->manifest = NULL;
|
||||
context->excluded_paths = NULL;
|
||||
context->size_skipped_paths = NULL;
|
||||
context->synced_dirs = NULL;
|
||||
context->plan_dirs = NULL;
|
||||
context->missing_args = NULL;
|
||||
context->scan_had_io_error = false;
|
||||
context->remove_source_files = NULL;
|
||||
context->early_delete = false;
|
||||
context->delete_plans = NULL;
|
||||
context->scan_stopped_early = false;
|
||||
context->total_files = 0;
|
||||
context->progress_bytes = 0;
|
||||
@@ -41,6 +48,7 @@ PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* que
|
||||
protocol_session_set_max_alloc(&context->allocation_session, config->max_alloc);
|
||||
context->dir_entries = NULL;
|
||||
context->dir_entries_mutex_init = false;
|
||||
context->delete_limit = false;
|
||||
int init = 0;
|
||||
if (config->use_metadata) {
|
||||
context->dir_entries = array_list_create(file_destroy);
|
||||
@@ -96,12 +104,101 @@ fail:
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void pipeline_context_sender_set_queue_byte_limit(PipelineContextSender* context,
|
||||
size_t max_bytes) {
|
||||
if (context == NULL)
|
||||
return;
|
||||
mtx_lock(&context->mutex_loader);
|
||||
context->max_queue_bytes = max_bytes;
|
||||
context->queued_bytes = 0;
|
||||
cnd_broadcast(&context->condition_not_full_loader);
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
}
|
||||
|
||||
size_t pipeline_context_sender_chunk_bytes(const Chunk* chunk) {
|
||||
if (chunk == NULL || chunk->items == NULL)
|
||||
return 0;
|
||||
size_t total = 0;
|
||||
for (int i = 0; i < chunk->element_count; i++) {
|
||||
const File* file = chunk->items[i];
|
||||
if (file == NULL || file->data == NULL || file->data->data == NULL)
|
||||
continue;
|
||||
if (file->data->size > SIZE_MAX - total)
|
||||
return SIZE_MAX;
|
||||
total += file->data->size;
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
void pipeline_context_sender_note_bytes_released(PipelineContextSender* context,
|
||||
size_t released_bytes) {
|
||||
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
|
||||
return;
|
||||
mtx_lock(&context->mutex_loader);
|
||||
if (released_bytes >= context->queued_bytes)
|
||||
context->queued_bytes = 0;
|
||||
else
|
||||
context->queued_bytes -= released_bytes;
|
||||
cnd_signal(&context->condition_not_full_loader);
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
}
|
||||
|
||||
bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk* chunk) {
|
||||
if (context == NULL || chunk == NULL)
|
||||
return false;
|
||||
size_t chunk_bytes = pipeline_context_sender_chunk_bytes(chunk);
|
||||
mtx_lock(&context->mutex_loader);
|
||||
while (!atomic_load(&context->cancelled)) {
|
||||
bool blocked_by_count = queue_is_full(context->queue_loader);
|
||||
bool blocked_by_budget = false;
|
||||
if (context->max_queue_bytes > 0) {
|
||||
size_t budget = context->max_queue_bytes;
|
||||
size_t used = context->queued_bytes;
|
||||
if (used >= budget) {
|
||||
blocked_by_budget = true;
|
||||
} else if (chunk_bytes > budget - used) {
|
||||
/* A single payload larger than the whole budget is only admitted to an
|
||||
empty pipeline so the wait can never deadlock. */
|
||||
blocked_by_budget = used != 0;
|
||||
}
|
||||
}
|
||||
if (!blocked_by_count && !blocked_by_budget)
|
||||
break;
|
||||
cnd_wait(&context->condition_not_full_loader, &context->mutex_loader);
|
||||
}
|
||||
if (atomic_load(&context->cancelled)) {
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
chunk_destroy(chunk);
|
||||
return false;
|
||||
}
|
||||
if (!queue_enqueue(context->queue_loader, chunk)) {
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
chunk_destroy(chunk);
|
||||
return false;
|
||||
}
|
||||
context->queued_bytes += chunk_bytes;
|
||||
cnd_signal(&context->condition_not_empty_loader);
|
||||
mtx_unlock(&context->mutex_loader);
|
||||
return true;
|
||||
}
|
||||
|
||||
void pipeline_context_sender_destroy(PipelineContextSender* context) {
|
||||
/* `config` is borrowed: the caller retains ownership and frees it after the
|
||||
pipeline has been destroyed (the worker threads are already joined, so no
|
||||
config access can outlive this call). */
|
||||
if (context->manifest) {
|
||||
array_list_delete(context->manifest);
|
||||
}
|
||||
if (context->delete_plans)
|
||||
delete_plan_sender_destroy(context->delete_plans);
|
||||
if (context->excluded_paths)
|
||||
array_list_delete(context->excluded_paths);
|
||||
if (context->size_skipped_paths)
|
||||
array_list_delete(context->size_skipped_paths);
|
||||
if (context->synced_dirs)
|
||||
array_list_delete(context->synced_dirs);
|
||||
if (context->plan_dirs)
|
||||
array_list_delete(context->plan_dirs);
|
||||
if (context->missing_args)
|
||||
array_list_delete(context->missing_args);
|
||||
if (context->remove_source_files)
|
||||
@@ -110,7 +207,6 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) {
|
||||
array_list_delete(context->dir_entries);
|
||||
if (context->dir_entries_mutex_init)
|
||||
mtx_destroy(&context->dir_entries_mutex);
|
||||
config_delete(context->config);
|
||||
queue_destroy(context->queue_scanner);
|
||||
queue_destroy(context->queue_loader);
|
||||
mtx_destroy(&context->mutex_scanner);
|
||||
@@ -122,248 +218,3 @@ void pipeline_context_sender_destroy(PipelineContextSender* context) {
|
||||
mtx_destroy(&context->mutex_progress);
|
||||
free(context);
|
||||
}
|
||||
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue,
|
||||
int file_descriptor, SSL* ssl) {
|
||||
PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver));
|
||||
if (context == NULL)
|
||||
return NULL;
|
||||
context->config = config;
|
||||
context->queue = queue;
|
||||
context->file_descriptor = file_descriptor;
|
||||
context->ssl = ssl;
|
||||
context->outcomes.entries = NULL;
|
||||
context->outcomes.count = 0;
|
||||
context->outcomes.capacity = 0;
|
||||
dir_time_list_init(&context->dir_times);
|
||||
protocol_session_init(&context->session, file_descriptor, file_descriptor);
|
||||
protocol_session_set_ssl(&context->session, ssl);
|
||||
context->receiver_done = false;
|
||||
context->queued_bytes = 0;
|
||||
context->max_queue_bytes = 0;
|
||||
context->deferred_manifest = NULL;
|
||||
atomic_init(&context->cancelled, false);
|
||||
int init = 0;
|
||||
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_full) != thrd_success)
|
||||
goto fail;
|
||||
init++;
|
||||
if (cnd_init(&context->condition_not_empty) != thrd_success)
|
||||
goto fail;
|
||||
// cppcheck-suppress unreadVariable
|
||||
init++;
|
||||
return context;
|
||||
|
||||
fail:
|
||||
log_perror("Error initializing synchronization objects");
|
||||
if (init >= 3)
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
if (init >= 2)
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
if (init >= 1)
|
||||
mtx_destroy(&context->mutex);
|
||||
free(context);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
|
||||
config_delete(context->config);
|
||||
if (context->deferred_manifest)
|
||||
delete_manifest_free(context->deferred_manifest);
|
||||
queue_destroy(context->queue);
|
||||
receiver_outcomes_destroy(&context->outcomes);
|
||||
dir_time_list_free(&context->dir_times);
|
||||
mtx_destroy(&context->mutex);
|
||||
cnd_destroy(&context->condition_not_full);
|
||||
cnd_destroy(&context->condition_not_empty);
|
||||
free(context);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes) {
|
||||
if (context == NULL)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
context->max_queue_bytes = max_bytes;
|
||||
context->queued_bytes = 0;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
size_t released_bytes) {
|
||||
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
|
||||
return;
|
||||
mtx_lock(&context->mutex);
|
||||
if (released_bytes >= context->queued_bytes)
|
||||
context->queued_bytes = 0;
|
||||
else
|
||||
context->queued_bytes -= released_bytes;
|
||||
cnd_signal(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) {
|
||||
if (context == NULL || file == NULL)
|
||||
return false;
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
mtx_lock(&context->mutex);
|
||||
while (!atomic_load(&context->cancelled)) {
|
||||
bool blocked_by_count = queue_is_full(context->queue);
|
||||
bool blocked_by_budget = false;
|
||||
if (context->max_queue_bytes > 0) {
|
||||
size_t budget = context->max_queue_bytes;
|
||||
size_t used = context->queued_bytes;
|
||||
if (used >= budget) {
|
||||
blocked_by_budget = true;
|
||||
} else if (file_bytes > budget - used) {
|
||||
/* A single payload larger than the whole budget (not possible with
|
||||
the per-file receive cap) is only admitted to an empty pipeline so
|
||||
the wait can never deadlock. */
|
||||
blocked_by_budget = used != 0;
|
||||
}
|
||||
}
|
||||
if (!blocked_by_count && !blocked_by_budget)
|
||||
break;
|
||||
cnd_wait(&context->condition_not_full, &context->mutex);
|
||||
}
|
||||
if (atomic_load(&context->cancelled)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
if (!queue_enqueue(context->queue, file)) {
|
||||
mtx_unlock(&context->mutex);
|
||||
file_destroy(file);
|
||||
return false;
|
||||
}
|
||||
context->queued_bytes += file_bytes;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
return true;
|
||||
}
|
||||
|
||||
static bool receiver_enqueue_file(File* file, void* context_pointer) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
|
||||
return pipeline_context_receiver_enqueue_file(context, file);
|
||||
}
|
||||
|
||||
static void receiver_thread_fail(PipelineContextReceiver* context) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
mtx_unlock(&context->mutex);
|
||||
}
|
||||
|
||||
int receive_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
int file_descriptor = context->file_descriptor;
|
||||
const Config* config = context->config;
|
||||
mtx_unlock(&context->mutex);
|
||||
|
||||
ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL};
|
||||
if (receiver_process_pending((Config*)config, file_descriptor, &sink,
|
||||
&context->deferred_manifest) != 0) {
|
||||
receiver_thread_fail(context);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
mtx_lock(&context->mutex);
|
||||
context->receiver_done = true;
|
||||
cnd_signal(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
|
||||
int write_thread(void* pipeline_context) {
|
||||
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
|
||||
protocol_session_bind(&context->session);
|
||||
mtx_lock(&context->mutex);
|
||||
bool save_to_disk = context->config->save_to_disk;
|
||||
char* root_directory = str_dup(context->config->receive_root_directory);
|
||||
mtx_unlock(&context->mutex);
|
||||
if (save_to_disk && !root_directory) {
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
|
||||
while (true) {
|
||||
File* file =
|
||||
queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty,
|
||||
&context->condition_not_full, &context->receiver_done);
|
||||
if (file == NULL) {
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_success;
|
||||
}
|
||||
size_t file_bytes = file->data ? file->data->size : 0;
|
||||
FileSaveResult result = FILE_SAVE_SKIPPED;
|
||||
if (save_to_disk) {
|
||||
result = file_save_to_disk_full(root_directory, file, context->config);
|
||||
if (result == FILE_SAVE_ERROR) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
}
|
||||
/* P7 Wave D: a directory's times are never applied inline (a later child
|
||||
write would clobber them); accumulate the metadata here and let the
|
||||
caller apply it once every writer has drained. */
|
||||
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
|
||||
context->config->use_metadata && !context->config->omit_dir_times &&
|
||||
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
/* Record the per-file outcome so a --remove-source-files sender learns
|
||||
which sources were actually written versus skipped on the receiver.
|
||||
Explicit directory entries and recreated device/special nodes have no
|
||||
source and are never acknowledged (mirrors receiver.c). */
|
||||
if (context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip &&
|
||||
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
mtx_lock(&context->mutex);
|
||||
atomic_store(&context->cancelled, true);
|
||||
context->receiver_done = true;
|
||||
cnd_broadcast(&context->condition_not_full);
|
||||
cnd_broadcast(&context->condition_not_empty);
|
||||
mtx_unlock(&context->mutex);
|
||||
free(root_directory);
|
||||
protocol_session_unbind();
|
||||
return thrd_error;
|
||||
}
|
||||
file_destroy(file);
|
||||
pipeline_context_receiver_note_bytes_released(context, file_bytes);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,11 +5,12 @@
|
||||
#include <stdatomic.h>
|
||||
|
||||
#include "array_list.h"
|
||||
#include "chunk.h"
|
||||
#include "config.h"
|
||||
#include "delete_plan.h"
|
||||
#include "file.h"
|
||||
#include "protocol.h"
|
||||
#include "queue.h"
|
||||
#include "receiver.h"
|
||||
#include "stop_condition.h"
|
||||
#include <openssl/ssl.h>
|
||||
|
||||
@@ -25,6 +26,15 @@ typedef struct {
|
||||
cnd_t condition_not_full_loader;
|
||||
cnd_t condition_not_empty_loader;
|
||||
bool loader_done;
|
||||
/* Aggregate loaded payload bytes queued on queue_loader but not yet released
|
||||
by the sender. Guarded by `mutex_loader`. When `max_queue_bytes` is
|
||||
non-zero the loader blocks before enqueueing a chunk that would push this
|
||||
total over it, so the sender buffers a bounded number of bytes rather than
|
||||
an unbounded count of chunks that may each be up to chunk_size (or a single
|
||||
file) in size. Files streamed straight from disk by sendfile hold no
|
||||
payload, so only in-memory (`data->data`) payloads are counted. */
|
||||
size_t queued_bytes;
|
||||
size_t max_queue_bytes;
|
||||
ArrayList* manifest;
|
||||
/* Protected prefixes (paths the source scan excluded by user rules) sent
|
||||
with the keep-set manifest so --delete leaves them alone unless
|
||||
@@ -33,6 +43,23 @@ typedef struct {
|
||||
scanner's exclusion sink) or, in the early modes, by the path-only pre-scan
|
||||
on the calling thread before the pipeline starts. */
|
||||
ArrayList* excluded_paths;
|
||||
/* --max-size/--min-size pruned source paths. These are ALWAYS sent as
|
||||
protected prefixes (even with --delete-excluded), so the destination
|
||||
mirrors of size-skipped files survive --delete like rsync. Populated by
|
||||
the scanner thread (workers append under mutex_scanner) or, in the early
|
||||
modes, by the path-only pre-scan on the calling thread. */
|
||||
ArrayList* size_skipped_paths;
|
||||
/* Destination-relative paths of the directories the source scan synchronized
|
||||
for this run (the receive root is the "." sentinel). Sent with the
|
||||
manifest so the receiver confines its extras walk to them, matching rsync's
|
||||
"delete only in synchronized directories" (notably for --files-from).
|
||||
Populated by the scanner thread or the early pre-scan. */
|
||||
ArrayList* synced_dirs;
|
||||
/* Destination-relative paths of every traversed source directory, for the
|
||||
per-directory delete plan keep set (so an empty source directory survives
|
||||
--delete rather than being removed as an extra). Prebuilt by the path-only
|
||||
pre-scan on the calling thread. */
|
||||
ArrayList* plan_dirs;
|
||||
/* --delete-missing-args: the destination-relative mirrors of the --files-from
|
||||
entries that are missing under the source. Computed by the preflight on
|
||||
the calling thread before the pipeline starts; the sender thread transmits
|
||||
@@ -45,11 +72,16 @@ typedef struct {
|
||||
--ignore-errors kept the run going. */
|
||||
bool scan_had_io_error;
|
||||
ArrayList* remove_source_files;
|
||||
/* True when --delete-before/--delete-during require the keep-set manifest to
|
||||
be transmitted before any file data: context->manifest is then prebuilt by
|
||||
a path-only pre-scan on the calling thread and the pipeline scanner must
|
||||
not append to it. Set once before the worker threads start. */
|
||||
/* True when --delete-before requires the whole-tree keep-set manifest to be
|
||||
transmitted before any file data: context->manifest is then prebuilt by a
|
||||
path-only pre-scan on the calling thread and the pipeline scanner must not
|
||||
append to it. Set once before the worker threads start. */
|
||||
bool early_delete;
|
||||
/* Non-NULL for --delete-during/--delete-delay: the per-directory plan set
|
||||
prebuilt by the path-only pre-scan on the calling thread. The sender
|
||||
thread transmits the root plan before any data and the remaining plans
|
||||
alongside the chunks. Set once before the worker threads start. */
|
||||
DeletePlanSender* delete_plans;
|
||||
mtx_t mutex_progress;
|
||||
int total_files;
|
||||
unsigned long long progress_bytes;
|
||||
@@ -74,60 +106,31 @@ typedef struct {
|
||||
ArrayList* dir_entries;
|
||||
mtx_t dir_entries_mutex;
|
||||
bool dir_entries_mutex_init;
|
||||
/* Set by the sender thread when the receiver reported a --max-delete-capped
|
||||
deletion (STATUS_DELETE_LIMIT): the transfer succeeded and the process must
|
||||
exit 25 like rsync. Read by the caller after the sender thread is joined. */
|
||||
bool delete_limit;
|
||||
} PipelineContextSender;
|
||||
|
||||
typedef struct PipelineContextReceiver {
|
||||
Queue* queue;
|
||||
Config* config;
|
||||
int file_descriptor;
|
||||
SSL* ssl;
|
||||
ProtocolSession session;
|
||||
ReceiverOutcomes outcomes;
|
||||
mtx_t mutex;
|
||||
cnd_t condition_not_full;
|
||||
cnd_t condition_not_empty;
|
||||
bool receiver_done;
|
||||
atomic_bool cancelled;
|
||||
/* Aggregate payload bytes that have been received but not yet released by
|
||||
the disk writer (queued or in the writer's hand). Guarded by `mutex`.
|
||||
When `max_queue_bytes` is non-zero the receiver blocks before enqueuing
|
||||
once this total would exceed it, so decompressed/copied file payloads
|
||||
buffered ahead of a slow disk writer respect the per-connection memory
|
||||
budget instead of growing without bound. */
|
||||
size_t queued_bytes;
|
||||
size_t max_queue_bytes;
|
||||
/* Keep-set manifest for the commit-style (late) deletion
|
||||
(--delete/--delete-after/--delete-delay). receive_thread parses the whole
|
||||
protocol stream but hands the manifest here instead of deleting while the
|
||||
disk writer may still be draining; the caller (server.c) commits the
|
||||
deletion after both threads have joined, so no extra is removed unless the
|
||||
transfer truly succeeded. NULL in the early delete modes (which delete at
|
||||
the manifest). */
|
||||
DeleteManifest* deferred_manifest;
|
||||
/* P7 Wave D: directory metadata collected by write_thread from received
|
||||
directory entries. Only write_thread mutates it (before it joins); the
|
||||
caller (server.c) applies it after the delete/delay-updates phase. */
|
||||
DirTimeList dir_times;
|
||||
} PipelineContextReceiver;
|
||||
|
||||
/* `config` is borrowed and must outlive the context: destroy does NOT free it,
|
||||
so the caller owns it and frees it with config_delete() afterwards. */
|
||||
PipelineContextSender* pipeline_context_sender_create(Config* config, Queue* queue_scanner,
|
||||
Queue* queue_loader);
|
||||
void pipeline_context_sender_destroy(PipelineContextSender* context);
|
||||
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
|
||||
int file_descriptor, SSL* ssl);
|
||||
void pipeline_context_receiver_destroy(PipelineContextReceiver* context);
|
||||
/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */
|
||||
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
|
||||
size_t max_bytes);
|
||||
/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue
|
||||
is full by element count or when adding `file` would push queued_bytes over
|
||||
the configured byte limit; waits until the disk writer releases bytes.
|
||||
Takes ownership of `file` on success and destroys it on failure/cancel. */
|
||||
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file);
|
||||
/* Account for `released_bytes` of payload memory that has been freed by the
|
||||
disk writer, unblocking a receiver that is waiting on the byte limit. */
|
||||
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
|
||||
/* Bound the loaded payload bytes the sender may buffer ahead of the network
|
||||
writer (see max_queue_bytes). */
|
||||
void pipeline_context_sender_set_queue_byte_limit(PipelineContextSender* context, size_t max_bytes);
|
||||
/* Total payload bytes a chunk currently holds in memory (loaded file data
|
||||
only; zero for entries with no payload or data streamed from disk). */
|
||||
size_t pipeline_context_sender_chunk_bytes(const Chunk* chunk);
|
||||
/* Blocking enqueue used by the sender's loader stage. Blocks while
|
||||
queue_loader is full by element count or when adding `chunk` would push the
|
||||
queued payload bytes over the configured byte limit; waits until the sender
|
||||
releases bytes. Takes ownership of `chunk` on success and destroys it on
|
||||
failure/cancel. */
|
||||
bool pipeline_context_sender_enqueue_chunk(PipelineContextSender* context, Chunk* chunk);
|
||||
/* Account for `released_bytes` of payload memory that the sender freed after
|
||||
destroying a chunk, unblocking a loader waiting on the byte limit. */
|
||||
void pipeline_context_sender_note_bytes_released(PipelineContextSender* context,
|
||||
size_t released_bytes);
|
||||
int receive_thread(void* pipeline_context);
|
||||
int write_thread(void* pipeline_context);
|
||||
#endif
|
||||
|
||||
+389
-27
@@ -12,8 +12,7 @@
|
||||
#include <time.h>
|
||||
#include <unistd.h>
|
||||
|
||||
#define RECEIVE_TIMEOUT_SEC 60 /* 60 second per-message timeout */
|
||||
#define SEND_TIMEOUT_SEC 60
|
||||
#define RECEIVE_TIMEOUT_SEC 60 /* built-in fallback for explicit -timed calls only */
|
||||
|
||||
static __thread int io_read_fd = -1;
|
||||
static __thread int io_write_fd = -1;
|
||||
@@ -22,10 +21,21 @@ static __thread ProtocolSession* bound_session;
|
||||
static __thread ProtocolSession legacy_io_session = {
|
||||
.read_fd = -1, .write_fd = -1, .max_alloc = DEFAULT_MAX_ALLOC};
|
||||
|
||||
/* Last STATUS_ERROR_DETAIL reason received on this thread (protocol 2.21.0).
|
||||
* Empty when the last status read carried no detail. */
|
||||
static __thread char io_error_detail[MAX_ERROR_DETAIL_BYTES + 1];
|
||||
|
||||
static unsigned long long io_bwlimit = 0;
|
||||
static mtx_t bw_mutex;
|
||||
static once_flag bw_mutex_once = ONCE_FLAG_INIT;
|
||||
|
||||
/* Process-wide wire byte counters, used by the client to render rsync's
|
||||
* --stats/--progress totals and the --out-format %b/%c tokens. The zero-copy
|
||||
* sendfile path bypasses protocol_send_n_data, so it reports its bytes through
|
||||
* protocol_note_bytes_written. */
|
||||
static atomic_ullong io_bytes_written = 0;
|
||||
static atomic_ullong io_bytes_read = 0;
|
||||
|
||||
static unsigned long long global_bwlimit(void);
|
||||
|
||||
static bool protocol_reserve_memory(ProtocolSession* session, size_t charge) {
|
||||
@@ -40,7 +50,9 @@ static bool protocol_reserve_memory(ProtocolSession* session, size_t charge) {
|
||||
}
|
||||
}
|
||||
|
||||
static void protocol_release_memory_for_session(ProtocolSession* session, size_t charge) {
|
||||
void protocol_release_memory_for_session(ProtocolSession* session, size_t charge) {
|
||||
if (!session)
|
||||
return;
|
||||
unsigned long long allocated = atomic_load(&session->total_allocated_bytes);
|
||||
while (true) {
|
||||
unsigned long long remaining = (unsigned long long)charge >= allocated ? 0 : allocated - charge;
|
||||
@@ -57,6 +69,10 @@ void io_set_fds(int read_fd, int write_fd) {
|
||||
bound_session = NULL;
|
||||
io_read_fd = read_fd;
|
||||
io_write_fd = write_fd;
|
||||
/* A descriptor switch starts a new connection on this thread: a stale
|
||||
rejection detail captured from the previous transport must not leak into
|
||||
the new one. */
|
||||
io_error_detail[0] = '\0';
|
||||
/* A descriptor switch starts a new transport; never reuse a TLS object
|
||||
belonging to a previous connection or test pipe. */
|
||||
io_ssl = NULL;
|
||||
@@ -76,10 +92,29 @@ void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd)
|
||||
session->read_fd = read_fd;
|
||||
session->write_fd = write_fd;
|
||||
session->max_alloc = DEFAULT_MAX_ALLOC;
|
||||
session->io_timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
atomic_init(&session->total_allocated_bytes, 0);
|
||||
protocol_session_set_bwlimit(session, global_bwlimit());
|
||||
}
|
||||
|
||||
void protocol_session_set_io_timeout(ProtocolSession* session, int sec) {
|
||||
if (!session)
|
||||
return;
|
||||
session->io_timeout_sec = sec;
|
||||
}
|
||||
|
||||
int protocol_get_io_timeout_sec(void) {
|
||||
const ProtocolSession* session = bound_session ? bound_session : &legacy_io_session;
|
||||
/* 0 (or negative) means the session timeout is disabled, matching rsync's
|
||||
* --timeout=0 default. Callers must treat a non-positive result as "wait
|
||||
* without a deadline" instead of substituting a built-in window. */
|
||||
return session->io_timeout_sec > 0 ? session->io_timeout_sec : 0;
|
||||
}
|
||||
|
||||
int protocol_server_io_timeout_sec(int client_timeout) {
|
||||
return client_timeout > 0 ? client_timeout : SERVER_IO_TIMEOUT_SEC;
|
||||
}
|
||||
|
||||
void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc) {
|
||||
if (!session)
|
||||
session = bound_session ? bound_session : &legacy_io_session;
|
||||
@@ -87,7 +122,8 @@ void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long
|
||||
}
|
||||
|
||||
static bool allocation_allowed(const ProtocolSession* session, size_t size) {
|
||||
return (unsigned long long)size <= session->max_alloc;
|
||||
/* max_alloc == 0 is rsync's --max-alloc=0 "no limit". */
|
||||
return session->max_alloc == 0 || (unsigned long long)size <= session->max_alloc;
|
||||
}
|
||||
|
||||
static void* protocol_alloc_for_session(const ProtocolSession* session, size_t size) {
|
||||
@@ -213,6 +249,18 @@ SSL* io_get_ssl(void) {
|
||||
return io_ssl;
|
||||
}
|
||||
|
||||
unsigned long long protocol_bytes_written(void) {
|
||||
return atomic_load(&io_bytes_written);
|
||||
}
|
||||
|
||||
unsigned long long protocol_bytes_read(void) {
|
||||
return atomic_load(&io_bytes_read);
|
||||
}
|
||||
|
||||
void protocol_note_bytes_written(unsigned long long bytes) {
|
||||
atomic_fetch_add(&io_bytes_written, bytes);
|
||||
}
|
||||
|
||||
static ProtocolSession* legacy_session(int read_fd, int write_fd) {
|
||||
if (bound_session)
|
||||
return bound_session;
|
||||
@@ -257,10 +305,15 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
log_debug_message(LOG_DEBUG_IO, " Sending n Data: %zu", data_size);
|
||||
if (!session)
|
||||
return false;
|
||||
/* A non-positive session timeout disables the deadline entirely (rsync's
|
||||
* --timeout=0 default); poll then blocks until the socket becomes writable. */
|
||||
int timeout_sec = session->io_timeout_sec > 0 ? session->io_timeout_sec : 0;
|
||||
int fd = session->write_fd;
|
||||
struct timespec deadline;
|
||||
if (timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += SEND_TIMEOUT_SEC;
|
||||
deadline.tv_sec += timeout_sec;
|
||||
}
|
||||
short wait_events = POLLOUT;
|
||||
ssize_t total_bytes_send = 0;
|
||||
while ((size_t)total_bytes_send < data_size) {
|
||||
@@ -268,7 +321,7 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
if (session->bwlimit > 0 && chunk > 65536)
|
||||
chunk = 65536;
|
||||
struct pollfd pfd = {.fd = fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, deadline_remaining_ms(&deadline));
|
||||
int poll_result = poll(&pfd, 1, timeout_sec > 0 ? deadline_remaining_ms(&deadline) : -1);
|
||||
if (poll_result == 0 || (poll_result < 0 && errno != EINTR)) {
|
||||
log_message(LOG_LEVEL_ERROR, "Send timeout or poll failure");
|
||||
return false;
|
||||
@@ -278,10 +331,14 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
if (pfd.revents & (POLLERR | POLLNVAL))
|
||||
return false;
|
||||
ssize_t bytes_send;
|
||||
if (session->ssl)
|
||||
bytes_send = SSL_write(session->ssl, (const char*)data + total_bytes_send, chunk);
|
||||
else
|
||||
if (session->ssl) {
|
||||
/* SSL_write takes an int length; clamp a >INT_MAX request into chunks so
|
||||
* the size_t downcast can never truncate into a negative/partial write. */
|
||||
size_t ssl_chunk = chunk > (size_t)INT_MAX ? (size_t)INT_MAX : chunk;
|
||||
bytes_send = SSL_write(session->ssl, (const char*)data + total_bytes_send, (int)ssl_chunk);
|
||||
} else {
|
||||
bytes_send = write(fd, (const char*)data + total_bytes_send, chunk);
|
||||
}
|
||||
if (bytes_send <= 0) {
|
||||
if (session->ssl) {
|
||||
int ssl_err = SSL_get_error(session->ssl, (int)bytes_send);
|
||||
@@ -289,6 +346,12 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN;
|
||||
continue;
|
||||
}
|
||||
/* A signal (e.g. Ctrl-C) interrupts the blocking TLS write: retry so
|
||||
the send loop can observe the abort flag at the next checkpoint. */
|
||||
if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR)
|
||||
continue;
|
||||
} else if (errno == EINTR) {
|
||||
continue;
|
||||
}
|
||||
log_message(LOG_LEVEL_ERROR, "Could not send data");
|
||||
return false;
|
||||
@@ -299,37 +362,49 @@ bool protocol_send_n_data(ProtocolSession* session, const void* data, size_t dat
|
||||
wait_events = POLLOUT;
|
||||
}
|
||||
log_debug_message(LOG_DEBUG_IO, " Send n Data: %zu", total_bytes_send);
|
||||
atomic_fetch_add(&io_bytes_written, (unsigned long long)total_bytes_send);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size,
|
||||
int timeout_sec);
|
||||
static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, size_t data_size,
|
||||
const struct timespec* deadline);
|
||||
|
||||
bool protocol_receive_n_data(ProtocolSession* session, void* data, size_t data_size) {
|
||||
return protocol_receive_n_data_timed(session, data, data_size, RECEIVE_TIMEOUT_SEC);
|
||||
/* Honor the session's configured deadline. A non-positive value disables the
|
||||
* deadline (rsync's --timeout=0 default): wait without a poll timeout. The
|
||||
* explicit _timed variants keep their own 0 -> built-in-default contract. */
|
||||
if (!session)
|
||||
return false;
|
||||
if (session->io_timeout_sec <= 0)
|
||||
return protocol_receive_n_data_until(session, data, data_size, NULL);
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += session->io_timeout_sec;
|
||||
return protocol_receive_n_data_until(session, data, data_size, &deadline);
|
||||
}
|
||||
|
||||
bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size,
|
||||
int timeout_sec) {
|
||||
/* Read exactly `data_size` bytes from `session` before `deadline` elapses
|
||||
* (CLOCK_MONOTONIC). Shared by the ordinary timed primitive and the error-detail
|
||||
* body reader so the latter can clamp itself to whatever deadline its caller
|
||||
* already established instead of always applying the session's 60 s window. */
|
||||
static bool protocol_receive_n_data_until(ProtocolSession* session, void* data, size_t data_size,
|
||||
const struct timespec* deadline) {
|
||||
log_debug_message(LOG_DEBUG_IO, " Receiving n Data: %zu", data_size);
|
||||
if (!session)
|
||||
return false;
|
||||
int fd = session->read_fd;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
|
||||
size_t total_bytes_received = 0;
|
||||
short wait_events = POLLIN;
|
||||
while (total_bytes_received < data_size) {
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
struct pollfd pfd = {.fd = fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, deadline_remaining_ms(&deadline));
|
||||
/* A NULL deadline means "wait indefinitely" (timeout disabled). */
|
||||
int poll_result = poll(&pfd, 1, deadline ? deadline_remaining_ms(deadline) : -1);
|
||||
if (poll_result == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", timeout_sec);
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout");
|
||||
return false;
|
||||
}
|
||||
if (poll_result < 0) {
|
||||
@@ -356,6 +431,13 @@ bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t
|
||||
wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN;
|
||||
continue;
|
||||
}
|
||||
/* A signal interrupts the blocking TLS read: retry (mirrors the send
|
||||
path and protocol_read_status_until) so the loop reaches its next
|
||||
abort/deadline checkpoint instead of failing spuriously. */
|
||||
if (ssl_err == SSL_ERROR_SYSCALL && errno == EINTR)
|
||||
continue;
|
||||
} else if (errno == EINTR) {
|
||||
continue;
|
||||
}
|
||||
if (bytes_received == 0)
|
||||
log_message(LOG_LEVEL_ERROR, "Connection closed while receiving data");
|
||||
@@ -368,9 +450,22 @@ bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t
|
||||
wait_events = POLLIN;
|
||||
}
|
||||
log_debug_message(LOG_DEBUG_IO, " Received n Data: %zu", total_bytes_received);
|
||||
atomic_fetch_add(&io_bytes_read, (unsigned long long)total_bytes_received);
|
||||
return true;
|
||||
}
|
||||
|
||||
bool protocol_receive_n_data_timed(ProtocolSession* session, void* data, size_t data_size,
|
||||
int timeout_sec) {
|
||||
if (!session)
|
||||
return false;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
return protocol_receive_n_data_until(session, data, data_size, &deadline);
|
||||
}
|
||||
|
||||
static const char* status_to_string(Status status) {
|
||||
switch (status) {
|
||||
case STATUS_OK:
|
||||
@@ -421,6 +516,14 @@ static const char* status_to_string(Status status) {
|
||||
return "AUTH_OK";
|
||||
case STATUS_AUTH_FAILED:
|
||||
return "AUTH_FAILED";
|
||||
case STATUS_ERROR_DETAIL:
|
||||
return "ERROR_DETAIL";
|
||||
case STATUS_DRY_RUN_TRANSFER:
|
||||
return "DRY_RUN_TRANSFER";
|
||||
case STATUS_DELETE_LIMIT:
|
||||
return "DELETE_LIMIT";
|
||||
case STATUS_DEST_INFO:
|
||||
return "DEST_INFO";
|
||||
default:
|
||||
return "UNKNOWN";
|
||||
}
|
||||
@@ -550,13 +653,10 @@ Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long
|
||||
return NULL;
|
||||
}
|
||||
result->protocol_charge = allocation_size;
|
||||
result->owner = session;
|
||||
return result;
|
||||
}
|
||||
|
||||
Data* protocol_receive_data(ProtocolSession* session) {
|
||||
return protocol_receive_data_limited(session, MAX_DATA_PAYLOAD_SIZE);
|
||||
}
|
||||
|
||||
bool protocol_send_int(ProtocolSession* session, int data) {
|
||||
if (!protocol_send_n_data(session, &data, sizeof(int)))
|
||||
return false;
|
||||
@@ -578,8 +678,85 @@ bool protocol_send_status(ProtocolSession* session, Status status) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Read the bounded, length-prefixed body of a STATUS_ERROR_DETAIL frame within
|
||||
* `deadline` (CLOCK_MONOTONIC), polling `abort_check` (may be NULL) between
|
||||
* drain chunks. The declared length is validated BEFORE any allocation:
|
||||
*
|
||||
* - `size > MAX_STRING_SIZE`: an absurd framing error. Reading/draining that
|
||||
* many bytes could never finish, so it is fatal (the caller tears the
|
||||
* connection down) rather than drained.
|
||||
* - `MAX_ERROR_DETAIL_BYTES < size <= MAX_STRING_SIZE`: drain exactly `size`
|
||||
* bytes through a small fixed scratch buffer so the stream stays in sync,
|
||||
* leaving the captured detail empty. No allocation happens.
|
||||
* - `size <= MAX_ERROR_DETAIL_BYTES`: read straight into the thread-local
|
||||
* `io_error_detail` buffer (size+1 capacity, already reserved), so the
|
||||
* session's --max-alloc / MAX_CONNECTION_MEMORY budgets are never touched.
|
||||
*
|
||||
* Returns false on a fatal framing problem or any I/O failure; the terminal
|
||||
* detail is then empty. The body is consumed on every non-fatal path even when
|
||||
* the caller ignores protocol_last_error(), so the stream never desyncs. */
|
||||
static bool protocol_receive_error_detail_until(ProtocolSession* session,
|
||||
const struct timespec* deadline,
|
||||
ProtocolWaitAbort abort_check) {
|
||||
io_error_detail[0] = '\0';
|
||||
size_t size = 0;
|
||||
if (!protocol_receive_n_data_until(session, &size, sizeof(size), deadline))
|
||||
return false;
|
||||
if (size > MAX_STRING_SIZE) {
|
||||
log_message(LOG_LEVEL_ERROR, "Error detail length %zu exceeds maximum %llu", size,
|
||||
(unsigned long long)MAX_STRING_SIZE);
|
||||
return false;
|
||||
}
|
||||
if (size > MAX_ERROR_DETAIL_BYTES) {
|
||||
char scratch[256];
|
||||
size_t remaining = size;
|
||||
while (remaining > 0) {
|
||||
if (abort_check && abort_check())
|
||||
return false;
|
||||
size_t chunk = remaining < sizeof(scratch) ? remaining : sizeof(scratch);
|
||||
if (!protocol_receive_n_data_until(session, scratch, chunk, deadline))
|
||||
return false;
|
||||
remaining -= chunk;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
if (!protocol_receive_n_data_until(session, io_error_detail, size, deadline))
|
||||
return false;
|
||||
io_error_detail[size] = '\0';
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Consume the optional detail body of a STATUS_ERROR_DETAIL frame and map the
|
||||
* status back to STATUS_ERROR for existing callers. Invoked for EVERY status
|
||||
* read so a stale detail from an earlier exchange is never reported for a later
|
||||
* one -- except for STATUS_KEEPALIVE, which carries no body and whose drain
|
||||
* (protocol_receive_status_keepalive) must NOT erase the terminal detail that
|
||||
* arrived just before it. Returns false on a fatal framing error. */
|
||||
static bool protocol_capture_error_detail(ProtocolSession* session, Status* status,
|
||||
const struct timespec* deadline,
|
||||
ProtocolWaitAbort abort_check) {
|
||||
if (*status == STATUS_KEEPALIVE)
|
||||
return true;
|
||||
io_error_detail[0] = '\0';
|
||||
if (*status != STATUS_ERROR_DETAIL)
|
||||
return true;
|
||||
*status = STATUS_ERROR;
|
||||
return protocol_receive_error_detail_until(session, deadline, abort_check);
|
||||
}
|
||||
|
||||
bool protocol_receive_status(ProtocolSession* session, Status* status) {
|
||||
if (!protocol_receive_n_data(session, status, sizeof(Status)))
|
||||
if (!session || !status)
|
||||
return false;
|
||||
struct timespec deadline;
|
||||
const struct timespec* deadline_ptr = NULL;
|
||||
if (session->io_timeout_sec > 0) {
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += session->io_timeout_sec;
|
||||
deadline_ptr = &deadline;
|
||||
}
|
||||
if (!protocol_receive_n_data_until(session, status, sizeof(Status), deadline_ptr))
|
||||
return false;
|
||||
if (!protocol_capture_error_detail(session, status, deadline_ptr, NULL))
|
||||
return false;
|
||||
log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status));
|
||||
return true;
|
||||
@@ -588,10 +765,169 @@ bool protocol_receive_status(ProtocolSession* session, Status* status) {
|
||||
/* protocol_receive_status with an explicit per-message deadline (seconds).
|
||||
Used where a single reply may legitimately take far longer than the default
|
||||
60 s receive window - e.g. the sender waiting for the early-delete ACK after
|
||||
the receiver committed a large (up to MAX_SERVER_DELETE_COUNT) deletion. */
|
||||
the receiver committed a large (up to MAX_SERVER_DELETE_COUNT) deletion. The
|
||||
error-detail body shares the same deadline as the status header. */
|
||||
bool protocol_receive_status_timed(ProtocolSession* session, Status* status, int timeout_sec) {
|
||||
if (!protocol_receive_n_data_timed(session, status, sizeof(Status), timeout_sec))
|
||||
if (!session || !status)
|
||||
return false;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
if (!protocol_receive_n_data_until(session, status, sizeof(Status), &deadline))
|
||||
return false;
|
||||
if (!protocol_capture_error_detail(session, status, &deadline, NULL))
|
||||
return false;
|
||||
log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status));
|
||||
return true;
|
||||
}
|
||||
|
||||
/* Read exactly one Status frame within `deadline` (CLOCK_MONOTONIC). Unlike
|
||||
* protocol_receive_status_keepalive this never emits a keepalive: it is used
|
||||
* to consume the first byte(s) of an already-signalled frame and to drain the
|
||||
* peer's outstanding keepalive replies, where injecting a write could split a
|
||||
* reply across a frame boundary. Returns false on timeout/EOF/error. */
|
||||
static bool protocol_read_status_until(ProtocolSession* session, Status* status,
|
||||
const struct timespec* deadline) {
|
||||
Status received = STATUS_ERROR;
|
||||
size_t got = 0;
|
||||
short wait_events = POLLIN;
|
||||
while (got < sizeof(Status)) {
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
int remaining_ms = deadline ? deadline_remaining_ms(deadline) : -1;
|
||||
if (remaining_ms == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status");
|
||||
return false;
|
||||
}
|
||||
struct pollfd pfd = {.fd = session->read_fd, .events = wait_events};
|
||||
int poll_result = poll(&pfd, 1, remaining_ms);
|
||||
if (poll_result == 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout while reading status");
|
||||
return false;
|
||||
}
|
||||
if (poll_result < 0) {
|
||||
if (errno == EINTR)
|
||||
continue;
|
||||
return false;
|
||||
}
|
||||
if (pfd.revents & (POLLERR | POLLNVAL))
|
||||
return false;
|
||||
}
|
||||
ssize_t bytes_received;
|
||||
if (session->ssl)
|
||||
bytes_received = SSL_read(session->ssl, (char*)&received + got, sizeof(Status) - got);
|
||||
else
|
||||
bytes_received = read(session->read_fd, (char*)&received + got, sizeof(Status) - got);
|
||||
if (bytes_received <= 0) {
|
||||
if (session->ssl) {
|
||||
int ssl_err = SSL_get_error(session->ssl, (int)bytes_received);
|
||||
if (ssl_err == SSL_ERROR_WANT_READ || ssl_err == SSL_ERROR_WANT_WRITE) {
|
||||
wait_events = ssl_err == SSL_ERROR_WANT_WRITE ? POLLOUT : POLLIN;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
if (bytes_received < 0 && errno == EINTR)
|
||||
continue;
|
||||
log_message(LOG_LEVEL_ERROR, "Connection closed while receiving status");
|
||||
return false;
|
||||
}
|
||||
got += (size_t)bytes_received;
|
||||
}
|
||||
*status = received;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, int timeout_sec,
|
||||
int keepalive_interval_sec, ProtocolWaitAbort abort_check) {
|
||||
if (!session || !status)
|
||||
return false;
|
||||
if (timeout_sec <= 0)
|
||||
timeout_sec = RECEIVE_TIMEOUT_SEC;
|
||||
if (keepalive_interval_sec <= 0)
|
||||
keepalive_interval_sec = timeout_sec;
|
||||
|
||||
struct timespec deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &deadline);
|
||||
deadline.tv_sec += timeout_sec;
|
||||
|
||||
unsigned long keepalives_sent = 0;
|
||||
unsigned long replies_seen = 0;
|
||||
Status final = STATUS_ERROR;
|
||||
while (true) {
|
||||
if (abort_check && abort_check())
|
||||
return false;
|
||||
if (!session->ssl || SSL_pending(session->ssl) == 0) {
|
||||
int remaining_ms = deadline_remaining_ms(&deadline);
|
||||
if (remaining_ms <= 0) {
|
||||
log_message(LOG_LEVEL_ERROR, "Receive timeout after %ds", timeout_sec);
|
||||
return false;
|
||||
}
|
||||
/* Only interleave a keepalive while waiting for the FIRST byte of a
|
||||
* frame; once part of a frame is buffered a write could race the peer's
|
||||
* reply into the middle of it. */
|
||||
long long interval_ms_ll = (long long)keepalive_interval_sec * 1000LL;
|
||||
int interval_ms = interval_ms_ll > INT_MAX ? INT_MAX : (int)interval_ms_ll;
|
||||
int wait_ms = interval_ms < remaining_ms ? interval_ms : remaining_ms;
|
||||
struct pollfd pfd = {.fd = session->read_fd, .events = POLLIN};
|
||||
int poll_result = poll(&pfd, 1, wait_ms);
|
||||
if (poll_result == 0) {
|
||||
if (abort_check && abort_check())
|
||||
return false;
|
||||
if (!protocol_send_status(session, STATUS_KEEPALIVE))
|
||||
return false;
|
||||
keepalives_sent++;
|
||||
continue;
|
||||
}
|
||||
if (poll_result < 0) {
|
||||
if (errno == EINTR)
|
||||
continue;
|
||||
return false;
|
||||
}
|
||||
if (pfd.revents & (POLLERR | POLLNVAL))
|
||||
return false;
|
||||
}
|
||||
Status received;
|
||||
if (!protocol_read_status_until(session, &received, &deadline))
|
||||
return false;
|
||||
if (!protocol_capture_error_detail(session, &received, &deadline, abort_check))
|
||||
return false;
|
||||
if (received == STATUS_KEEPALIVE) {
|
||||
/* The receiver's answer to one of our keepalives. */
|
||||
replies_seen++;
|
||||
continue;
|
||||
}
|
||||
final = received;
|
||||
break;
|
||||
}
|
||||
/* Drain the replies the receiver still owes for keepalives we sent while it
|
||||
* was busy. It answers them only after the real status, so leaving them
|
||||
* unread would put stale KEEPALIVE frames ahead of the next exchange and
|
||||
* desynchronize the protocol. */
|
||||
if (replies_seen < keepalives_sent) {
|
||||
/* A short separate grace, not the (possibly exhausted) main deadline: the
|
||||
terminal status already arrived, so a peer that never answers its owed
|
||||
keepalives must not turn a successful ack into a reported failure. */
|
||||
struct timespec drain_deadline;
|
||||
clock_gettime(CLOCK_MONOTONIC, &drain_deadline);
|
||||
drain_deadline.tv_sec += 1;
|
||||
while (replies_seen < keepalives_sent) {
|
||||
Status drained;
|
||||
if (!protocol_read_status_until(session, &drained, &drain_deadline)) {
|
||||
log_message(LOG_LEVEL_WARNING, "peer did not answer %lu keepalive(s); continuing",
|
||||
keepalives_sent - replies_seen);
|
||||
break;
|
||||
}
|
||||
if (!protocol_capture_error_detail(session, &drained, &drain_deadline, abort_check))
|
||||
return false;
|
||||
if (drained != STATUS_KEEPALIVE) {
|
||||
log_message(LOG_LEVEL_ERROR, "Unexpected status while draining keepalive replies");
|
||||
return false;
|
||||
}
|
||||
replies_seen++;
|
||||
}
|
||||
}
|
||||
*status = final;
|
||||
log_debug_message(LOG_DEBUG_PROTO, "Received Status: %s", status_to_string(*status));
|
||||
return true;
|
||||
}
|
||||
@@ -635,3 +971,29 @@ bool receive_status(int fd, Status* status) {
|
||||
bool receive_status_timed(int fd, Status* status, int timeout_sec) {
|
||||
return protocol_receive_status_timed(legacy_session(fd, -1), status, timeout_sec);
|
||||
}
|
||||
bool receive_status_keepalive(int fd, Status* status, int timeout_sec, int keepalive_interval_sec,
|
||||
ProtocolWaitAbort abort_check) {
|
||||
return protocol_receive_status_keepalive(legacy_session(fd, -1), status, timeout_sec,
|
||||
keepalive_interval_sec, abort_check);
|
||||
}
|
||||
|
||||
bool send_error_detail(int fd, const char* message) {
|
||||
if (!message)
|
||||
message = "";
|
||||
char bounded[MAX_ERROR_DETAIL_BYTES + 1];
|
||||
size_t len = strlen(message);
|
||||
if (len > MAX_ERROR_DETAIL_BYTES) {
|
||||
memcpy(bounded, message, MAX_ERROR_DETAIL_BYTES);
|
||||
bounded[MAX_ERROR_DETAIL_BYTES] = '\0';
|
||||
message = bounded;
|
||||
}
|
||||
return send_status(fd, STATUS_ERROR_DETAIL) && send_str(fd, message);
|
||||
}
|
||||
|
||||
const char* protocol_last_error(void) {
|
||||
return io_error_detail;
|
||||
}
|
||||
|
||||
void protocol_clear_last_error(void) {
|
||||
io_error_detail[0] = '\0';
|
||||
}
|
||||
|
||||
+132
-2
@@ -9,6 +9,12 @@
|
||||
/* Maximum allowed string size for receive_str (64 KB) */
|
||||
#define MAX_STRING_SIZE (64 * 1024)
|
||||
|
||||
/* Hard cap on the optional server->client rejection detail carried by
|
||||
* STATUS_ERROR_DETAIL (protocol 2.21.0). A longer message is sliced to this
|
||||
* many bytes before it is sent, so a peer can never be made to retain more than
|
||||
* this for a rejection and the detail frame stays a small, fixed bound. */
|
||||
#define MAX_ERROR_DETAIL_BYTES 4096
|
||||
|
||||
/* Maximum uncompressed file payload accepted by the receiver's whole-file
|
||||
* paths. A single whole file is charged against the per-connection memory
|
||||
* reservation (MAX_CONNECTION_MEMORY) and against the server allocation
|
||||
@@ -28,6 +34,11 @@
|
||||
#define DEFAULT_MAX_ALLOC (1ULL * 1024 * 1024 * 1024)
|
||||
/* Server policy ceiling for a client-provided allocation limit. */
|
||||
#define MAX_SERVER_ALLOC (256ULL * 1024 * 1024)
|
||||
/* Server-owned floor for the per-message I/O deadline. A client --timeout=0
|
||||
(rsync's default) disables the client's own deadlines, but a server session
|
||||
must never be held open forever by a silent peer (slow-loris), so the server
|
||||
floors the effective deadline at this value. */
|
||||
#define SERVER_IO_TIMEOUT_SEC 60
|
||||
/* Bounded cumulative per-connection receive budget. In-flight wire buffers,
|
||||
decompression buffers and queued (not yet written) file payloads for a
|
||||
connection must stay within this ceiling. */
|
||||
@@ -52,6 +63,14 @@ typedef struct ProtocolSession {
|
||||
atomic_ullong total_allocated_bytes;
|
||||
bool eight_bit_output;
|
||||
unsigned long long max_alloc;
|
||||
/* Per-session deadline (seconds) applied to every protocol send/receive by
|
||||
* protocol_send_n_data / protocol_receive_n_data. The initialized default is
|
||||
* the built-in 60 s window; a value <= 0 disables the deadline (rsync's
|
||||
* --timeout=0). Set from the negotiated Config->timeout so --timeout is
|
||||
* honored by the poll()-driven protocol I/O, not just the socket
|
||||
* SO_RCVTIMEO/SO_SNDTIMEO. The server does not propagate a client 0 here: it
|
||||
* installs protocol_server_io_timeout_sec() so its sessions keep a floor. */
|
||||
int io_timeout_sec;
|
||||
} ProtocolSession;
|
||||
|
||||
typedef int Status;
|
||||
@@ -126,7 +145,64 @@ enum NET_STATUS {
|
||||
STATUS_AUTH_CHALLENGE,
|
||||
STATUS_AUTH_RESPONSE,
|
||||
STATUS_AUTH_OK,
|
||||
STATUS_AUTH_FAILED
|
||||
STATUS_AUTH_FAILED,
|
||||
/* Optional server->client rejection detail (protocol 2.21.0). When the
|
||||
* server refuses a transfer for a concrete reason it may send
|
||||
* STATUS_ERROR_DETAIL followed by a length-prefixed, bounded string instead
|
||||
* of a bare STATUS_ERROR. receive_status() consumes the string and maps the
|
||||
* status back to STATUS_ERROR, so every pre-2.21 call site keeps working;
|
||||
* callers that want the human-readable reason consult protocol_last_error().
|
||||
* Appended immediately after STATUS_AUTH_FAILED so the existing wire values
|
||||
* never move. */
|
||||
STATUS_ERROR_DETAIL,
|
||||
/* Server-contacting --dry-run (protocol 2.21.0). Sent by the receiver in
|
||||
* response to a per-file STATUS_CHECK when the wire config carries
|
||||
* dry_run=true and the file is NOT already up to date: it tells the sender
|
||||
* the file WOULD be transferred, and the sender must NOT transmit any data
|
||||
* (the receiver reads none in dry-run). STATUS_OK keeps its meaning in this
|
||||
* path ("already up to date / nothing to do"). Appended after
|
||||
* STATUS_ERROR_DETAIL so no existing status is renumbered. */
|
||||
STATUS_DRY_RUN_TRANSFER,
|
||||
/* --max-delete budget exhausted (protocol 2.23.0). Sent by the receiver as
|
||||
* the terminal success status INSTEAD of STATUS_OK when a --delete/
|
||||
* --delete-missing-args commit removed up to the --max-delete bound but had
|
||||
* to skip further extras. The transfer itself succeeded and all file data is
|
||||
* stored; the sender maps this to rsync's exit code 25 ("the --max-delete
|
||||
* limit stopped deletions"). Appended after STATUS_DRY_RUN_TRANSFER so no
|
||||
* existing status is renumbered. */
|
||||
STATUS_DELETE_LIMIT,
|
||||
/* Destination-state report for output parity (protocol 2.23.0). When the
|
||||
* wire config carries report_dest_info=true, the receiver answers every
|
||||
* per-file STATUS_CHECK request with STATUS_DEST_INFO FIRST, followed by a
|
||||
* fixed record describing the pre-transfer destination entry
|
||||
* (int32 has_old; uint64 size; int64 mtime; int64 mtime_nsec; uint32 mode;
|
||||
* int32 uid; int32 gid). The ordinary STATUS_OK/STATUS_NEXT/... verdict
|
||||
* follows, so the sender can render rsync-accurate -i/--out-format columns
|
||||
* (new vs modified, and which of size/time/perms/owner/group differ) without
|
||||
* changing the transfer decision itself. Appended after
|
||||
* STATUS_DELETE_LIMIT so no existing status is renumbered. */
|
||||
STATUS_DEST_INFO,
|
||||
/* Per-directory delete plan (protocol 2.24.0). The sender of a
|
||||
* --delete-during/--delete-delay transfer streams one frame per source
|
||||
* directory in directory order instead of a single whole-tree keep-set
|
||||
* manifest. The receiver applies the plan when it arrives
|
||||
* (--delete-during removes that directory's extras immediately) or records
|
||||
* the extras and applies them only after the whole transfer succeeded
|
||||
* (--delete-delay). Payload: an int32 has_config flag (1 on the first plan
|
||||
* of the run, 0 afterwards); when set, the three global config sections
|
||||
* (protected-prefix count+paths, size-skipped count+paths, missing-args
|
||||
* count+paths); then the destination-relative directory path wire string
|
||||
* ("." for the receive root); then the child-directory count + names and the
|
||||
* child-file count + names that must be kept. Appended after
|
||||
* STATUS_DEST_INFO so no existing status is renumbered. */
|
||||
STATUS_DELETE_PLAN,
|
||||
/* End-of-transfer receiver counter report (protocol 2.25.0). When the wire
|
||||
* config carries report_stats=true, the receiver sends this status once,
|
||||
* immediately before its terminal success status, followed by a fixed stats
|
||||
* record (see format_stats_send/receive in format.h) and, when the run is a
|
||||
* --dry-run with --delete, the would-delete path list. Appended after
|
||||
* STATUS_DELETE_PLAN so no existing status is renumbered. */
|
||||
STATUS_STATS
|
||||
};
|
||||
|
||||
void io_set_fds(int read_fd, int write_fd);
|
||||
@@ -134,6 +210,14 @@ void io_set_bwlimit(unsigned long long bytes_per_sec);
|
||||
void io_set_ssl(SSL* ssl);
|
||||
SSL* io_get_ssl(void);
|
||||
|
||||
/* Process-wide wire byte counters. protocol_send_n_data/protocol_receive_n_data
|
||||
* update them; the zero-copy sendfile path reports through
|
||||
* protocol_note_bytes_written. Used by the client to render rsync's
|
||||
* --stats/--progress totals and the --out-format %b/%c tokens. */
|
||||
unsigned long long protocol_bytes_written(void);
|
||||
unsigned long long protocol_bytes_read(void);
|
||||
void protocol_note_bytes_written(unsigned long long bytes);
|
||||
|
||||
void protocol_session_init(ProtocolSession* session, int read_fd, int write_fd);
|
||||
/* Transitional bridge for helpers whose signatures still carry only an fd. */
|
||||
void protocol_session_bind(ProtocolSession* session);
|
||||
@@ -141,6 +225,20 @@ void protocol_session_unbind(void);
|
||||
void protocol_session_set_ssl(ProtocolSession* session, SSL* ssl);
|
||||
void protocol_session_set_bwlimit(ProtocolSession* session, unsigned long long bytes_per_sec);
|
||||
void protocol_session_set_max_alloc(ProtocolSession* session, unsigned long long max_alloc);
|
||||
/* Override the per-message send/receive deadline for this session. The value
|
||||
* is stored verbatim: a positive value sets the deadline, `sec` <= 0 disables
|
||||
* it (rsync's --timeout=0). An explicit long deadline (e.g. the delete-ack
|
||||
* wait) is applied per-call by protocol_receive_status_timed and is unaffected
|
||||
* by this setter. */
|
||||
void protocol_session_set_io_timeout(ProtocolSession* session, int sec);
|
||||
/* Effective per-message I/O deadline (seconds) for the currently-bound session.
|
||||
* Zero means the deadline is disabled (rsync's --timeout=0). Used by the
|
||||
* plaintext sendfile path which bypasses the protocol send primitive. */
|
||||
int protocol_get_io_timeout_sec(void);
|
||||
/* The server-side effective deadline for a client-requested timeout: a positive
|
||||
* client value is honored, otherwise the SERVER_IO_TIMEOUT_SEC floor applies so
|
||||
* a silent peer can never hold a session open forever. */
|
||||
int protocol_server_io_timeout_sec(int client_timeout);
|
||||
void* protocol_alloc(size_t size);
|
||||
void* protocol_realloc(void* ptr, size_t size);
|
||||
void protocol_session_set_8_bit_output(ProtocolSession* session, bool enabled);
|
||||
@@ -157,7 +255,6 @@ char* protocol_receive_str(ProtocolSession* session);
|
||||
bool protocol_send_str_redacted(ProtocolSession* session, const char* data);
|
||||
char* protocol_receive_str_redacted(ProtocolSession* session);
|
||||
bool protocol_send_data(ProtocolSession* session, const Data* data);
|
||||
Data* protocol_receive_data(ProtocolSession* session);
|
||||
Data* protocol_receive_data_limited(ProtocolSession* session, unsigned long long maximum_size);
|
||||
bool protocol_send_int(ProtocolSession* session, int data);
|
||||
bool protocol_receive_int(ProtocolSession* session, int* data);
|
||||
@@ -178,10 +275,43 @@ bool send_int(int file_descriptor, int data);
|
||||
bool receive_int(int file_descriptor, int* data);
|
||||
bool send_status(int file_descriptor, Status status);
|
||||
bool receive_status(int file_descriptor, Status* status);
|
||||
/* Send STATUS_ERROR_DETAIL followed by a bounded (<= MAX_ERROR_DETAIL_BYTES)
|
||||
* length-prefixed string. Over-long messages are sliced and NULL is treated
|
||||
* as "". Returns false if the status or the string could not be sent. */
|
||||
bool send_error_detail(int file_descriptor, const char* message);
|
||||
/* Human-readable reason captured from the most recent STATUS_ERROR_DETAIL
|
||||
* received on this thread, or "" when the last status was a bare STATUS_ERROR
|
||||
* (or no detail was seen). Thread-local, and valid until the next non-keepalive
|
||||
* status read on the same thread; a later STATUS_KEEPALIVE does NOT clear it.
|
||||
* The detail body is bounded by MAX_ERROR_DETAIL_BYTES: an over-cap declared
|
||||
* length is drained and yields "" (so the stream never desyncs), while an
|
||||
* absurd length is a fatal framing error that fails the status read. */
|
||||
const char* protocol_last_error(void);
|
||||
/* Clear the thread-local last-error buffer. */
|
||||
void protocol_clear_last_error(void);
|
||||
/* receive_status with an explicit per-message deadline in seconds, instead of
|
||||
the default RECEIVE_TIMEOUT_SEC. A reply that may legitimately take longer
|
||||
(e.g. the early-delete ACK after a large receiver-side deletion) must use
|
||||
this so the sender does not abort after the deletion already committed. */
|
||||
bool receive_status_timed(int file_descriptor, Status* status, int timeout_sec);
|
||||
|
||||
/* Callback polled by protocol_receive_status_keepalive once per keepalive
|
||||
interval. Return true to stop waiting (e.g. a SIGINT/SIGTERM abort flag was
|
||||
set). Kept as a function pointer so the protocol layer does not depend on
|
||||
client signal state. */
|
||||
typedef bool (*ProtocolWaitAbort)(void);
|
||||
|
||||
/* Like receive_status_timed, but while the peer is silent it emits
|
||||
STATUS_KEEPALIVE every keepalive_interval_sec (the receiver answers each with
|
||||
STATUS_KEEPALIVE, which this function consumes and skips) so a long
|
||||
server-side operation does not look like a dead connection. The total wait
|
||||
is still bounded by timeout_sec; abort_check (may be NULL) is polled every
|
||||
interval and, when it returns true, ends the wait immediately with false.
|
||||
Runs entirely on the calling thread: the protocol send path is NOT safe for
|
||||
concurrent writers, so this must not be paired with a helper thread. */
|
||||
bool receive_status_keepalive(int file_descriptor, Status* status, int timeout_sec,
|
||||
int keepalive_interval_sec, ProtocolWaitAbort abort_check);
|
||||
bool protocol_receive_status_keepalive(ProtocolSession* session, Status* status, int timeout_sec,
|
||||
int keepalive_interval_sec, ProtocolWaitAbort abort_check);
|
||||
|
||||
#endif
|
||||
|
||||
+181
-21
@@ -44,6 +44,181 @@ static bool parse_two_digits(const char* s, int* out) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* True when the current character of the cursor is a decimal digit. */
|
||||
static bool is_digit(const char* cp) {
|
||||
return *cp >= '0' && *cp <= '9';
|
||||
}
|
||||
|
||||
/* rsync 3.4.1's flexible --stop-at date parser (ported from
|
||||
* options.c:parse_time). Returns a time_t, or (time_t)-1 on a malformed value.
|
||||
* Accepted forms include Y-M-DTh:m, Y/M/DTh:m, Y-M-D, M-D, D, h:m, :m and
|
||||
* "T h:m"; a 1- or 2-digit year and omitted fields are resolved to the next
|
||||
* matching point in time in the local timezone. Seconds are NOT accepted
|
||||
* (rsync rejects them too); FastSync keeps its own HH:MM:SS spelling as an
|
||||
* extension handled by the caller. `now` is passed in so tests are
|
||||
* deterministic; production passes time(NULL). */
|
||||
static time_t parse_time_rsync(const char* value, time_t now) {
|
||||
const char* cp;
|
||||
time_t val;
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return (time_t)-1;
|
||||
struct tm t;
|
||||
int in_date, old_mday, n;
|
||||
|
||||
memset(&t, 0, sizeof t);
|
||||
t.tm_year = t.tm_mon = t.tm_mday = -1;
|
||||
t.tm_hour = t.tm_min = t.tm_isdst = -1;
|
||||
cp = value;
|
||||
if (*cp == 'T' || *cp == 't' || *cp == ':') {
|
||||
in_date = *cp == ':' ? 0 : -1;
|
||||
cp++;
|
||||
} else
|
||||
in_date = 1;
|
||||
for (;; cp++) {
|
||||
if (!is_digit(cp))
|
||||
return (time_t)-1;
|
||||
n = 0;
|
||||
do {
|
||||
n = n * 10 + *cp++ - '0';
|
||||
} while (is_digit(cp));
|
||||
if (*cp == ':')
|
||||
in_date = 0;
|
||||
if (in_date > 0) {
|
||||
if (t.tm_year != -1)
|
||||
return (time_t)-1;
|
||||
t.tm_year = t.tm_mon;
|
||||
t.tm_mon = t.tm_mday;
|
||||
t.tm_mday = n;
|
||||
if (!*cp)
|
||||
break;
|
||||
if (*cp == 'T' || *cp == 't') {
|
||||
if (!cp[1])
|
||||
break;
|
||||
in_date = -1;
|
||||
} else if (*cp != '-' && *cp != '/')
|
||||
return (time_t)-1;
|
||||
continue;
|
||||
}
|
||||
if (t.tm_hour != -1)
|
||||
return (time_t)-1;
|
||||
t.tm_hour = t.tm_min;
|
||||
t.tm_min = n;
|
||||
if (!*cp) {
|
||||
if (in_date < 0)
|
||||
return (time_t)-1;
|
||||
break;
|
||||
}
|
||||
if (*cp != ':')
|
||||
return (time_t)-1;
|
||||
in_date = 0;
|
||||
}
|
||||
|
||||
in_date = 0;
|
||||
if (t.tm_year < 0) {
|
||||
t.tm_year = today.tm_year;
|
||||
in_date = 1;
|
||||
} else if (t.tm_year < 100) {
|
||||
while (t.tm_year < today.tm_year)
|
||||
t.tm_year += 100;
|
||||
} else
|
||||
t.tm_year -= 1900;
|
||||
if (t.tm_mon < 0) {
|
||||
t.tm_mon = today.tm_mon;
|
||||
in_date = 2;
|
||||
} else
|
||||
t.tm_mon--;
|
||||
if (t.tm_mday < 0) {
|
||||
t.tm_mday = today.tm_mday;
|
||||
in_date = 3;
|
||||
}
|
||||
|
||||
n = 0;
|
||||
if (t.tm_min < 0) {
|
||||
t.tm_hour = t.tm_min = 0;
|
||||
} else if (t.tm_hour < 0) {
|
||||
if (in_date != 3)
|
||||
return (time_t)-1;
|
||||
in_date = 0;
|
||||
t.tm_hour = today.tm_hour;
|
||||
n = 60 * 60;
|
||||
}
|
||||
|
||||
/* mktime() may roll a too-large tm_mday into the following month; undo that
|
||||
* in the "next match" loop below. */
|
||||
old_mday = t.tm_mday;
|
||||
if (t.tm_hour > 23 || t.tm_min > 59 || t.tm_mon < 0 || t.tm_mon >= 12 || t.tm_mday < 1 ||
|
||||
t.tm_mday > 31 || (val = mktime(&t)) == (time_t)-1)
|
||||
return (time_t)-1;
|
||||
|
||||
while (in_date && (val <= now || t.tm_mday < old_mday)) {
|
||||
switch (in_date) {
|
||||
case 3:
|
||||
old_mday = ++t.tm_mday;
|
||||
break;
|
||||
case 2:
|
||||
if (t.tm_mday < old_mday)
|
||||
t.tm_mday = old_mday; /* the month already got bumped forward */
|
||||
else if (++t.tm_mon == 12) {
|
||||
t.tm_mon = 0;
|
||||
t.tm_year++;
|
||||
}
|
||||
break;
|
||||
case 1:
|
||||
if (t.tm_mday < old_mday) {
|
||||
/* mon==1 mday==29 got bumped to mon==2 */
|
||||
if (t.tm_mon != 2 || old_mday != 29)
|
||||
return (time_t)-1;
|
||||
t.tm_mon = 1;
|
||||
t.tm_mday = 29;
|
||||
}
|
||||
t.tm_year++;
|
||||
break;
|
||||
}
|
||||
if ((val = mktime(&t)) == (time_t)-1) {
|
||||
if (in_date != 3 || t.tm_mday <= 28)
|
||||
return (time_t)-1;
|
||||
t.tm_mday = old_mday = 1;
|
||||
in_date = 2;
|
||||
}
|
||||
}
|
||||
if (n) {
|
||||
while (val <= now)
|
||||
val += n;
|
||||
}
|
||||
return val;
|
||||
}
|
||||
|
||||
/* FastSync's HH:MM or HH:MM:SS spelling on the current local day. rsync's own
|
||||
* --stop-at accepts only HH:MM, so this is a strict superset extension. */
|
||||
static bool parse_clock_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
size_t len = strlen(value);
|
||||
if (len != 5 && len != 8)
|
||||
return false;
|
||||
if (value[2] != ':' || (len == 8 && value[5] != ':'))
|
||||
return false;
|
||||
int hh, mm, ss = 0;
|
||||
if (!parse_two_digits(value, &hh) || !parse_two_digits(value + 3, &mm))
|
||||
return false;
|
||||
if (len == 8 && !parse_two_digits(value + 6, &ss))
|
||||
return false;
|
||||
if (hh > 23 || mm > 59 || ss > 59)
|
||||
return false;
|
||||
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return false;
|
||||
today.tm_hour = hh;
|
||||
today.tm_min = mm;
|
||||
today.tm_sec = ss;
|
||||
today.tm_isdst = -1;
|
||||
time_t deadline = mktime(&today);
|
||||
if (deadline == (time_t)-1)
|
||||
return false;
|
||||
*out_deadline = deadline;
|
||||
return true;
|
||||
}
|
||||
|
||||
bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
if (!value || !out_deadline)
|
||||
return false;
|
||||
@@ -91,28 +266,13 @@ bool stop_parse_at_time(const char* value, time_t now, time_t* out_deadline) {
|
||||
return true;
|
||||
}
|
||||
|
||||
/* HH:MM or HH:MM:SS on the current local day. */
|
||||
size_t len = strlen(value);
|
||||
if (len != 5 && len != 8)
|
||||
return false;
|
||||
if (value[2] != ':' || (len == 8 && value[5] != ':'))
|
||||
return false;
|
||||
int hh, mm, ss = 0;
|
||||
if (!parse_two_digits(value, &hh) || !parse_two_digits(value + 3, &mm))
|
||||
return false;
|
||||
if (len == 8 && !parse_two_digits(value + 6, &ss))
|
||||
return false;
|
||||
if (hh > 23 || mm > 59 || ss > 59)
|
||||
return false;
|
||||
/* HH:MM or HH:MM:SS on the current local day (FastSync extension). */
|
||||
if (parse_clock_time(value, now, out_deadline))
|
||||
return true;
|
||||
|
||||
struct tm today;
|
||||
if (!localtime_r(&now, &today))
|
||||
return false;
|
||||
today.tm_hour = hh;
|
||||
today.tm_min = mm;
|
||||
today.tm_sec = ss;
|
||||
today.tm_isdst = -1;
|
||||
time_t deadline = mktime(&today);
|
||||
/* rsync's full/partial date-and-time form (e.g. 2000-12-31T23:59, 12-31,
|
||||
* 14:00, :59, 1, 1-30). */
|
||||
time_t deadline = parse_time_rsync(value, now);
|
||||
if (deadline == (time_t)-1)
|
||||
return false;
|
||||
*out_deadline = deadline;
|
||||
|
||||
@@ -75,6 +75,15 @@ static int parse_remote_dest(const char* dest, RemoteDest* r) {
|
||||
memcpy(r->host, dest, host_len);
|
||||
r->host[host_len] = '\0';
|
||||
}
|
||||
/* The user@host token is handed to ssh in option position. Reject anything
|
||||
* that ssh would consume as an option (a leading '-') or an empty host, so a
|
||||
* crafted destination can never inject an ssh option such as
|
||||
* -oProxyCommand=... . This mirrors config_parse_ssh_dest's validation and
|
||||
* is defense-in-depth for callers that bypass it. */
|
||||
if (r->host[0] == '\0' || r->host[0] == '-' || r->user[0] == '-') {
|
||||
remote_dest_destroy(r);
|
||||
return -1;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
@@ -128,7 +137,7 @@ char* ssh_build_remote_command(const char* server_path, bool old_args, char* con
|
||||
q++;
|
||||
len++;
|
||||
}
|
||||
if (len > SIZE_MAX - q * 3 || len + q * 3 + 3 > SIZE_MAX - command_len)
|
||||
if (q > (SIZE_MAX - len) / 3 || len + q * 3 + 3 > SIZE_MAX - command_len)
|
||||
return NULL;
|
||||
command_len += len + q * 3 + 3;
|
||||
}
|
||||
@@ -216,10 +225,10 @@ char** ssh_build_client_argv(const char* rsh_command, int port, const char* user
|
||||
nwords = 1;
|
||||
}
|
||||
|
||||
/* Fixed tail: three -o pairs (6) + optional -p/value (2) + user@host +
|
||||
* remote command + terminating NULL. */
|
||||
/* Fixed tail: three -o pairs (6) + optional -p/value (2) + the "--" end of
|
||||
* options marker + user@host + remote command + terminating NULL. */
|
||||
int port_extra = (port > 0 && port != 22) ? 2 : 0;
|
||||
size_t total = (size_t)nwords + 6 + (size_t)port_extra + 3;
|
||||
size_t total = (size_t)nwords + 6 + (size_t)port_extra + 4;
|
||||
char** argv = calloc(total, sizeof(char*));
|
||||
if (!argv) {
|
||||
for (int i = 0; i < nwords; i++)
|
||||
@@ -253,6 +262,13 @@ char** ssh_build_client_argv(const char* rsh_command, int port, const char* user
|
||||
goto fail_argv;
|
||||
ac++;
|
||||
}
|
||||
/* End of options: guarantees the user@host token that follows is treated as
|
||||
* the destination and never re-interpreted as an ssh option, even if every
|
||||
* caller-side validation were bypassed. */
|
||||
argv[ac] = str_dup("--");
|
||||
if (!argv[ac])
|
||||
goto fail_argv;
|
||||
ac++;
|
||||
argv[ac] = str_dup(userhost);
|
||||
if (!argv[ac])
|
||||
goto fail_argv;
|
||||
|
||||
+141
-17
@@ -1,4 +1,5 @@
|
||||
#include "transport_tcp.h"
|
||||
#include "daemon_limits.h"
|
||||
#include "log.h"
|
||||
#include "protocol.h"
|
||||
#include "utils.h"
|
||||
@@ -8,6 +9,7 @@
|
||||
#include <netinet/in.h>
|
||||
#include <netinet/tcp.h>
|
||||
#include <openssl/ssl.h>
|
||||
#include <pthread.h>
|
||||
#include <signal.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
@@ -18,18 +20,45 @@
|
||||
|
||||
static volatile sig_atomic_t g_active_connections = 0;
|
||||
|
||||
/* Shared registry installed on the active server; the SIGCHLD handler needs a
|
||||
* file-scope pointer so it can reclaim the dead child's slot. Set once by
|
||||
* accept_loop before the fork loop (single-threaded parent). */
|
||||
static DaemonLimitRegistry* g_limit_registry = NULL;
|
||||
/* Slot reserved by the parent for the connection child currently being forked.
|
||||
* Written before fork(), read by the child (which inherits the value). */
|
||||
static int g_current_slot = DAEMON_LIMITS_NO_SLOT;
|
||||
|
||||
static void tcp_apply_socket_timeout(int fd);
|
||||
static void tcp_enable_nodelay_default(int fd, int family);
|
||||
|
||||
static void sigchld_handler(int sig) {
|
||||
(void)sig;
|
||||
int saved_errno = errno;
|
||||
while (waitpid(-1, NULL, WNOHANG) > 0) {
|
||||
pid_t pid;
|
||||
while ((pid = waitpid(-1, NULL, WNOHANG)) > 0) {
|
||||
if (g_active_connections > 0)
|
||||
g_active_connections--;
|
||||
daemon_limits_reclaim_pid(g_limit_registry, (long)pid);
|
||||
}
|
||||
/* Re-derive the occupancy counters once for the whole reap batch. The slot
|
||||
* table is the source of truth, so this self-heals any count leaked by a child
|
||||
* SIGKILLed mid-registration. Atomics only: async-signal-safe. */
|
||||
if (g_limit_registry)
|
||||
daemon_limits_recompute(g_limit_registry);
|
||||
errno = saved_errno;
|
||||
}
|
||||
|
||||
/* Reset a signal to its default action with sigaction (preferred over
|
||||
* signal(3), whose semantics are implementation-defined). Used in the forked
|
||||
* child before it can spawn any thread. */
|
||||
static void reset_signal_default(int sig) {
|
||||
struct sigaction action;
|
||||
memset(&action, 0, sizeof(action));
|
||||
action.sa_handler = SIG_DFL;
|
||||
sigemptyset(&action.sa_mask);
|
||||
sigaction(sig, &action, NULL);
|
||||
}
|
||||
|
||||
/* Map a listen socket's address to its numeric port for logging, independent
|
||||
* of whether it is an IPv4 or IPv6 sockaddr. */
|
||||
static unsigned short server_address_port(const struct sockaddr_storage* addr) {
|
||||
@@ -107,6 +136,7 @@ Server* server_create_ex(int port, const ServerBindOptions* bind_opts) {
|
||||
server->ssl_ctx = NULL;
|
||||
server->max_connections = 100;
|
||||
server->active_connections = 0;
|
||||
server->limit_registry = NULL;
|
||||
|
||||
return server;
|
||||
}
|
||||
@@ -115,6 +145,20 @@ Server* server_create(int port) {
|
||||
return server_create_ex(port, NULL);
|
||||
}
|
||||
|
||||
void server_set_max_connections(Server* server, unsigned int max_connections) {
|
||||
if (server && max_connections > 0)
|
||||
server->max_connections = max_connections;
|
||||
}
|
||||
|
||||
void server_set_limit_registry(Server* server, struct DaemonLimitRegistry* registry) {
|
||||
if (server)
|
||||
server->limit_registry = registry;
|
||||
}
|
||||
|
||||
int transport_tcp_current_slot(void) {
|
||||
return g_current_slot;
|
||||
}
|
||||
|
||||
void server_delete(Server** server) {
|
||||
if (server == NULL || *server == NULL)
|
||||
return;
|
||||
@@ -133,9 +177,19 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
log_perror("Could not listen on port!");
|
||||
return;
|
||||
}
|
||||
signal(SIGCHLD, sigchld_handler);
|
||||
/* SIGCHLD via sigaction (not signal(3)); SA_RESTART keeps accept(2) from
|
||||
* failing with EINTR, and SA_NOCLDSTOP only notifies on child exit. The
|
||||
* accept loop is single-threaded at this point, so installing here cannot race
|
||||
* a worker thread. */
|
||||
struct sigaction chld_action;
|
||||
memset(&chld_action, 0, sizeof(chld_action));
|
||||
chld_action.sa_handler = sigchld_handler;
|
||||
sigemptyset(&chld_action.sa_mask);
|
||||
chld_action.sa_flags = SA_RESTART | SA_NOCLDSTOP;
|
||||
sigaction(SIGCHLD, &chld_action, NULL);
|
||||
g_limit_registry = server->limit_registry;
|
||||
while (1) {
|
||||
struct sockaddr_in client_addr;
|
||||
struct sockaddr_storage client_addr;
|
||||
socklen_t client_len = sizeof(client_addr);
|
||||
int fd = accept(server->file_descriptor, (struct sockaddr*)&client_addr, &client_len);
|
||||
if (fd < 0) {
|
||||
@@ -143,22 +197,69 @@ static void accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
continue;
|
||||
}
|
||||
tcp_apply_socket_timeout(fd);
|
||||
tcp_enable_nodelay_default(fd, client_addr.ss_family);
|
||||
char peer[128];
|
||||
if (!utils_sockaddr_to_string((const struct sockaddr*)&client_addr, peer, sizeof(peer)))
|
||||
snprintf(peer, sizeof(peer), "unknown");
|
||||
if ((unsigned int)g_active_connections >= server->max_connections) {
|
||||
log_message(LOG_LEVEL_WARNING, "Max connections (%u) reached, rejecting",
|
||||
server->max_connections);
|
||||
log_message(LOG_LEVEL_WARNING, "Max connections (%u) reached, rejecting %s",
|
||||
server->max_connections, peer);
|
||||
close(fd);
|
||||
continue;
|
||||
}
|
||||
log_message(LOG_LEVEL_INFO, "%s", log_fmt);
|
||||
int slot = DAEMON_LIMITS_NO_SLOT;
|
||||
if (server->limit_registry) {
|
||||
slot = daemon_limits_claim_slot(server->limit_registry);
|
||||
if (slot == DAEMON_LIMITS_NO_SLOT) {
|
||||
/* The global cap bounds live children, so this only happens when the
|
||||
* fixed registry is smaller than the configured cap; fail closed. */
|
||||
log_message(LOG_LEVEL_WARNING, "Connection registry slots exhausted (max %u), rejecting %s",
|
||||
server->max_connections, peer);
|
||||
close(fd);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
log_message(LOG_LEVEL_INFO, "%s from %s", log_fmt, peer);
|
||||
g_current_slot = slot;
|
||||
/* Block SIGCHLD across fork() and the parent's pid publication: a child
|
||||
* that exits immediately must not be reaped before its slot records its
|
||||
* pid, which would leak the slot and its module/source counts. Use
|
||||
* pthread_sigmask rather than sigprocmask so the behavior is well defined
|
||||
* even if this process ever gains threads: the mask is per-thread, the fork
|
||||
* copies only the calling thread, and the child inherits this thread's
|
||||
* blocked mask until it restores `previous` below. No thread exists yet at
|
||||
* this point, and none is created before the mask is restored, so the
|
||||
* critical window is race-free. */
|
||||
sigset_t blocked;
|
||||
sigset_t previous;
|
||||
sigemptyset(&blocked);
|
||||
sigaddset(&blocked, SIGCHLD);
|
||||
pthread_sigmask(SIG_BLOCK, &blocked, &previous);
|
||||
pid_t pid = fork();
|
||||
if (pid == 0) {
|
||||
pthread_sigmask(SIG_SETMASK, &previous, NULL);
|
||||
/* Connection children must not run the parent's global cleanup(): it
|
||||
* frees state (credentials / daemon conf) that the child's worker
|
||||
* threads may still be reading and closes fd numbers the child could
|
||||
* already have reused. Reset the inherited handlers so a signal
|
||||
* terminates the child directly; SIGCHLD is reset too since a child
|
||||
* must never reap the parent's children. This runs before the child
|
||||
* spawns any thread, so it cannot race one. */
|
||||
reset_signal_default(SIGINT);
|
||||
reset_signal_default(SIGTERM);
|
||||
reset_signal_default(SIGCHLD);
|
||||
close(server->file_descriptor);
|
||||
child_fn(fd, child_ctx);
|
||||
close(fd);
|
||||
_exit(0);
|
||||
} else if (pid > 0) {
|
||||
g_active_connections++;
|
||||
if (server->limit_registry)
|
||||
daemon_limits_set_slot_pid(server->limit_registry, slot, (long)pid);
|
||||
} else if (server->limit_registry) {
|
||||
/* fork() failed: release the reservation so the slot is not leaked. */
|
||||
daemon_limits_reclaim_slot(server->limit_registry, slot);
|
||||
}
|
||||
pthread_sigmask(SIG_SETMASK, &previous, NULL);
|
||||
close(fd);
|
||||
}
|
||||
}
|
||||
@@ -169,6 +270,9 @@ struct plain_ctx {
|
||||
|
||||
static void plain_child_fn(int fd, void* ctx) {
|
||||
((struct plain_ctx*)ctx)->handler(fd);
|
||||
/* handler() never closes the connection fd; the child owns its single
|
||||
* close here after the handler has fully torn down. */
|
||||
close(fd);
|
||||
}
|
||||
|
||||
bool server_listen(Server* server, void (*handler)(int file_descriptor)) {
|
||||
@@ -185,14 +289,14 @@ void server_accept_loop(Server* server, void (*child_fn)(int, void*), void* chil
|
||||
accept_loop(server, child_fn, child_ctx, log_fmt);
|
||||
}
|
||||
|
||||
static int g_timeout_sec = 30;
|
||||
static int g_contimeout_sec = 10;
|
||||
/* rsync defaults: --timeout=0 (disabled) and --contimeout=60. A non-positive
|
||||
* value means "no timeout" rather than "leave the built-in value in place". */
|
||||
static int g_timeout_sec = 0;
|
||||
static int g_contimeout_sec = 60;
|
||||
|
||||
void tcp_set_timeouts(int timeout_sec, int contimeout_sec) {
|
||||
if (timeout_sec > 0)
|
||||
g_timeout_sec = timeout_sec;
|
||||
if (contimeout_sec > 0)
|
||||
g_contimeout_sec = contimeout_sec;
|
||||
g_timeout_sec = timeout_sec > 0 ? timeout_sec : 0;
|
||||
g_contimeout_sec = contimeout_sec > 0 ? contimeout_sec : 0;
|
||||
}
|
||||
|
||||
int tcp_get_contimeout_sec(void) {
|
||||
@@ -204,6 +308,10 @@ int tcp_get_timeout_sec(void) {
|
||||
}
|
||||
|
||||
static void tcp_apply_socket_timeout(int fd) {
|
||||
/* timeout 0 means no timeout: leave the socket in its default (blocking)
|
||||
* mode instead of installing a zero SO_RCVTIMEO/SO_SNDTIMEO. */
|
||||
if (g_timeout_sec <= 0)
|
||||
return;
|
||||
struct timeval tv;
|
||||
tv.tv_sec = g_timeout_sec;
|
||||
tv.tv_usec = 0;
|
||||
@@ -211,6 +319,19 @@ static void tcp_apply_socket_timeout(int fd) {
|
||||
setsockopt(fd, SOL_SOCKET, SO_SNDTIMEO, &tv, sizeof(tv));
|
||||
}
|
||||
|
||||
/* Enable TCP_NODELAY by default on a transfer socket: the protocol emits many
|
||||
* small messages and Nagle's algorithm would otherwise coalesce/delay them.
|
||||
* Best-effort only: the family guard keeps this to IP/TCP sockets, and a
|
||||
* setsockopt failure is ignored. A caller-provided --sockopts TCP_NODELAY=0
|
||||
* is applied afterwards on the connect path, so an explicit user choice still
|
||||
* wins. */
|
||||
static void tcp_enable_nodelay_default(int fd, int family) {
|
||||
if (family != AF_INET && family != AF_INET6)
|
||||
return;
|
||||
int value = 1;
|
||||
setsockopt(fd, IPPROTO_TCP, TCP_NODELAY, &value, sizeof(value));
|
||||
}
|
||||
|
||||
Client* client_create() {
|
||||
Client* client = (Client*)malloc(sizeof(Client));
|
||||
if (client == NULL) {
|
||||
@@ -356,6 +477,9 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
if (client->file_descriptor < 0)
|
||||
continue;
|
||||
|
||||
/* Default first; a user --sockopts TCP_NODELAY=0 applied below overrides. */
|
||||
tcp_enable_nodelay_default(client->file_descriptor, rp->ai_family);
|
||||
|
||||
if (opts && opts->sockopt_count > 0 &&
|
||||
!tcp_apply_sockopts(client->file_descriptor, opts->sockopts, opts->sockopt_count)) {
|
||||
close(client->file_descriptor);
|
||||
@@ -363,11 +487,15 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
break;
|
||||
}
|
||||
|
||||
/* --contimeout=0 disables the connect timeout: skip the pre-connect socket
|
||||
* timeouts entirely. */
|
||||
if (g_contimeout_sec > 0) {
|
||||
struct timeval ct;
|
||||
ct.tv_sec = g_contimeout_sec;
|
||||
ct.tv_usec = 0;
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_RCVTIMEO, &ct, sizeof(ct));
|
||||
setsockopt(client->file_descriptor, SOL_SOCKET, SO_SNDTIMEO, &ct, sizeof(ct));
|
||||
}
|
||||
|
||||
if (bind_addr_family != 0) {
|
||||
if (rp->ai_family != bind_addr_family) {
|
||||
@@ -402,10 +530,6 @@ bool tcp_connect_socket_ex(Client* client, const char* host, int port,
|
||||
return true;
|
||||
}
|
||||
|
||||
bool tcp_connect_socket(Client* client, const char* host, int port) {
|
||||
return tcp_connect_socket_ex(client, host, port, NULL);
|
||||
}
|
||||
|
||||
bool client_connect_ex(Client* client, const char* host, int port, const TcpConnectOptions* opts) {
|
||||
if (!tcp_connect_socket_ex(client, host, port, opts))
|
||||
return false;
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user