Author SHA1 Message Date
TapTap 2f07895fae docs: update existing compatibility status
CI / lint (pull_request) Successful in 11s
CI / sanitizers (undefined) (pull_request) Successful in 37s
CI / sanitizers (address) (pull_request) Successful in 38s
CI / fuzz-build (pull_request) Successful in 14s
CI / coverage (pull_request) Successful in 31s
CI / build-and-test (pull_request) Successful in 1m15s
CI / valgrind (pull_request) Successful in 33s
2026-09-03 22:02:51 +02:00
TapTap f75db195bd Add rsync-compatible existing option
CI / lint (pull_request) Successful in 11s
CI / sanitizers (undefined) (pull_request) Successful in 38s
CI / sanitizers (address) (pull_request) Successful in 39s
CI / fuzz-build (pull_request) Successful in 14s
CI / coverage (pull_request) Successful in 31s
CI / build-and-test (pull_request) Successful in 1m15s
CI / valgrind (pull_request) Successful in 33s
2026-09-03 17:11:33 +02:00
193 changed files with 3194 additions and 54445 deletions
+14 -28
View File
@@ -4,15 +4,14 @@ on:
push: push:
branches: [main, dev] branches: [main, dev]
pull_request: pull_request:
workflow_dispatch:
jobs: jobs:
lint: lint:
runs-on: ubuntu-latest runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v10 container: gitea.tap-tap.win/taptap/fastsync-ci:v9
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@v4
- name: clang-format check - name: clang-format check
run: find src/ tests/ -name '*.c' -o -name '*.h' | xargs clang-format --dry-run --Werror run: find src/ tests/ -name '*.c' -o -name '*.h' | xargs clang-format --dry-run --Werror
@@ -20,17 +19,13 @@ jobs:
- name: cppcheck - name: cppcheck
run: cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/ run: cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/
# Fast PR gate: build + unit tests + a representative subset of integration
# tests (marked `ci`), parallelized with pytest-xdist. Only the full coverage
# jobs below (sanitizers/fuzz/coverage/valgrind and the FULL integration
# suite) run on merge to dev/main, so PR CI stays well under ~3 minutes.
build-and-test: build-and-test:
runs-on: ubuntu-latest runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v10 container: gitea.tap-tap.win/taptap/fastsync-ci:v9
needs: lint needs: lint
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@v4
- name: Configure - name: Configure
run: cmake -B build -S . -DSTRICT_WARNINGS=ON run: cmake -B build -S . -DSTRICT_WARNINGS=ON
@@ -41,25 +36,19 @@ jobs:
- name: Unit Tests - name: Unit Tests
run: ctest --test-dir build --output-on-failure -j$(nproc) run: ctest --test-dir build --output-on-failure -j$(nproc)
- name: Integration Tests (PR smoke subset) - name: Integration Tests
if: github.event_name == 'pull_request' run: python3 -m pytest tests/integration/ -v --tb=short
run: python3 -m pytest tests/integration/ -n 4 --dist=load -m ci --durations=25 --tb=short -q
- name: Integration Tests (full suite)
if: github.event_name == 'push'
run: python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" --durations=25 --tb=short -q
sanitizers: sanitizers:
runs-on: ubuntu-latest runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v10 container: gitea.tap-tap.win/taptap/fastsync-ci:v9
needs: lint needs: lint
if: github.event_name == 'push'
strategy: strategy:
matrix: matrix:
sanitizer: [address, undefined] sanitizer: [address, undefined]
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@v4
- name: Configure - name: Configure
run: cmake -B build-${{ matrix.sanitizer }} -S . -DSANITIZER=${{ matrix.sanitizer }} run: cmake -B build-${{ matrix.sanitizer }} -S . -DSANITIZER=${{ matrix.sanitizer }}
@@ -72,12 +61,11 @@ jobs:
fuzz-build: fuzz-build:
runs-on: ubuntu-latest runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v10 container: gitea.tap-tap.win/taptap/fastsync-ci:v9
needs: lint needs: lint
if: github.event_name == 'push'
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@v4
- name: Configure (clang + fuzz) - name: Configure (clang + fuzz)
run: CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON run: CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON
@@ -94,12 +82,11 @@ jobs:
coverage: coverage:
runs-on: ubuntu-latest runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v10 container: gitea.tap-tap.win/taptap/fastsync-ci:v9
needs: lint needs: lint
if: github.event_name == 'push'
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@v4
- name: Configure - name: Configure
run: cmake -B build -S . -DENABLE_COVERAGE=ON run: cmake -B build -S . -DENABLE_COVERAGE=ON
@@ -118,12 +105,11 @@ jobs:
valgrind: valgrind:
runs-on: ubuntu-latest runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v10 container: gitea.tap-tap.win/taptap/fastsync-ci:v9
needs: lint needs: lint
if: github.event_name == 'push'
steps: steps:
- name: Checkout - name: Checkout
uses: actions/checkout@11d5960a326750d5838078e36cf38b85af677262 # v4 uses: actions/checkout@v4
- name: Configure - name: Configure
run: cmake -B build -S . -DSTRICT_WARNINGS=ON run: cmake -B build -S . -DSTRICT_WARNINGS=ON
-4
View File
@@ -8,7 +8,3 @@ build-*/
build2/ build2/
build3/ build3/
build_docker2/ build_docker2/
# Test/run artifacts
root/
test_partial_install_tmp/
+1 -1
View File
@@ -128,7 +128,7 @@ Do not wait for the user to tell you CI failed — check proactively. The user s
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+1 -1
View File
@@ -92,7 +92,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+12 -13
View File
@@ -27,19 +27,16 @@ FetchContent_Declare(xxhash GIT_REPOSITORY https://github.com/Cyan4973/xxHash GI
FetchContent_MakeAvailable(xxhash) FetchContent_MakeAvailable(xxhash)
# Sanitizer option # Sanitizer option
set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, undefined, none)") set(SANITIZER "none" CACHE STRING "Sanitizer to enable (address, thread, none)")
set_property(CACHE SANITIZER PROPERTY STRINGS address thread undefined none) set_property(CACHE SANITIZER PROPERTY STRINGS address thread none)
if(SANITIZER STREQUAL "address") if(SANITIZER STREQUAL "address")
add_compile_options(-fsanitize=address -fno-omit-frame-pointer -g) add_compile_options(-fsanitize=address -fno-omit-frame-pointer -g)
add_link_options(-fsanitize=address) add_link_options(-fsanitize=address)
elseif(SANITIZER STREQUAL "thread") elseif(SANITIZER STREQUAL "thread")
add_compile_options(-fsanitize=thread -fno-omit-frame-pointer -g) add_compile_options(-fsanitize=thread -fno-omit-frame-pointer -g)
add_link_options(-fsanitize=thread) add_link_options(-fsanitize=thread)
elseif(SANITIZER STREQUAL "undefined")
add_compile_options(-fsanitize=undefined -fno-omit-frame-pointer -g)
add_link_options(-fsanitize=undefined)
elseif(NOT SANITIZER STREQUAL "none") elseif(NOT SANITIZER STREQUAL "none")
message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, undefined, none") message(FATAL_ERROR "Unknown sanitizer: ${SANITIZER}. Supported values: address, thread, none")
endif() endif()
option(STRICT_WARNINGS "Enable strict warnings" OFF) option(STRICT_WARNINGS "Enable strict warnings" OFF)
@@ -87,7 +84,7 @@ tests/integration/ — Python pytest integration tests
### Dependencies ### Dependencies
- **zstd** — found via `find_library(ZSTD_LIBRARY zstd)` - **zstd** — found via `find_library(ZSTD_LIBRARY zstd)`
- **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport) - **OpenSSL** — found via `find_package(OpenSSL REQUIRED)` (TLS 1.2+ transport)
- **xxHash** — fetched via `FetchContent` from the upstream repository (delta transfer hashing, v0.8.3) - **xxHash** — fetched via `FetchContent` from GitHub (delta transfer hashing, v0.8.3)
- **pthreads** — found via `find_package(Threads REQUIRED)` - **pthreads** — found via `find_package(Threads REQUIRED)`
- **C11 standard** — required - **C11 standard** — required
- **CMake 3.22+** — minimum version - **CMake 3.22+** — minimum version
@@ -97,7 +94,7 @@ tests/integration/ — Python pytest integration tests
- Use `file(GLOB ...)` for source collection (existing pattern). - Use `file(GLOB ...)` for source collection (existing pattern).
- All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`. - All targets link `Threads::Threads`, `${ZSTD_LIBRARY}`, `OpenSSL::SSL`, `OpenSSL::Crypto`, and `xxhash`.
- Include directories: `src/shared`, `src/server`, `src/client`, `tests` (for test target). - Include directories: `src/shared`, `src/server`, `src/client`, `tests` (for test target).
- Sanitizer support: pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (live option in CMakeLists.txt). - Sanitizer support: pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (live option in CMakeLists.txt).
- Build with `cmake -B build -S . && cmake --build build -j$(nproc)`. - Build with `cmake -B build -S . && cmake --build build -j$(nproc)`.
- For CI, dependencies are provided by the project's custom Docker image (repo-root `Dockerfile`, same image CI uses). For local development, use `nix-shell`. Never add `apt-get install` / `pip install` to CI workflows. See `AGENTS.md`. - For CI, dependencies are provided by the project's custom Docker image (repo-root `Dockerfile`, same image CI uses). For local development, use `nix-shell`. Never add `apt-get install` / `pip install` to CI workflows. See `AGENTS.md`.
@@ -108,7 +105,7 @@ tests/integration/ — Python pytest integration tests
3. Add new dependencies with `find_package` or `find_library`. 3. Add new dependencies with `find_package` or `find_library`.
4. When adding a new executable target, follow the pattern of existing targets. 4. When adding a new executable target, follow the pattern of existing targets.
5. When adding a new library (static/shared), use `add_library` and follow the project's naming. 5. When adding a new library (static/shared), use `add_library` and follow the project's naming.
6. For sanitizer builds, pass `-DSANITIZER=address`, `-DSANITIZER=thread`, or `-DSANITIZER=undefined` to cmake (matching CI's matrix strategy). 6. For sanitizer builds, pass `-DSANITIZER=address` or `-DSANITIZER=thread` to cmake (matching CI's matrix strategy).
7. Always verify the build compiles after changes. 7. Always verify the build compiles after changes.
## Sanitizer Configurations ## Sanitizer Configurations
@@ -122,9 +119,11 @@ cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (race conditions)
cmake --build build -j$(nproc) cmake --build build -j$(nproc)
``` ```
UndefinedBehaviorSanitizer uses the same built-in option: For UndefinedBehaviorSanitizer (no `-DSANITIZER=undefined` option in CMakeLists.txt yet), use the manual flag approach:
```bash ```bash
cmake -B build -S . -DSANITIZER=undefined cmake -B build -S . \
-DCMAKE_C_FLAGS="-fsanitize=undefined -fno-omit-frame-pointer -g" \
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=undefined"
cmake --build build -j$(nproc) cmake --build build -j$(nproc)
``` ```
@@ -160,7 +159,7 @@ cmake -B build -S . -DCMAKE_BUILD_TYPE=RelWithDebInfo
```bash ```bash
cmake -B build -S . cmake -B build -S .
cmake --build build -j$(nproc) cmake --build build -j$(nproc)
./build/server -p 8080 --allow-unauthenticated ./build/server
./build/client ./build/client
./build/tests ./build/tests
``` ```
@@ -188,7 +187,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+5 -5
View File
@@ -27,7 +27,7 @@ FastSync is a file synchronization tool (like rsync, but faster). It transfers f
cmake -B build -S . && cmake --build build -j$(nproc) cmake -B build -S . && cmake --build build -j$(nproc)
# Server (TCP mode) # Server (TCP mode)
./build/server -p 8080 --allow-unauthenticated ./build/server
# Client (TCP mode) # Client (TCP mode)
./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk ./build/client --source-dir /path/to/send --dest-dir /path/to/receive --save-to-disk
@@ -37,13 +37,13 @@ cmake -B build -S . && cmake --build build -j$(nproc)
# Run tests # Run tests
./build/tests # unit tests ./build/tests # unit tests
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # integration tests python3 test.py # integration tests
``` ```
## Code Walkthrough ## Code Walkthrough
### Client Entry Point (`src/client/client_cli.c`) ### Client Entry Point (`src/client/client_cli.c`)
- Parses CLI arguments using a custom option-table parser (`OPTION_TABLE` in `src/client/client_cli.c`); there is no `getopt*` usage - Parses CLI arguments using `getopt_long`
- Creates `Config` struct with all options - Creates `Config` struct with all options
- Detects SSH destinations (contains `:`) - Detects SSH destinations (contains `:`)
- Calls into `client_send.c` for the actual transfer - Calls into `client_send.c` for the actual transfer
@@ -109,7 +109,7 @@ Collection of files for batch transfer. Serialized with file count, then per-fil
zstd streaming compression via `ZSTD_compressStream2`/`ZSTD_decompressStream`. Compression happens per-chunk in the sender stage. Level 1-22 (default 5). Streaming means memory usage stays bounded regardless of file size. zstd streaming compression via `ZSTD_compressStream2`/`ZSTD_decompressStream`. Compression happens per-chunk in the sender stage. Level 1-22 (default 5). Streaming means memory usage stays bounded regardless of file size.
### "How does sendfile() work?" ### "How does sendfile() work?"
On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `--sendfile` (long form only). Only works with TCP (not SSH, not compression). On Linux, `sendfile()` copies data directly from kernel file buffer to socket, bypassing userspace. ~2x faster for large files. Enabled with `-f` flag. Only works with TCP (not SSH, not compression).
### "How does incremental sync work?" ### "How does incremental sync work?"
Client sends file metadata (path, size, mtime) to server. Server checks if destination file has same size+mtime. If match, server responds `STATUS_OK` (skip). If mismatch, server responds `STATUS_NEXT` (send). Client sends file metadata (path, size, mtime) to server. Server checks if destination file has same size+mtime. If match, server responds `STATUS_OK` (skip). If mismatch, server responds `STATUS_NEXT` (send).
@@ -138,7 +138,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+1 -1
View File
@@ -316,7 +316,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+10 -8
View File
@@ -14,9 +14,10 @@ Diagnose crashes, memory errors, hangs, and logic bugs. You use structured debug
### Memory Errors ### Memory Errors
```bash ```bash
# AddressSanitizer (fast, recommended first) # AddressSanitizer (fast, recommended first)
cmake -B build-asan -S . -DSANITIZER=address cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
cmake --build build-asan -j$(nproc) -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
./build-asan/client # or ./build-asan/server -p 8080 --allow-unauthenticated cmake --build build -j$(nproc)
./build/client # or ./build/server
# Valgrind (slower, more thorough) # Valgrind (slower, more thorough)
valgrind --leak-check=full --show-leak-kinds=all --track-origins=yes \ valgrind --leak-check=full --show-leak-kinds=all --track-origins=yes \
@@ -31,9 +32,10 @@ valgrind --tool=drd ./build/client ...
### Thread Sanitizer ### Thread Sanitizer
```bash ```bash
cmake -B build-tsan -S . -DSANITIZER=thread cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=thread" \
cmake --build build-tsan -j$(nproc) -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
./build-tsan/tests cmake --build build -j$(nproc)
./build/tests
``` ```
### GDB ### GDB
@@ -141,7 +143,7 @@ gprof ./build/client gmon.out
### Step 5: Verify ### Step 5: Verify
- Run `./build/tests` (unit tests) - Run `./build/tests` (unit tests)
- Run `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` (integration tests) - Run `python3 test.py` (integration tests)
- Run under valgrind again to confirm clean - Run under valgrind again to confirm clean
- Test under ASan again - Test under ASan again
@@ -160,7 +162,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+1 -1
View File
@@ -96,7 +96,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+24 -53
View File
@@ -16,18 +16,12 @@ Scan the codebase for patterns that suggest new feature opportunities. You ident
### Module Map ### Module Map
``` ```
src/client/ Client-side: CLI parsing, scanning, sending src/client/ Client-side: CLI parsing, scanning, sending
client_cli.c Entry point, OPTION_TABLE parser, config setup client_cli.c Entry point, argument parsing, config setup
usage.c Usage/help text (authoritative CLI flag list)
client_send.c Transfer orchestration, pipeline management client_send.c Transfer orchestration, pipeline management
client_validation.c Destination/CLI validation
scanner.c BFS directory traversal, chunk building scanner.c BFS directory traversal, chunk building
change_list.c File change-list bookkeeping
src/server/ Server-side: listening, receiving, writing src/server/ Server-side: listening, receiving, writing
server.c TCP accept loop, per-connection handling server.c TCP accept loop, per-connection handling
server_cli.c Server option-table CLI parsing
receiver.c Receiver-side file handling
receiver_pipeline.c Receiver worker pipeline
src/shared/ Shared libraries (used by both client and server) src/shared/ Shared libraries (used by both client and server)
protocol.c/h Wire protocol: status codes, send/receive primitives protocol.c/h Wire protocol: status codes, send/receive primitives
@@ -38,63 +32,40 @@ src/shared/ Shared libraries (used by both client and server)
data.c/h Generic buffer type (Data) data.c/h Generic buffer type (Data)
metadata.c/h File metadata (mode, uid, gid, mtime) metadata.c/h File metadata (mode, uid, gid, mtime)
file.c/h File representation file.c/h File representation
file_send.c/h Sender-side file transfer
file_receive.c/h Receiver-side file transfer
file_list.c/h File list model
file_store.c/h Destination file store
array_list.c/h Dynamic array array_list.c/h Dynamic array
delta.c/h Delta transfer algorithm
checksum.c/h Whole-file/block checksums (xxHash, md5)
filter.c/h rsync-style filter rules
batch.c/h Batch files (--write-batch/--read-batch)
charset.c/h Filename charset conversion (--iconv)
chmod.c/h Permission modification (--chmod)
xattr.c/h Extended attributes
hardlink.c/h Hard-link handling
identity.c/h uid/gid mapping (--usermap/--groupmap/--chown)
credentials.c/h Daemon credentials
daemon_conf.c/h Daemon module configuration
motd.c/h Daemon MOTD
delay_updates.c/h Delayed update staging
stop_condition.c/h Stop-after/stop-at handling
transport_tcp.c/h TCP client/server with sendfile() zero-copy transport_tcp.c/h TCP client/server with sendfile() zero-copy
transport_ssh.c/h SSH transport with ControlMaster transport_ssh.c/h SSH transport with ControlMaster
transport_tls.c/h TLS encryption via OpenSSL transport_tls.c/h TLS encryption via OpenSSL
multiprocessing.c/h Fork-based concurrency multiprocessing.c/h Fork-based concurrency
log.c/h Logging utilities log.c/h Logging utilities
utils.c/h Shared utilities utils.c/h Shared utilities
file_types.h Shared file type definitions
``` ```
### Existing CLI Flags (authoritative source: `src/client/usage.c`) ### Existing CLI Flags (from client_cli.c)
``` ```
--source-dir <dir> Source directory --source-dir <dir> Source directory to sync (required)
--dest-dir <dir> Destination directory on server --dest-dir <dir> Destination directory on server (required)
--server-host <ip> Server IP address (default: 127.0.0.1) --host <host> Server hostname/IP (required)
--server-port <n> Server port (default: 8080); --port is an alias --port <port> Server TCP port
-c, --checksum Verify content by checksum instead of size+mtime --server-mode Listen as server
-z, --compress [level] Enable compression (level 1-22, default 5) --use-compression, -c Enable zstd compression
-j, --threads[=N] Enable multithreaded scanner/loader/sender pipeline --use-multithreading, -m Enable multithreaded transfer
--chunk-serialization Enable chunk serialization (long form only) --use-sendfile, -s Use sendfile() zero-copy TCP
--sendfile sendfile() zero-copy (TCP only; long form only) --use-ssh, -S Use SSH transport
-s, --secluded-args Protect-args compatibility option (no effect) --use-tls, -T Enable TLS encryption
--tls Enable TLS encryption; --cert/--key/--ca give PEMs --cert <file> TLS certificate file
--bwlimit <KB/s> Bandwidth limit in kilobytes per second --key <file> TLS key file
--delete Delete files on receiver not in source --ca <file> TLS CA certificate file
--incremental Skip files unchanged since last transfer --insecure Skip TLS verification
--delta Delta transfer for changed files (needs --incremental) --bwlimit <bytes/s> Bandwidth limit
-f, --filter=RULE rsync-style filter rule (+/- include/exclude) --delete Delete files not in source
--exclude <pattern> Exclude files matching pattern --include <pattern> Include filter pattern
--include <pattern> Only include files matching pattern --exclude <pattern> Exclude filter pattern
-m, --prune-empty-dirs Do not transfer empty directory entries --dry-run Print what would be transferred
-n, --dry-run Show what would be transferred --save-to-disk Save transferred files to disk (for server tests)
--save-to-disk Write received files to disk
--version Print version and exit --version Print version and exit
--help Show help --help Print help
``` ```
> Always confirm the current flags with `./build/client --help`; the table above
> is a representative subset. `src/client/usage.c` is the authoritative list and
> `OPTION_TABLE` in `src/client/client_cli.c` is the parser (there is no `getopt*`).
## Feature Scout Checklist ## Feature Scout Checklist
@@ -317,7 +288,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+9 -10
View File
@@ -21,7 +21,7 @@ Design integration tests that verify the full transfer pipeline works end-to-end
- Multiple configurations (TCP, SSH, TLS, compression, multithreading) - Multiple configurations (TCP, SSH, TLS, compression, multithreading)
- Network shaping (LAN, WAN profiles) - Network shaping (LAN, WAN profiles)
- Feature tests (dry run, archive, exclude, delete, incremental, bandwidth limit) - Feature tests (dry run, archive, exclude, delete, incremental, bandwidth limit)
- Run: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` - Run: `python3 -m pytest tests/ -v --tb=short`
### 3. New: Focused Integration Tests ### 3. New: Focused Integration Tests
When adding new features or fixing bugs, write targeted integration tests. When adding new features or fixing bugs, write targeted integration tests.
@@ -35,14 +35,13 @@ mkdir -p /tmp/fastsync_test/src
echo "test content" > /tmp/fastsync_test/src/file.txt echo "test content" > /tmp/fastsync_test/src/file.txt
# Start server # Start server
./build/server -p 8080 --allow-unauthenticated & ./build/server &
SERVER_PID=$! SERVER_PID=$!
sleep 0.5 sleep 0.5
# Run client # Run client
./build/client --source-dir /tmp/fastsync_test/src \ ./build/client --source-dir /tmp/fastsync_test/src \
--dest-dir /tmp/fastsync_test/dst \ --dest-dir /tmp/fastsync_test/dst \
--server-port 8080 \
--save-to-disk --save-to-disk
# Verify # Verify
@@ -77,7 +76,7 @@ openssl req -x509 -newkey rsa:2048 -keyout /tmp/key.pem -out /tmp/cert.pem \
### Pattern 4: Incremental Sync ### Pattern 4: Incremental Sync
```bash ```bash
# First sync # First sync
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk ./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
# Modify source # Modify source
echo "updated" >> /tmp/src/file.txt echo "updated" >> /tmp/src/file.txt
@@ -90,14 +89,14 @@ echo "updated" >> /tmp/src/file.txt
### Pattern 5: Delete Verification ### Pattern 5: Delete Verification
```bash ```bash
# Initial sync # Initial sync
./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk ./build/client --source-dir /tmp/src --dest-dir /tmp/dst --save-to-disk -M
# Add extra file to dest # Add extra file to dest
echo "extra" > /tmp/dst/.../extra.txt echo "extra" > /tmp/dst/.../extra.txt
# Sync with --delete # Sync with --delete
./build/client --source-dir /tmp/src --dest-dir /tmp/dst \ ./build/client --source-dir /tmp/src --dest-dir /tmp/dst \
--save-to-disk --delete --save-to-disk --delete -M
# Verify extra.txt is gone # Verify extra.txt is gone
test ! -f /tmp/dst/.../extra.txt test ! -f /tmp/dst/.../extra.txt
@@ -116,7 +115,7 @@ The project uses Gitea Actions. Key jobs:
jobs: jobs:
new-job: new-job:
runs-on: ubuntu-latest runs-on: ubuntu-latest
container: gitea.tap-tap.win/taptap/fastsync-ci:v10 container: gitea.tap-tap.win/taptap/fastsync-ci:v7
steps: steps:
- uses: actions/checkout@v4 - uses: actions/checkout@v4
- name: Configure - name: Configure
@@ -128,7 +127,7 @@ jobs:
- name: Unit Tests - name: Unit Tests
run: ./build-${{ matrix.sanitizer }}/tests run: ./build-${{ matrix.sanitizer }}/tests
- name: Integration Tests - name: Integration Tests
run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" run: LSAN_OPTIONS=suppressions=.lsan-suppressions.txt python3 -m pytest tests/ -v --tb=short
``` ```
The symlink step is required because `tests/conftest.py` expects `./build` to exist. The symlink step is required because `tests/conftest.py` expects `./build` to exist.
@@ -136,7 +135,7 @@ The symlink step is required because `tests/conftest.py` expects `./build` to ex
After any code change: After any code change:
- [ ] Unit tests pass: `./build/tests` - [ ] Unit tests pass: `./build/tests`
- [ ] Integration tests pass: `python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv"` - [ ] Integration tests pass: `python3 -m pytest tests/ -v --tb=short`
- [ ] Build clean: no warnings with `-Wall` - [ ] Build clean: no warnings with `-Wall`
- [ ] No memory errors: ASan clean - [ ] No memory errors: ASan clean
- [ ] No thread errors: TSan clean (if threading involved) - [ ] No thread errors: TSan clean (if threading involved)
@@ -157,7 +156,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+16 -17
View File
@@ -1,5 +1,5 @@
--- ---
description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates Gitea issues from their findings. description: Top-level orchestrator that analyzes the FastSync codebase by delegating to specialized sub-agents and creates GitHub issues from their findings.
mode: subagent mode: subagent
--- ---
@@ -12,7 +12,7 @@ You are the primary orchestrator agent. Your job is to:
2. Decide which specialized sub-agents to dispatch for analysis 2. Decide which specialized sub-agents to dispatch for analysis
3. Delegate analysis work using the task tool 3. Delegate analysis work using the task tool
4. Receive structured findings from sub-agents 4. Receive structured findings from sub-agents
5. Create Gitea issues from those findings using `tea issues create` 5. Create GitHub issues from those findings using `gh issue create`
6. Coordinate the overall analysis workflow end-to-end 6. Coordinate the overall analysis workflow end-to-end
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`. > **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
@@ -97,7 +97,7 @@ First, read the repository structure to understand what exists:
### Phase 2: Determine Analysis Scope ### Phase 2: Determine Analysis Scope
Based on what the user requests or what needs attention: Based on what the user requests or what needs attention:
- **New features wanted?** → Dispatch `feature-scout` sub-agent - **New features wanted?** → Dispatch `feature-scout` sub-agent
- **Security audit needed?** → Dispatch `security-auditor` sub-agent - **Security audit needed?** → Dispatch `security-screener` sub-agent
- **Code quality review?** → Dispatch `code-quality-guardian` sub-agent - **Code quality review?** → Dispatch `code-quality-guardian` sub-agent
- **All of the above?** → Run all three in parallel - **All of the above?** → Run all three in parallel
@@ -110,7 +110,7 @@ Context: <provide summary of what was found in Phase 1>
``` ```
``` ```
Task: Ask the security-auditor agent to analyze the codebase. Task: Ask the security-screener agent to analyze the codebase.
Context: <provide summary of what was found in Phase 1> Context: <provide summary of what was found in Phase 1>
``` ```
@@ -138,14 +138,14 @@ Each sub-agent returns findings in this structured format:
- **Labels**: comma-separated labels for the issue - **Labels**: comma-separated labels for the issue
``` ```
### Phase 5: Create Gitea Issues ### Phase 5: Create GitHub Issues
For each finding, create a Gitea issue: For each finding, create a GitHub issue:
```bash ```bash
tea issues create --repo TapTap/FastSync \ gh issue create \
--title "<Finding Title>" \ --title "<Finding Title>" \
--labels "<labels>" \ --label "<labels>" \
--description "## Description --body "## Description
<description> <description>
## Location ## Location
@@ -175,13 +175,11 @@ _This issue was automatically generated by the issue-creator agent._"
### Duplicate Detection ### Duplicate Detection
Before creating an issue: Before creating an issue:
1. Check existing open issues: `tea issues list --repo TapTap/FastSync --state open --labels "<label>"` 1. Check existing open issues: `gh issue list --state open --label "<label>"`
2. Search for similar titles using `tea issues list --repo TapTap/FastSync --keyword "<keywords>"` 2. Search for similar titles using `gh issue list --search "<keywords>"`
3. If a similar issue exists, add a comment instead of creating a duplicate: 3. If a similar issue exists, add a comment instead of creating a duplicate:
```bash ```bash
tea comment --repo TapTap/FastSync <issue-number> "Additional finding from automated analysis: <details>" gh issue comment <issue-number> --body "Additional finding from automated analysis: <details>"
# or POST to the Gitea API:
# POST https://gitea.tap-tap.win/api/v1/repos/TapTap/FastSync/issues/<n>/comments
``` ```
## Sub-Agent Reference ## Sub-Agent Reference
@@ -191,12 +189,13 @@ Before creating an issue:
| Agent | File | Purpose | | Agent | File | Purpose |
|---|---|---| |---|---|---|
| feature-scout | `.opencode/agents/feature-scout.md` | Scans for feature opportunities | | feature-scout | `.opencode/agents/feature-scout.md` | Scans for feature opportunities |
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits and vulnerability scans | | security-screener | `.opencode/agents/security-screener.md` | Scans for security vulnerabilities |
| code-quality-guardian | `.opencode/agents/code-quality-guardian.md` | Scans for code quality improvements | | code-quality-guardian | `.opencode/agents/code-quality-guardian.md` | Scans for code quality improvements |
| architect | `.opencode/agents/architect.md` | Architecture reviews | | architect | `.opencode/agents/architect.md` | Architecture reviews |
| c-reviewer | `.opencode/agents/c-reviewer.md` | C code correctness reviews | | c-reviewer | `.opencode/agents/c-reviewer.md` | C code correctness reviews |
| debugger | `.opencode/agents/debugger.md` | Bug diagnosis | | debugger | `.opencode/agents/debugger.md` | Bug diagnosis |
| refactorer | `.opencode/agents/refactorer.md` | Code refactoring | | refactorer | `.opencode/agents/refactorer.md` | Code refactoring |
| security-auditor | `.opencode/agents/security-auditor.md` | Security audits |
| test-writer | `.opencode/agents/test-writer.md` | Test development | | test-writer | `.opencode/agents/test-writer.md` | Test development |
| perf-analyst | `.opencode/agents/perf-analyst.md` | Performance analysis | | perf-analyst | `.opencode/agents/perf-analyst.md` | Performance analysis |
| protocol-designer | `.opencode/agents/protocol-designer.md` | Protocol design | | protocol-designer | `.opencode/agents/protocol-designer.md` | Protocol design |
@@ -243,7 +242,7 @@ tests/test_file.c — File tests
tests/test_transport_tcp.c — TCP transport tests tests/test_transport_tcp.c — TCP transport tests
tests/test_transport_tls.c — TLS transport tests tests/test_transport_tls.c — TLS transport tests
tests/test_array_list.c — Array list tests tests/test_array_list.c — Array list tests
tests/integration/ — Python pytest integration tests tests/pytest/ — Python integration tests
``` ```
### Build & Config Files ### Build & Config Files
@@ -260,7 +259,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+5 -7
View File
@@ -56,11 +56,9 @@ DirectoryScanner → Queue(Scanner→Loader) → ChunkBuilder → Queue(Loader
### Benchmark Context ### Benchmark Context
Use the maintained benchmark tool — do not cite stale README numbers: From README benchmarks (25MB mixed files, localhost):
- `python3 benchmark/bench.py` runs the repeatable throughput benchmark. - Best config: `-m -c` (multithread + compression) → 0.20s, 11.2× faster than rsync
- The real flags are `-j` (multithreading) and `-z` (compression); a fast loopback - `sendfile()` bypasses userspace → ~2× faster on localhost
config combines `-j -z`.
- `sendfile()` (via `--sendfile`) bypasses userspace → ~2× faster on localhost
- Compression reduces wire data enough that transfer becomes latency-bound on WAN - Compression reduces wire data enough that transfer becomes latency-bound on WAN
## Output Format ## Output Format
@@ -122,7 +120,6 @@ time ./build/client [args...]
# High precision # High precision
perf stat -e task-clock ./build/client [args...] perf stat -e task-clock ./build/client [args...]
```
## CI & Task Execution ## CI & Task Execution
@@ -130,8 +127,9 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details. **CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
```
+1 -1
View File
@@ -91,7 +91,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+1 -1
View File
@@ -160,7 +160,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+45 -307
View File
@@ -3,64 +3,25 @@ description: Audits FastSync for security vulnerabilities — TLS config, input
mode: subagent mode: subagent
--- ---
You are the security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport. This is the single canonical security agent. You are a security auditor for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
## Your Role ## Your Role
Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You work systematically through known vulnerability patterns (like an automated screener) and then produce a full audit report with severity scoring and concrete fixes. Audit the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices.
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`. ## Attack Surface
## Project Architecture ### Network Input Points
1. **TCP server** (`src/server/server.c`) — accepts connections from any client
2. **SSH transport** (`src/shared/transport_ssh.c`) — receives data via stdio pipe
3. **Protocol parsing** (`src/shared/protocol.c`) — deserializes all incoming data
4. **Config deserialization** (`src/shared/config.c`) — receives remote config
5. **Chunk deserialization** (`src/shared/chunk.c`) — receives file batches
### Module Map ### TLS Configuration
``` - OpenSSL TLS 1.2+ via `src/shared/transport_tls.c`
src/client/ Client-side: CLI parsing, scanning, sending - Certificate/key loading, CA verification
client_cli.c Entry point, argument parsing, config setup - SSL context setup, cipher suite selection
client_send.c Transfer orchestration, pipeline management
client_validation.c Destination/CLI validation
scanner.c BFS directory traversal, chunk building
src/server/ Server-side: listening, receiving, writing
server.c TCP accept loop, per-connection handling
receiver.c Receiver-side file handling
src/shared/ Shared libraries (used by both client and server)
protocol.c/h Wire protocol: status codes, send/receive primitives
compression.c/h zstd streaming compression/decompression
chunk.c/h File grouping and batch serialization
queue.c/h Thread-safe bounded queue (producer-consumer)
config.c/h Runtime configuration, serialization, parsing
data.c/h Generic buffer type (Data)
metadata.c/h File metadata (mode, uid, gid, mtime)
file.c/h File representation
file_receive.c/h Receiver-side file transfer
file_store.c/h Destination file store
delta.c/h Delta transfer algorithm
checksum.c/h Whole-file/block checksums (xxHash, md5)
filter.c/h rsync-style filter rules
xattr.c/h Extended attributes
identity.c/h uid/gid mapping
credentials.c/h Daemon credentials
transport_tcp.c/h TCP client/server with sendfile() zero-copy
transport_ssh.c/h SSH transport with ControlMaster
transport_tls.c/h TLS encryption via OpenSSL
multiprocessing.c/h Fork-based concurrency
log.c/h Logging utilities
utils.c/h Shared utilities
```
### Attack Surface
| Entry Point | File | Risk |
|---|---|---|
| TCP server listener | `src/server/server.c` | Externally reachable on network |
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
| File writer | `src/server/server.c` / `receiver.c` | Writes received files to disk |
## Security Audit Checklist ## Security Audit Checklist
@@ -72,182 +33,51 @@ src/shared/ Shared libraries (used by both client and server)
- [ ] Chunk count and file count validated before allocation - [ ] Chunk count and file count validated before allocation
- [ ] Config field lengths bounded - [ ] Config field lengths bounded
### 2. Buffer Overflow Risks ### 2. Buffer Safety
- [ ] No `strcpy` — use `snprintf` or `strncpy` with null termination
- [ ] `malloc` size calculations don't overflow (e.g., `count * sizeof(...)`)
- [ ] No fixed-size stack buffers for unbounded input
- [ ] `receive_n_data` always checks return value
- [ ] Off-by-one in path concatenation
Search for these dangerous patterns in all `.c` and `.h` files: ### 3. Memory Safety in Error Paths
- [ ] All error paths free allocated resources
- [ ] No use-after-free on error paths
- [ ] No double-free on error paths
- [ ] Partial reads handled (don't use incomplete data)
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data ### 4. TLS/SSL Security
```c - [ ] TLS 1.2 minimum enforced (no SSLv3, TLS 1.0, TLS 1.1)
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary - [ ] Certificate verification enabled when CA provided
char buf[1024]; // SUSPICIOUS — what limits the input to 1024? - [ ] Certificate verification disabled only with explicit warning
char line[4096]; // SUSPICIOUS — what limits the line length? - [ ] Private key file permissions checked
``` - [ ] No hardcoded certificates or keys
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent - [ ] Cipher suites restricted to strong algorithms
```bash - [ ] SSL error codes checked after `SSL_read`/`SSL_write`
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
```
- [ ] **Unbounded `sprintf` to fixed buffer**
```c
char buf[256];
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
```
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
```c
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
```
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
### 3. Path Traversal in File Operations ### 5. Authentication & Authorization
Check all paths constructed from received data:
- [ ] **Files constructed with client-provided filenames + destination directory**
```c
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
```
Check for `../` filtering:
```bash
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
```
- [ ] **`realpath()` usage** for path canonicalization
- [ ] **Symlink following** — does the server follow symlinks in the destination?
- [ ] **Null byte injection** — received filenames with embedded `\0`
### 4. Unchecked Return Values from Critical Functions
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
```bash
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
```
For each match, verify NULL check exists before use.
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
- [ ] **`fopen()` / `open()`** return values not checked
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
### 5. TLS / SSL Security
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
```c
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
```
- [ ] **Certificate verification disabled** without explicit `--ca`/warning
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
- [ ] **Private key file permissions** not checked before loading
- [ ] **Hostname verification** not performed on server certificate
- [ ] **Session renegotiation** not limited (DoS vector)
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
- [ ] **No hardcoded certificates or keys**
- [ ] **SSL error codes checked after `SSL_read`/`SSL_write`**
### 6. Memory Safety Issues
- [ ] **Use-after-free** — object freed but pointer still used later
- [ ] **Double-free** — `free()` called twice on same pointer
- [ ] **Memory leaks** on error paths — allocated but not freed before return
- [ ] **Integer overflow** in allocation size computation
```c
// DANGER: count * sizeof(Type) can overflow
void *arr = malloc(count * sizeof(Element));
// SAFE:
if (count > SIZE_MAX / sizeof(Element)) return NULL;
void *arr = malloc(count * sizeof(Element));
```
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
```c
// BAD: leaks original pointer on failure
buf = realloc(buf, new_size);
// GOOD:
void *tmp = realloc(buf, new_size);
if (!tmp) { free(buf); return NULL; }
buf = tmp;
```
- [ ] **All error paths free allocated resources** (no leaks / UAF / double-free)
- [ ] **Partial reads handled** (don't use incomplete data)
### 7. Integer Overflow in Allocation
Check all size calculations:
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
- [ ] Allocations where size is multiplied by count
```bash
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
```
- [ ] Loop counters that could wrap (unsigned underflow)
- [ ] Signed integer overflow in size checks
### 8. Format String Vulnerabilities
- [ ] User-controlled data passed as format string
```c
printf(user_input); // VULNERABLE
fprintf(stderr, user_input); // VULNERABLE
syslog(LOG_INFO, user_input); // VULNERABLE
printf("%s", user_input); // SAFE
```
```bash
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
```
### 9. Authentication & Authorization
- [ ] SSH transport relies on SSH authentication (not custom auth) - [ ] SSH transport relies on SSH authentication (not custom auth)
- [ ] No password/credential storage in plaintext - [ ] No password/credential storage in plaintext
- [ ] Server doesn't trust client-supplied paths blindly - [ ] Server doesn't trust client-supplied paths blindly
- [ ] Destination directory validated before writing - [ ] Destination directory validated before writing
### 10. TOCTOU Race Conditions ### 6. Denial of Service
- [ ] File existence check followed by open (Time-of-check to Time-of-use) - [ ] Bounded memory allocation (can't OOM server with huge chunk)
```c - [ ] Timeout on connections (no indefinite blocking)
if (access(path, F_OK) == 0) { // CHECK - [ ] Maximum connection limit or rate limiting
fd = open(path, O_RDWR); // USE — file could have changed - [ ] Malformed protocol messages handled gracefully (no crash)
}
```
- [ ] `stat()` followed by `open()` with different permissions
- [ ] Temporary file creation with predictable names
### 11. Insecure Temporary File Usage ### 7. Cryptographic Practices
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead - [ ] No custom crypto — uses OpenSSL only
- [ ] Temporary files created in world-writable directories - [ ] No hardcoded keys, IVs, or salts
- [ ] Temporary files not cleaned up on error paths - [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
- [ ] Predictable temp file names (race + symlink attack)
### 12. Hardcoded Secrets / Credentials ### 8. File System Security
- [ ] Hardcoded passwords, API keys, or tokens
- [ ] Hardcoded TLS private keys or certificates
- [ ] Hardcoded connection strings with embedded credentials
- [ ] Test certificates/keys in source tree (should be documented if intentional)
### 13. Denial of Service Vectors
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
- Check `chunk.c` for chunk count limits
- Check `protocol.c` for message size limits
- Check `config.c` for config field size limits
- [ ] **No connection limits** — server doesn't cap concurrent connections
- [ ] **No timeouts** — connections can hang indefinitely
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
- [ ] **Repeated slow reads** — slow loris style attack
- [ ] **Fork bomb** — server forks per connection without limit
### 14. Information Disclosure
- [ ] Server sends detailed error messages to client (path disclosure, version info)
- [ ] Debug logging enabled in production
- [ ] Stack traces leaked to users
- [ ] Timing side channels in authentication or comparison
### 15. File System Security
- [ ] Received file permissions validated (no SUID/SGID injection) - [ ] Received file permissions validated (no SUID/SGID injection)
- [ ] Symlink attack prevention (don't follow symlinks in destination) - [ ] Symlink attack prevention (don't follow symlinks in destination)
- [ ] Race conditions in file creation (TOCTOU) - [ ] Race conditions in file creation (TOCTOU)
- [ ] Temporary file security (if any) - [ ] Temporary file security (if any)
### 16. Cryptographic Practices
- [ ] No custom crypto — uses OpenSSL only
- [ ] No hardcoded keys, IVs, or salts
- [ ] Random data from `/dev/urandom` or OpenSSL `RAND_bytes`
## Common Vulnerability Patterns ## Common Vulnerability Patterns
### Format String Bugs ### Format String Bugs
@@ -288,59 +118,9 @@ receive_n_data(fd, buffer, expected_size);
if (!receive_n_data(fd, buffer, expected_size)) { /* handle error */ } if (!receive_n_data(fd, buffer, expected_size)) { /* handle error */ }
``` ```
## How to Scan
### Automated Pattern Search
Run these searches across the codebase:
```bash
# Buffer overflow risks
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
# Fixed size stack buffers
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
# Format string risks
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
# Malloc without null check pattern
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
# Integer overflow in allocation
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
# Path construction
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
```
### Manual Code Review
After automated scanning, manually review high-risk files:
1. `src/shared/protocol.c` — all receive paths
2. `src/shared/config.c` — deserialization logic
3. `src/shared/chunk.c` — chunk parsing
4. `src/shared/transport_tls.c` — TLS configuration
5. `src/server/server.c` — file writing and connection handling
## Output Format ## Output Format
Return findings in this structured format, one per vulnerability: For each vulnerability found:
```
## Finding: <Short descriptive title>
- **Severity**: critical/high/medium/low
- **Category**: security
- **Location**: file:line range
- **Description**: what the vulnerability is, including:
- How it can be triggered
- What the impact is (RCE, DoS, info leak, etc.)
- Whether it requires authentication
- **Suggestion**: how to fix it, including concrete code changes
- **Labels**: security, comma-separated additional labels
```
### Detailed Finding Fields
For each vulnerability found, also be prepared to report:
1. **Location** — file:line 1. **Location** — file:line
2. **Severity** — critical / high / medium / low / informational 2. **Severity** — critical / high / medium / low / informational
3. **Category** — input-validation / buffer / memory / tls / auth / dos / crypto / fs 3. **Category** — input-validation / buffer / memory / tls / auth / dos / crypto / fs
@@ -349,31 +129,6 @@ For each vulnerability found, also be prepared to report:
6. **Fix** — concrete code change 6. **Fix** — concrete code change
7. **CVSS estimate** — rough severity score if exploitable 7. **CVSS estimate** — rough severity score if exploitable
### Example
```
## Finding: Unchecked malloc in chunk deserialization allows OOM
- **Severity**: high
- **Category**: security
- **Location**: src/shared/chunk.c:45-50
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
where `count` comes directly from the network. An attacker can send a crafted
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
to either fail (crash if unchecked) or allocate enormous memory (OOM).
No authentication needed — the attack works on the initial connection.
- **Suggestion**: Add bounds checking before allocation:
```c
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
log_error("Invalid chunk file count: %u", count);
return NULL;
}
```
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
- **Labels**: security, dos
```
### Audit Summary
Also provide a summary: Also provide a summary:
``` ```
=== SECURITY AUDIT SUMMARY === === SECURITY AUDIT SUMMARY ===
@@ -385,30 +140,13 @@ Low: <count>
Informational: <count> Informational: <count>
``` ```
### No Findings
If no security issues are found, return:
```
## No security findings
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
```
## Severity Guidelines
| Severity | Definition | Example |
|---|---|---|
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
## CI & Task Execution ## CI & Task Execution
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success. When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+310
View File
@@ -0,0 +1,310 @@
---
description: Scans the FastSync codebase for security vulnerabilities — buffer overflows, path traversal, TLS issues, memory safety, and cryptographic hygiene.
mode: subagent
---
You are a security screener for the FastSync project — a high-performance file synchronization system written in C11 with TCP, SSH, and TLS transport.
## Your Role
Scan the codebase for security vulnerabilities. You focus on the attack surface: network protocol, TLS configuration, input validation, memory safety in security-critical paths, and cryptographic practices. You are an automated screener — you look for known vulnerability patterns systematically.
> **Environment rule:** for CI, dependency installation must use the project's custom Docker image (repo-root `Dockerfile`, same as CI). For local development, use `nix-shell` (see `README.md`). See `AGENTS.md`.
## Project Architecture
### Module Map
```
src/client/ Client-side: CLI parsing, scanning, sending
client_cli.c Entry point, argument parsing, config setup
client_send.c Transfer orchestration, pipeline management
scanner.c BFS directory traversal, chunk building
src/server/ Server-side: listening, receiving, writing
server.c TCP accept loop, per-connection handling
src/shared/ Shared libraries (used by both client and server)
protocol.c/h Wire protocol: status codes, send/receive primitives
compression.c/h zstd streaming compression/decompression
chunk.c/h File grouping and batch serialization
queue.c/h Thread-safe bounded queue (producer-consumer)
config.c/h Runtime configuration, serialization, parsing
data.c/h Generic buffer type (Data)
metadata.c/h File metadata (mode, uid, gid, mtime)
file.c/h File representation
array_list.c/h Dynamic array
transport_tcp.c/h TCP client/server with sendfile() zero-copy
transport_ssh.c/h SSH transport with ControlMaster
transport_tls.c/h TLS encryption via OpenSSL
multiprocessing.c/h Fork-based concurrency
log.c/h Logging utilities
utils.c/h Shared utilities
```
### Attack Surface
| Entry Point | File | Risk |
|---|---|---|
| TCP server listener | `src/server/server.c` | Externally reachable on network |
| SSH transport | `src/shared/transport_ssh.c` | Accepts data via stdio pipe |
| Protocol parser | `src/shared/protocol.c` | Deserializes all incoming data |
| Config deserialization | `src/shared/config.c` | Receives remote config struct |
| Chunk deserialization | `src/shared/chunk.c` | Receives file batches |
| TLS handshake | `src/shared/transport_tls.c` | SSL context and cert validation |
| File writer | `src/server/server.c` | Writes received files to disk |
## Security Screener Checklist
### 1. Buffer Overflow Risks
Search for these dangerous patterns in all `.c` and `.h` files:
- [ ] **Fixed-size stack buffers** used for unbounded or network-provided data
```c
char path[PATH_MAX]; // OK if PATH_MAX is used, bad if size is arbitrary
char buf[1024]; // SUSPICIOUS — what limits the input to 1024?
char line[4096]; // SUSPICIOUS — what limits the line length?
```
- [ ] **`strcpy` / `strcat` / `sprintf` calls** — all should be `snprintf` or equivalent
```bash
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c" --include="*.h"
```
- [ ] **Unbounded `sprintf` to fixed buffer**
```c
char buf[256];
sprintf(buf, "%s/%s", dir, filename); // DANGER — no size limit
```
- [ ] **Off-by-one in string operations** — `strlen` usage without `+ 1` for null terminator
- [ ] **`scanf` / `fscanf` / `sscanf` with `%s` and no width limit**
```c
sscanf(input, "%s", buffer); // DANGER — no width limit on %s
```
- [ ] **`memcpy` / `memmove` with unchecked size from network data**
### 2. Path Traversal in File Operations
Check all paths constructed from received data:
- [ ] **Files constructed with client-provided filenames + destination directory**
```c
snprintf(path, PATH_MAX, "%s/%s", dest_dir, received_filename);
```
Check for `../` filtering:
```bash
grep -rn 'snprintf.*%s.*%s.*path\|snprintf.*dest_dir\|snprintf.*base_dir' src/ --include="*.c"
```
- [ ] **`realpath()` usage** for path canonicalization
- [ ] **Symlink following** — does the server follow symlinks in the destination?
- [ ] **Null byte injection** — received filenames with embedded `\0`
### 3. Unchecked Return Values from Critical Functions
- [ ] **`malloc` / `calloc` / `realloc` return values not checked** before dereference
```bash
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
```
For each match, verify NULL check exists before use.
- [ ] **`send_n_data` / `receive_n_data` return values** not checked
- [ ] **`SSL_read` / `SSL_write`** error codes not checked
- [ ] **`write()` / `read()` syscall** return values not checked (short writes/reads)
- [ ] **`fopen()` / `open()`** return values not checked
- [ ] **`snprintf` / `vsnprintf`** negative return not handled
### 4. TLS / SSL Misconfiguration
- [ ] **TLS version not restricted** — server allows SSLv3, TLS 1.0, or TLS 1.1
```c
SSL_CTX_set_min_proto_version(ctx, TLS1_2_VERSION); // REQUIRED
```
- [ ] **Certificate verification disabled** without explicit `--insecure` flag
- [ ] **`SSL_CTX_set_verify` not called** — default is no verification
- [ ] **Weak cipher suites allowed** — need to call `SSL_CTX_set_cipher_list()`
- [ ] **Private key file permissions** not checked before loading
- [ ] **Hostname verification** not performed on server certificate
- [ ] **Session renegotiation** not limited (DoS vector)
- [ ] **TLS certificate/key paths from untrusted input** — can client specify arbitrary paths?
### 5. Memory Safety Issues
- [ ] **Use-after-free** — object freed but pointer still used later
- [ ] **Double-free** — `free()` called twice on same pointer
- [ ] **Memory leaks** on error paths — allocated but not freed before return
- [ ] **Integer overflow** in allocation size computation
```c
// DANGER: count * sizeof(Type) can overflow
void *arr = malloc(count * sizeof(Element));
// SAFE:
if (count > SIZE_MAX / sizeof(Element)) return NULL;
void *arr = malloc(count * sizeof(Element));
```
- [ ] **`realloc` return value** not saved to temporary pointer (leak on failure)
```c
// BAD: leaks original pointer on failure
buf = realloc(buf, new_size);
// GOOD:
void *tmp = realloc(buf, new_size);
if (!tmp) { free(buf); return NULL; }
buf = tmp;
```
### 6. Integer Overflow in Allocation
Check all size calculations:
- [ ] Allocations where count comes from network data (chunk count, file count, etc.)
- [ ] Allocations where size is multiplied by count
```bash
grep -rn 'malloc.*\*.*sizeof\|calloc(.*sizeof' src/ --include="*.c"
```
- [ ] Loop counters that could wrap (unsigned underflow)
- [ ] Signed integer overflow in size checks
### 7. Format String Vulnerabilities
- [ ] User-controlled data passed as format string
```c
printf(user_input); // VULNERABLE
fprintf(stderr, user_input); // VULNERABLE
syslog(LOG_INFO, user_input); // VULNERABLE
printf("%s", user_input); // SAFE
```
```bash
grep -rn 'printf(\|fprintf(\|syslog(\|snprintf(' src/ --include="*.c" | grep -v '"[^"]*%'
```
### 8. TOCTOU Race Conditions
- [ ] File existence check followed by open (Time-of-check to Time-of-use)
```c
if (access(path, F_OK) == 0) { // CHECK
fd = open(path, O_RDWR); // USE — file could have changed
}
```
- [ ] `stat()` followed by `open()` with different permissions
- [ ] Temporary file creation with predictable names
### 9. Insecure Temporary File Usage
- [ ] `mktemp` / `tmpnam` — use `mkstemp` instead
- [ ] Temporary files created in world-writable directories
- [ ] Temporary files not cleaned up on error paths
- [ ] Predictable temp file names (race + symlink attack)
### 10. Hardcoded Secrets / Credentials
- [ ] Hardcoded passwords, API keys, or tokens
- [ ] Hardcoded TLS private keys or certificates
- [ ] Hardcoded connection strings with embedded credentials
- [ ] Test certificates/keys in source tree (should be documented if intentional)
### 11. Denial of Service Vectors
- [ ] **Unbounded memory allocation** — can client request huge allocation that OOMs server?
- Check `chunk.c` for chunk count limits
- Check `protocol.c` for message size limits
- Check `config.c` for config field size limits
- [ ] **No connection limits** — server doesn't cap concurrent connections
- [ ] **No timeouts** — connections can hang indefinitely
- [ ] **Recursive parsing** — could cause stack overflow with crafted input
- [ ] **Repeated slow reads** — slow loris style attack
- [ ] **Fork bomb** — server forks per connection without limit
### 12. Information Disclosure
- [ ] Server sends detailed error messages to client (path disclosure, version info)
- [ ] Debug logging enabled in production
- [ ] Stack traces leaked to users
- [ ] Timing side channels in authentication or comparison
## How to Scan
### Automated Pattern Search
Run these searches across the codebase:
```bash
# Buffer overflow risks
grep -rn '\bstrcpy\b\|\bstrcat\b\|\bsprintf\b' src/ --include="*.c"
# Fixed size stack buffers
grep -rn 'char [a-z_]*\[[0-9]*\];' src/ --include="*.c" --include="*.h"
# Format string risks
grep -rn 'printf(\|fprintf(\|syslog(' src/ --include="*.c" | grep -v '"[^"]*%'
# Malloc without null check pattern
grep -rn '= malloc\|= calloc\|= realloc' src/ --include="*.c"
# Integer overflow in allocation
grep -rn 'malloc.*\*\|calloc.*<' src/ --include="*.c"
# Path construction
grep -rn 'snprintf.*path\|snprintf.*dir' src/ --include="*.c"
```
### Manual Code Review
After automated scanning, manually review high-risk files:
1. `src/shared/protocol.c` — all receive paths
2. `src/shared/config.c` — deserialization logic
3. `src/shared/chunk.c` — chunk parsing
4. `src/shared/transport_tls.c` — TLS configuration
5. `src/server/server.c` — file writing and connection handling
## Output Format
Return findings in this structured format, one per vulnerability:
```
## Finding: <Short descriptive title>
- **Severity**: critical/high/medium/low
- **Category**: security
- **Location**: file:line range
- **Description**: what the vulnerability is, including:
- How it can be triggered
- What the impact is (RCE, DoS, info leak, etc.)
- Whether it requires authentication
- **Suggestion**: how to fix it, including concrete code changes
- **Labels**: security, comma-separated additional labels
```
### Example
```
## Finding: Unchecked malloc in chunk deserialization allows OOM
- **Severity**: high
- **Category**: security
- **Location**: src/shared/chunk.c:45-50
- **Description**: `chunk_deserialize()` calls `malloc(count * sizeof(File))`
where `count` comes directly from the network. An attacker can send a crafted
chunk header with an extremely large count (e.g., UINT32_MAX), causing malloc
to either fail (crash if unchecked) or allocate enormous memory (OOM).
No authentication needed — the attack works on the initial connection.
- **Suggestion**: Add bounds checking before allocation:
```c
if (count > MAX_CHUNK_FILES || count > SIZE_MAX / sizeof(File)) {
log_error("Invalid chunk file count: %u", count);
return NULL;
}
```
Define `MAX_CHUNK_FILES` as a reasonable limit (e.g., 100000).
- **Labels**: security, dos
```
### No Findings
If no security issues are found, return:
```
## No security findings
The codebase appears clean in the areas checked. No vulnerabilities found at this time.
```
## Severity Guidelines
| Severity | Definition | Example |
|---|---|---|
| **critical** | Remote code execution, unauthenticated compromise | Buffer overflow on network input |
| **high** | Significant impact but requires specific conditions | DoS via unbounded allocation, path traversal |
| **medium** | Limited impact, requires auth or other conditions | TOCTOU race in file operations |
| **low** | Minor issues, defense in depth | Missing null check that's unlikely to trigger |
| **informational** | Not exploitable but violates best practice | Hardcoded value that could be configurable |
## CI & Task Execution
When using `tea` (the task execution agent) to run CI or tests, always set a sufficient timeout (e.g., 600000ms) to allow the workflow to finish. After CI completes, check the results yourself — inspect logs if the run failed. Never assume success.
## Branch Strategy
Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. See `AGENTS.md` for details.
+5 -3
View File
@@ -138,9 +138,11 @@ int LLVMFuzzerTestOneInput(const uint8_t *data, size_t size) {
Build for fuzzing: Build for fuzzing:
```bash ```bash
CC=clang CXX=clang++ cmake -B build-fuzz -S . -DENABLE_FUZZ=ON cmake -B build-fuzz -S . \
-DCMAKE_C_FLAGS="-fsanitize=fuzzer,address,undefined -g" \
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=fuzzer,address,undefined"
cmake --build build-fuzz -j$(nproc) cmake --build build-fuzz -j$(nproc)
./build-fuzz/fuzz_chunk_deserialize corpus/ -max_len=1048576 ./build-fuzz/tests/fuzz_chunk_deserialize corpus/ -max_len=1048576
``` ```
### AFL++ Harness ### AFL++ Harness
@@ -214,7 +216,7 @@ When using `tea` (the task execution agent) to run CI or tests, always set a suf
## Branch Strategy ## Branch Strategy
Never push directly to `dev` or `main`. All changes must be developed on a feature branch and merged via a pull request targeting `dev`. Create a branch (`git checkout -b <branch-name>`), push it, and open the PR with `tea pr create --repo TapTap/FastSync --base dev --head <branch-name>`. Wait for CI to pass before merging. Never push directly to `main`. All changes must be developed on a feature branch and merged via a pull request. Always create a new branch (`git checkout -b <branch-name>`) before making changes, push it, and open a PR with `gh pr create --fill`. Wait for CI to pass before merging.
## Dependency Installation ## Dependency Installation
+21 -34
View File
@@ -38,62 +38,50 @@ dd if=/dev/urandom of=/tmp/fastsync_bench/src/large.bin bs=1M count=10 2>/dev/nu
Test each configuration 3 times, record median: Test each configuration 3 times, record median:
```bash ```bash
# Real FastSync flags: -z=compression, -j=multithreading,
# --chunk-serialization, --sendfile (long form only). The old rsync-style
# spellings -c/-m/-s/-f are NOT the same options (-c=--checksum,
# -m=--prune-empty-dirs, -s=--secluded-args, -f=--filter) and must not be used.
CONFIGS=( CONFIGS=(
"Standard|" "Standard|"
"Compression|-z" "Compression|-c"
"Multithreading|-j" "Multithreading|-m"
"MT+Compression|-j -z" "MT+Compression|-m -c"
"Chunk Serialization|-j -z --chunk-serialization" "Chunk Serialization|-s"
"Sendfile|--sendfile" "MT+Compression+Chunk|-m -c -s"
"Sendfile|-f"
) )
PORT=18080
for config in "${CONFIGS[@]}"; do for config in "${CONFIGS[@]}"; do
IFS='|' read -r name flags <<< "$config" IFS='|' read -r name flags <<< "$config"
echo "=== $name ===" echo "=== $name ==="
for run in 1 2 3; do for run in 1 2 3; do
rm -rf /tmp/fastsync_bench/dst rm -rf /tmp/fastsync_bench/dst
mkdir -p /tmp/fastsync_bench/dst mkdir -p /tmp/fastsync_bench/dst
./build/server -p "$PORT" --allow-unauthenticated & ./build/server &
SERVER_PID=$! SERVER_PID=$!
sleep 0.5 sleep 0.5
START=$(date +%s%N) START=$(date +%s%N)
./build/client --source-dir /tmp/fastsync_bench/src \ ./build/client --source-dir /tmp/fastsync_bench/src \
--dest-dir /tmp/fastsync_bench/dst \ --dest-dir /tmp/fastsync_bench/dst \
--server-port "$PORT" \
--save-to-disk $flags --save-to-disk $flags
END=$(date +%s%N) END=$(date +%s%N)
ELAPSED=$(( (END - START) / 1000000 )) ELAPSED=$(( (END - START) / 1000000 ))
echo " Run $run: ${ELAPSED}ms" echo " Run $run: ${ELAPSED}ms"
kill $SERVER_PID 2>/dev/null kill $SERVER_PID 2>/dev/null
wait $SERVER_PID 2>/dev/null wait $SERVER_PID 2>/dev/null
done done
done done
``` ```
### Step 4: Full Benchmark Tool (Preferred) ### Step 4: Full Integration Benchmark (Optional)
The maintained benchmark tool is `benchmark/bench.py`. It handles building,
data generation, network shaping (LAN/WAN profiles or custom `--delay`/`--jitter`/
`--throughput`/`--loss`), rsync comparison, and JSON/table reporting:
For comprehensive benchmarking with network shaping:
```bash ```bash
python3 benchmark/bench.py --help python3 test.py --full
python3 benchmark/bench.py --runs 5 --profiles unlimited
python3 benchmark/bench.py --size-mb 100 --random-ratio 0.5 --output json
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
``` ```
Network shaping needs root (`tc`/`netem` on `lo`). SSH and TLS coverage lives in This tests LAN/WAN profiles, SSH, TLS, and compares against rsync.
the pytest integration suite, not the benchmark tool.
### Step 5: Report Results ### Step 5: Report Results
@@ -104,14 +92,13 @@ Platform: <OS, CPU, network>
Configuration | Run 1 | Run 2 | Run 3 | Median Configuration | Run 1 | Run 2 | Run 3 | Median
-----------------------|---------|---------|---------|-------- -----------------------|---------|---------|---------|--------
Standard | 0.12s | 0.11s | 0.12s | 0.12s Standard | 0.12s | 0.11s | 0.12s | 0.12s
Compression (-z) | 0.09s | 0.08s | 0.09s | 0.09s Compression (-c) | 0.09s | 0.08s | 0.09s | 0.09s
Multithreading (-j) | 0.07s | 0.07s | 0.08s | 0.07s Multithreading (-m) | 0.07s | 0.07s | 0.08s | 0.07s
MT+Compression (-j -z) | 0.05s | 0.05s | 0.06s | 0.05s MT+Compression (-m -c) | 0.05s | 0.05s | 0.06s | 0.05s
Chunk Serialization (--chunk-serialization) | 0.05s | 0.04s | 0.05s | 0.05s Sendfile (-f) | 0.04s | 0.04s | 0.04s | 0.04s
Sendfile (--sendfile) | 0.04s | 0.04s | 0.04s | 0.04s
Best configuration: Sendfile (--sendfile) Best configuration: MT+Compression (-m -c)
Throughput: <X> MB/s Throughput: <X> MB/s
``` ```
+17 -12
View File
@@ -32,19 +32,23 @@ Try to reproduce the issue with the exact command the user provides.
**Memory errors (first priority):** **Memory errors (first priority):**
```bash ```bash
rm -rf build-asan rm -rf build
cmake -B build-asan -S . -DSANITIZER=address cmake -B build -S . \
cmake --build build-asan -j$(nproc) -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
./build-asan/tests -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
cmake --build build -j$(nproc)
./build/tests
# or run the failing command # or run the failing command
``` ```
**Thread errors:** **Thread errors:**
```bash ```bash
rm -rf build-tsan rm -rf build
cmake -B build-tsan -S . -DSANITIZER=thread cmake -B build -S . \
cmake --build build-tsan -j$(nproc) -DCMAKE_C_FLAGS="-fsanitize=thread -g" \
./build-tsan/tests -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
cmake --build build -j$(nproc)
./build/tests
``` ```
**Valgrind (if ASan doesn't find it):** **Valgrind (if ASan doesn't find it):**
@@ -104,12 +108,13 @@ cmake -B build -S . && cmake --build build -j$(nproc)
./build/tests ./build/tests
# If integration test needed # If integration test needed
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" python3 test.py
# Re-run under sanitizer to confirm fix # Re-run under sanitizer to confirm fix
rm -rf build-asan rm -rf build
cmake -B build-asan -S . -DSANITIZER=address cmake -B build -S . -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
cmake --build build-asan -j$(nproc) -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
cmake --build build -j$(nproc)
# reproduce the original failing command # reproduce the original failing command
``` ```
+9 -5
View File
@@ -19,7 +19,7 @@ tea pr checkout <number>
If already on a PR branch, verify with: If already on a PR branch, verify with:
```bash ```bash
git branch --show-current git branch --show-current
git log dev..HEAD --oneline git log main..HEAD --oneline
``` ```
### Step 2: Clean build ### Step 2: Clean build
@@ -39,13 +39,17 @@ If the PR touches threading, memory management, or network code, also build with
```bash ```bash
# AddressSanitizer # AddressSanitizer
rm -rf build-asan rm -rf build-asan
cmake -B build-asan -S . -DSANITIZER=address cmake -B build-asan -S . \
-DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer -g" \
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
cmake --build build-asan -j$(nproc) cmake --build build-asan -j$(nproc)
./build-asan/tests ./build-asan/tests
# ThreadSanitizer (if threading changes) # ThreadSanitizer (if threading changes)
rm -rf build-tsan rm -rf build-tsan
cmake -B build-tsan -S . -DSANITIZER=thread cmake -B build-tsan -S . \
-DCMAKE_C_FLAGS="-fsanitize=thread -g" \
-DCMAKE_EXE_LINKER_FLAGS="-fsanitize=thread"
cmake --build build-tsan -j$(nproc) cmake --build build-tsan -j$(nproc)
./build-tsan/tests ./build-tsan/tests
``` ```
@@ -87,10 +91,10 @@ If tests fail:
### Step 6: Run integration tests (optional) ### Step 6: Run integration tests (optional)
```bash ```bash
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" python3 test.py
``` ```
This runs the integration suite (benchmarking is `benchmark/bench.py`). It takes longer — only run if the user asks or if unit tests pass. This runs the integration + benchmark suite. It takes longer — only run if the user asks or if unit tests pass.
### Step 7: Fix and commit ### Step 7: Fix and commit
+3 -3
View File
@@ -19,13 +19,13 @@ tea pr checkout <number>
If already on a PR branch, verify with: If already on a PR branch, verify with:
```bash ```bash
git branch --show-current git branch --show-current
git log dev..HEAD --oneline git log main..HEAD --oneline
``` ```
### Step 2: Get changed files ### Step 2: Get changed files
```bash ```bash
git diff dev --name-only -- '*.c' '*.h' git diff main --name-only -- '*.c' '*.h'
``` ```
This gives the list of C source and header files changed in the PR. This gives the list of C source and header files changed in the PR.
@@ -125,7 +125,7 @@ STYLE: <count>
If the user wants to post the review as a PR comment: If the user wants to post the review as a PR comment:
```bash ```bash
tea comment --repo TapTap/FastSync <number> "<review report>" tea pr comment <number> --comment "<review report>"
``` ```
## Rules ## Rules
+10 -19
View File
@@ -16,7 +16,7 @@ Ask the user or determine from context:
- **Minor** (x.Y.0) — new features, backward compatible - **Minor** (x.Y.0) — new features, backward compatible
- **Patch** (x.y.Z) — bug fixes, no protocol changes - **Patch** (x.y.Z) — bug fixes, no protocol changes
Current version: `PROTOCOL_VERSION "2.21.0"` in `src/shared/config.h` Current version: `PROTOCOL_VERSION "1.1.0"` in `src/shared/config.h`
### Step 2: Check Protocol Version ### Step 2: Check Protocol Version
@@ -37,7 +37,7 @@ rm -rf build
cmake -B build -S . cmake -B build -S .
cmake --build build -j$(nproc) cmake --build build -j$(nproc)
./build/tests ./build/tests
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" python3 test.py
``` ```
ALL tests must pass before release. ALL tests must pass before release.
@@ -46,10 +46,12 @@ ALL tests must pass before release.
```bash ```bash
# ASan # ASan
rm -rf build-asan rm -rf build
cmake -B build-asan -S . -DSANITIZER=address cmake -B build -S . \
cmake --build build-asan -j$(nproc) -DCMAKE_C_FLAGS="-fsanitize=address -fno-omit-frame-pointer" \
./build-asan/tests -DCMAKE_EXE_LINKER_FLAGS="-fsanitize=address"
cmake --build build -j$(nproc)
./build/tests
``` ```
### Step 5: Update README (If Needed) ### Step 5: Update README (If Needed)
@@ -77,23 +79,12 @@ git commit -m "Release vX.Y.Z
git tag -a vX.Y.Z -m "Release vX.Y.Z" git tag -a vX.Y.Z -m "Release vX.Y.Z"
``` ```
### Step 8: Push and Open dev → main PR ### Step 8: Push
`main` is protected and only receives changes via `dev` → `main` PRs (see AGENTS.md). Never push directly to `main`.
```bash ```bash
# Push the release commit and tag to dev git push origin main --tags
git push origin dev
git push origin vX.Y.Z
# Open the dev → main release PR for review + CI
tea pr create --repo TapTap/FastSync --head dev --base main \
--title "Release vX.Y.Z" \
--description "Release vX.Y.Z"
``` ```
Then wait for the full CI to pass and request review before the PR is merged to `main`.
### Step 9: Report ### Step 9: Report
``` ```
+2 -2
View File
@@ -102,9 +102,9 @@ Informational: <count>
... ...
=== VERDICT === === VERDICT ===
[PASS] No critical/high-severity issues found [PASS] No critical/high issues found
— or — — or —
[FAIL] <N> critical/high-severity issues must be fixed [FAIL] <N> critical/high issues must be fixed
``` ```
## Rules ## Rules
+14 -15
View File
@@ -4,31 +4,31 @@ FastSync is a high-performance file synchronization system written in C11. It su
## Dependency installation ## Dependency installation
**CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v10`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest + pytest-xdist, openssh-client, and Node.js. **CI rule:** never add `apt-get install` / `pip install` steps to CI workflows — use the custom Docker image instead. The image is built from the repo-root `Dockerfile` and is the same image CI uses: `gitea.tap-tap.win/taptap/fastsync-ci:v7`. It contains the full toolchain: gcc/g++, CMake, libzstd-dev, libssl-dev, make, git, cppcheck, clang-format, python3 + pytest, openssh-client, and Node.js.
**Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity. **Host rule:** for local development, use `nix-shell` (see `README.md`) which provides zstd, OpenSSL, CMake, and gcc. The Docker image can also be used locally for CI parity.
```bash ```bash
# Use the prebuilt CI image directly (faster, guaranteed CI parity) # Use the prebuilt CI image directly (faster, guaranteed CI parity)
docker pull gitea.tap-tap.win/taptap/fastsync-ci:v10 docker pull gitea.tap-tap.win/taptap/fastsync-ci:v7
docker tag gitea.tap-tap.win/taptap/fastsync-ci:v10 fastsync-ci:local docker tag gitea.tap-tap.win/taptap/fastsync-ci:v7 fastsync-ci:local
# Or build the image from the repo-root Dockerfile # Or build the image from the repo-root Dockerfile
# (Note: the prebuilt :v10 image reflects the previous Dockerfile state; # (Note: the prebuilt :v7 image reflects the previous Dockerfile state;
# rebuild from source to pick up any newly added packages like lcov/valgrind.) # rebuild from source to pick up any newly added packages like lcov/valgrind.)
docker build -t fastsync-ci:local . docker build -t fastsync-ci:local .
# Build, run unit tests, and run integration tests inside the container # Build, run unit tests, and run integration tests inside the container
docker run --rm -v "$PWD:/workspace" -w /workspace fastsync-ci:local \ docker run --rm -v "$PWD:/workspace" -w /workspace fastsync-ci:local \
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load' sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/'
# Avoid root-owned build/ artifacts by matching your host UID/GID # Avoid root-owned build/ artifacts by matching your host UID/GID
docker run --rm --user "$(id -u):$(id -g)" -v "$PWD:/workspace" \ docker run --rm --user "$(id -u):$(id -g)" -v "$PWD:/workspace" \
-w /workspace fastsync-ci:local \ -w /workspace fastsync-ci:local \
sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/integration/ -n 4 --dist=load' sh -c 'cmake -B build -S . && cmake --build build -j$(nproc) && ./build/tests && python3 -m pytest tests/'
``` ```
> **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source. > **Note:** The first `cmake configure` (`cmake -B build -S .`) fetches xxHash from GitHub via `FetchContent` — network access is required. Subsequent reconfigures reuse the cached source.
If a dependency is missing from the CI image, add it to the `Dockerfile` (and rebuild) rather than adding an install step to the CI workflow. If a dependency is missing from the CI image, add it to the `Dockerfile` (and rebuild) rather than adding an install step to the CI workflow.
@@ -41,7 +41,7 @@ cmake -B build -S . -DSANITIZER=address # AddressSanitizer (ASan)
cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan) cmake -B build -S . -DSANITIZER=thread # ThreadSanitizer (TSan)
``` ```
The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), then a **fast PR gate** — build + unit + a representative subset of integration tests marked `@pytest.mark.ci`, parallelized with pytest-xdist (`-n 4 --dist=load`). The full coverage jobs (full integration suite as `-m "not setpriv"`, sanitizer, fuzz, coverage, valgrind) run **only on push to `dev`/`main`**; pull requests skip them to keep PR CI under ~3 minutes. The two `setpriv` privilege tests are excluded from CI via a marker because their result depends on the runner/container uid and host mount permissions. The CI workflow (`.gitea/workflows/ci.yaml`) runs lint (clang-format, cppcheck), build + test (unit + integration), and sanitizer (currently only `address`) jobs sequentially.
## Build ## Build
@@ -53,34 +53,33 @@ cmake -B build -S . && cmake --build build -j$(nproc)
```bash ```bash
./build/tests # unit tests ./build/tests # unit tests
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" # full integration suite (CI excludes env-dependent privilege tests) python3 -m pytest tests/ # integration tests
python3 -m pytest tests/integration/ -n 4 --dist=load -m ci # PR-gate subset only
``` ```
## CI Workflow — Waiting for Results ## CI Workflow — Waiting for Results
When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Monitor CI status via the Gitea API (see below) or `tea actions`, then inspect logs on failure. When running the CI workflow via `tea` (the task execution agent), always set a sufficient timeout (e.g., 600000ms) to allow CI to finish. After CI completes, check the results yourself — do not assume success. Use `gh run watch` or similar to monitor CI status, then inspect logs on failure.
## CI Troubleshooting ## CI Troubleshooting
### If lint (clang-format) fails ### If lint (clang-format) fails
Run clang-format in the CI Docker image to match the exact CI version: Run clang-format in the CI Docker image to match the exact CI version:
```bash ```bash
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \ docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v9 \
sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i' sh -c 'find src/ tests/ -name "*.c" -o -name "*.h" | xargs clang-format -i'
``` ```
### If cppcheck fails ### If cppcheck fails
Fix reported issues locally, then verify with: Fix reported issues locally, then verify with:
```bash ```bash
docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 \ docker run --rm -v "$PWD:/workspace" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v9 \
sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/' sh -c 'cppcheck --enable=warning,style,performance,portability --suppress=missingIncludeSystem --error-exitcode=1 --inline-suppr src/ tests/'
``` ```
### If integration tests fail ### If integration tests fail
Run locally before pushing: Run locally before pushing:
```bash ```bash
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" python3 -m pytest tests/ -v --tb=short
``` ```
## Branch Strategy ## Branch Strategy
@@ -165,7 +164,7 @@ This can be cron'd locally if desired (e.g., `crontab -e` with `opencode run`).
## Is opencode a good option? ## Is opencode a good option?
**Yes, for FastSync's needs.** The hybrid model works well: **Yes, for FastSync's needs.** The hybrid model works well:
- opencode's 16 specialized agents handle deep code analysis, fixes, tests, and reviews - opencode's 17 specialized agents handle deep code analysis, fixes, tests, and reviews
- The assistant orchestrates subagents, merges branches, and iterates on CI - The assistant orchestrates subagents, merges branches, and iterates on CI
- You only review the final output - You only review the final output
-199
View File
@@ -1,199 +0,0 @@
# Changelog
All notable changes to FastSync are documented here. Versions match
`PROTOCOL_VERSION` (printed by `fastsync --version`); the client and server must
run the same version because the handshake is strict.
## [2.21.0] - 2026-09-14
### Added
- Optional server→client rejection detail (protocol 2.21.0). A rejected
operation may now carry a bounded human-readable reason via
`STATUS_ERROR_DETAIL` instead of a bare `STATUS_ERROR`, so the client can
report *why* the server refused (daemon module gate, config validation,
receiver-side path/node validation). `receive_status()` transparently maps the
new status back to `STATUS_ERROR` for every existing call site and captures
the reason into a thread-local buffer exposed by `protocol_last_error()`. The
detail body is always consumed, so the stream cannot desynchronize, and
messages are sliced to `MAX_ERROR_DETAIL_BYTES` (4096) on send.
- **Server-contacting `--dry-run` (protocol 2.21.0).** `--dry-run` now performs
a real handshake with a remote/daemon receiver and reports exactly what WOULD
change based on receiver state (existing destination files, mtimes, checksums,
basis dirs). The wire config carries the dry-run intent (`Config.dry_run`) and
the receiver answers each per-file check with `STATUS_DRY_RUN_TRANSFER` (would
transfer) or `STATUS_OK` (already up to date); the sender prints the
would-transfer set and its trailer without sending any file data. The receiver
performs the normal read-only incremental decision but mutates nothing: no temp
files, writes, renames, deletes, metadata/xattr/chown, or directory creation.
A plain local destination (no explicit `--server-port`/remote) keeps the
original client-side dry-run. Would-delete reporting for `--delete*` is
deferred to a follow-up; dry-run never deletes.
- Daemon `max connections per host` (per-source-IP concurrent cap, default 0 =
unlimited), `auth lockout threshold` (default 10; 0 disables) and
`auth lockout duration` (default 300 s) config keys.
- `fastsync-server --allow-super` opt-in for a privileged standalone TCP server;
without it a root standalone receiver forces super-user activities off (device
nodes, `--write-devices`, ownership). The `--stdio` SSH argv is client-composed,
so super activities always stay off there.
### Changed
- Config wire fields are now declared once in an X-macro table
(`CONFIG_WIRE_FIELDS` in `src/shared/config.h`) that generates the struct
members, defaults, and the send/receive sequence, removing the manual
six-site field sync. Wire bytes and `PROTOCOL_VERSION` are unchanged.
- `receive_incremental_check()` (the per-file `STATUS_CHECK` fast path) is split
into small static helpers with a short linear orchestrator. Pure refactor: the
wire byte stream and all cleanup are unchanged.
- `authorized_root` state has a single owner (`utils.c`) with read accessors; the
duplicated statics in `file.c` and the server were removed.
- `Data` records its owning `ProtocolSession` so its memory charge is returned to
the session that reserved it, regardless of the destroying thread.
- The receiver pipeline moved out of `shared` into `server/receiver_pipeline.[ch]`;
the build now uses explicit `fastsync_shared` / `fastsync_client_core` /
`fastsync_server_core` targets instead of a GLOB, and the client no longer links
server code.
- The benchmark tool generates the requested random/compressible data mix
accurately, verifies each transfer before recording it, computes correct
percentiles, adds a MB/s column, handles `tc`/netem without requiring `sudo`
when already root, builds into a dedicated `build-bench/` directory, and adds a
`--warm` incremental-transfer mode.
- The `nix-shell` dev environment provides the full toolchain (clang-format,
cppcheck, pytest-xdist, OpenSSH, rsync, iproute2, valgrind, lcov) and no longer
builds on entry.
### Security
- Enforce the daemon's per-module `max connections` cap (0 = unlimited) and add
the shared per-source `max connections per host` cap plus a cross-process
`auth lockout`. Because the listener forks one child per connection, the
counters live in an anonymous shared mapping created before the accept loop and
reclaimed by the parent's `SIGCHLD` handler, so the per-module, per-source and
auth-failure state is shared across every child (including after `SIGKILL`). The
per-source table has a bounded lifetime (expired/idle entries are reclaimed,
with a rate-limited warning when genuinely full), and the occupancy counters are
re-derived from the shared slot table on every child exit. Trusted loopback
peers are exempt (they share one address); clients behind a shared NAT/proxy
share a single per-host budget and lockout, which is documented.
- Hardening from a full security audit:
- Fail a truncated zstd frame instead of spinning forever (remote DoS).
- Open receiver destination/basis/hard-link entries `O_NONBLOCK` so a
client-planted FIFO cannot block a worker indefinitely.
- Require a regular file before `--inplace` writes, closing a FIFO-hang and a
raw-device write that bypassed the `--write-devices` gate.
- Reject SSH destinations whose user/host begins with `-` and insert `--` before
the host token, closing `-o ProxyCommand=…` argument injection (RCE).
- Gate client `--force` recursive removal behind the server `--allow-delete`
policy.
- Reject empty `hosts allow`/`hosts deny`/`auth users` values instead of
silently meaning "unrestricted".
- Restrict TLS 1.2 to AEAD suites and set server cipher preference; load the
private key TOCTOU-safely from an `O_NOFOLLOW` fd; verify IP literals against
IP SANs; guard client-cert CN truncation.
- Make `--dry-run` content-blind: it neither reads destination files nor
hashes basis files, removing a 1-bit content oracle against `read only`
modules.
- Bound glob matching (iterative DP, no exponential backtracking) and bound
line reads for filter/`--files-from`/pattern files.
- Gate `system.posix_acl_*` xattrs on `--acls` and charge decompression/chunk
allocations against the per-connection memory budget.
### Fixed
- Pre-auth NULL dereference in `config_delete()` when an over-long
`basis_count` (and the analogous count fields) was received and then failed
validation; received counts are now validated before being published.
- Leaked inherited `Data` in the forked compression-truncation unit test
(valgrind definite leak).
- `receive_status()` no longer loses a captured rejection reason when owed
keepalives are drained.
## [2.20.0] - 2026-09-13
### Security
- Cap cumulative `DirTimeList` growth and bound pre-auth config-string memory
(remote memory-exhaustion DoS).
- Daemon host access control (`hosts allow`/`hosts deny`, IPv4/IPv6/CIDR),
configurable global `max connections`, connection audit logging, and a
bounded `auth failure delay` throttle. IPv4-mapped peers are normalized and
invalid patterns are rejected at parse time (no silent fail-open).
- Honor `--timeout` for protocol I/O and bound idle/session time to defeat
keepalive slowloris; child-safe signal handling in the forked daemon.
- Compiler/linker hardening (`_FORTIFY_SOURCE`, stack protector, PIE, RELRO)
and pinned build dependencies.
### Fixed
- Use-after-free in the basis-dir oversize preflight.
- Placeholder `Data` leaks, `missing_args` leak, scanner chunk leak.
- Thread-safe logging; single fd owner and cleanup epilogue in the server
handler.
### Performance
- Metadata now crosses the wire as one packed frame (protocol 2.20.0).
- Delete keep-set and `--files-from` lookups indexed (O(n*m) → O(n)).
- Reused per-thread zstd contexts; `TCP_NODELAY` by default.
- Byte-bounded sender queues; removed a redundant scanner `stat()`.
## [2.19.0] - 2026-09-12
### Security
- **Daemon authentication rewritten as SCRAM-SHA-256 challenge/response**
(`STATUS_AUTH_CHALLENGE` → `STATUS_AUTH_RESPONSE` → `STATUS_AUTH_OK`/`STATUS_AUTH_FAILED`),
replacing the old replayable static `SHA-256(password)` bearer credential.
Each proof is bound to a fresh per-connection server nonce plus a client
nonce, so a captured response can never be reused.
- **Salted verifier store.** `--password-file`/`--early-input` now hold
`user:$fastsync$1$pbkdf2-sha256$<iters>$<salt>$<stored_key>$<server_key>`
(PBKDF2-HMAC-SHA256, default 600000 iterations, range 100000–10000000). The
legacy `user:SHA256HEX` form is hard-rejected; there is no auto-upgrade.
Generate stores offline with `fastsync-server --hash-credentials FILE
[--iterations N]`.
- **Username-enumeration hardening.** Unknown/off-list users are answered with a
dummy verifier whose salt is a deterministic per-username value
(`HMAC-SHA256(dummy_key, username)`), using the store-wide uniform iteration
count and a constant-time full-length membership scan. The dummy key is
persisted in an owner-only `<store>.dummykey` sidecar (atomic publish, exact
mode 0600) so challenges are stable across restarts.
- **Verified transport for auth-required modules.** A module with `auth users`
accepts credentials only over verified TLS whose client certificate matches
`--client-cn`, or — when `--allow-unauthenticated` is explicitly set —
plaintext from a loopback peer. Remote plaintext is refused before any
challenge. Clients must use `--tls` to send `--password-file` credentials to a
non-loopback daemon; `--client-cn` is mandatory with `--tls`.
- **Secret hygiene.** The plaintext password, derived keys, nonces/proofs and
the dummy key are wiped from memory on every path and never logged.
- Carried-over hardening: `-K` TOCTOU-safe directory walk
(`openat(O_NOFOLLOW)` per component), always shell-quoted SSH remote path,
TLS compression/renegotiation disabled, race-free (open-then-`fstat`)
`--password-file`/`--early-input` checks, log-injection escaping, and lazy
protocol debug escaping.
### Added
- `fastsync-server --hash-credentials FILE [--iterations N]` offline tool.
- `<store>.dummykey` sidecar (auto-created, owner-only, 0600).
- Integration tests for auth replay rejection, malformed frames, legacy-store
refusal, and the loopback/TLS transport policy; fuzz targets for config
receive and daemon-auth parsing.
### Changed
- **Protocol version 2.18.0 → 2.19.0 (breaking).** The config-frame auth block
is now `[present][username]` (digest removed) and the auth challenge/response
frames are interleaved between the config frame and its `STATUS_OK`. A 2.19.0
client and a 2.18.0 server (or vice versa) fail cleanly at the handshake.
- Daemon modules declaring `auth users` require a configured credential store at
startup (fail closed); operators regenerate stores from plaintext with
`--hash-credentials`.
### Notes
- First tagged release. FastSync implements rsync-compatible file
synchronization over TCP and SSH with TLS (OpenSSL), streaming zstd
compression, multithreaded transfers, and incremental sync. See
[RSYNC_COMPAT.md](RSYNC_COMPAT.md) for the flag-parity matrix.
+26 -199
View File
@@ -1,6 +1,6 @@
cmake_minimum_required(VERSION 3.22) cmake_minimum_required(VERSION 3.22)
project(FastFileTransfer VERSION 2.21.0) project(FastFileTransfer)
set(CMAKE_EXPORT_COMPILE_COMMANDS ON) set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
set(CMAKE_C_STANDARD 11) set(CMAKE_C_STANDARD 11)
@@ -38,24 +38,11 @@ if(ENABLE_COVERAGE)
add_link_options(--coverage) add_link_options(--coverage)
endif() endif()
# --- Build hardening option ---
# Production hardening is applied to the shipping server/client binaries only,
# and only when no sanitizer or coverage instrumentation is active: sanitizers
# carry their own instrumentation, and _FORTIFY_SOURCE requires an optimising
# build (never the -O0 used for coverage).
option(ENABLE_HARDENING "Enable compiler/linker hardening for production targets" ON)
set(HARDENING_ACTIVE OFF)
if(ENABLE_HARDENING AND SANITIZER STREQUAL "none" AND NOT ENABLE_COVERAGE)
set(HARDENING_ACTIVE ON)
endif()
include(FetchContent) include(FetchContent)
FetchContent_Declare( FetchContent_Declare(
xxhash xxhash
GIT_REPOSITORY https://github.com/Cyan4973/xxHash GIT_REPOSITORY https://github.com/Cyan4973/xxHash
# v0.8.3 is a lightweight tag pointing at this exact commit (no ^{} peel GIT_TAG v0.8.3
# entry); pin the commit SHA instead of the mutable tag.
GIT_TAG e626a72bc2321cd320e953a0ccf1584cad60f363 # v0.8.3
SOURCE_SUBDIR cmake_unofficial SOURCE_SUBDIR cmake_unofficial
) )
FetchContent_MakeAvailable(xxhash) FetchContent_MakeAvailable(xxhash)
@@ -70,179 +57,35 @@ endif()
find_package(OpenSSL REQUIRED) find_package(OpenSSL REQUIRED)
# --- Explicit source lists --- file(GLOB SHARED_SRCS "src/shared/*.c")
# The shared library is self-contained: it must never depend on the client or set(FILE_STORE_SRCS "${CMAKE_CURRENT_SOURCE_DIR}/src/shared/file_store.c")
# server modules. In particular, the receiver pipeline (receive_thread / list(REMOVE_ITEM SHARED_SRCS ${FILE_STORE_SRCS})
# write_thread) lives under src/server, not here, so the client executable can file(GLOB SERVER_SRCS "src/server/*.c")
# link the shared library without pulling in any server code. set(SERVER_RECEIVER_SRCS src/server/receiver.c)
set(SHARED_SRCS file(GLOB CLIENT_SRCS "src/client/*.c")
src/shared/array_list.c
src/shared/batch.c
src/shared/charset.c
src/shared/checksum.c
src/shared/chmod.c
src/shared/chunk.c
src/shared/compression.c
src/shared/config.c
src/shared/credentials.c
src/shared/daemon_conf.c
src/shared/daemon_limits.c
src/shared/data.c
src/shared/delay_updates.c
src/shared/delta.c
src/shared/file.c
src/shared/file_list.c
src/shared/file_receive.c
src/shared/file_send.c
src/shared/file_store.c
src/shared/filter.c
src/shared/hardlink.c
src/shared/identity.c
src/shared/log.c
src/shared/metadata.c
src/shared/motd.c
src/shared/multiprocessing.c
src/shared/protocol.c
src/shared/queue.c
src/shared/stop_condition.c
src/shared/transport_ssh.c
src/shared/transport_tcp.c
src/shared/transport_tls.c
src/shared/utils.c
src/shared/xattr.c
)
# Server implementation (no main): the receiver read/write pipeline plus the
# CLI parser. The server executable adds its own main (server.c).
set(SERVER_CORE_SRCS
src/server/receiver.c
src/server/receiver_pipeline.c
src/server/server_cli.c
)
set(SERVER_MAIN_SRCS src/server/server.c)
# Client implementation (no main): everything except the CLI entry point.
set(CLIENT_CORE_SRCS
src/client/change_list.c
src/client/client_send.c
src/client/client_validation.c
src/client/scanner.c
src/client/usage.c
)
set(CLIENT_MAIN_SRCS src/client/client_cli.c)
# --- Library targets ---
add_library(fastsync_shared STATIC ${SHARED_SRCS})
target_include_directories(fastsync_shared PUBLIC src/shared)
target_link_libraries(fastsync_shared PUBLIC Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL
OpenSSL::Crypto xxhash)
add_library(fastsync_client_core STATIC ${CLIENT_CORE_SRCS})
target_include_directories(fastsync_client_core PUBLIC src/client)
target_link_libraries(fastsync_client_core PUBLIC fastsync_shared)
add_library(fastsync_server_core STATIC ${SERVER_CORE_SRCS})
target_include_directories(fastsync_server_core PUBLIC src/server)
target_link_libraries(fastsync_server_core PUBLIC fastsync_shared)
# --- Main executables --- # --- Main executables ---
# The client links only the shared library and its own core; it deliberately add_executable(server ${SERVER_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS})
# does NOT get src/server on its include path nor compile receiver.c. target_include_directories(server PRIVATE src/shared src/server src/client)
add_executable(server ${SERVER_MAIN_SRCS}) target_link_libraries(server PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
target_link_libraries(server PRIVATE fastsync_server_core)
add_executable(client ${CLIENT_MAIN_SRCS}) add_executable(client ${CLIENT_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
target_link_libraries(client PRIVATE fastsync_client_core) target_include_directories(client PRIVATE src/shared src/server src/client)
target_link_libraries(client PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
# --- Production hardening ---
# Each compile flag is probed so a compiler/architecture that lacks it still
# configures cleanly. _FORTIFY_SOURCE is guarded separately because it only
# works in an optimising build. xxHash is a static archive built by
# FetchContent, so it must be position-independent for the -pie link; the same
# applies to the first-party static libraries linked into the -pie binaries.
if(HARDENING_ACTIVE)
set_target_properties(xxhash fastsync_shared fastsync_server_core fastsync_client_core
PROPERTIES POSITION_INDEPENDENT_CODE ON)
include(CheckCCompilerFlag)
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
check_c_compiler_flag("${flag}" ${_harden_var})
endforeach()
check_c_compiler_flag("-D_FORTIFY_SOURCE=2" HARDEN_FORTIFY_SOURCE)
foreach(target fastsync_shared fastsync_server_core fastsync_client_core server client)
foreach(flag -fstack-protector-strong -fstack-clash-protection -fPIE)
string(MAKE_C_IDENTIFIER "HARDEN_${flag}" _harden_var)
if(${_harden_var})
target_compile_options(${target} PRIVATE ${flag})
endif()
endforeach()
if(HARDEN_FORTIFY_SOURCE)
target_compile_options(${target} PRIVATE -D_FORTIFY_SOURCE=2)
endif()
endforeach()
foreach(target server client)
target_link_options(${target} PRIVATE -pie -Wl,-z,relro -Wl,-z,now -Wl,-z,noexecstack)
endforeach()
endif()
# --- Testing --- # --- Testing ---
enable_testing() enable_testing()
# --- Unit tests --- # Common test libraries
# The monolithic test binary exercises both client and server code, so it is set(TEST_LIBS Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL OpenSSL::Crypto xxhash)
# the one place that legitimately sees both include directories and links both set(TEST_INCLUDES tests src/shared src/server src/client)
# core libraries. client_cli.c is compiled here directly (with the test build
# define) rather than linked from fastsync_client_core so its test-only shims
# and the absence of main() are preserved.
set(TEST_SRCS
tests/runner.c
tests/test_array_list.c
tests/test_batch.c
tests/test_change_list.c
tests/test_checksum.c
tests/test_chunk.c
tests/test_client_cli.c
tests/test_compression.c
tests/test_config.c
tests/test_credentials.c
tests/test_daemon_conf.c
tests/test_daemon_limits.c
tests/test_data.c
tests/test_delay_updates.c
tests/test_delta.c
tests/test_file.c
tests/test_file_list.c
tests/test_file_sendfile.c
tests/test_fuzz_smoke.c
tests/test_glob.c
tests/test_hardlink.c
tests/test_iconv.c
tests/test_log.c
tests/test_metadata.c
tests/test_motd.c
tests/test_multiprocessing.c
tests/test_property.c
tests/test_protocol.c
tests/test_protocol_error.c
tests/test_queue.c
tests/test_receiver_timeout.c
tests/test_robustness.c
tests/test_scanner.c
tests/test_server.c
tests/test_server_cli.c
tests/test_shared_utils.c
tests/test_stop.c
tests/test_stress.c
tests/test_transport_ssh.c
tests/test_transport_tcp.c
tests/test_transport_tls.c
tests/test_xattr.c
)
add_executable(tests ${TEST_SRCS} src/client/client_cli.c) # Monolithic test binary (backward compatible)
target_include_directories(tests PRIVATE tests) file(GLOB TEST_SRCS "tests/test_*.c" "tests/runner.c")
add_executable(tests ${TEST_SRCS} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS} src/client/scanner.c src/client/client_cli.c src/client/client_validation.c src/client/usage.c)
target_include_directories(tests PRIVATE ${TEST_INCLUDES})
target_compile_definitions(tests PRIVATE FASTSYNC_TEST_BUILD) target_compile_definitions(tests PRIVATE FASTSYNC_TEST_BUILD)
target_link_libraries(tests PRIVATE fastsync_server_core fastsync_client_core) target_link_libraries(tests PRIVATE ${TEST_LIBS})
add_test(NAME unit_all COMMAND tests) add_test(NAME unit_all COMMAND tests)
# --- Fuzz targets (requires clang) --- # --- Fuzz targets (requires clang) ---
@@ -251,29 +94,13 @@ if(ENABLE_FUZZ)
if(NOT CMAKE_C_COMPILER_ID MATCHES "Clang") if(NOT CMAKE_C_COMPILER_ID MATCHES "Clang")
message(FATAL_ERROR "ENABLE_FUZZ requires Clang (compiler is ${CMAKE_C_COMPILER_ID})") message(FATAL_ERROR "ENABLE_FUZZ requires Clang (compiler is ${CMAKE_C_COMPILER_ID})")
endif() endif()
set(FUZZ_SRCS file(GLOB FUZZ_SRCS "tests/fuzz/*.c")
tests/fuzz/fuzz_chunk_deserialize.c
tests/fuzz/fuzz_compress_decompress.c
tests/fuzz/fuzz_config_receive.c
tests/fuzz/fuzz_delta_deserialize.c
tests/fuzz/fuzz_delta_signature_deserialize.c
tests/fuzz/fuzz_glob_match.c
tests/fuzz/fuzz_identity_parse.c
tests/fuzz/fuzz_manifest.c
tests/fuzz/fuzz_metadata_from_buf.c
tests/fuzz/fuzz_protocol_framing.c
tests/fuzz/fuzz_xattr_block.c
)
# Compile the sources under test directly so libFuzzer's coverage
# instrumentation sees them (static libraries would be uninstrumented).
set(FUZZ_CORE_SRCS ${SHARED_SRCS} src/server/receiver.c src/server/receiver_pipeline.c)
foreach(FUZZ_SRC ${FUZZ_SRCS}) foreach(FUZZ_SRC ${FUZZ_SRCS})
get_filename_component(FUZZ_NAME ${FUZZ_SRC} NAME_WE) get_filename_component(FUZZ_NAME ${FUZZ_SRC} NAME_WE)
add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${FUZZ_CORE_SRCS}) add_executable(${FUZZ_NAME} ${FUZZ_SRC} ${SHARED_SRCS} ${FILE_STORE_SRCS} ${SERVER_RECEIVER_SRCS})
target_include_directories(${FUZZ_NAME} PRIVATE tests src/shared src/server) target_include_directories(${FUZZ_NAME} PRIVATE ${TEST_INCLUDES})
target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer) target_compile_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined -fno-omit-frame-pointer)
target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined) target_link_options(${FUZZ_NAME} PRIVATE -fsanitize=fuzzer,address,undefined)
target_link_libraries(${FUZZ_NAME} PRIVATE Threads::Threads ${ZSTD_LIBRARY} OpenSSL::SSL target_link_libraries(${FUZZ_NAME} PRIVATE ${TEST_LIBS})
OpenSSL::Crypto xxhash)
endforeach() endforeach()
endif() endif()
+1 -1
View File
@@ -3,7 +3,7 @@ RUN apt-get update && apt-get install -y --no-install-recommends \
gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \ gcc g++ make libc6-dev cmake libzstd-dev libssl-dev git ca-certificates curl cppcheck clang-format \
python3 python3-pip python3-venv openssl openssh-client \ python3 python3-pip python3-venv openssl openssh-client \
lcov valgrind clang libclang-rt-18-dev && \ lcov valgrind clang libclang-rt-18-dev && \
pip3 install --break-system-packages pytest pytest-xdist && \ pip3 install --break-system-packages pytest && \
curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \ curl -fsSL https://deb.nodesource.com/setup_20.x | bash - && \
apt-get install -y --no-install-recommends nodejs && \ apt-get install -y --no-install-recommends nodejs && \
rm -rf /var/lib/apt/lists/* rm -rf /var/lib/apt/lists/*
+52 -227
View File
@@ -6,10 +6,6 @@ source/destination model and rsync-style options while adding optional
multithreading, streaming zstd compression, chunking, zero-copy TCP transfers, multithreading, streaming zstd compression, chunking, zero-copy TCP transfers,
and native TCP/TLS transports. and native TCP/TLS transports.
The release version is FastSync's client/server protocol version (printed by
`fastsync --version`); client and server must match. See
[CHANGELOG.md](CHANGELOG.md) for the history.
The compatibility target is straightforward: The compatibility target is straightforward:
- Existing rsync commands should keep the same meaning. - Existing rsync commands should keep the same meaning.
@@ -67,26 +63,13 @@ replacement for every rsync feature or protocol mode.
- Archive mode does not yet provide all of rsync's `-rlptgoD` behavior. - Archive mode does not yet provide all of rsync's `-rlptgoD` behavior.
- Symlink transfer is incomplete; link targets are not yet recreated in all - Symlink transfer is incomplete; link targets are not yet recreated in all
modes. modes.
- Owner/group, ACL, xattr, and hard-link handling is incomplete or - Owner/group, ACL, xattr, hard-link, device, and special-file handling is
unavailable. incomplete or unavailable.
- Device and special-file preservation is implemented with documented - Sparse-file handling does not yet preserve all holes correctly.
divergences: recreated device nodes require `CAP_MKNOD` on the receiver (a - `--partial`, `--partial-dir`, `--append`, and `--append-verify` are not yet
non-root receiver skips the entry), and sockets cannot be recreated (FIFOs full rsync-style resumable transfers.
are). - Several rsync short options currently have FastSync-specific meanings. Do
- Sparse-file hole preservation (`-S`, `--sparse`) is implemented receiver-side: not assume every short option is interchangeable yet.
long all-zero runs are written as holes (no wire change; the full file image
is already in memory).
- `--partial`, `--partial-dir`, `-P`, `--append`, and `--append-verify` keep
the write atomic (temp + rename). With `--partial`, a failed/interrupted write
now retains the already-written temp at the destination path (best-effort) so
a later `--append`/`--append-verify` run can resume it.
- `--dirs` is not implemented. Its compatibility aliases `--old-dirs` and
`--old-d` are recognized but rejected explicitly rather than silently using
FastSync's recursive directory behavior.
- Short-option names are now rsync-parity (Phase 7 Wave A): FastSync's former
collisions were renamed (`-j`/`--threads`, `--preserve`, `--sendfile`,
`--chunk-serialization`, `--timeout`, `--ssh-port`), so `-m`, `-M`, `-f`,
`-s`, `-T`, `-p`, `-c`, `-a`, and `-z` follow rsync. See `RSYNC_COMPAT.md`.
The detailed flag matrix is maintained in The detailed flag matrix is maintained in
[`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented, [`RSYNC_COMPAT.md`](RSYNC_COMPAT.md). It distinguishes implemented,
@@ -96,50 +79,37 @@ partial, alternate, and planned behavior.
### Build ### Build
`compile_commands.json` is a symlink to `build/compile_commands.json` and is used by clangd/editor tooling; its target is generated by the build, so it dangles until the first build.
### Client ### Client
| Argument | Description | | Argument | Description |
|----------|-------------| |----------|-------------|
| Positional | `<source> <dest>` — automatic SSH detection if dest contains `:` | | Positional | `<source> <dest>` — automatic SSH detection if dest contains `:` |
| `-c, --checksum` | Verify content by checksum instead of size+mtime | | `-c [level]` | Compression with optional level (1–22, default 5) |
| `-z, --compress [level]` | Enable streaming zstd compression (level 1–22, default 5) | | `-z [level]` | Alias for `-c` |
| `-a, --archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials (not compression/multithreading) | | `-a, --archive` | Archive mode: enables `-c -m -M` (no `-s`) |
| `-j, --threads[=N]` | Multithreading mode; `N` (1–256) sets the parallel scanner worker count, bare `-j`/`--threads` uses the default | | `-m` | Multithreading mode |
| `-m` | rsync `--prune-empty-dirs` (short form now rsync-parity) | | `-s` | Chunk serialization (batch all files per chunk) |
| `--chunk-serialization` | Chunk serialization (batch all files per chunk; long form only) | | `-f, --sendfile` | Sendfile zero-copy. Incompatible with `-c` / `-s`. TCP only. |
| `-s` | rsync `--secluded-args` compatibility no-op (remote SSH argv is already injection-safe) | | `-M, --preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) |
| `--sendfile` | Sendfile zero-copy. Incompatible with compression / chunk serialization. TCP only. Long form only. |
| `--preserve` | Preserve supported file metadata (mode and mtime; ownership and atime are unsupported) |
| `-n, --dry-run` | Scan and print what would be transferred | | `-n, --dry-run` | Scan and print what would be transferred |
| `-p, --perms` | Preserve permission bits (part of the metadata bundle) | | `-p <port>` | SSH port (default: 22) |
| `--ssh-port <port>` | SSH port (default: 22) |
| `-v, --verbose` | Enable debug logging | | `-v, --verbose` | Enable debug logging |
| `-q, --quiet` | Suppress non-error output |
| `--progress` | Show real-time transfer speed | | `--progress` | Show real-time transfer speed |
| `-P` | Enables partial-transfer mode + progress output; interrupted writes retain the already-written temp for resumption | | `--delete` | Delete files on receiver not present in source |
| `--delete` | Delete files on receiver not present in source (default timing: delete-after, i.e. only after the whole transfer succeeded) |
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`) |
| `--delete-during`, `--del` | Delete extras once the keep-set is known, before data is applied (implies `--delete`) |
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`) |
| `--delete-after` | Explicit delete-after timing (implies `--delete`) |
| `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) | | `--exclude <pattern>` | Exclude files matching glob pattern (repeatable) |
| `--exclude-from <file>` | Read exclude patterns from a file (one per line) | | `--exclude-from <file>` | Read exclude patterns from a file (one per line) |
| `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) | | `--include <pattern>` | Only transfer files matching glob pattern (repeatable, whitelist) |
| `--max-size <n>` | Skip files larger than n bytes | | `--max-size <n>` | Skip files larger than n bytes |
| `--min-size <n>` | Skip files smaller than n bytes | | `--min-size <n>` | Skip files smaller than n bytes |
| `--max-alloc <SIZE>` | Maximum single allocation (binary units: B, K, M, G, T, P, E; default 1G) | | `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `-s`. |
| `--incremental` | Skip files unchanged since last transfer (size + mtime). Auto-enables `--preserve`. Incompatible with `--chunk-serialization`. |
| `--existing` | Skip files not already present at the destination; update existing files normally. | | `--existing` | Skip files not already present at the destination; update existing files normally. |
| `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second | | `--bwlimit <KB/s>` | Bandwidth limit in kilobytes per second |
| `--chunk-size <n>` | Chunk size in bytes (default: 10485760) | | `--chunk-size <n>` | Chunk size in bytes (default: 10485760) |
| `--timeout <sec>` | Positive I/O timeout in seconds, applied to both the socket (`SO_RCVTIMEO`/`SO_SNDTIMEO`, built-in default 30 s) and the per-message protocol poll deadline (built-in default 60 s). Omit the option to keep both built-ins; `0` is rejected. The server side keeps the built-in 60 s protocol window (the value is not sent on the wire). | | `--timeout <sec>` | I/O timeout in seconds (default: 30) |
| `--contimeout <sec>` | Connection timeout in seconds (default: 10) | | `--contimeout <sec>` | Connection timeout in seconds (default: 10) |
| `--backup` | Backup existing destination files before overwriting | | `--backup` | Backup existing destination files before overwriting |
| `--backup-dir <dir>` | Target directory for backups (requires `--backup`) | | `--backup-dir <dir>` | Target directory for backups (requires `--backup`) |
| `--stats` | Print transfer statistics at end (bytes, files, timing) | | `--stats` | Print transfer statistics at end (bytes, files, timing) |
| `-h, --human-readable` | Format transfer byte sizes with binary units |
| `--max-depth <n>` | Maximum directory depth to recurse (0 = unlimited, default: 0) | | `--max-depth <n>` | Maximum directory depth to recurse (0 = unlimited, default: 0) |
| `--log-file <path>` | Write log messages to file instead of stderr | | `--log-file <path>` | Write log messages to file instead of stderr |
| `--source-dir <path>` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) | | `--source-dir <path>` | Source directory (overrides `FASTSYNC_SOURCE_DIR`) |
@@ -151,20 +121,7 @@ partial, alternate, and planned behavior.
| `--cert <path>` | TLS certificate file (PEM) | | `--cert <path>` | TLS certificate file (PEM) |
| `--key <path>` | TLS private key file (PEM) | | `--key <path>` | TLS private key file (PEM) |
| `--ca <path>` | TLS CA certificate file for verification (PEM) | | `--ca <path>` | TLS CA certificate file for verification (PEM) |
| `--client-cn <name>` | TLS client certificate common name; mandatory with `--tls` (a TLS connection always verifies the client CN) | | `--client-cn <name>` | Required TLS client certificate common name |
**Per-message vs. connection timeouts.** `--timeout` bounds each individual protocol
send/receive (the `poll()` deadline), so a peer that stops mid-frame is dropped. It
does not, by itself, stop a peer that keeps sending well-formed frames forever. The
receiver therefore also enforces two wall-clock (`CLOCK_MONOTONIC`) bounds on a
connection: a **1 hour** idle limit and a **24 hour** overall session cap. Only
frames that move real work (not `STATUS_KEEPALIVE`/`STATUS_ABORT` and not an
empty `STATUS_CHECK_BATCH`/`STATUS_DIR_TIMES`) refresh the idle timestamp, so a
peer cannot hold a connection slot by emitting cheap empty frames; a peer that
fabricates minimal non-empty frames can still occupy a slot until the 24 hour
cap, since no bound can require actual payload without risking a legitimate
long operation. Both are deliberately generous so a legitimate long-running
transfer is never aborted.
### Server ### Server
@@ -178,8 +135,7 @@ transfer is never aborted.
| `--ca <path>` | TLS CA certificate file for verification (PEM) | | `--ca <path>` | TLS CA certificate file for verification (PEM) |
| `--destination-root <path>` | Authorized destination root (default: `.`) | | `--destination-root <path>` | Authorized destination root (default: `.`) |
| `--allow-delete` | Permit manifest deletion | | `--allow-delete` | Permit manifest deletion |
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. **Rejected with `--stdio`** (the SSH remote argv is client-composed, so a client could otherwise pass it and defeat the secure default; operators exposing `fastsync-server --stdio` over SSH must use a forced command if the default must hold). No effect when not root. | | `--allow-unauthenticated` | Permit plaintext TCP clients |
| `--allow-unauthenticated` | Permit plaintext TCP clients. For an `auth users` module this opts in **loopback plaintext only**; remote auth still requires verified TLS, so the flag never permits remote plaintext auth. |
| `-v, --verbose` | Enable debug logging | | `-v, --verbose` | Enable debug logging |
| `--help` | Show help | | `--help` | Show help |
@@ -205,7 +161,7 @@ transfer is never aborted.
3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`; 3. **FileMetadata** — `mode`, `uid`, `gid`, `mtime_sec`, `mtime_nsec`;
uid / gid are advisory wire fields and are never applied by the receiver; uid / gid are advisory wire fields and are never applied by the receiver;
atime is unsupported atime is unsupported
4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`. 4. **Config** — runtime parameters (transported over wire, TLS settings excluded). Includes `timeout`, `contimeout`, `quiet`, `backup`, `backup_dir`, `stats`, `max_depth`, `log_file`, `queue_size`.
5. **Queue** — thread-safe bounded queue with condition variables 5. **Queue** — thread-safe bounded queue with condition variables
6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement 6. **DirectoryScanner** — recursive BFS traversal with exclude and include pattern support, max-depth enforcement
@@ -305,28 +261,17 @@ The remote host must have `fastsync-server` available in `PATH`, or use
working directory, so use a destination below that directory unless the working directory, so use a destination below that directory unless the
remote server is otherwise configured with a matching authorized root. remote server is otherwise configured with a matching authorized root.
The remote `--stdio` server argv is composed by the client, so it must never
be trusted to opt a root receiver into super-user activities: `--allow-super`
is rejected with `--stdio` and super stays off on that path. Operators
exposing `fastsync-server --stdio` over SSH must use a forced command (e.g. an
`authorized_keys` `command=` entry) if the default must hold.
```bash ```bash
ssh user@host 'mkdir -p destination' ssh user@host 'mkdir -p destination'
./build/client /path/to/source user@host:destination ./build/client /path/to/source user@host:destination
``` ```
FastSync is **push-only**: the source (first argument) is always a local
directory and only the destination may be remote. A remote source such as
`client user@host:src ./local` (a "pull") is intentionally not supported; see
[RSYNC_COMPAT.md](RSYNC_COMPAT.md#direction).
### TCP transfer ### TCP transfer
Start the FastSync server: Start the FastSync server:
```bash ```bash
./build/server --destination-root /path/to -p 8080 --allow-unauthenticated ./build/server --destination-root /path/to -p 8080
``` ```
Then run the client: Then run the client:
@@ -378,7 +323,7 @@ FastSync-native are optional performance or transport extensions.
./build/client --incremental --checksum /source/ user@host:destination/ ./build/client --incremental --checksum /source/ user@host:destination/
#Preserve supported mode and timestamp metadata #Preserve supported mode and timestamp metadata
./build/client --preserve /source/ user@host:destination/ ./build/client -M /source/ user@host:destination/
#Keep backups of overwritten destination files #Keep backups of overwritten destination files
./build/client --backup --backup-dir backups \ ./build/client --backup --backup-dir backups \
@@ -392,40 +337,28 @@ features without changing the meaning of ordinary compatibility options.
| Option | Purpose | | Option | Purpose |
|---|---| |---|---|
| `-j`, `--threads[=N]` | Enable the multithreaded scanner/loader/sender pipeline. `N` (1–256) sets the parallel scanner worker count; bare `-j`/`--threads` uses the default. | | `-m` | Enable the multithreaded scanner/loader/sender pipeline. |
| `-z [level]`, `--compress [level]` | Enable streaming zstd compression, levels 1-22. | | `-c [level]`, `-z [level]` | Enable streaming zstd compression, levels 1-22. |
| `--compress-level <n>` | Set the zstd compression level. | | `--compress-level <n>` | Set the zstd compression level. |
| `--zc <alg>` | Alias for `--compress-choice`. FastSync supports `zstd` and `none`. |
| `--zl <n>` | Alias for `--compress-level`. |
| `--skip-compress <list>` | Skip compression for comma-separated suffixes; incompatible with `--chunk-serialization`. |
| `--compress-threads <n>` | Use `n` zstd compression workers. Requires compression and a zstd build with threaded support; the setting affects sender CPU work only. |
| `--chunk-size <bytes>` | Set the transfer chunk size. | | `--chunk-size <bytes>` | Set the transfer chunk size. |
| `--chunk-serialization` | Enable FastSync chunk serialization (long form only; `-s` is rsync's `--secluded-args`). | | `-s` | Enable FastSync chunk serialization. |
| `--sendfile` | Use TCP `sendfile()` zero-copy transfer. Incompatible with compression and chunk serialization. Long form only. | | `-f`, `--sendfile` | Use TCP `sendfile()` zero-copy transfer. Incompatible with compression and chunk serialization. |
| `--delta` | Use FastSync-native block delta transfer. Requires `--incremental`. | | `--delta` | Use FastSync-native block delta transfer. Requires `--incremental`. |
| `--delta-block <bytes>` | Set the FastSync delta block size (`--block-size` is an alias). | | `--delta-block <bytes>` | Set the FastSync delta block size. |
| `--delta-max <bytes>` | Limit files eligible for FastSync delta transfer. | | `--delta-max <bytes>` | Limit files eligible for FastSync delta transfer. |
| `--server-host <host>` | Select the TCP server host. | | `--server-host <host>` | Select the TCP server host. |
| `--server-port <port>` | Select the TCP server port (`--port <port>` and `--port=<port>` are rsync-friendly aliases). | | `--server-port <port>` | Select the TCP server port. |
| `--tls` | Enable TLS for TCP transport. | | `--tls` | Enable TLS for TCP transport. |
| `--bwlimit <KB/s>` | Apply token-bucket bandwidth limiting. | | `--bwlimit <KB/s>` | Apply token-bucket bandwidth limiting. |
| `--progress` | Show transfer progress and throughput. | | `--progress` | Show transfer progress and throughput. |
| `--stats` | Print transfer statistics. | | `--stats` | Print transfer statistics. |
| `--timeout <seconds>` | Set the socket **and** per-message protocol I/O timeout (positive seconds). Omit to keep the built-in 30 s socket / 60 s protocol defaults. | | `--timeout <seconds>` | Set I/O timeout. |
| `--contimeout <seconds>` | Set connection timeout. | | `--contimeout <seconds>` | Set connection timeout. |
Short-option conflicts with rsync have been resolved for the CLI namespace Current short-option conflicts are tracked as compatibility work. In
(Phase 7): `-c` is now rsync's `--checksum`, `-m` is `--prune-empty-dirs`, `-M` particular, FastSync currently uses `-p` for SSH port, `-s` for chunk
is `--remote-option`, `-f` is `--filter`, `-s` is `--secluded-args`, `-p` is serialization, and `-S` for sparse handling. These meanings must be reconciled
`--perms`, and `-T` is `--temp-dir`. FastSync's own flags were renamed to before FastSync can claim full rsync CLI compatibility.
long-form-only or new shorts: multithreading is `-j`/`--threads`, metadata
is `--preserve`, sendfile is `--sendfile`, chunk serialization is
`--chunk-serialization`, timeout is `--timeout`, and SSH port is `--ssh-port`.
`-a`/`--archive` is now real rsync archive (`-rlptgoD`).
`--secluded-args` (and its short form `-s`) is accepted as a compatibility
no-op. It does not change FastSync's transport or protocol behavior, because
remote SSH argv is already built injection-safe.
## Client Options ## Client Options
@@ -433,13 +366,9 @@ remote SSH argv is already built injection-safe.
| Option | Description | | Option | Description |
|---|---| |---|---|
| `-a`, `--archive` | rsync archive mode (`-rlptgoD`): links, metadata, devices and specials. | | `-a`, `--archive` | Enable current archive preset. Full rsync archive semantics are planned. |
| `-n`, `--dry-run` | Scan and report without writing files. | | `-n`, `--dry-run` | Scan and report without writing files. |
| `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. Default timing is delete-after: extras are removed only after the whole transfer succeeded. | | `--delete` | Request removal of destination entries absent from the source. The server must allow deletion. |
| `--delete-before` | Delete extras before the transfer starts (implies `--delete`). |
| `--delete-during`, `--del` | Delete extras once the keep-set manifest is known, before data is applied (implies `--delete`; early mode, same engine behaviour as `--delete-before`). |
| `--delete-delay` | Delete extras only after a successful transfer (implies `--delete`; commit mode, same behaviour as `--delete-after`). |
| `--delete-after` | Explicit delete-after timing: delete only after the transfer succeeded (implies `--delete`). |
| `--exclude <pattern>` | Exclude matching paths. Repeatable. | | `--exclude <pattern>` | Exclude matching paths. Repeatable. |
| `--include <pattern>` | Include matching paths. Repeatable. | | `--include <pattern>` | Include matching paths. Repeatable. |
| `--exclude-from <file>` | Read exclude patterns from a file. | | `--exclude-from <file>` | Read exclude patterns from a file. |
@@ -450,26 +379,25 @@ remote SSH argv is already built injection-safe.
zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.| zero means unlimited.| | `--incremental` | Skip files matching destination size and mtime.|
| `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` | | `--checksum` | Include xxHash64 content checks in incremental comparisons.| | `--backup` |
Back up overwritten files.| | `--backup - dir<dir>` | Store backups under a separate directory.| Back up overwritten files.| | `--backup - dir<dir>` | Store backups under a separate directory.|
| `--suffix<suffix>` | Set the backup filename suffix.| | `--partial` | | `--suffix<suffix>` | Set the backup filename suffix.| | `--partial` |
Select partial - transfer handling. On failed/interrupted writes the Select partial - transfer handling.With `--partial - dir`,
already-written temp file is retained (best-effort) for resumption.| completed files are written there;
With `--partial --partial-dir <dir>`, completed files are written under the resumable transfers are not implemented.| | `--partial - dir<dir>` |
partial directory and installed atomically. | | `--partial - dir<dir>` | Set a relative partial - transfer directory below the server destination root;
Set a relative partial - transfer directory below the server destination root. use with `--partial`. |
Use with `--partial`. |
| `--inplace` | Write directly to the destination instead of using a temporary file. | | `--inplace` | Write directly to the destination instead of using a temporary file. |
### Metadata and links ### Metadata and links
| Option | Description | | Option | Description |
|---|---| |---|---|
| `--preserve` | Preserve supported file metadata, currently mode and modification time (long form only). | | `-M`, `--preserve` | Preserve supported file metadata, currently mode and modification time. |
| `-l`, `--links` | Request symlink preservation; | `-l`, `--links` | Request symlink preservation;
link-target transfer remains incomplete. | link-target transfer remains incomplete. |
| `--copy-links` | Copy symlink referents. | | `--copy-links` | Copy symlink referents. |
| `--safe-links` | Skip symlinks that point outside the transfer tree. | | `--safe-links` | Skip symlinks that point outside the transfer tree. |
| `--copy-unsafe-links` | Copy unsafe symlink referents. | | `--copy-unsafe-links` | Copy unsafe symlink referents. |
| `-S`, `--sparse` | Sparse-file handling: receiver preserves holes (zero runs are written as holes; no wire change). | | `-S`, `--sparse` | Request sparse-file handling; full hole preservation is planned. |
### Output and logging ### Output and logging
@@ -486,13 +414,13 @@ link-target transfer remains incomplete. |
| Option | Description | | Option | Description |
|---|---| |---|---|
| `--ssh-port <port>` | SSH port for the SSH transport (default: 22). Note the short `-p` is now rsync's `--perms`. | | `-p <port>` | SSH port in the current CLI. This conflicts with rsync's `-p` permissions option and is planned for correction. |
| `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode. | | `--fastsync-server-path <path>` | Remote FastSync server path for SSH mode. |
| `--source-dir <path>` | Set the source directory explicitly. | | `--source-dir <path>` | Set the source directory explicitly. |
| `--dest-dir <path>` | Set the destination directory explicitly. | | `--dest-dir <path>` | Set the destination directory explicitly. |
| `--save-to-disk` | Enable server-side disk persistence. | | `--save-to-disk` | Enable server-side disk persistence. |
| `--server-host <host>` | TCP server address. | | `--server-host <host>` | TCP server address. |
| `--server-port <port>` | TCP server port. `--port <port>` / `--port=<port>` is an alias. | | `--server-port <port>` | TCP server port. |
| `--tls` | Enable TLS. Requires `--cert` and `--key`. | | `--tls` | Enable TLS. Requires `--cert` and `--key`. |
| `--cert <path>` | TLS certificate file. | | `--cert <path>` | TLS certificate file. |
| `--key <path>` | TLS private key file. | | `--key <path>` | TLS private key file. |
@@ -510,67 +438,10 @@ link-target transfer remains incomplete. |
| `--ca <path>` | CA file for peer verification. | | `--ca <path>` | CA file for peer verification. |
| `--destination-root <path>` | Confine received files to this server-side root; | `--destination-root <path>` | Confine received files to this server-side root;
defaults to the current directory. | defaults to the current directory. |
| `--allow-delete` | Permit client delete manifests. Deletion is refused by default. This also gates `--force` (which can recursively replace/remove a destination directory tree). | | `--allow-delete` | Permit client delete manifests. Deletion is refused by default. |
| `--allow-super` | Standalone TCP listener only: keep super-user activities enabled for a **root** receiver. Without it a root standalone server forces `SUPER_MODE_OFF`, so client `--devices`/`--write-devices`/`--super` and client-chosen ownership requests are skipped/refused. Rejected with `--stdio` (the SSH remote argv is client-composed; use a forced command if the default must hold). No effect when not root. Daemon modules opt in per module with `client owner = yes`. |
| `-v`, `--verbose` | Enable debug logging. | | `-v`, `--verbose` | Enable debug logging. |
| `--help` | Print server usage. | | `--help` | Print server usage. |
### Daemon configuration
`fastsync-server --daemon --config FILE` reads a line-based module config (an
implicit global section, then `[module]` sections). Besides `port`, `motd file`,
and `address`, the global section accepts:
- `max connections = N` — global cap on concurrent connections, default 100. The
listener enforces it; `0`, negative, and non-numeric values are parse errors.
- `max connections per host = N` — cap on concurrent connections from a single
source IP, default 0 (unlimited). Enforced across all forked connection
children through a shared registry.
- `auth failure delay = MS` — milliseconds to sleep after a failed
authentication, default 500. `0` disables it and the value is capped at 5000,
so online password guessing is rate-limited per connection. Successful auths
are never delayed.
- `auth lockout threshold = N` — number of failed authentications from one source
IP before that source is locked out, default 10; `0` disables the lockout. The
failure counter is shared across every connection child, so the lockout holds
even when the next attempt is handled by a different forked child.
- `auth lockout duration = SECONDS` — how long a locked-out source is refused
(default 300). A locked-out client is refused before any SCRAM challenge is
sent; a successful authentication clears the counter.
- `hosts allow` / `hosts deny` — comma- and/or whitespace-separated host access
patterns.
A `[module]` may also set `max connections` (0 = unlimited; enforced per module
across all connection children) and its own `hosts allow`/`hosts deny`.
The per-host cap and the shared auth lockout identify a source by its numeric
peer IP. **Loopback peers (127.0.0.0/8, IPv6 `::1`) are exempt**: every local
client shares that one address, so counting or locking them out would let one
local process deny service to all the others. The per-module and global
`max connections` caps still apply to loopback. Because the key is the peer IP,
`max connections per host` and `auth lockout` also cannot distinguish clients
behind the same NAT, proxy, or reverse-proxy address — they share one budget and
one lockout counter, so an over-aggressive lockout can affect unrelated users
behind that address. Prefer TLS client certificates (`--client-cn`) plus
`hosts allow`/`hosts deny` for per-client policy when clients share an address,
and size `auth lockout threshold` accordingly.
The shared per-source table has a bounded lifetime: an entry with no live
connection is reclaimed once its lockout has expired, or after it has been idle
(300 s). If every entry is still live or locked, a new source is admitted without
per-host accounting (fail open) and a rate-limited warning is logged; the
per-module cap and host ACLs still apply. The occupancy counters are re-derived
from the shared slot table after every child exit, so a child killed mid-transfer
(or mid-registration) cannot leak a slot or an occupancy count.
Host patterns are `*` (match all), IPv4/IPv6 literals, or IPv4/IPv6 CIDR
(`10.0.0.0/8`, `2001:db8::/32`). Hostnames are not resolved, so hostname globs
are rejected at parse time rather than silently never matching. A matching
`hosts deny` rejects; if any `hosts allow` entries exist, a peer matching none of
them is rejected; deny takes precedence over allow. The global list is checked
before the module list, before authentication, and the connecting peer address
(IPv4 or IPv6) appears in the connection and authentication audit log lines.
## Architecture ## Architecture
### Client ### Client
@@ -595,57 +466,11 @@ before the module list, before authentication, and the connecting peer address
## Protocol and Security ## Protocol and Security
FastSync protocol version `2.21.0` is shared by the client and server. The FastSync protocol version `2.3.0` is shared by the client and server. The
current protocol is sender-driven and includes configuration negotiation, current protocol is sender-driven and includes configuration negotiation,
including the maximum allocation limit, incremental checks, checksums, incremental checks, checksums, manifests, keep-alives, abort handling, and
manifests, keep-alives, abort handling, per-file remove-source results, and FastSync-native delta messages. Client and server versions must currently
FastSync-native delta messages. match exactly.
Client and server versions must currently match exactly.
Daemon modules that declare `auth users` authenticate with a SCRAM-SHA-256-style
challenge/response against a salted PBKDF2 verifier store: no password and no
replayable bearer credential crosses the wire or is stored on the daemon. All
store entries share one iteration count, and an unknown user is answered with a
deterministic per-username dummy challenge, so probing the daemon cannot
enumerate users. Store lines are generated with
`fastsync-server --hash-credentials <plaintext-file>` (see `RSYNC_COMPAT.md`);
redirect that output to an owner-only (mode 0600) file, and note that legacy
`user:SHA256HEX` stores are rejected. FastSync also maintains an owner-only
(mode 0600) `<store>.dummykey` sidecar next to the store: it holds the store-wide
dummy key, is auto-created on first load, and must be preserved across daemon
restarts so the dummy challenge for an unknown user stays stable (the key is
never regenerated while the sidecar exists). The sidecar is secret material and
must be protected like the credential store: keep it owner-only (mode 0600) and
include it with the store in backups and credential rotation. If the sidecar
cannot be created (a process-substitution/FIFO store path such as `/dev/fd/N`, a
read-only filesystem, a missing directory, or a create, write, fsync, link, or
fchmod failure), the daemon logs a warning and uses a transient key, so the
cross-restart guarantee does not hold for those deployments. One residual is
accepted: the store
iteration count is observable pre-auth by design, since the miss path must match
a hit.
An `auth users` module accepts credentials only when one of two conditions
holds: (a) the connection is an encrypted, verified TLS connection whose client
certificate matches the server's `--client-cn`, or (b) the connection is
plaintext from a loopback peer **and** the operator explicitly passed
`--allow-unauthenticated`. A remote plaintext peer is refused before any
challenge is sent, and `--allow-unauthenticated` never permits remote plaintext
auth: remote peers still require verified TLS regardless of the flag. Clients
sending daemon credentials with `--password-file` to a non-loopback daemon must
therefore use `--tls`; the client rejects a non-local plaintext credential
destination before any network I/O. Daemon modules are a `--daemon`-only
feature: the SSH `--stdio` path never loads a daemon config and is not an auth
transport for them.
Because the loopback allowance trusts whichever peer the kernel reports as
`127.0.0.1`, it assumes nothing relays remote connections to the daemon. A local
TCP forwarder or a TLS-terminating proxy in front of an auth-module listener
makes remote clients appear as loopback and bypasses the mutual-TLS identity
check, so do not front an auth-module listener with such a relay. `--tls` always
mandates `--client-cn`, so a TLS connection to an auth-required module always
has its client CN verified (`--client-cn` matches the certificate's CN only, not
a subjectAltName, which is acceptable for a private CA).
TLS provides encrypted TCP transport. Supplying `--ca` enables certificate TLS provides encrypted TCP transport. Supplying `--ca` enables certificate
verification; without it, traffic is encrypted but peer identity is not verification; without it, traffic is encrypted but peer identity is not
@@ -685,7 +510,7 @@ Run the unit test binary:
Run the Python integration suite: Run the Python integration suite:
```bash ```bash
python3 -m pytest tests/integration/ -n 4 --dist=load -m "not setpriv" python3 -m pytest tests/
``` ```
For stricter local validation: For stricter local validation:
+127 -760
View File
File diff suppressed because it is too large Load Diff
+70 -415
View File
@@ -3,24 +3,19 @@
Compares FastSync configs against rsync (no compression) and rsync+zstd. Compares FastSync configs against rsync (no compression) and rsync+zstd.
Data is ~75% random/incompressible and ~25% structured/compressible by default, Data is ~75% random/incompressible and ~25% structured/compressible by default,
controllable via --random-ratio. Transfers are verified by default (source and controllable via --random-ratio.
destination must match) so a fast-but-broken copy is never counted.
Usage: Usage:
python3 benchmark/bench.py python3 benchmark/bench.py
python3 benchmark/bench.py --runs 5 --profiles lan wan python3 benchmark/bench.py --runs 5 --profiles lan wan
python3 benchmark/bench.py --random-ratio 0.5 --size-mb 50 python3 benchmark/bench.py --random-ratio 0.5 --size-mb 50
python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit python3 benchmark/bench.py --delay 50ms --jitter 10ms --throughput 100mbit
python3 benchmark/bench.py --warm --runs 3
python3 benchmark/bench.py --output json python3 benchmark/bench.py --output json
""" """
import argparse import argparse
import filecmp
import json import json
import math
import os import os
import random import random
import shlex
import shutil import shutil
import socket import socket
import statistics import statistics
@@ -30,11 +25,8 @@ import tempfile
import time import time
PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), "..")) PROJECT_ROOT = os.path.abspath(os.path.join(os.path.dirname(__file__), ".."))
DEFAULT_BUILD_DIR = "build-bench" BUILD_DIR = os.path.join(PROJECT_ROOT, "build")
# Populated by configure_build_dirs(); default to the dedicated bench dir so SERVER_CMD = [os.path.join(BUILD_DIR, "server")]
# importing this module never depends on the user's existing build/ tree.
BUILD_DIR = os.path.join(PROJECT_ROOT, DEFAULT_BUILD_DIR)
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")] CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
BENCH_DIR = os.path.join(PROJECT_ROOT, "bench_data") BENCH_DIR = os.path.join(PROJECT_ROOT, "bench_data")
@@ -51,12 +43,11 @@ NETWORK_PROFILES = {
} }
FASTSYNC_CONFIGS = [ FASTSYNC_CONFIGS = [
{"name": "fastsync", "flags": [], "tool": "fastsync"}, {"name": "fastsync", "flags": [], "tool": "fastsync"},
{"name": "fastsync -z", "flags": ["-z"], "tool": "fastsync"}, {"name": "fastsync -c", "flags": ["-c"], "tool": "fastsync"},
{"name": "fastsync -j", "flags": ["-j"], "tool": "fastsync"}, {"name": "fastsync -m", "flags": ["-m"], "tool": "fastsync"},
{"name": "fastsync -j -z", "flags": ["-j", "-z"], "tool": "fastsync"}, {"name": "fastsync -m -c", "flags": ["-m", "-c"], "tool": "fastsync"},
{"name": "fastsync -j -z --chunk-serialization", "flags": ["-j", "-z", "--chunk-serialization"], "tool": "fastsync"}, {"name": "fastsync -m -c -s", "flags": ["-m", "-c", "-s"], "tool": "fastsync"},
{"name": "fastsync --sendfile", "flags": ["--sendfile"], "tool": "fastsync"},
] ]
RSYNC_CONFIGS = [ RSYNC_CONFIGS = [
@@ -65,7 +56,6 @@ RSYNC_CONFIGS = [
{"name": "rsync -z --zstd", "flags": ["-z", "--zc", "zstd"],"tool": "rsync"}, {"name": "rsync -z --zstd", "flags": ["-z", "--zc", "zstd"],"tool": "rsync"},
] ]
class RsyncDaemon: class RsyncDaemon:
"""Manages an rsync daemon for network-fair benchmarking.""" """Manages an rsync daemon for network-fair benchmarking."""
@@ -127,12 +117,6 @@ STRUCTURED_FILES = {
"nested/another.txt": b"another nested file\n" * 50, "nested/another.txt": b"another nested file\n" * 50,
} }
# Repeated text used to synthesize genuinely compressible filler of any size.
COMPRESSIBLE_TEXT = (
b"FastSync benchmark payload: the quick brown fox jumps over the lazy dog. "
b"0123456789 ABCDEFGHIJKLMNOPQRSTUVWXYZ abcdefghijklmnopqrstuvwxyz\n"
)
class Progress: class Progress:
"""Simple progress bar with ETA.""" """Simple progress bar with ETA."""
@@ -167,127 +151,35 @@ class Progress:
sys.stderr.flush() sys.stderr.flush()
def write_compressible(path, nbytes):
"""Write exactly nbytes of highly compressible, repeated text content."""
if nbytes <= 0:
return
block = COMPRESSIBLE_TEXT * (max(1, 8192 // len(COMPRESSIBLE_TEXT)) + 1)
remaining = nbytes
with open(path, "wb") as f:
while remaining > 0:
piece = block if remaining >= len(block) else block[:remaining]
f.write(piece)
remaining -= len(piece)
def generate_bench_data(source_dir, size_mb=25, random_ratio=0.75): def generate_bench_data(source_dir, size_mb=25, random_ratio=0.75):
"""Generate test data honouring the requested random/compressible split. """Generate test data. ~random_ratio is incompressible, rest is structured."""
Exactly ``random_ratio * target`` bytes are incompressible random data and
the remainder is genuinely compressible structured/repeated content. The
measured byte counts are returned so callers can report the real mix.
"""
if os.path.exists(source_dir): if os.path.exists(source_dir):
shutil.rmtree(source_dir) shutil.rmtree(source_dir)
os.makedirs(source_dir) os.makedirs(source_dir)
target = size_mb * 1024 * 1024 target = size_mb * 1024 * 1024
random_budget = int(target * random_ratio) structured_budget = int(target * (1 - random_ratio))
compressible_budget = target - random_budget written = 0
compressible_written = 0
random_written = 0
files = 0
# A handful of fixed, human-meaningful files (directories, small files, a
# binary blob) as long as they fit inside the compressible budget.
for rel_path, content in STRUCTURED_FILES.items(): for rel_path, content in STRUCTURED_FILES.items():
if compressible_written + len(content) > compressible_budget: if written >= structured_budget:
break break
full_path = os.path.join(source_dir, rel_path) full_path = os.path.join(source_dir, rel_path)
os.makedirs(os.path.dirname(full_path), exist_ok=True) os.makedirs(os.path.dirname(full_path), exist_ok=True)
with open(full_path, "wb") as f: with open(full_path, "wb") as f:
f.write(content) f.write(content)
compressible_written += len(content) written += len(content)
files += 1
# Fill the rest of the compressible share with generated repeated content. os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
if compressible_written < compressible_budget: i = 0
os.makedirs(os.path.join(source_dir, "compressible"), exist_ok=True) while written < target:
i = 0 chunk_size = min(5 * 1024 * 1024, target - written)
while compressible_written < compressible_budget: with open(os.path.join(source_dir, f"bulk/file_{i}.dat"), "wb") as f:
chunk = min(1024 * 1024, compressible_budget - compressible_written) f.write(random.randbytes(chunk_size))
write_compressible(os.path.join(source_dir, "compressible", f"text_{i}.dat"), chunk) written += chunk_size
compressible_written += chunk i += 1
files += 1
i += 1
# Incompressible share. return written
if random_written < random_budget:
os.makedirs(os.path.join(source_dir, "bulk"), exist_ok=True)
i = 0
while random_written < random_budget:
chunk = min(5 * 1024 * 1024, random_budget - random_written)
with open(os.path.join(source_dir, "bulk", f"file_{i}.dat"), "wb") as f:
f.write(random.randbytes(chunk))
random_written += chunk
files += 1
i += 1
return {
"total_bytes": compressible_written + random_written,
"compressible_bytes": compressible_written,
"random_bytes": random_written,
"files": files,
}
def list_relative_files(root):
"""Return the set of file paths (relative to root) under a directory."""
found = set()
for dirpath, _dirnames, filenames in os.walk(root):
for name in filenames:
full = os.path.join(dirpath, name)
found.add(os.path.relpath(full, root))
return found
def verify_transfer(source_dir, dest_dir):
"""Recursively check dest matches source (paths, sizes, content).
Returns (ok, detail). Content is compared byte-for-byte, never hashed, so
collisions are impossible. This is intentionally not part of the timing.
"""
if not os.path.isdir(dest_dir):
return False, "destination directory missing"
src_files = list_relative_files(source_dir)
dst_files = list_relative_files(dest_dir)
if src_files != dst_files:
missing = src_files - dst_files
extra = dst_files - src_files
return False, f"path set mismatch (missing {len(missing)}, extra {len(extra)})"
for rel in sorted(src_files):
src = os.path.join(source_dir, rel)
dst = os.path.join(dest_dir, rel)
if os.path.getsize(src) != os.path.getsize(dst):
return False, f"size mismatch: {rel}"
if not filecmp.cmp(src, dst, shallow=False):
return False, f"content mismatch: {rel}"
return True, ""
def percentile(values, pct):
"""Linear-interpolation percentile (matches numpy's default method)."""
if not values:
return None
ordered = sorted(values)
if len(ordered) == 1:
return ordered[0]
rank = (len(ordered) - 1) * (pct / 100.0)
low = math.floor(rank)
high = math.ceil(rank)
if low == high:
return ordered[int(rank)]
return ordered[low] + (ordered[high] - ordered[low]) * (rank - low)
def find_free_port(): def find_free_port():
@@ -315,45 +207,18 @@ def wait_proc(proc, timeout=5):
proc.wait() proc.wait()
def _tc_base_cmd():
"""Return the command prefix for tc, honouring root vs sudo."""
tc = shutil.which("tc")
if not tc:
raise RuntimeError(
"tc (iproute2) not found in PATH; install iproute2 to use network profiles")
if os.geteuid() == 0:
return [tc]
sudo = shutil.which("sudo")
if sudo:
return [sudo, tc]
raise RuntimeError(
"applying network limits requires root or sudo; "
"re-run as root or install sudo")
def _run_tc(args, check=True):
return subprocess.run(_tc_base_cmd() + args, check=check, capture_output=True)
def netem_apply(delay=None, jitter=None, throughput=None, loss=None): def netem_apply(delay=None, jitter=None, throughput=None, loss=None):
"""Apply tc/netem rules to loopback. Pass None to skip a parameter.""" """Apply tc/netem rules to loopback. Pass None to skip a parameter."""
netem_reset() netem_reset()
params = [] cmd = ["sudo", "tc", "qdisc", "add", "dev", "lo", "root", "netem"]
if throughput: if throughput:
params += ["rate", throughput] cmd += ["rate", throughput]
if delay: if delay:
params += ["delay", delay, jitter or "0ms"] cmd += ["delay", delay, jitter or "0ms"]
if loss: if loss:
params += ["loss", loss] cmd += ["loss", loss]
if not params: if len(cmd) > 6:
return subprocess.run(cmd, check=True, capture_output=True)
try:
_run_tc(["qdisc", "add", "dev", "lo", "root", "netem"] + params)
except subprocess.CalledProcessError as exc:
detail = exc.stderr.decode(errors="replace").strip() if exc.stderr else str(exc)
raise RuntimeError(f"failed to apply network profile via tc/netem: {detail}") from exc
except RuntimeError:
raise
def netem_apply_profile(profile_name): def netem_apply_profile(profile_name):
@@ -370,11 +235,7 @@ def netem_apply_profile(profile_name):
def netem_reset(): def netem_reset():
"""Best-effort removal of any loopback qdisc. Always safe to call.""" subprocess.run("sudo tc qdisc del dev lo root".split(), capture_output=True)
try:
_run_tc(["qdisc", "del", "dev", "lo", "root"], check=False)
except Exception:
pass
def run_fastsync(source_dir, dest_dir, flags, port): def run_fastsync(source_dir, dest_dir, flags, port):
@@ -391,10 +252,8 @@ def run_fastsync(source_dir, dest_dir, flags, port):
duration = time.monotonic() - start duration = time.monotonic() - start
if result.returncode == 0: if result.returncode == 0:
return duration return duration
sys.stderr.write(f" fastsync failed (exit {result.returncode}): "
f"{result.stderr.strip()[:500]}\n")
except subprocess.TimeoutExpired: except subprocess.TimeoutExpired:
sys.stderr.write(" fastsync timed out after 120s\n") pass
return None return None
@@ -410,10 +269,8 @@ def run_rsync(source_dir, dest_dir, flags, rsync_daemon=None):
duration = time.monotonic() - start duration = time.monotonic() - start
if result.returncode == 0: if result.returncode == 0:
return duration return duration
sys.stderr.write(f" rsync failed (exit {result.returncode}): "
f"{result.stderr.strip()[:500]}\n")
except subprocess.TimeoutExpired: except subprocess.TimeoutExpired:
sys.stderr.write(" rsync timed out after 120s\n") pass
return None return None
@@ -425,81 +282,7 @@ def run_transfer(config, source_dir, dest_dir, port=None, rsync_daemon=None):
return run_fastsync(source_dir, dest_dir, config["flags"], port) return run_fastsync(source_dir, dest_dir, config["flags"], port)
def apply_incremental_changes(source_dir, target_bytes): def run_benchmark(source_dir, dest_dir, configs, runs, profile_name, progress=None):
"""Add and modify a few files so a warm transfer has real work to do.
Returns a mutation record (changed byte count plus enough data to revert
and re-apply it) so every warm run can start from a pristine source.
"""
modified_n = 3
added_n = 2
per_file = max(4096, target_bytes // (modified_n + added_n))
modified = {}
added = {}
changed = 0
existing = sorted(list_relative_files(source_dir))
if existing:
step = max(1, len(existing) // modified_n)
for rel in existing[::step][:modified_n]:
path = os.path.join(source_dir, rel)
original_size = os.path.getsize(path)
with open(path, "ab") as f:
f.write(random.randbytes(per_file))
modified[rel] = (original_size, per_file)
changed += per_file
for i in range(added_n):
os.makedirs(os.path.join(source_dir, "incremental"), exist_ok=True)
rel = os.path.join("incremental", f"new_{i}.dat")
write_compressible(os.path.join(source_dir, rel), per_file)
added[rel] = per_file
changed += per_file
return {"changed": changed, "modified": modified, "added": added}
def revert_incremental_changes(source_dir, mutation):
"""Undo apply_incremental_changes so the source is pristine again."""
if not mutation:
return
for rel, (original_size, _appended) in mutation["modified"].items():
path = os.path.join(source_dir, rel)
if os.path.exists(path):
with open(path, "r+b") as f:
f.truncate(original_size)
for rel in mutation["added"]:
path = os.path.join(source_dir, rel)
if os.path.exists(path):
os.remove(path)
def reapply_incremental_changes(source_dir, mutation):
"""Re-apply a mutation after an untimed pristine seed transfer."""
if not mutation:
return
for rel, (_original_size, appended) in mutation["modified"].items():
with open(os.path.join(source_dir, rel), "ab") as f:
f.write(random.randbytes(appended))
for rel, size in mutation["added"].items():
write_compressible(os.path.join(source_dir, rel), size)
def expected_received_root(dest_dir, source_dir, tool):
"""Where a tool places transferred files inside dest_dir.
FastSync mirrors the absolute source path under dest_dir (see the
integration suite's get_dest_received_dir); rsync copies the source tree
contents directly into dest_dir.
"""
if tool == "rsync":
return dest_dir
return os.path.join(dest_dir, os.path.abspath(source_dir).lstrip(os.sep))
def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
measure_bytes, verify=True, warm=False, mutation=None,
progress=None):
"""Run benchmark for all configs, returns list of results.""" """Run benchmark for all configs, returns list of results."""
is_limited = profile_name != "unlimited" is_limited = profile_name != "unlimited"
has_rsync = any(c["tool"] == "rsync" for c in configs) has_rsync = any(c["tool"] == "rsync" for c in configs)
@@ -515,10 +298,7 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
results = [] results = []
for config in configs: for config in configs:
times = [] times = []
invalid = 0
for run_idx in range(runs): for run_idx in range(runs):
if warm:
revert_incremental_changes(source_dir, mutation)
if os.path.exists(dest_dir): if os.path.exists(dest_dir):
shutil.rmtree(dest_dir) shutil.rmtree(dest_dir)
os.makedirs(dest_dir, exist_ok=True) os.makedirs(dest_dir, exist_ok=True)
@@ -526,33 +306,15 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
port = find_free_port() port = find_free_port()
server = None server = None
try: try:
if config["tool"] == "fastsync" or warm: if config["tool"] == "fastsync":
server = subprocess.Popen( server = subprocess.Popen(
SERVER_CMD + ["-p", str(port)], SERVER_CMD + ["-p", str(port)],
stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL, stdout=subprocess.DEVNULL, stderr=subprocess.DEVNULL,
) )
wait_for_port(port) wait_for_port(port)
if warm:
seed = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
if seed is None:
invalid += 1
sys.stderr.write(" warm-mode seeding failed; run not counted\n")
continue
reapply_incremental_changes(source_dir, mutation)
t = run_transfer(config, source_dir, dest_dir, port, rsync_daemon) t = run_transfer(config, source_dir, dest_dir, port, rsync_daemon)
if t is None: if t is not None:
invalid += 1
elif verify:
root = expected_received_root(dest_dir, source_dir, config["tool"])
ok, detail = verify_transfer(source_dir, root)
if ok:
times.append(t)
else:
invalid += 1
sys.stderr.write(f" verification FAILED ({detail}); run not counted\n")
else:
times.append(t) times.append(t)
finally: finally:
if server: if server:
@@ -565,22 +327,15 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
"config": config["name"], "config": config["name"],
"tool": config["tool"], "tool": config["tool"],
"profile": profile_name, "profile": profile_name,
"warm": warm,
"runs": len(times), "runs": len(times),
"invalid": invalid,
"times": [round(t, 4) for t in times], "times": [round(t, 4) for t in times],
} }
if times: if times:
p50 = percentile(times, 50) entry["p50"] = round(statistics.median(times), 4)
p95 = percentile(times, 95) entry["p95"] = round(sorted(times)[int(len(times) * 0.95)], 4) if len(times) > 1 else entry["p50"]
entry["p50"] = round(p50, 4)
entry["p95"] = round(p95, 4)
entry["min"] = round(min(times), 4) entry["min"] = round(min(times), 4)
entry["max"] = round(max(times), 4) entry["max"] = round(max(times), 4)
entry["stdev"] = round(statistics.stdev(times), 4) if len(times) > 1 else 0.0 entry["stdev"] = round(statistics.stdev(times), 4) if len(times) > 1 else 0.0
if measure_bytes:
entry["throughput_mbps"] = round(
(measure_bytes / (1024 * 1024)) / p50, 3)
results.append(entry) results.append(entry)
return results return results
finally: finally:
@@ -590,59 +345,44 @@ def run_benchmark(source_dir, dest_dir, configs, runs, profile_name,
netem_reset() netem_reset()
def print_table(results, measure_bytes, stats, warm): def print_table(results, total_bytes, random_ratio):
"""Print results as a human-readable table grouped by profile.""" """Print results as a human-readable table grouped by profile."""
profiles = {} profiles = {}
for r in results: for r in results:
profiles.setdefault(r["profile"], []).append(r) profiles.setdefault(r["profile"], []).append(r)
total = stats["total_bytes"]
comp_pct = stats["compressible_bytes"] / total * 100 if total else 0
rand_pct = stats["random_bytes"] / total * 100 if total else 0
for profile, entries in profiles.items(): for profile, entries in profiles.items():
params = NETWORK_PROFILES.get(profile, {}) params = NETWORK_PROFILES.get(profile, {})
print(f"\n{'=' * 95}") print(f"\n{'=' * 85}")
print(f" Profile: {profile.upper()}") print(f" Profile: {profile.upper()}")
if params.get("rate"): if params.get("rate"):
print(f" Network: {params['rate']}, {params['delay']} +/- {params['jitter']}, loss {params['loss']}") print(f" Network: {params['rate']}, {params['delay']} +/- {params['jitter']}, loss {params['loss']}")
else: else:
print(f" Network: unlimited") print(f" Network: unlimited")
print(f" Data: {total / (1024*1024):.1f} MB " print(f" Data: {total_bytes / (1024*1024):.1f} MB ({random_ratio*100:.0f}% random, {(1-random_ratio)*100:.0f}% compressible)")
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)") print(f"{'=' * 85}")
if warm:
print(f" Mode: warm (incremental) — measured {measure_bytes / (1024*1024):.2f} MB "
f"changed after an untimed full seed")
else:
print(" Mode: cold (full copy)")
print(f"{'=' * 95}")
fs_entries = [e for e in entries if e.get("tool") == "fastsync"] fs_entries = [e for e in entries if e.get("tool") == "fastsync"]
rsync_entries = [e for e in entries if e.get("tool") == "rsync"] rsync_entries = [e for e in entries if e.get("tool") == "rsync"]
header = (f" {'Config':<38} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} "
f"{'stdev':>8} {'MB/s':>9} {'runs':>5} {'bad':>4}")
rule = (f" {'-' * 38} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} "
f"{'-' * 8} {'-' * 9} {'-' * 5} {'-' * 4}")
if fs_entries: if fs_entries:
print(f"\n FastSync:") print(f"\n FastSync:")
print(header) print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
print(rule) print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
for e in sorted(fs_entries, key=lambda x: x.get("p50", 999)): for e in sorted(fs_entries, key=lambda x: x.get("p50", 999)):
_print_entry(e) _print_entry(e)
if rsync_entries: if rsync_entries:
print(f"\n rsync:") print(f"\n rsync:")
print(header) print(f" {'Config':<25} {'p50':>8} {'p95':>8} {'min':>8} {'max':>8} {'stdev':>8} {'runs':>5}")
print(rule) print(f" {'-' * 25} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 8} {'-' * 5}")
for e in sorted(rsync_entries, key=lambda x: x.get("p50", 999)): for e in sorted(rsync_entries, key=lambda x: x.get("p50", 999)):
_print_entry(e) _print_entry(e)
if params.get("rate_bps") and fs_entries and rsync_entries: if params.get("rate_bps") and fs_entries and rsync_entries:
fs_best = min((e["p50"] for e in fs_entries if "p50" in e), default=None) fs_best = min((e["p50"] for e in fs_entries if "p50" in e), default=None)
rsync_best = min((e["p50"] for e in rsync_entries if "p50" in e), default=None) rsync_best = min((e["p50"] for e in rsync_entries if "p50" in e), default=None)
theoretical = measure_bytes / params["rate_bps"] theoretical = total_bytes / params["rate_bps"]
if fs_best and rsync_best: if fs_best and rsync_best:
print(f"\n Theoretical max (line rate): {theoretical:.4f}s") print(f"\n Theoretical max (line rate): {theoretical:.4f}s")
print(f" FastSync best: {fs_best:.4f}s ({theoretical/fs_best:.2f}x vs line rate)") print(f" FastSync best: {fs_best:.4f}s ({theoretical/fs_best:.2f}x vs line rate)")
@@ -652,43 +392,10 @@ def print_table(results, measure_bytes, stats, warm):
def _print_entry(e): def _print_entry(e):
if "p50" in e: if "p50" in e:
tp = f"{e['throughput_mbps']:.2f}" if "throughput_mbps" in e else "N/A" print(f" {e['config']:<25} {e['p50']:>7.4f}s {e['p95']:>7.4f}s "
print(f" {e['config']:<38} {e['p50']:>7.4f}s {e['p95']:>7.4f}s " f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} {e['runs']:>5}")
f"{e['min']:>7.4f}s {e['max']:>7.4f}s {e['stdev']:>7.4f} "
f"{tp:>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
else: else:
print(f" {e['config']:<38} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} " print(f" {e['config']:<25} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {'N/A':>8} {e['runs']:>5}")
f"{'N/A':>8} {'N/A':>9} {e['runs']:>5} {e.get('invalid', 0):>4}")
def configure_build_dirs(build_dir):
"""Install the selected build directory and derived binary paths."""
global BUILD_DIR, SERVER_CMD, CLIENT_CMD
if not os.path.isabs(build_dir):
build_dir = os.path.join(PROJECT_ROOT, build_dir)
BUILD_DIR = os.path.abspath(build_dir)
SERVER_CMD = [os.path.join(BUILD_DIR, "server"), "--allow-unauthenticated"]
CLIENT_CMD = [os.path.join(BUILD_DIR, "client")]
def build_project():
"""Configure (Release) and build into the dedicated bench build dir."""
if shutil.which("cmake") is None:
sys.stderr.write("cmake not found in PATH; cannot build\n")
sys.exit(1)
os.makedirs(BUILD_DIR, exist_ok=True)
configure = ["cmake", "-B", BUILD_DIR, "-S", PROJECT_ROOT,
"-DCMAKE_BUILD_TYPE=Release"]
result = subprocess.run(configure, capture_output=True, text=True)
if result.returncode != 0:
sys.stderr.write("CMake configure failed:\n" + result.stdout + result.stderr + "\n")
sys.exit(1)
jobs = str(os.cpu_count() or 1)
result = subprocess.run(["cmake", "--build", BUILD_DIR, "-j", jobs],
capture_output=True, text=True)
if result.returncode != 0:
sys.stderr.write("Build failed:\n" + result.stdout + result.stderr + "\n")
sys.exit(1)
def main(): def main():
@@ -705,20 +412,12 @@ Custom network limits (--delay/--jitter/--throughput) override profiles.
Data mix: Data mix:
Default is ~75%% random/incompressible + ~25%% structured/compressible, Default is ~75%% random/incompressible + ~25%% structured/compressible,
reflecting typical real-world file sets. The actual mix is measured and reflecting typical real-world file sets.
reported. Transfers are verified (destination must match source) unless
--no-verify is given.
Warm mode:
--warm seeds the destination with an untimed full copy of a pristine base,
then measures only the incremental transfer after modifying a few files.
Examples: Examples:
%(prog)s --profiles wan --runs 5 %(prog)s --profiles wan --runs 5
%(prog)s --throughput 50mbit --delay 30ms --jitter 5ms %(prog)s --throughput 50mbit --delay 30ms --jitter 5ms
%(prog)s --random-ratio 0.5 --size-mb 100 %(prog)s --random-ratio 0.5 --size-mb 100
%(prog)s --warm --runs 3 --no-rsync
%(prog)s --dry-run --size-mb 4 --random-ratio 0.25
""") """)
parser.add_argument("--runs", type=int, default=3, parser.add_argument("--runs", type=int, default=3,
help="Number of runs per config (default: 3)") help="Number of runs per config (default: 3)")
@@ -726,8 +425,7 @@ Examples:
choices=list(NETWORK_PROFILES.keys()), choices=list(NETWORK_PROFILES.keys()),
help="Predefined network profiles (default: unlimited)") help="Predefined network profiles (default: unlimited)")
parser.add_argument("--configs", nargs="+", default=None, parser.add_argument("--configs", nargs="+", default=None,
help="Custom FastSync config flags (shell-quoted, e.g. " help="Custom FastSync config flags")
"\"-j -z --chunk-serialization\")")
parser.add_argument("--size-mb", type=int, default=25, parser.add_argument("--size-mb", type=int, default=25,
help="Test data size in MB (default: 25)") help="Test data size in MB (default: 25)")
parser.add_argument("--random-ratio", type=float, default=0.75, parser.add_argument("--random-ratio", type=float, default=0.75,
@@ -742,14 +440,6 @@ Examples:
help="Custom packet loss (e.g. 1%%)") help="Custom packet loss (e.g. 1%%)")
parser.add_argument("--no-rsync", action="store_true", parser.add_argument("--no-rsync", action="store_true",
help="Skip rsync comparison") help="Skip rsync comparison")
parser.add_argument("--no-verify", action="store_true",
help="Skip source/destination verification after each run")
parser.add_argument("--warm", action="store_true",
help="Incremental mode: seed dest first, measure only changes")
parser.add_argument("--build-dir", default=DEFAULT_BUILD_DIR,
help=f"Build directory (default: {DEFAULT_BUILD_DIR})")
parser.add_argument("--dry-run", action="store_true",
help="Only generate data and report its composition, then exit")
parser.add_argument("--progress", action="store_true", parser.add_argument("--progress", action="store_true",
help="Show progress bar with ETA") help="Show progress bar with ETA")
parser.add_argument("--output", choices=["table", "json"], default="table", parser.add_argument("--output", choices=["table", "json"], default="table",
@@ -758,48 +448,12 @@ Examples:
help="Don't clean up test data") help="Don't clean up test data")
args = parser.parse_args() args = parser.parse_args()
if not 0.0 <= args.random_ratio <= 1.0: # Build
parser.error("--random-ratio must be between 0.0 and 1.0") print("Building...")
if args.size_mb <= 0: if os.system(f"cmake -B {BUILD_DIR} -S {PROJECT_ROOT} > /dev/null 2>&1") != 0:
parser.error("--size-mb must be positive") print("CMake configure failed"); sys.exit(1)
if os.system(f"cmake --build {BUILD_DIR} -j$(nproc) > /dev/null 2>&1") != 0:
configure_build_dirs(args.build_dir) print("Build failed"); sys.exit(1)
# Generate data
source_dir = os.path.join(BENCH_DIR, "source")
dest_dir = os.path.join(BENCH_DIR, "dest")
stats = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
total_bytes = stats["total_bytes"]
comp_pct = stats["compressible_bytes"] / total_bytes * 100 if total_bytes else 0
rand_pct = stats["random_bytes"] / total_bytes * 100 if total_bytes else 0
print(f"Generated {total_bytes / (1024*1024):.1f} MB in {stats['files']} files "
f"({rand_pct:.0f}% random, {comp_pct:.0f}% compressible actual)",
file=sys.stderr)
if args.dry_run:
print(f"size_mb={args.size_mb} random_ratio={args.random_ratio:.4f} "
f"total_bytes={stats['total_bytes']} "
f"compressible_bytes={stats['compressible_bytes']} "
f"random_bytes={stats['random_bytes']} files={stats['files']}")
if not args.keep_data:
shutil.rmtree(BENCH_DIR, ignore_errors=True)
return
# Warm mode: keep a pristine base copy, then mutate the live source.
base_dir = None
measure_bytes = total_bytes
mutation = None
if args.warm:
change_target = max(64 * 1024, min(int(total_bytes * 0.01), 4 * 1024 * 1024))
mutation = apply_incremental_changes(source_dir, change_target)
measure_bytes = mutation["changed"]
revert_incremental_changes(source_dir, mutation)
print(f"Warm mode: each run seeds a full copy, then measures "
f"{measure_bytes / (1024*1024):.3f} MB of add/change deltas", file=sys.stderr)
# Build (Release: benchmarking a debug build is meaningless)
print(f"Building (Release) into {BUILD_DIR}...", file=sys.stderr)
build_project()
# Determine active profile for display # Determine active profile for display
has_custom_net = args.delay or args.jitter or args.throughput or args.loss has_custom_net = args.delay or args.jitter or args.throughput or args.loss
@@ -819,10 +473,18 @@ Examples:
else: else:
profiles_to_run = args.profiles or ["unlimited"] profiles_to_run = args.profiles or ["unlimited"]
# Build config list (shlex so quoted/space-separated flags survive) # Generate data
source_dir = os.path.join(BENCH_DIR, "source")
dest_dir = os.path.join(BENCH_DIR, "dest")
total_bytes = generate_bench_data(source_dir, args.size_mb, args.random_ratio)
compressible_pct = (1 - args.random_ratio) * 100
random_pct = args.random_ratio * 100
print(f"Generated {total_bytes / (1024*1024):.1f} MB "
f"({random_pct:.0f}% random, {compressible_pct:.0f}% compressible)")
# Build config list
if args.configs: if args.configs:
fastsync_configs = [{"name": c, "flags": shlex.split(c), "tool": "fastsync"} fastsync_configs = [{"name": c, "flags": c.split(), "tool": "fastsync"} for c in args.configs]
for c in args.configs]
else: else:
fastsync_configs = list(FASTSYNC_CONFIGS) fastsync_configs = list(FASTSYNC_CONFIGS)
@@ -834,21 +496,14 @@ Examples:
total_runs = len(configs) * args.runs * len(profiles_to_run) total_runs = len(configs) * args.runs * len(profiles_to_run)
progress = Progress(total_runs, "Benchmarking") if args.progress else None progress = Progress(total_runs, "Benchmarking") if args.progress else None
if progress: if progress:
print(f"Running {total_runs} transfers...", file=sys.stderr) print(f"Running {total_runs} transfers...")
all_results = [] all_results = []
try: try:
for profile in profiles_to_run: for profile in profiles_to_run:
results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile, results = run_benchmark(source_dir, dest_dir, configs, args.runs, profile, progress)
measure_bytes, verify=not args.no_verify,
warm=args.warm, mutation=mutation,
progress=progress)
all_results.extend(results) all_results.extend(results)
except RuntimeError as exc:
sys.stderr.write(f"error: {exc}\n")
sys.exit(1)
finally: finally:
netem_reset()
if not args.keep_data: if not args.keep_data:
shutil.rmtree(BENCH_DIR, ignore_errors=True) shutil.rmtree(BENCH_DIR, ignore_errors=True)
@@ -856,7 +511,7 @@ Examples:
if args.output == "json": if args.output == "json":
print(json.dumps(all_results, indent=2)) print(json.dumps(all_results, indent=2))
else: else:
print_table(all_results, measure_bytes, stats, args.warm) print_table(all_results, total_bytes, args.random_ratio)
print() print()
-10
View File
@@ -1,10 +0,0 @@
[pytest]
; Fast integration subset run on every pull request (see .gitea/workflows/ci.yaml).
markers =
ci: fast, representative integration tests run on the PR CI gate
setpriv: privilege-dependent tests (drop to an unprivileged user); excluded
from CI because their result depends on the runner/container uid and the
host mount permissions, but run locally as root
daemon_detach: real double-fork backgrounding path (--daemon without
--no-detach); slower/fragile, so it runs in the full suite but not the
fast PR gate
+3 -34
View File
@@ -3,35 +3,11 @@
}: }:
pkgs.mkShell { pkgs.mkShell {
# Development shell for FastSync. Provides the host-side toolchain needed to
# build, lint, unit-test, integration-test and benchmark the project.
# It deliberately does NOT build on entry: run the CMake commands in README.md
# (or use the CI Docker image for exact CI parity).
nativeBuildInputs = with pkgs; [ nativeBuildInputs = with pkgs; [
# build
gcc gcc
cmake cmake
gnumake gnumake
pkg-config pkg-config
# lint / static analysis (matches CI)
clang-tools # clang-format
cppcheck
# tests
(python3.withPackages (ps: with ps; [ pytest pytest-xdist psutil ]))
openssh # SSH transport integration tests
# debugging
gdb
valgrind
# coverage
lcov
# benchmark tooling
rsync
iproute2 # tc/netem for network shaping
# misc
git
curl
nodejs
nixpkgs-fmt
docker docker
tea tea
]; ];
@@ -39,21 +15,14 @@ pkgs.mkShell {
buildInputs = with pkgs; [ buildInputs = with pkgs; [
zstd zstd
openssl openssl
(python3.withPackages (ps: with ps; [ pytest ]))
]; ];
# The CMake configure step fetches xxHash via FetchContent, which needs
# network access; NIX_ENFORCE_PURITY must be off so the sandbox does not block.
NIX_ENFORCE_PURITY = 0; NIX_ENFORCE_PURITY = 0;
shellHook = '' shellHook = ''
export NIX_ENFORCE_PURITY=0 export NIX_ENFORCE_PURITY=0
# Make an existing build tree available on PATH, but never build here. cmake -B build
if [ -d "$PWD/build" ]; then export PATH="$PWD/build:$PATH"
export PATH="$PWD/build:$PATH"
fi
echo "FastSync dev shell ready."
echo " Build: cmake -B build -S . && cmake --build build -j\$(nproc)"
echo " Unit: ./build/tests"
echo " CI parity: docker run --rm --user \"\$(id -u):\$(id -g)\" -v \"\$PWD:/workspace\" -w /workspace gitea.tap-tap.win/taptap/fastsync-ci:v10 ..."
''; '';
} }
-329
View File
@@ -1,329 +0,0 @@
#include "change_list.h"
#include "utils.h"
#include <limits.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/stat.h>
#include <time.h>
/* Itemize code emitted for a transferred regular file.
*
* Layout (rsync-compatible 11-char item): `>f` marks a regular file that was
* transferred to the remote host; the trailing nine markers are, in order,
* c(hecksum) s(ize) t(ime) p(erms) o(wner) g(roup) u(ser/acl) a(ttrs) x(attrs).
* Every marker is `+` (FastSync does not compare each attribute on the
* receiving side, so a sent file is reported as fully updated). Files that
* are already up to date print no line at all, matching rsync's single -i
* which only itemizes changes.
*
* Because the scanner only yields regular-file transfer candidates, `>d`
* (directory) lines are never produced; directories are not transferred as
* items by FastSync. */
#define ITEMIZE_SENT_FILE ">f+++++++++"
typedef struct {
char* data;
size_t length;
size_t capacity;
} StrBuf;
static void strbuf_free(StrBuf* buf) {
if (buf == NULL)
return;
free(buf->data);
buf->data = NULL;
buf->length = 0;
buf->capacity = 0;
}
static bool strbuf_reserve(StrBuf* buf, size_t extra) {
if (buf->length > SIZE_MAX - extra - 1)
return false;
size_t need = buf->length + extra + 1;
if (need <= buf->capacity)
return true;
size_t capacity = buf->capacity > 0 ? buf->capacity : 32;
while (capacity < need) {
if (capacity > SIZE_MAX / 2) {
capacity = need;
break;
}
capacity *= 2;
}
char* grown = realloc(buf->data, capacity);
if (!grown)
return false;
buf->data = grown;
buf->capacity = capacity;
return true;
}
static bool strbuf_append_char(StrBuf* buf, char c) {
if (!strbuf_reserve(buf, 1))
return false;
buf->data[buf->length++] = c;
buf->data[buf->length] = '\0';
return true;
}
static bool strbuf_append(StrBuf* buf, const char* text) {
if (text == NULL)
return true;
size_t length = strlen(text);
if (!strbuf_reserve(buf, length))
return false;
memcpy(buf->data + buf->length, text, length);
buf->length += length;
buf->data[buf->length] = '\0';
return true;
}
static bool strbuf_append_ull(StrBuf* buf, unsigned long long value) {
char digits[32];
int written = snprintf(digits, sizeof(digits), "%llu", value);
if (written < 0 || (size_t)written >= sizeof(digits))
return false;
return strbuf_append(buf, digits);
}
static bool strbuf_append_longlong(StrBuf* buf, long long value) {
char digits[32];
int written = snprintf(digits, sizeof(digits), "%lld", value);
if (written < 0 || (size_t)written >= sizeof(digits))
return false;
return strbuf_append(buf, digits);
}
bool change_list_enabled(const Config* config) {
return config != NULL && (config->itemize_changes || config->out_format != NULL ||
(config->log_file != NULL && config->log_file_format != NULL));
}
char* change_render_itemize(const ChangeEvent* event) {
if (event == NULL || event->decision != CHANGE_SENT)
return str_dup("");
const char* code = event->is_directory ? ">d+++++++++" : ITEMIZE_SENT_FILE;
StrBuf line = {0};
bool ok = strbuf_append(&line, code) && strbuf_append(&line, " ") &&
strbuf_append(&line, event->path != NULL ? event->path : "");
if (!ok) {
strbuf_free(&line);
return NULL;
}
return line.data;
}
static const char* leaf_name(const char* path) {
if (path == NULL)
return "";
const char* slash = strrchr(path, '/');
return slash != NULL && slash[1] != '\0' ? slash + 1 : path;
}
char* change_render_format(const char* format, const ChangeEvent* event) {
if (format == NULL)
return NULL;
StrBuf line = {0};
bool ok = true;
for (const char* p = format; *p != '\0' && ok;) {
if (*p != '%') {
ok = strbuf_append_char(&line, *p);
p++;
continue;
}
char token = p[1];
if (token == '\0') {
ok = strbuf_append_char(&line, '%');
break;
}
switch (token) {
case '%':
ok = strbuf_append_char(&line, '%');
break;
case 'f':
ok = strbuf_append(&line, event->path != NULL ? event->path : "");
break;
case 'n':
ok = strbuf_append(&line, leaf_name(event->path));
break;
case 'l':
ok = strbuf_append_ull(&line, event->size);
break;
case 'b':
ok = strbuf_append_ull(&line, event->bytes_sent);
break;
case 'M':
ok = strbuf_append_longlong(&line, (long long)event->mtime_sec);
break;
default:
/* Unknown escape sequences are preserved verbatim. */
ok = strbuf_append_char(&line, '%') && strbuf_append_char(&line, token);
break;
}
p += 2;
}
if (!ok) {
strbuf_free(&line);
return NULL;
}
if (line.data == NULL) {
line.data = str_dup("");
if (!line.data)
return NULL;
}
return line.data;
}
/* Format a mode as an `ls -l` permission string, e.g. `-rw-r--r--`. */
static void mode_to_ls_string(mode_t mode, char out[11]) {
out[0] = S_ISDIR(mode) ? 'd'
: S_ISLNK(mode) ? 'l'
: S_ISCHR(mode) ? 'c'
: S_ISBLK(mode) ? 'b'
: S_ISFIFO(mode) ? 'p'
: S_ISSOCK(mode) ? 's'
: '-';
mode_t bits = mode & 07777;
out[1] = (bits & S_IRUSR) ? 'r' : '-';
out[2] = (bits & S_IWUSR) ? 'w' : '-';
out[3] = (bits & S_IXUSR) ? (bits & S_ISUID ? 's' : 'x') : (bits & S_ISUID ? 'S' : '-');
out[4] = (bits & S_IRGRP) ? 'r' : '-';
out[5] = (bits & S_IWGRP) ? 'w' : '-';
out[6] = (bits & S_IXGRP) ? (bits & S_ISGID ? 's' : 'x') : (bits & S_ISGID ? 'S' : '-');
out[7] = (bits & S_IROTH) ? 'r' : '-';
out[8] = (bits & S_IWOTH) ? 'w' : '-';
out[9] = (bits & S_IXOTH) ? (bits & S_ISVTX ? 't' : 'x') : (bits & S_ISVTX ? 'T' : '-');
out[10] = '\0';
}
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime,
const char* path) {
char permission[11];
mode_to_ls_string(mode, permission);
char date[32];
struct tm broken_down;
if (localtime_r(&mtime, &broken_down) != NULL) {
if (strftime(date, sizeof(date), "%Y/%m/%d %H:%M:%S", &broken_down) == 0)
snprintf(date, sizeof(date), "?");
} else {
snprintf(date, sizeof(date), "?");
}
StrBuf line = {0};
char size_field[32];
int written = snprintf(size_field, sizeof(size_field), "%llu", size);
if (written < 0 || (size_t)written >= sizeof(size_field)) {
strbuf_free(&line);
return NULL;
}
bool ok = strbuf_append(&line, permission) && strbuf_append_char(&line, ' ') &&
strbuf_append(&line, size_field) && strbuf_append_char(&line, ' ') &&
strbuf_append(&line, date) && strbuf_append_char(&line, ' ') &&
strbuf_append(&line, path != NULL ? path : "");
if (!ok) {
strbuf_free(&line);
return NULL;
}
return line.data;
}
static void print_escaped_line(FILE* stream, const char* line, bool eight_bit_output) {
char* escaped = output_escape(line, eight_bit_output);
if (escaped != NULL) {
fprintf(stream, "%s\n", escaped);
free(escaped);
} else {
fprintf(stream, "%s\n", line);
}
fflush(stream);
}
void change_emit(const Config* config, const ChangeEvent* event) {
if (event == NULL || !change_list_enabled(config))
return;
if (event->decision == CHANGE_UP_TO_DATE)
return;
bool to_stdout = config->itemize_changes || config->out_format != NULL;
bool to_log = config->log_file != NULL && config->log_file_format != NULL;
if (to_stdout) {
char* line = config->out_format != NULL ? change_render_format(config->out_format, event)
: change_render_itemize(event);
if (line != NULL) {
print_escaped_line(stdout, line, config->eight_bit_output);
free(line);
}
}
if (to_log) {
char* line = change_render_format(config->log_file_format, event);
if (line != NULL) {
print_escaped_line(config->log_file, line, config->eight_bit_output);
free(line);
}
}
}
static bool format_uses_mtime(const char* format) {
if (format == NULL)
return false;
/* Mirror change_render_format's tokenizer: "%%" is a literal percent (so
* "%%M" does NOT expand %M) and unknown "%X" escapes consume both chars.
* This keeps the optional stat() fallback below in step with the renderer. */
for (const char* p = format; *p != '\0';) {
if (*p != '%') {
p++;
continue;
}
char token = p[1];
if (token == '\0')
break;
if (token == 'M')
return true;
p += 2;
}
return false;
}
void change_emit_file_sent(const Config* config, const File* file) {
if (file == NULL || !change_list_enabled(config))
return;
ChangeEvent event;
memset(&event, 0, sizeof(event));
/* The displayed path is the one transmitted (with -R + --files-from this is
the bare relative destination path); the metadata fallback below still
stats the local absolute path. */
event.path = file_wire_path(file);
event.decision = CHANGE_SENT;
event.is_directory = false;
event.size = file->data != NULL ? file->data->size : 0;
/* FastSync has no wire-byte counter yet, so %b reports the source length
* that had to be delivered (always equal to %l); the actual bytes written
* to the socket (compressed/delta) are not measured. */
event.bytes_sent = event.size;
if (file->metadata != NULL) {
event.mtime_sec = file->metadata->mtime_sec;
} else if (format_uses_mtime(config->out_format) || format_uses_mtime(config->log_file_format)) {
/* Best-effort fallback for %M when no metadata was captured (no -M): the
* path is stat()ed just to fill the field, and any failure leaves 0. */
struct stat st;
if (file->path != NULL && stat(file->path, &st) == 0)
event.mtime_sec = st.st_mtime;
}
change_emit(config, &event);
}
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
void change_emit_dir_sent(const Config* config, const File* file) {
if (file == NULL || !change_list_enabled(config))
return;
ChangeEvent event;
memset(&event, 0, sizeof(event));
event.path = file_wire_path(file);
event.decision = CHANGE_SENT;
event.is_directory = true;
event.size = 0;
event.bytes_sent = 0;
if (file->metadata != NULL)
event.mtime_sec = file->metadata->mtime_sec;
change_emit(config, &event);
}
-78
View File
@@ -1,78 +0,0 @@
#ifndef CHANGE_LIST_H
#define CHANGE_LIST_H
#include "config.h"
#include "file_types.h"
#include <stdbool.h>
#include <sys/stat.h>
#include <time.h>
/*
* Shared per-file change-event / output model (rsync --itemize-changes,
* --out-format, --log-file-format, and --list-only all render from here).
*
* FastSync is a push-style tool: the client sends files from the source tree
* to a server that writes them under the destination root. Events are
* emitted by whichever code path decides a file's fate (the single-threaded
* send loop and the `-m` sender thread both call the same per-file sender), so
* all change events are emitted by exactly one thread and itemize/out-format
* lines never interleave with each other. They may still interleave with
* legacy log messages (log.c) that share the same stdout/log-file stream.
*/
typedef enum {
CHANGE_SENT, /* file data (full or delta) was transmitted */
CHANGE_UP_TO_DATE, /* receiver already had an identical file; skipped */
} ChangeDecision;
typedef struct {
const char* path; /* full source path */
ChangeDecision decision;
bool is_directory;
unsigned long long size; /* source file length in bytes */
/* The number of bytes reported for a sent file. FastSync has no wire-byte
* counter, so this is always the source length (== size / %l); actual
* post-compression/delta bytes on the wire are not counted. */
unsigned long long bytes_sent;
time_t mtime_sec; /* 0 when unknown */
} ChangeEvent;
/* True when any output mode is active and per-file events matter. */
bool change_list_enabled(const Config* config);
/* Render the rsync-style itemize line for a transferred file:
* `>f+++++++++ <path>`
* The 11-char code is `>f` (regular file transferred to the remote host)
* followed by c/s/t/p/o/g/u/a/x markers that are all `+` (value will be set
* / differs) because FastSync does not separately compare checksums, size,
* mtime, perms, owner, group, uid, acl, or xattr on the receiving side, so a
* sent file is reported as fully updated. Up-to-date files print no line
* (rsync single `-i` only shows changes). Caller frees the result. */
char* change_render_itemize(const ChangeEvent* event);
/* Expand an --out-format/--log-file-format template. Tokens:
* %f full source path %b "bytes sent" == the source length (%l);
* %n leaf (base) name actual post-compression/delta wire bytes
* %l file length in bytes are not counted
* %M mtime in whole seconds %% a literal percent sign
* Unknown %X sequences are preserved verbatim. Caller frees the result. */
char* change_render_format(const char* format, const ChangeEvent* event);
/* Render one --list-only long-listing entry:
* `-rw-r--r-- 12 2026/09/06 10:00:00 <path>`
* (ls -l style columns; mtime in the local time zone). Caller frees it. */
char* change_render_list_line(mode_t mode, unsigned long long size, time_t mtime, const char* path);
/* Emit an event to every active destination:
* stdout: --itemize-changes line, or the --out-format expansion when set;
* log file: the --log-file-format expansion (requires --log-file).
* CHANGE_UP_TO_DATE events produce no output. */
void change_emit(const Config* config, const ChangeEvent* event);
/* Build and emit a CHANGE_SENT event for a file the client just sent. */
void change_emit_file_sent(const Config* config, const File* file);
/* Build and emit a CHANGE_SENT event for an explicit directory entry (-d). */
void change_emit_dir_sent(const Config* config, const File* file);
#endif
+183 -1974
View File
File diff suppressed because it is too large Load Diff
+154 -2052
View File
File diff suppressed because it is too large Load Diff
+2 -19
View File
@@ -4,27 +4,10 @@
#include "chunk.h" #include "chunk.h"
#include "config.h" #include "config.h"
#include "transport_tcp.h" #include "transport_tcp.h"
#include <signal.h>
#include <stdbool.h>
/* Set ONLY by the client's SIGINT/SIGTERM handler (async-signal-safe: the int send_chunk(Client* client, Chunk* chunk, Config* config);
* handler stores 1 and does nothing else). The send loops poll it via
* client_abort_pending() and, when set, best-effort send STATUS_ABORT so the
* receiver can clean up before the client exits. */
extern volatile sig_atomic_t client_abort_requested;
bool client_abort_pending(void);
/* Arm/disarm abort handling around the network phase. While disarmed, a
* SIGINT/SIGTERM takes the default action (immediate termination) so local-only
* modes are not left unresponsive. Defined in client_cli.c. */
void client_set_abort_armed(bool armed);
/* Both sender entry points BORROW `config` for the duration of the call; they
* never free it, and the caller retains ownership (freeing it with
* config_delete() once the call returns). */
int send_files(Config* config); int send_files(Config* config);
/* Takes ownership only when *config is set to NULL on return. */
int send_files_multithreaded(Config** config); int send_files_multithreaded(Config** config);
/* Phase 6 residual-batch (client-only). See client_send.c. */
int write_batch_from_source(const Config* config, const char* batch_path);
int apply_batch_to_dest(const Config* config, const char* batch_path, const char* dest_root);
#endif #endif
+22 -77
View File
@@ -1,72 +1,44 @@
#include "client_validation.h" #include "client_validation.h"
#include "log.h" #include "log.h"
#include "usage.h" #include "usage.h"
#include "utils.h"
#include <string.h>
#include <stdio.h> #include <stdio.h>
/* Validate config after parsing. Returns true if valid. */ /* Validate config after parsing. Returns true if valid. */
bool validate_config(const Config* config) { bool validate_config(const Config* config) {
/* Phase 6 residual-batch modes relax the normal source+destination pair: the if (!config->send_directory || !config->receive_root_directory) {
batch driver is local and needs only what it consumes. --only-write-batch
emits a batch from the source (no destination, no server);
--read-batch applies a batch to the destination (no source, no server);
--write-batch runs the live transfer AND emits a batch, so it keeps the
full pair. */
bool write_batch = config->write_batch != NULL;
bool only_write_batch = config->only_write_batch != NULL;
bool read_batch = config->read_batch != NULL;
if ((write_batch && only_write_batch) || (write_batch && read_batch) ||
(only_write_batch && read_batch)) {
log_message(LOG_LEVEL_ERROR,
"--write-batch, --only-write-batch, and --read-batch are mutually exclusive");
return false;
}
/* A dry-run of a local batch apply is not meaningful: --read-batch bypasses
the client-side scan/server decision entirely, so dry-run would have no
wire state to report (and must not be used as a mutation escape hatch).
--only-write-batch likewise never contacts a receiver. --write-batch DOES
run a live transfer but additionally mutates the filesystem by emitting the
batch file, so a dry-run must not write it either. Reject all three up
front instead of silently ignoring --dry-run. */
if (config->dry_run && (read_batch || only_write_batch || write_batch)) {
log_message(LOG_LEVEL_ERROR,
"--dry-run cannot be combined with --read-batch, --only-write-batch, or "
"--write-batch; a dry-run must not mutate anything, including batch files");
return false;
}
if (read_batch) {
if (!config->receive_root_directory) {
log_message(LOG_LEVEL_ERROR, "--read-batch requires a destination directory");
print_usage();
return false;
}
} else if (only_write_batch) {
if (!config->send_directory) {
log_message(LOG_LEVEL_ERROR, "--only-write-batch requires a source directory");
print_usage();
return false;
}
} else if (!config->send_directory || !config->receive_root_directory) {
log_message(LOG_LEVEL_ERROR, "source and destination directories are required"); log_message(LOG_LEVEL_ERROR, "source and destination directories are required");
print_usage(); print_usage();
return false; return false;
} }
if (config->compression_threads > 0 && !config->use_compression) { if (config->use_sendfile && (config->use_chunk_serialization || config->use_compression)) {
log_message(LOG_LEVEL_ERROR, "--compress-threads requires compression (-c or -z)"); log_message(LOG_LEVEL_ERROR, "-f/--sendfile cannot be combined with -c (compression) or -s "
"(chunk serialization)");
return false; return false;
} }
if (config->transport == TRANSPORT_SSH && config->use_sendfile) { if (config->transport == TRANSPORT_SSH && config->use_sendfile) {
log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport"); log_message(LOG_LEVEL_ERROR, "-f/--sendfile is not supported with SSH transport");
return false; return false;
} }
/* -4 and -6 are mutually exclusive: a socket address family cannot be both. */ if (config->use_incremental && config->use_chunk_serialization) {
if (config->ipv4 && config->ipv6) { log_message(LOG_LEVEL_ERROR, "--incremental is not supported with -s (chunk serialization)");
log_message(LOG_LEVEL_ERROR, "-4/--ipv4 and -6/--ipv6 are mutually exclusive");
return false; return false;
} }
if (config->log_file_format && !config->log_file) { if (config->use_delta && !config->use_incremental) {
log_message(LOG_LEVEL_ERROR, "--log-file-format requires --log-file"); log_message(LOG_LEVEL_ERROR, "--delta requires --incremental");
return false;
}
if (config->use_delta && config->use_chunk_serialization) {
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -s (chunk serialization)");
return false;
}
if (config->use_delta && config->use_sendfile) {
log_message(LOG_LEVEL_ERROR, "--delta cannot be combined with -f (sendfile)");
return false;
}
if (config->append || config->append_verify) {
fprintf(
stderr,
"Error: --append and --append-verify are not supported yet; refusing to ignore option\n");
return false; return false;
} }
if (config->use_tls) { if (config->use_tls) {
@@ -75,32 +47,5 @@ bool validate_config(const Config* config) {
return false; return false;
} }
} }
/* Daemon credentials (A7, protocol 2.19.0): a --password-file would send the
username in the clear and derive a SCRAM proof a network sniffer could
attack offline, so it is only allowed over TLS (which itself mandates a
verified --cert/--key/--ca set above) or to a loopback destination. A
remote plaintext daemon is refused here, before any network I/O. */
if (config->password_file && !config->use_tls && !utils_host_is_loopback(config->server_host)) {
log_message(LOG_LEVEL_ERROR, "sending daemon credentials to a non-local server requires --tls");
return false;
}
/* Every cross-field invariant the receiver enforces lives in one shared
predicate so the client and the server can never disagree. The client
reports the specific reason here, before any network I/O. */
const char* invariants_error = config_invariants_error(config);
if (invariants_error) {
log_message(LOG_LEVEL_ERROR, "%s", invariants_error);
return false;
}
/* --protocol: FastSync has exactly one wire format, so the forced version
must equal the current PROTOCOL_VERSION exactly. Rejected here, before any
network I/O, rather than letting the server hit its own mismatch check. */
if (strcmp(config->version, PROTOCOL_VERSION) != 0) {
log_message(LOG_LEVEL_ERROR,
"--protocol must be %s (FastSync supports only its current wire "
"protocol version and cannot speak an older or virtual one)",
PROTOCOL_VERSION);
return false;
}
return true; return true;
} }
+94 -1061
View File
File diff suppressed because it is too large Load Diff
+16 -139
View File
@@ -2,32 +2,14 @@
#define SCANNER_H #define SCANNER_H
#include "chunk.h" #include "chunk.h"
#include "file_list.h"
#include "filter.h"
#include "hardlink.h"
#include "protocol.h"
#include "queue.h" #include "queue.h"
#include "stop_condition.h"
#include <dirent.h> #include <dirent.h>
#include <stdbool.h> #include <stdbool.h>
#include <stdatomic.h>
#include <sys/types.h>
#include <threads.h> #include <threads.h>
#include <stdatomic.h>
/* Upper bound on the configurable parallel scanner worker count (--threads=N):
* keeps one transfer from spawning an unbounded pool on a very large machine. */
#define MAX_SCANNER_THREADS 256
typedef struct { typedef struct {
bool use_metadata; bool use_metadata;
/* Phase 4 metadata capture: -U/--atimes and -N/--crtimes tell the scanner to
* capture the source access / birth time into each entry's FileMetadata. */
bool preserve_atimes;
bool preserve_crtimes;
/* Phase 4 xattrs: when preserve_xattrs || preserve_acls is set the scanner
* captures each regular file's whitelisted xattr set onto the File. */
bool preserve_xattrs;
bool preserve_acls;
unsigned long long chunk_size; unsigned long long chunk_size;
char** exclude_patterns; char** exclude_patterns;
int exclude_count; int exclude_count;
@@ -41,114 +23,29 @@ typedef struct {
bool copy_links; bool copy_links;
bool safe_links; bool safe_links;
bool copy_unsafe_links; bool copy_unsafe_links;
/* Phase 4 symlink-trust sender options: -k/--copy-dirlinks (dereference a
* symlink to a directory as a directory, keeping symlinks-to-files as
* symlinks) and --munge-links (rewrite each transmitted symlink target with a
* marker; escaping targets are never transmitted). Both are client/sender
* side only and never serialized to the wire (keep_dirlinks is the
* receiver-side counterpart). */
bool copy_dirlinks;
bool munge_links;
bool checksum; bool checksum;
bool one_file_system;
/* Phase 4 special/devices: whether device nodes (--devices) and special files
* (--specials) are preserved via recreation, and whether --copy-devices
* copies a device's content as an ordinary regular file. */
bool preserve_devices;
bool preserve_specials;
bool copy_devices;
/* Phase 2 (files-from / filter layer). All pointers are shared read-only
* across scanner instances and worker threads; ownership stays with the
* caller (client_send). */
const FileListSet* file_list; /* --files-from allow-set, or NULL */
const FilterRuleList* base_filters; /* command-line + -C rules, or NULL */
bool per_dir_filters; /* -F: read .rsync-filter per directory */
bool dirs; /* -d/--dirs: transfer dir entries, no recursion */
bool relative; /* -R/--relative (dest rel paths, with --files-from) */
/* --prune-empty-dirs (long only): in --dirs mode an empty source directory's
explicit entry is omitted from the transfer file list (so nothing is
created at the destination and it can be pruned by --delete); explicitly
--files-from-listed directories always pass through. Recursive transfers
never emit empty directories, so the flag has no additional effect there. */
bool prune_empty_dirs;
/* Delete-excluded protection sink (optional): when non-NULL the scanner
* appends the destination-relative path of every entry it prunes because a
* USER SELECTION rule excluded it (--filter/-C/per-dir rules, the legacy
* --exclude/--include layer, and --max-size/--min-size). The sender turns
* this list into the manifest's protected prefixes so `--delete` leaves the
* destination mirror of excluded source paths alone (rsync's default), and
* empties it when --delete-excluded opts back into deleting them. NOT
* recorded for --files-from subset pruning (whose delete semantics stay
* keep-set-only) or for -R/--files-from relative wire paths. When
* `excluded_mutex` is non-NULL it is taken around every append (the parallel
* scanner shares one list across its worker threads). */
ArrayList* excluded_paths;
mtx_t* excluded_mutex;
/* --ignore-errors: an unreadable directory during the scan is recorded as an
* I/O error and skipped instead of aborting the scan. Client-only. */
bool ignore_io_errors;
/* --ignore-missing-args (implied by --delete-missing-args): an explicitly
* --files-from-listed entry that does not exist under the source is skipped
* instead of failing (the --dirs generator is the only scanner path that
* observes a listed-but-missing entry). */
bool ignore_missing_args;
/* --hard-links (-H): shared, mutable (mutex-guarded) link-group detection
* table, NULL when -H is off. Owned by the caller (client_send), shared
* read-only here; the parallel scanner passes it unchanged to every worker so
* one table detects every group across all subdirectories. */
HardLinkTable* hardlinks;
/* Phase 6: optional sender stop deadline. When non-NULL the scanner checks
* it at natural loop boundaries and stops emitting chunks once reached
* (without marking the scan as failed), so a busy scan itself stops early.
* Client-only, never serialized to the wire. */
const StopCondition* stop_condition;
/* P7 Wave D (protocol 2.17.0): directory-time capture sink. When
* `capture_dir_times` is true the recursive scan appends one is_dir File
* (with metadata, no payload) per source directory it traverses to
* `dir_entries`, so the sender can transmit trailing STATUS_DIR_TIMES
* frame(s) and the receiver can apply directory mtimes AFTER all children
* are written. `dir_entries_mutex` (optional) guards the list
* for the parallel scanner's shared worker threads; the caller owns both.
* The --dirs generator does not use this (its directory entries carry their
* metadata inline through STATUS_MKDIR). */
bool capture_dir_times;
ArrayList* dir_entries;
mtx_t* dir_entries_mutex;
} ScannerOptions; } ScannerOptions;
/* Internal per-scanner filter state. FilterNode chains represent the ordered
* per-directory .rsync-filter rules that apply below a directory. */
typedef struct FilterNode FilterNode;
typedef struct { typedef struct {
/* Scan inputs, copied once at create time. Everything that is also a
ScannerOptions field lives here (with the normalized chunk_size); only
scanner-owned bookkeeping stays as direct members below. */
ScannerOptions options;
Queue* directories; Queue* directories;
DIR* current_dir; DIR* current_dir;
char* current_path; char* current_path;
bool use_metadata;
unsigned long long chunk_size;
char** exclude_patterns;
int exclude_count;
char** include_patterns;
int include_count;
unsigned long long max_size;
unsigned long long min_size;
int max_depth;
int current_depth; int current_depth;
dev_t root_dev; bool follow_symlinks;
bool copy_links;
bool safe_links;
bool copy_unsafe_links;
bool checksum;
bool failed; bool failed;
/* Phase 2 (files-from / filter layer). */
char* root_path; /* transfer root (fs path) for rel computation */
char* current_rel; /* rel path of the open directory ("" == root) */
bool at_seed_dir; /* next open is the seed directory */
FilterNode* seed_node; /* inherited context of the seed dir, or NULL */
FilterNode* current_node; /* filter context of the open directory */
ArrayList* filter_nodes; /* owned FilterNode arena (may be NULL) */
/* --dirs / -R state for the directory-entry generator (options.dirs replaces
the recursive scan). */
bool relative_mode; /* file_list && relative: send bare relative wire paths */
bool dirs_root_emitted;
int list_index;
ArrayList* dirs_batch; /* owned when non-NULL */
unsigned long long dirs_batch_size;
/* A directory could not be opened (I/O error, e.g. EACCES). With
--ignore-errors the scan continues past it and the caller decides what to
do; `failed` is reserved for fatal errors that always abort the scan. */
bool io_error;
} DirectoryScanner; } DirectoryScanner;
typedef struct { typedef struct {
@@ -162,13 +59,9 @@ typedef struct {
thrd_t* threads; thrd_t* threads;
bool done; bool done;
bool failed; bool failed;
/* A worker skipped an unreadable directory under --ignore-errors (non-fatal). */
bool io_error;
atomic_bool cancelled; atomic_bool cancelled;
int completed; int completed;
Chunk* initial_chunk; Chunk* initial_chunk;
ProtocolSession* allocation_session;
FilterNode* root_filter_node; /* root .rsync-filter context (owned by ps) */
} ParallelScanner; } ParallelScanner;
DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata, DirectoryScanner* directory_scanner_create(const char* root_directory, bool use_metadata,
@@ -184,26 +77,10 @@ Chunk* directory_scanner_next(DirectoryScanner* scanner);
bool directory_scanner_failed(const DirectoryScanner* scanner); bool directory_scanner_failed(const DirectoryScanner* scanner);
void directory_scanner_destroy(DirectoryScanner* scanner); void directory_scanner_destroy(DirectoryScanner* scanner);
/* --one-file-system (-x) decision: a directory entry may be descended into
* only when the option is disabled or the entry lives on the same device as
* the transfer root. Exposed so tests can exercise the rule directly. */
bool scanner_same_filesystem(bool one_file_system, dev_t root_device, dev_t entry_device);
/* Relative path of an on-disk path below `root` ("" == the root itself, NULL
* when `fs_path` is not under `root`). Handles trailing slashes and a root of
* "/". Exposed so tests can exercise the mapping directly. */
char* scanner_path_relative(const char* root, const char* fs_path);
ParallelScanner* parallel_scanner_create_with_options(const char* root_directory, ParallelScanner* parallel_scanner_create_with_options(const char* root_directory,
const ScannerOptions* options, const ScannerOptions* options);
ProtocolSession* allocation_session);
Chunk* parallel_scanner_next(ParallelScanner* scanner); Chunk* parallel_scanner_next(ParallelScanner* scanner);
bool parallel_scanner_failed(const ParallelScanner* scanner); bool parallel_scanner_failed(const ParallelScanner* scanner);
bool parallel_scanner_had_io_error(const ParallelScanner* scanner);
void parallel_scanner_destroy(ParallelScanner* scanner); void parallel_scanner_destroy(ParallelScanner* scanner);
/* True when a directory could not be opened during the scan (an I/O error,
recorded even when --ignore-errors keeps the scan going past it). */
bool directory_scanner_had_io_error(const DirectoryScanner* scanner);
#endif #endif
+12 -240
View File
@@ -2,7 +2,6 @@
#include <stdio.h> #include <stdio.h>
#include <delta.h> #include <delta.h>
#include <chunk.h> #include <chunk.h>
#include "scanner.h"
void print_usage(void) { void print_usage(void) {
printf("Usage:\n"); printf("Usage:\n");
@@ -12,292 +11,65 @@ void print_usage(void) {
printf("Destination formats:\n"); printf("Destination formats:\n");
printf(" user@host:/path SSH transport (rsync-style)\n"); printf(" user@host:/path SSH transport (rsync-style)\n");
printf(" host:/path SSH transport (current user)\n"); printf(" host:/path SSH transport (current user)\n");
printf(" host::module/path Daemon TCP transport (fastsync-server --daemon);\n");
printf(" module names a server-side module, path is relative\n");
printf(" within it (connect with --server-port)\n");
printf(" /local/path TCP transport (requires server on localhost:8080)\n"); printf(" /local/path TCP transport (requires server on localhost:8080)\n");
printf("\n"); printf("\n");
printf("Options:\n"); printf("Options:\n");
printf(" -c, --checksum Verify content by checksum instead of size+mtime\n"); printf(" -c [level] Enable compression (level 1-22, default 5)\n");
printf(" -z, --compress [level] Enable compression (level 1-22, default 5)\n"); printf(" -z [level] Alias for -c\n");
printf(" -a, --archive rsync archive mode (-rlptgoD): links, metadata,\n"); printf(" -a, --archive Archive mode (-c -m -M)\n");
printf(" devices and specials (not compression/multithreading)\n");
printf(" -n, --dry-run Show what would be transferred\n"); printf(" -n, --dry-run Show what would be transferred\n");
printf(" --remove-source-files Remove regular source files after successful transfer\n"); printf(" -p <port> SSH port (default: 22)\n");
printf(" -p, --perms Preserve permission bits (part of the metadata bundle)\n");
printf(" --ssh-port <port> SSH port (default: 22)\n");
printf(" -e, --rsh <command> Remote shell to launch on the client for the SSH\n");
printf(" transport (default: ssh). The command may include\n");
printf(" arguments, e.g. -e \"ssh -p 2222\"\n");
printf(" --rsync-path <path> Alias for --fastsync-server-path (path to the\n");
printf(" fastsync server binary on the remote side)\n");
printf(" --blocking-io Leave the SSH transport socket without read/write\n");
printf(" timeouts so it blocks naturally\n");
printf(" --outbuf=MODE stdout/stderr buffering: N (none/unbuffered),\n");
printf(" L (line-buffered), or B (block-buffered, default)\n");
printf(" --progress Show transfer progress\n"); printf(" --progress Show transfer progress\n");
printf(" -P Partial mode with progress (retention incomplete)\n");
printf(" -8, --8-bit-output Leave high-bit characters unescaped in output\n");
printf(" --iconv=LOCAL[,REMOTE] Convert file-NAME charsets at the wire boundary:\n");
printf(" LOCAL is the charset of our file names, REMOTE is the\n");
printf(" remote side's charset (defaults to LOCAL). Names are\n");
printf(" converted before transmission and back on receipt; a\n");
printf(" name that cannot be represented in the target charset\n");
printf(" fails that transfer cleanly (rsync-compatible)\n");
printf(" --protocol=NUM Force the wire protocol version (must equal the current\n");
printf(" PROTOCOL_VERSION; FastSync cannot speak older/virtual\n");
printf(" wire formats)\n");
printf(" --write-batch=FILE Run the normal live transfer AND also emit a\n");
printf(" self-contained batch file of the whole source tree\n");
printf(" (implies the single-threaded transfer path)\n");
printf(" --only-write-batch=FILE\n");
printf(" Emit the batch file only (no destination, no server)\n");
printf(" --read-batch=FILE Apply the batch file to the destination (no source, no\n");
printf(" server); takes only the destination as an argument\n");
printf(" --delete Delete files on receiver not in source\n"); printf(" --delete Delete files on receiver not in source\n");
printf(" (default timing: delete only after the whole\n");
printf(" transfer has succeeded)\n");
printf(" --delete-before Delete extras before the transfer starts\n");
printf(" (implies --delete)\n");
printf(" --delete-during Delete extras once the keep-set manifest is known,\n");
printf(" before the data is applied (implies --delete)\n");
printf(" --del Alias for --delete-during\n");
printf(" --delete-delay Delete extras only after a successful transfer\n");
printf(" (implies --delete)\n");
printf(" --delete-after Delete only after the whole transfer succeeded\n");
printf(" (the default --delete timing; implies --delete)\n");
printf(" --delete-excluded Also delete destination files that were excluded on\n");
printf(" the source (default protects them, matching rsync)\n");
printf(" --max-delete=NUM Never delete more than NUM destination entries per run;\n");
printf(" if the extras would exceed NUM, nothing is deleted and\n");
printf(" the run fails with a clear error (implies --delete only\n");
printf(" when used with it)\n");
printf(" --ignore-errors Continue (and still delete) when a source directory is\n");
printf(" unreadable during the scan, instead of aborting with no\n");
printf(" deletion\n");
printf(" --force A file may replace a destination directory by removing\n");
printf(" that (non-empty) directory first\n");
printf(" --ignore-missing-args A --files-from entry that does not exist under the\n");
printf(" source is silently skipped instead of failing the run\n");
printf(" --delete-missing-args Implies --ignore-missing-args; also deletes each missing\n");
printf(" entry's destination mirror receiver-side. Independent of\n");
printf(" --delete (it does not imply --delete; a non-empty directory\n");
printf(" mirror is removed only with --force or --delete)\n");
printf(" -m, --prune-empty-dirs Do not transfer empty directory entries (--dirs mode);\n");
printf(" recursive transfers never send empty dirs\n");
printf(" Note: each timing flag implies --delete. Combining a timing flag with\n");
printf(" --no-delete (in either order) is rejected as a config error.\n");
printf(" --ignore-existing Skip files that already exist on receiver\n");
printf(" --delay-updates Put updated files into place only at the end of transfer\n");
printf(" --dirs, -d, --old-dirs, --old-d Transfer the named directory entries without\n");
printf(" recursing into their contents (-d <dir> mirrors the source\n");
printf(" directory empty; with --files-from listed dirs are created\n");
printf(" empty and listed files are transferred)\n");
printf(" -R, --relative With --files-from, preserve each listed entry's relative path\n");
printf(" below the destination root instead of mirroring the full\n");
printf(" source path (no effect without --files-from)\n");
printf(" --no-implied-dirs With -R --files-from, refuse to place a listed file whose\n");
printf(" parent directory is not itself listed\n");
printf(" --mkpath Create the destination root directory on the server when it\n");
printf(" does not exist yet\n");
printf(" --exclude <pattern> Exclude files matching pattern\n"); printf(" --exclude <pattern> Exclude files matching pattern\n");
printf(" --include <pattern> Only include files matching pattern\n"); printf(" --include <pattern> Only include files matching pattern\n");
printf(" --exclude-from <file> Read exclude patterns from file\n"); printf(" --exclude-from <file> Read exclude patterns from file\n");
printf(" --include-from <file> Read include patterns from file\n"); printf(" --include-from <file> Read include patterns from file\n");
printf(" --files-from <file> Read the source file list from FILE (paths relative to the "
"source root)\n");
printf(" -0, --from0 Entries in --files-from are NUL-delimited\n");
printf(" -f, --filter=RULE rsync-style filter rule (+/- include/exclude; repeatable;\n");
printf(" both --filter=RULE and the -f RULE / -f=RULE short forms work)\n");
printf(" -C, --cvs-exclude Auto-ignore common CVS/SCM files (.git/, .svn/, *.o, *~, ...)\n");
printf(" -F Apply per-directory .rsync-filter files during the scan\n");
printf(" --max-size <n> Skip files larger than n bytes\n"); printf(" --max-size <n> Skip files larger than n bytes\n");
printf(" --min-size <n> Skip files smaller than n bytes\n"); printf(" --min-size <n> Skip files smaller than n bytes\n");
printf(" --max-alloc <SIZE> Maximum single allocation (default: 1G)\n");
printf(" --incremental Skip files unchanged since last transfer\n"); printf(" --incremental Skip files unchanged since last transfer\n");
printf(" --size-only Skip incremental files matching in size, ignoring mtime\n");
printf(" -I, --ignore-times Transfer files even when size and mtime match\n");
printf(" -@, --modify-window <sec> Modification time tolerance\n");
printf(" -u, --update Skip files newer than the source on receiver\n");
printf(" --existing Skip files not already present at destination\n"); printf(" --existing Skip files not already present at destination\n");
printf(" --compare-dest <dir> Treat DIR (relative to destination root) as an extra\n");
printf(" comparison basis: unchanged files are not transferred\n");
printf(" (requires --incremental, which is implied)\n");
printf(" --copy-dest <dir> Like --compare-dest, but copies the unchanged file from DIR\n");
printf(" into the destination instead of transferring its data\n");
printf(" --link-dest <dir> Like --copy-dest, but hard-links the unchanged file from DIR\n");
printf(" into the destination (repeatable; earlier DIRs win)\n");
printf(" --checksum-choice, --cc <alg> Whole-file checksum algorithm for --incremental/\n");
printf(" --checksum compares (xxh64/xxhash or md5; default xxh64 with\n");
printf(" seed 0). The seed comes from --checksum-seed\n");
printf(" --checksum-seed <num> Seed for the whole-file xxHash64 digest (and the delta\n");
printf(" block strong hash, low 32 bits); md5 ignores the seed. The\n");
printf(" digest algorithm and seed must match on sender and receiver\n");
printf(" --delta Delta transfer for changed files (requires --incremental)\n"); printf(" --delta Delta transfer for changed files (requires --incremental)\n");
printf(" -W, --whole-file Transfer changed files without delta processing\n"); printf(" --delta-block <n> Delta block size in bytes (default: %d)\n",
printf(" -y, --fuzzy Use a similar-named file already in the destination\n"); DELTA_BLOCK_SIZE_DEFAULT);
printf(" directory as the delta basis when the destination has no\n");
printf(" usable file at the exact path (saves bandwidth; implies\n");
printf(" --incremental and --delta; inert with --whole-file,\n");
printf(" --no-delta, or --no-incremental)\n");
printf(" --no-fuzzy Disable --fuzzy\n");
printf(" --delta-block <n>, --block-size <n>\n");
printf(" Delta block size in bytes (default: %d)\n", DELTA_BLOCK_SIZE_DEFAULT);
printf(" --delta-max <n> Max file size for delta transfer (default: %llu)\n", printf(" --delta-max <n> Max file size for delta transfer (default: %llu)\n",
DELTA_MAX_FILE_SIZE); DELTA_MAX_FILE_SIZE);
printf(" -j, --threads[=N] Enable the multithreaded scanner/loader/sender\n"); printf(" -m Enable multithreading\n");
printf(" pipeline; N (1-%d) sets the parallel scanner worker\n", printf(" -s Enable chunk serialization\n");
MAX_SCANNER_THREADS); printf(" -f Enable sendfile (TCP only, not with -c or -s)\n");
printf(" count (bare -j/--threads uses the default)\n");
printf(" --chunk-serialization Enable chunk serialization (long form only)\n");
printf(" -s, --secluded-args Protect-args compatibility option (no effect; remote\n");
printf(" SSH argv is already built injection-safe)\n");
printf(" --sendfile Enable sendfile zero-copy (TCP only; long form only)\n");
printf(" --compress-choice <alg> Compression algorithm (default: zstd)\n");
printf(" --zc <alg> Alias for --compress-choice\n");
printf(" -v, --verbose Enable debug logging\n"); printf(" -v, --verbose Enable debug logging\n");
printf(" -q, --quiet Suppress non-error output\n"); printf(" -M, --preserve Preserve file metadata\n");
printf(" --debug=FLAGS Fine-grained debug logging (use --debug=help for flags)\n");
printf(" --info=FLAGS Fine-grained info: copy,misc,skip,stats,all,none\n");
printf(" none suppresses info even with --verbose\n");
printf(" --preserve Preserve file metadata (long form only)\n");
printf(" -E, --executability Preserve executable permission bits\n");
printf(" -X, --xattrs Preserve user extended attributes (user.* only;\n");
printf(" privileged security.*/trusted.* namespaces are\n");
printf(" never captured or applied)\n");
printf(" -A, --acls Preserve POSIX ACLs (the system.posix_acl_* xattrs;\n");
printf(" setting an ACL the receiver is not permitted to\n");
printf(" set is warned and skipped, never fatal)\n");
printf(" --fake-super Store the source uid/gid/mode/mtime in a reserved\n");
printf(" user.fastsync.stat xattr on each written file and\n");
printf(" re-apply it (fd-relative) on a privileged run; the\n");
printf(" recording format diverges from rsync's user.rsync.%%stat%%\n");
printf(" --super Permit the receiver to attempt super-user activities\n");
printf(" (char/block device-node creation, --write-devices)\n");
printf(" within the confined receive root. Never elevates\n");
printf(" privileges and never bypasses confinement; ownership\n");
printf(" is still applied only with an explicit identity flag\n");
printf(" (--numeric-ids/--chown/--usermap/--groupmap/--copy-as)\n");
printf(" --no-super Forbid those super-user activities even when the\n");
printf(" receiver is running as root\n");
printf(" --chmod <changes> Modify transferred permissions (rsync syntax)\n");
printf(" --numeric-ids Do not map uid/gid by name: use the source numeric\n");
printf(" ids directly when applying ownership\n");
printf(" --usermap=MAP Map usernames when applying ownership: comma-separated\n");
printf(" FROM:TO rules, first match wins. FROM/TO are names\n");
printf(" (resolved on the source machine), * (match any /\n");
printf(" current user), or @N numeric ids. e.g. *:nobody\n");
printf(" --groupmap=MAP Map group names when applying ownership (same syntax)\n");
printf(" --chown=USER:GROUP Override the ownership of transferred files. Forms:\n");
printf(" USER:GROUP, USER (owner only), :GROUP (group only); a\n");
printf(" value of * means the current/root user as appropriate.\n");
printf(" Names resolve on the source machine; @N for numerics.\n");
printf(" (Metadata is enabled with --preserve; -M now means\n");
printf(" rsync's --remote-option.)\n");
printf(" --copy-as=USER[:GROUP] Force every written entry (files, dirs, symlinks\n");
printf(" and special nodes) to USER[:GROUP], resolved on the\n");
printf(" source machine like --chown. Requires a privileged\n");
printf(" (root) receiver and implies --preserve; an\n");
printf(" unprivileged receiver refuses the transfer. Never\n");
printf(" switches process credentials (safe-subset; see\n");
printf(" RSYNC_COMPAT.md). A daemon refuses it.\n");
printf(" --chunk-size <n> Chunk size in bytes (default: %d)\n", DEFAULT_CHUNK_SIZE); printf(" --chunk-size <n> Chunk size in bytes (default: %d)\n", DEFAULT_CHUNK_SIZE);
printf(" --source-dir <path> Source directory\n"); printf(" --source-dir <path> Source directory\n");
printf(" --dest-dir <path> Destination directory\n"); printf(" --dest-dir <path> Destination directory\n");
printf(" --save-to-disk Write received files to disk\n"); printf(" --save-to-disk Write received files to disk\n");
printf(" --server-host <ip> Server IP address (default: 127.0.0.1)\n"); printf(" --server-host <ip> Server IP address (default: 127.0.0.1)\n");
printf(" --server-port <n> Server port (default: 8080)\n"); printf(" --server-port <n> Server port (default: 8080)\n");
printf(" --port <n> Alias for --server-port\n");
printf(" --password-file <f> Authenticate a host::module/path daemon destination.\n");
printf(" The file's first user:password line supplies the\n");
printf(" username and password (only a SHA-256 digest of the\n");
printf(" password is sent; keep the file mode 0600)\n");
printf(" --no-motd Suppress display of the daemon's MOTD (the server\n");
printf(" still sends it; the client just does not show it)\n");
printf(" --bwlimit <KB/s> Bandwidth limit in kilobytes per second\n"); printf(" --bwlimit <KB/s> Bandwidth limit in kilobytes per second\n");
printf(" --tls Enable TLS encryption\n"); printf(" --tls Enable TLS encryption\n");
printf(" --cert <path> TLS certificate file (PEM)\n"); printf(" --cert <path> TLS certificate file (PEM)\n");
printf(" --key <path> TLS private key file (PEM)\n"); printf(" --key <path> TLS private key file (PEM)\n");
printf(" --ca <path> TLS CA certificate file (PEM)\n"); printf(" --ca <path> TLS CA certificate file (PEM)\n");
printf(" --timeout <sec> I/O timeout in seconds (default: 30; long form only)\n"); printf(" --timeout <sec> I/O timeout in seconds (default: 30)\n");
printf(" -T <sec> Alias for --timeout\n");
printf(" --contimeout <sec> Connection timeout in seconds (default: 10)\n"); printf(" --contimeout <sec> Connection timeout in seconds (default: 10)\n");
printf(" --stop-after=MINS Stop the transfer after MINS minutes (a positive\n");
printf(" integer); whatever was already transferred is kept\n");
printf(" --stop-at=TIME Stop at an absolute time: HH:MM, HH:MM:SS, or\n");
printf(" now+N[smhd] (a time already in the past stops the\n");
printf(" transfer immediately; client-only). An early stop\n");
printf(" skips the late --delete keep-set so it cannot delete\n");
printf(" source mirrors that were not yet scanned\n");
printf(" --address <ip> Bind the outgoing client socket to this source address\n");
printf(" -4, --ipv4 Force IPv4 for destination resolution\n");
printf(" -6, --ipv6 Force IPv6 for destination resolution\n");
printf(" --sockopts=OPTS Comma-separated OPT=VAL socket options applied before connect:\n");
printf(" TCP_NODELAY, SO_KEEPALIVE, SO_RCVBUF, SO_SNDBUF, SO_REUSEADDR\n");
printf(" --backup Backup existing files before overwriting\n"); printf(" --backup Backup existing files before overwriting\n");
printf(" --backup-dir <dir> Directory for backups (requires --backup)\n"); printf(" --backup-dir <dir> Directory for backups (requires --backup)\n");
printf(" --suffix <str> Backup suffix (default: ~)\n"); printf(" --suffix <str> Backup suffix (default: ~)\n");
printf(" --stats Print transfer statistics at end\n"); printf(" --stats Print transfer statistics at end\n");
printf(" -i, --itemize-changes Print an rsync-style per-file change line\n");
printf(" --out-format=FORMAT Output format for changed files (%%f %%n %%l %%b %%M %%%%)\n");
printf(" --list-only List source files instead of transferring\n");
printf(" --log-file-format=FORMAT Per-file log line format (needs --log-file)\n");
printf(" -h, --human-readable Print byte sizes in human-readable form\n");
printf(" --max-depth <n> Maximum directory depth (0=unlimited)\n"); printf(" --max-depth <n> Maximum directory depth (0=unlimited)\n");
printf(" -x, --one-file-system Do not cross filesystem boundaries\n");
printf(" --log-file <path> Write log messages to file\n"); printf(" --log-file <path> Write log messages to file\n");
printf(" --stderr=MODE Route logging to stderr: errors or all\n");
printf(" --partial Keep partial files on interrupted transfer\n"); printf(" --partial Keep partial files on interrupted transfer\n");
printf(" --partial-dir <dir> Directory for partial files\n"); printf(" --partial-dir <dir> Directory for partial files\n");
printf(" -T, --temp-dir <dir> Scratch dir for temp files before atomic install\n");
printf(" --fastsync-server-path <path>\n"); printf(" --fastsync-server-path <path>\n");
printf(" Path to fastsync-server on remote (default: fastsync-server)\n"); printf(" Path to fastsync-server on remote (default: fastsync-server)\n");
printf(" --old-args Accepted for rsync CLI compatibility; no effect (the\n");
printf(" remote server path is always safely quoted now)\n");
printf(" -M, --remote-option=OPT Append OPT to the REMOTE server invocation over SSH\n");
printf(" (repeatable; each value is single-quote-escaped on the remote\n");
printf(" command line; empty values and values with control characters\n");
printf(" are rejected; -M OPT, -M=OPT and --remote-option=OPT work)\n");
printf(" --trust-sender Trust the remote sender's file list: the receiver skips its\n");
printf(" own up-front path-traversal/containment re-validation of the\n");
printf(" incoming file list (fewer checks, faster, potentially unsafe).\n");
printf(" Local receiver policy: never sent to the peer, off by default\n");
printf(" -l, --links Copy symlinks as symlinks\n"); printf(" -l, --links Copy symlinks as symlinks\n");
printf(" --copy-links Transform symlinks into referent files\n"); printf(" --copy-links Transform symlinks into referent files\n");
printf(" --safe-links Skip symlinks that point outside transfer tree\n"); printf(" --safe-links Skip symlinks that point outside transfer tree\n");
printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n"); printf(" --copy-unsafe-links Only transform unsafe symlinks into referent files\n");
printf(" -k, --copy-dirlinks Transform symlinks to directories into real dirs\n");
printf(" -K, --keep-dirlinks Keep an existing symlink-to-dir as that dir\n");
printf(" --munge-links Munge symlink targets on the wire (sender)\n");
printf(" -H, --hard-links Preserve hard-link relationships across the transfer\n");
printf(" -S, --sparse Handle sparse files efficiently\n"); printf(" -S, --sparse Handle sparse files efficiently\n");
printf(
" -D Preserve device and special files (implies --devices --specials)\n");
printf(
" --devices Recreate device nodes on the destination (privileged; skipped when\n");
printf(" the receiver lacks CAP_MKNOD)\n");
printf(" --specials Recreate special files (FIFOs) on the destination (sockets "
"skipped)\n");
printf(" --copy-devices Copy a source device's content as a regular file instead\n");
printf(" --write-devices Write received data into an existing destination device node\n");
printf(" --inplace Update files in-place (no temp+rename)\n"); printf(" --inplace Update files in-place (no temp+rename)\n");
printf(
" --preallocate Allocate destination file space up front (fail-fast on full disk)\n");
printf(" --append Resume a shorter destination by appending only its tail\n");
printf(" (prefix is not verified; requires --incremental)\n");
printf(" --append-verify Like --append, but verifies the retained prefix checksum\n");
printf(" before appending (falls back to a full transfer on mismatch)\n");
printf(" --fsync Fsync every written file before publication\n");
printf(" --compress-level <n> Compression level (default: 5)\n"); printf(" --compress-level <n> Compression level (default: 5)\n");
printf(" --zl <n> Alias for --compress-level\n");
printf(" --skip-compress=LIST Skip compression for comma-separated suffixes\n");
printf(" --compress-threads <n> Compression worker threads (requires zstd threaded support)\n");
printf(" --no-OPTION Disable a supported boolean option\n");
printf(" --help Show this help\n"); printf(" --help Show this help\n");
printf(" -V, --version Show version\n"); printf(" -V, --version Show version\n");
} }
void print_debug_usage(void) {
printf("Supported debug flags: IO,PROTO,PACK,UTIL,ALL,NONE\n");
printf("Flags may be comma-separated, for example: --debug=io,proto\n");
printf("Other rsync debug flags are unsupported and rejected.\n");
}
-1
View File
@@ -2,6 +2,5 @@
#define USAGE_H #define USAGE_H
void print_usage(void); void print_usage(void);
void print_debug_usage(void);
#endif #endif
+30 -419
View File
@@ -1,60 +1,11 @@
#include "receiver.h" #include "receiver.h"
#include "charset.h"
#include "chunk.h" #include "chunk.h"
#include "config.h"
#include "delay_updates.h"
#include "file.h"
#include "file_receive.h"
#include "log.h" #include "log.h"
#include "metadata.h"
#include "protocol.h" #include "protocol.h"
#include "utils.h" #include "utils.h"
#include <stdlib.h> #include <stdlib.h>
#include <sys/stat.h> #include <sys/stat.h>
#include <time.h>
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code) {
if (!outcomes)
return false;
if (outcomes->count == outcomes->capacity) {
size_t new_capacity = outcomes->capacity == 0 ? 64 : outcomes->capacity * 2;
if (new_capacity < outcomes->capacity)
return false;
unsigned char* grown = realloc(outcomes->entries, new_capacity);
if (!grown)
return false;
outcomes->entries = grown;
outcomes->capacity = new_capacity;
}
outcomes->entries[outcomes->count++] = code;
return true;
}
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes) {
if (!outcomes)
return;
free(outcomes->entries);
outcomes->entries = NULL;
outcomes->count = 0;
outcomes->capacity = 0;
}
/* End-of-transfer success frame. When --remove-source-files was negotiated
each processed data file is acknowledged first (STATUS_NEXT = written,
STATUS_OK = skipped) so the sender never removes a source the receiver did
not actually store. The frame always ends with a plain STATUS_OK. */
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes) {
if (!config->remove_source_files)
return send_status(fd, STATUS_OK);
size_t count = outcomes ? outcomes->count : 0;
for (size_t i = 0; i < count; i++) {
Status per_file = outcomes->entries[i] == FILE_SAVE_WRITTEN ? STATUS_NEXT : STATUS_OK;
if (!send_status(fd, per_file))
return false;
}
return send_status(fd, STATUS_OK);
}
static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) { static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) {
if (!chunk || !sink || !sink->store_file) if (!chunk || !sink || !sink->store_file)
@@ -75,55 +26,28 @@ static bool receiver_process_chunk(Chunk* chunk, const ReceiverSink* sink) {
return true; return true;
} }
/* P7 Wave D: read one STATUS_DIR_TIMES frame (a count followed by that many
* (path, metadata) directory entries) and route every entry through the regular
* store_file sink. A dir-time entry is RECORD-ONLY (file->dir_time_only): the
* sink accumulates its metadata for end-of-transfer application but creates
* nothing, so an empty/pruned source directory is never resurrected. A large
* tree arrives as repeated frames, each bounded by MAX_MANIFEST_ENTRIES; a
* malformed count or entry is a hard error. */
static bool receiver_process_dir_times(int fd, const Config* config, const ReceiverSink* sink) {
int count;
if (!receive_int(fd, &count) || count < 0 || count > MAX_MANIFEST_ENTRIES)
return false;
for (int i = 0; i < count; i++) {
File* dir = file_receive_dir_time(fd, config);
if (!dir || !sink->store_file(dir, sink->context))
return false;
}
return true;
}
static bool receiver_process_batch(Config* config, int file_descriptor) { static bool receiver_process_batch(Config* config, int file_descriptor) {
int count; int count;
if (config->checksum || !receive_int(file_descriptor, &count) || count < 0 || if (config->checksum || !receive_int(file_descriptor, &count) || count < 0 ||
count > MAX_MANIFEST_ENTRIES) count > MAX_MANIFEST_ENTRIES)
return false; return false;
for (int i = 0; i < count; i++) { for (int i = 0; i < count; i++) {
char* check_path = receive_wire_str(file_descriptor); char* check_path = receive_str(file_descriptor);
if (!check_path) if (!check_path)
return false; return false;
unsigned long long check_size; unsigned long long check_size;
long long check_mtime; long long check_mtime;
long long check_mtime_nsec;
if (!receive_n_data(file_descriptor, &check_size, sizeof(check_size)) || if (!receive_n_data(file_descriptor, &check_size, sizeof(check_size)) ||
!receive_n_data(file_descriptor, &check_mtime, sizeof(check_mtime)) || !receive_n_data(file_descriptor, &check_mtime, sizeof(check_mtime))) {
!receive_n_data(file_descriptor, &check_mtime_nsec, sizeof(check_mtime_nsec)) || free(check_path);
check_mtime_nsec < 0 || check_mtime_nsec >= 1000000000LL) { return false;
}
if (!utils_valid_batch_path(check_path)) {
free(check_path); free(check_path);
send_status(file_descriptor, STATUS_ERROR); send_status(file_descriptor, STATUS_ERROR);
return false; return false;
} }
/* --trust-sender: accept a ``..``/absolute check path (a trusted sender's if (check_size > MAX_RECEIVE_FILE_SIZE) {
odd-but-legit entry) and defer containment to the secure stat below;
an empty path is still always rejected. */
if (check_path[0] == '\0' ||
(!file_get_trust_sender() && !utils_valid_batch_path(check_path))) {
free(check_path);
send_status(file_descriptor, STATUS_ERROR);
return false;
}
if (check_size > MAX_RECEIVE_WHOLE_FILE_SIZE) {
free(check_path); free(check_path);
send_status(file_descriptor, STATUS_ERROR); send_status(file_descriptor, STATUS_ERROR);
return false; return false;
@@ -136,15 +60,8 @@ static bool receiver_process_batch(Config* config, int file_descriptor) {
} }
struct stat st; struct stat st;
bool has_old = file_stat_secure(full_path, &st); bool has_old = file_stat_secure(full_path, &st);
long long old_mtime_nsec = 0; bool match = has_old && (unsigned long long)st.st_size == check_size &&
if (has_old) { (long long)st.st_mtime == check_mtime;
#ifdef __linux__
old_mtime_nsec = st.st_mtim.tv_nsec;
#endif
}
bool match = !config->ignore_times && has_old && (unsigned long long)st.st_size == check_size &&
metadata_mtime_matches(st.st_mtime, old_mtime_nsec, (time_t)check_mtime,
(long)check_mtime_nsec, config->modify_window);
bool sent = send_status(file_descriptor, match ? STATUS_OK : STATUS_NEXT); bool sent = send_status(file_descriptor, match ? STATUS_OK : STATUS_NEXT);
free(full_path); free(full_path);
free(check_path); free(check_path);
@@ -154,226 +71,34 @@ static bool receiver_process_batch(Config* config, int file_descriptor) {
return true; return true;
} }
/* ---- Anti-slowloris connection bounds ----
* A legitimate transfer either streams data frames continuously or, when it
* must pause, sends STATUS_KEEPALIVE so the peer sees the connection is alive.
* An attacker can therefore squat on a connection slot indefinitely by sending
* only keepalives under the per-message timeout. Two CLOCK_MONOTONIC bounds
* defeat that without ever punishing a real transfer:
*
* MAX_SESSION_IDLE_SEC (1 h): the longest a stream may make no forward
* progress. Data/status frames count as progress and refresh the timer;
* keepalives do not. One hour is far longer than any real pause between
* data frames, yet small enough to reap a slowloris well before the 24 h
* session cap.
*
* MAX_SESSION_WALL_SEC (24 h): an absolute ceiling on one connection's
* lifetime as defense-in-depth against a trickle of progress frames that
* resets the idle timer just below its limit. Larger than any plausible
* single transfer while still bounding resource occupancy.
*
* Both are wall-clock deltas, so the per-message poll timeout (60 s by default,
* or --timeout) can never fool them, and both the single-threaded and the -m
* receiver paths (receiver_process_pending) share the same logic. */
#define MAX_SESSION_IDLE_SEC 3600u
#define MAX_SESSION_WALL_SEC 86400u
static unsigned int g_max_session_idle_sec = MAX_SESSION_IDLE_SEC;
static unsigned int g_max_session_wall_sec = MAX_SESSION_WALL_SEC;
void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec) {
g_max_session_idle_sec = idle_sec;
g_max_session_wall_sec = wall_sec;
}
void receiver_reset_time_limits(void) {
g_max_session_idle_sec = MAX_SESSION_IDLE_SEC;
g_max_session_wall_sec = MAX_SESSION_WALL_SEC;
}
bool receiver_time_limit_exceeded(const struct timespec* session_start,
const struct timespec* last_progress,
const struct timespec* now) {
if (!session_start || !last_progress || !now)
return false;
if (now->tv_sec - session_start->tv_sec >= (time_t)g_max_session_wall_sec)
return true;
if (now->tv_sec - last_progress->tv_sec >= (time_t)g_max_session_idle_sec)
return true;
return false;
}
/* A frame proves forward progress only when it cannot be fabricated for free.
* KEEPALIVE/ABORT are pure liveness, and CHECK_BATCH/DIR_TIMES may carry zero
* entries, so a peer must not be able to hold a connection slot forever by
* merely emitting empty frames. */
static bool status_counts_as_progress(Status status) {
switch (status) {
case STATUS_KEEPALIVE:
case STATUS_ABORT:
case STATUS_CHECK_BATCH:
case STATUS_DIR_TIMES:
return false;
default:
return true;
}
}
/* Refresh the progress timestamp for a forward-moving frame and enforce the
* bounds above. Returns false when the connection must be dropped; the
* terminal STATUS_ERROR is sent only when the sink owns error reporting (the
* -m sink sets send_error=false so the main thread emits exactly one). */
static bool receiver_note_status(const struct timespec* session_start,
struct timespec* last_progress, Status status, int file_descriptor,
const ReceiverSink* sink) {
struct timespec now;
if (clock_gettime(CLOCK_MONOTONIC, &now) != 0)
now = *last_progress;
if (status_counts_as_progress(status))
*last_progress = now;
if (!receiver_time_limit_exceeded(session_start, last_progress, &now))
return true;
log_message(LOG_LEVEL_ERROR,
"Receive session exceeded its time bound (idle %us / total %us); aborting connection",
g_max_session_idle_sec, g_max_session_wall_sec);
if (!sink || sink->send_error)
send_status(file_descriptor, STATUS_ERROR);
return false;
}
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) { int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink) {
return receiver_process_pending(config, file_descriptor, sink, NULL);
}
/* Runs the whole receive loop. The delete manifest may legitimately arrive
either FIRST (--delete-before / --delete-during: the sender transmits the
validated keep-set before any file data) or LAST (plain --delete /
--delete-after / --delete-delay: the manifest closes the data stream). In
the early modes the receiver deletes as soon as the manifest has been read
and acknowledges with STATUS_OK so the sender only starts streaming once the
deletion has committed (or failed); in the late modes the manifest is held
and the deletion is committed only after the terminal STATUS_FINISHED proves
the whole transfer succeeded. See receiver_process_pending() for how the -m
receiver defers that commit until its disk writer has drained. */
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
DeleteManifest** pending_manifest) {
Status status; Status status;
if (!receive_status(file_descriptor, &status)) if (!receive_status(file_descriptor, &status))
return -1; return -1;
/* Wall-clock (=CLOCK_MONOTONIC) anti-slowloris bookkeeping. session_start is
* fixed for the whole connection; last_progress is refreshed by every frame
* that is not a keepalive/abort. */
struct timespec session_start;
struct timespec last_progress;
clock_gettime(CLOCK_MONOTONIC, &session_start);
last_progress = session_start;
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
return -1;
bool early_delete = config_delete_timing_early(config);
/* Parked keep-set for the late/commit timing. Every exit path below frees it
exactly once; the only exception is the successful FINISHED handoff, which
transfers ownership to *pending_manifest (used by the -m receiver). */
DeleteManifest* deferred_manifest = NULL;
while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK || while (status == STATUS_NEXT || status == STATUS_CHUNK || status == STATUS_CHECK ||
status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH || status == STATUS_KEEPALIVE || status == STATUS_ABORT || status == STATUS_CHECK_BATCH) {
status == STATUS_MKDIR || status == STATUS_MANIFEST || status == STATUS_HARDLINK ||
status == STATUS_SYMLINK || status == STATUS_SPECIAL || status == STATUS_DIR_TIMES) {
if (status == STATUS_KEEPALIVE) { if (status == STATUS_KEEPALIVE) {
if (!send_status(file_descriptor, STATUS_KEEPALIVE)) if (!send_status(file_descriptor, STATUS_KEEPALIVE))
goto fail; return -1;
goto next_status; goto next;
} }
if (status == STATUS_ABORT) { if (status == STATUS_ABORT) {
log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up"); log_message(LOG_LEVEL_INFO, "Received abort from client, cleaning up");
goto fail; return -1;
} }
if (status == STATUS_CHECK) { if (status == STATUS_CHECK) {
bool skipped = false; bool skipped;
bool would_transfer = false; File* file = receive_incremental_check(file_descriptor, config, &skipped);
File* file = receive_incremental_check_ex(file_descriptor, config, &skipped, &would_transfer); if (!skipped && (!file || !sink->store_file(file, sink->context)))
if (config->dry_run) {
/* Server-contacting --dry-run: the reply has already been sent
(STATUS_OK = up to date, STATUS_DRY_RUN_TRANSFER = would transfer) and
nothing may be stored. Both flags false means a genuine protocol
error (STATUS_ERROR already sent or sent by receive_error below). */
if (!skipped && !would_transfer)
goto receive_error;
} else if (!skipped && (!file || !sink->store_file(file, sink->context))) {
goto receive_error; goto receive_error;
}
} else if (status == STATUS_CHUNK) { } else if (status == STATUS_CHUNK) {
Chunk* chunk = receive_chunk_data(file_descriptor, config); Chunk* chunk = receive_chunk_data(file_descriptor, config);
if (!chunk || !receiver_process_chunk(chunk, sink)) if (!chunk || !receiver_process_chunk(chunk, sink))
goto receive_error; goto receive_error;
} else if (status == STATUS_CHECK_BATCH) { } else if (status == STATUS_CHECK_BATCH) {
if (!receiver_process_batch(config, file_descriptor)) if (!receiver_process_batch(config, file_descriptor))
goto fail; return -1;
goto next_status; goto next;
} else if (status == STATUS_MKDIR) {
File* dir = file_receive_directory(file_descriptor, config);
if (!dir || !sink->store_file(dir, sink->context))
goto receive_error;
} else if (status == STATUS_DIR_TIMES) {
if (!receiver_process_dir_times(file_descriptor, config, sink))
goto receive_error;
} else if (status == STATUS_HARDLINK) {
File* file = file_receive_hardlink(file_descriptor);
if (!file || !sink->store_file(file, sink->context))
goto receive_error;
} else if (status == STATUS_SYMLINK) {
File* sym = file_receive_symlink(file_descriptor, config);
if (!sym || !sink->store_file(sym, sink->context))
goto receive_error;
} else if (status == STATUS_SPECIAL) {
File* file = file_receive_special(file_descriptor);
if (!file || !sink->store_file(file, sink->context))
goto receive_error;
} else if (status == STATUS_MANIFEST) {
DeleteManifest* manifest = receive_manifest_entries(file_descriptor);
if (!manifest)
goto fail; /* receive_manifest_entries already sent STATUS_ERROR */
if (config->dry_run) {
/* Server-contacting --dry-run mutates nothing, so a keep-set manifest
is consumed and discarded. The early-delete mode still needs its ACK
so a sender blocked on the delete handshake is not left hanging. */
delete_manifest_free(manifest);
if (early_delete && !send_status(file_descriptor, STATUS_OK))
goto fail;
goto next_status;
}
if (early_delete) {
/* --delete-before / --delete-during: the manifest is authoritative the
moment it arrives, before any file data. Delete now and acknowledge
so the sender only starts streaming once the deletion committed (or
failed). This is the rsync delete-before/delete-during window: a
later transfer failure does not restore these deletions. */
bool deletion_ok = (config->use_delete || config->delete_missing_args)
? manifest_delete_all(config, manifest)
: true;
delete_manifest_free(manifest);
if (!deletion_ok) {
send_status(file_descriptor, STATUS_ERROR);
goto fail;
}
if (!send_status(file_descriptor, STATUS_OK))
goto fail;
} else if (config->use_delete || config->delete_missing_args) {
/* Plain --delete / --delete-after / --delete-delay and the
--delete-missing-args exact-path deletions: hold the manifest and
commit it only after STATUS_FINISHED. */
if (deferred_manifest) {
log_message(LOG_LEVEL_ERROR, "Received a second delete manifest");
delete_manifest_free(deferred_manifest);
deferred_manifest = NULL;
delete_manifest_free(manifest);
send_status(file_descriptor, STATUS_ERROR);
goto fail;
}
deferred_manifest = manifest;
} else {
delete_manifest_free(manifest);
}
goto next_status;
} else { } else {
File* file = file_receive(config, file_descriptor); File* file = file_receive(config, file_descriptor);
if (!file) { if (!file) {
@@ -383,149 +108,35 @@ int receiver_process_pending(Config* config, int file_descriptor, const Receiver
if (!sink->store_file(file, sink->context)) if (!sink->store_file(file, sink->context))
goto receive_error; goto receive_error;
} }
next_status: next:
if (!receive_status(file_descriptor, &status)) if (!receive_status(file_descriptor, &status))
goto receive_error; goto receive_error;
if (!receiver_note_status(&session_start, &last_progress, status, file_descriptor, sink))
goto fail;
} }
if (status == STATUS_MANIFEST && receive_manifest(file_descriptor, config, &status) != 0)
return -1;
if (status != STATUS_FINISHED) { if (status != STATUS_FINISHED) {
log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status"); log_message(LOG_LEVEL_ERROR, "Did not receive FINISHED Status");
goto receive_error; goto receive_error;
} }
/* Commit-style (late) deletion: every data frame has been received and the if (sink->send_success && !send_status(file_descriptor, STATUS_OK))
sender proved the whole tree with STATUS_FINISHED. The single-threaded return -1;
receiver stores files synchronously, so everything is on disk here and the
deletion can be committed before the --delay-updates publication in
send_success (the walker skips the staging dir, so staged files are never
treated as extras). The -m receiver passes `pending_manifest` because its
disk writer may still be draining; the caller commits after the writer has
joined so no extra file is removed unless the transfer is known to have
succeeded. */
if (deferred_manifest) {
if (pending_manifest) {
*pending_manifest = deferred_manifest;
deferred_manifest = NULL;
} else {
bool deletion_ok = manifest_delete_all(config, deferred_manifest);
delete_manifest_free(deferred_manifest);
deferred_manifest = NULL;
if (!deletion_ok) {
send_status(file_descriptor, STATUS_ERROR);
goto fail;
}
}
}
if (sink->send_success) {
if (sink->send_success_frame) {
if (!sink->send_success_frame(file_descriptor, sink->context))
goto fail;
} else if (!send_status(file_descriptor, STATUS_OK)) {
goto fail;
}
}
return 0; return 0;
fail:
/* Failure exits that must not (or already did) report a STATUS_ERROR. The
parked keep-set is dropped: never commit a deletion for a failed stream. */
if (deferred_manifest) {
delete_manifest_free(deferred_manifest);
deferred_manifest = NULL;
}
return -1;
receive_error: receive_error:
if (deferred_manifest) {
delete_manifest_free(deferred_manifest);
deferred_manifest = NULL;
}
if (sink->send_error) if (sink->send_error)
send_status(file_descriptor, STATUS_ERROR); send_status(file_descriptor, STATUS_ERROR);
return -1; return -1;
} }
/* ---- Single-threaded sink (used by receiver_receive_files) ---- */ static bool receiver_save_file(File* file, void* context) {
Config* config = context;
typedef struct { bool success =
Config* config; !config->save_to_disk || file_save_to_disk(config->receive_root_directory, file, config);
ReceiverOutcomes outcomes;
/* P7 Wave D: directory metadata accumulated during the stream, applied only
after the whole transfer (and its delete/publication phases) has run so a
child write never clobbers a directory mtime. */
DirTimeList dir_times;
} ReceiverSaveContext;
static bool receiver_save_file(File* file, void* context_pointer) {
ReceiverSaveContext* context = context_pointer;
FileSaveResult result = FILE_SAVE_ERROR;
if (context->config->dry_run) {
/* Defense in depth: a dry-run receiver mutates nothing even if a data
frame reaches the sink (the sender is not supposed to send one). */
result = FILE_SAVE_SKIPPED;
} else if (!context->config->save_to_disk) {
/* Nothing is stored; report the file as not-written so a
--remove-source-files sender keeps its source. */
result = FILE_SAVE_SKIPPED;
} else {
result = file_save_to_disk_full(context->config->receive_root_directory, file, context->config);
}
/* A directory's times are deferred, never applied inline: collect the
metadata now and apply it at the end. -O/--omit-dir-times is honored by
dir_time_list_apply's caller (see receiver_send_success_frame). */
if (result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
dir_times_should_capture(context->config) &&
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
file_destroy(file);
return false;
}
/* A dry-run receiver mutates nothing AND records no per-file outcomes: a
hostile dry-run client that streamed data frames anyway must not be able to
grow `outcomes` without bound (receiver_outcomes_append reallocs uncharged)
or force a per-frame ack. */
if (!context->config->dry_run && result != FILE_SAVE_ERROR &&
context->config->remove_source_files && !file->is_dir && !file->is_special && !file->skip &&
!receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
file_destroy(file);
return false;
}
file_destroy(file); file_destroy(file);
return result != FILE_SAVE_ERROR; return success;
}
static bool receiver_send_success_frame(int fd, void* context_pointer) {
ReceiverSaveContext* context = context_pointer;
/* Server-contacting --dry-run: nothing was staged or written, so there is
nothing to publish and no directory times to stamp. */
if (context->config->dry_run)
return receiver_send_final_success(fd, context->config, &context->outcomes);
/* --delay-updates: the whole protocol stream (including manifest/delete
handling, which ran inside receiver_process) has succeeded and every
staged file was fully written. Publish them atomically now, before the
success/outcome frame tells a --remove-source-files sender it may delete
its sources. */
if (context->config->delay_updates && context->config->delay_context) {
if (!delay_updates_publish(context->config->delay_context, context->config)) {
send_status(fd, STATUS_ERROR);
return false;
}
}
/* P7 Wave D: every child is now written and the delete / --delay-updates
phases have committed, so it is finally safe to stamp directory times.
This runs after the deferred deletion because receiver_process commits it
before calling this success frame. */
dir_time_list_apply(&context->dir_times, context->config->receive_root_directory);
return receiver_send_final_success(fd, context->config, &context->outcomes);
} }
int receiver_receive_files(Config* config, int file_descriptor) { int receiver_receive_files(Config* config, int file_descriptor) {
ReceiverSaveContext context = {.config = config, .outcomes = {0}}; ReceiverSink sink = {receiver_save_file, config, true, true};
dir_time_list_init(&context.dir_times); return receiver_process(config, file_descriptor, &sink);
ReceiverSink sink = {receiver_save_file, &context, true, true, receiver_send_success_frame};
int ret = receiver_process(config, file_descriptor, &sink);
if (ret != 0 && config->delay_updates && config->delay_context)
delay_updates_cleanup(config->delay_context);
receiver_outcomes_destroy(&context.outcomes);
dir_time_list_free(&context.dir_times);
return ret;
} }
-49
View File
@@ -3,66 +3,17 @@
#include "config.h" #include "config.h"
#include "file.h" #include "file.h"
#include "file_receive.h"
#include <stdbool.h>
#include <time.h>
typedef bool (*ReceiverFileSink)(File* file, void* context); typedef bool (*ReceiverFileSink)(File* file, void* context);
/* Ordered per-file save outcomes for one connection. One entry is appended
for every data-bearing file the receiver processes (in the order the files
were sent) so the sender of a --remove-source-files transfer can be told
which sources were actually written versus skipped on the receiver. */
typedef struct {
unsigned char* entries; /* FILE_SAVE_WRITTEN or FILE_SAVE_SKIPPED */
size_t count;
size_t capacity;
} ReceiverOutcomes;
typedef bool (*ReceiverSuccessFrame)(int fd, void* context);
typedef struct { typedef struct {
ReceiverFileSink store_file; ReceiverFileSink store_file;
void* context; void* context;
bool send_error; bool send_error;
bool send_success; bool send_success;
/* Emits the end-of-transfer success frame. When the sender requested
--remove-source-files this includes one per-file status per processed
data file followed by the final STATUS_OK; otherwise just STATUS_OK. */
ReceiverSuccessFrame send_success_frame;
} ReceiverSink; } ReceiverSink;
bool receiver_outcomes_append(ReceiverOutcomes* outcomes, unsigned char code);
void receiver_outcomes_destroy(ReceiverOutcomes* outcomes);
bool receiver_send_final_success(int fd, const Config* config, const ReceiverOutcomes* outcomes);
int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink); int receiver_process(Config* config, int file_descriptor, const ReceiverSink* sink);
/* receiver_process with an escape hatch for the commit-style (late) deletion:
when `pending_manifest` is non-NULL the receiver does NOT delete at
STATUS_FINISHED itself; instead it stores the owned keep-set manifest there
(leaving *pending_manifest untouched on early modes/errors) so the caller can
commit the deletion only after its disk writer has fully drained. Pass NULL
to keep the default behaviour (delete before the success frame). */
int receiver_process_pending(Config* config, int file_descriptor, const ReceiverSink* sink,
DeleteManifest** pending_manifest);
int receiver_receive_files(Config* config, int file_descriptor); int receiver_receive_files(Config* config, int file_descriptor);
/* ---- Connection time bounds (anti-slowloris) ----
* receiver_process_pending() aborts a connection that makes no forward progress
* (only STATUS_KEEPALIVE/STATUS_ABORT frames) beyond a wall-clock idle limit,
* and enforces a hard cap on the whole session. Both are CLOCK_MONOTONIC
* deltas, independent of the per-message poll deadline, so a 60 s (or
* --timeout) receive window can never reset them. Defaults are deliberately
* generous (see MAX_SESSION_IDLE_SEC / MAX_SESSION_WALL_SEC in receiver.c). */
/* Test seam: override the idle/session wall-clock limits (0 = abort on the
* next status). Always restore with receiver_reset_time_limits(). */
void receiver_set_time_limits(unsigned int idle_sec, unsigned int wall_sec);
void receiver_reset_time_limits(void);
/* Pure predicate over explicit monotonic timestamps, exposed so the bound is
* unit-testable without sleeping. True when either the idle or the overall
* session limit has elapsed. */
bool receiver_time_limit_exceeded(const struct timespec* session_start,
const struct timespec* last_progress, const struct timespec* now);
#endif #endif
-258
View File
@@ -1,258 +0,0 @@
#include "receiver_pipeline.h"
#include "log.h"
#include "protocol.h"
#include "queue.h"
#include "utils.h"
#include <stdlib.h>
#include <string.h>
#include <threads.h>
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue,
int file_descriptor, SSL* ssl) {
PipelineContextReceiver* context = malloc(sizeof(PipelineContextReceiver));
if (context == NULL)
return NULL;
context->config = config;
context->queue = queue;
context->file_descriptor = file_descriptor;
context->ssl = ssl;
context->outcomes.entries = NULL;
context->outcomes.count = 0;
context->outcomes.capacity = 0;
dir_time_list_init(&context->dir_times);
protocol_session_init(&context->session, file_descriptor, file_descriptor);
protocol_session_set_ssl(&context->session, ssl);
context->receiver_done = false;
context->queued_bytes = 0;
context->max_queue_bytes = 0;
context->deferred_manifest = NULL;
atomic_init(&context->cancelled, false);
int init = 0;
if (mtx_init(&context->mutex, mtx_plain) != thrd_success)
goto fail;
init++;
if (cnd_init(&context->condition_not_full) != thrd_success)
goto fail;
init++;
if (cnd_init(&context->condition_not_empty) != thrd_success)
goto fail;
// cppcheck-suppress unreadVariable
init++;
return context;
fail:
log_perror("Error initializing synchronization objects");
if (init >= 3)
cnd_destroy(&context->condition_not_empty);
if (init >= 2)
cnd_destroy(&context->condition_not_full);
if (init >= 1)
mtx_destroy(&context->mutex);
free(context);
return NULL;
}
void pipeline_context_receiver_destroy(PipelineContextReceiver* context) {
config_delete(context->config);
if (context->deferred_manifest)
delete_manifest_free(context->deferred_manifest);
queue_destroy(context->queue);
receiver_outcomes_destroy(&context->outcomes);
dir_time_list_free(&context->dir_times);
mtx_destroy(&context->mutex);
cnd_destroy(&context->condition_not_full);
cnd_destroy(&context->condition_not_empty);
free(context);
}
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
size_t max_bytes) {
if (context == NULL)
return;
mtx_lock(&context->mutex);
context->max_queue_bytes = max_bytes;
context->queued_bytes = 0;
cnd_broadcast(&context->condition_not_full);
mtx_unlock(&context->mutex);
}
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
size_t released_bytes) {
if (context == NULL || context->max_queue_bytes == 0 || released_bytes == 0)
return;
mtx_lock(&context->mutex);
if (released_bytes >= context->queued_bytes)
context->queued_bytes = 0;
else
context->queued_bytes -= released_bytes;
cnd_signal(&context->condition_not_full);
mtx_unlock(&context->mutex);
}
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file) {
if (context == NULL || file == NULL)
return false;
size_t file_bytes = file->data ? file->data->size : 0;
mtx_lock(&context->mutex);
while (!atomic_load(&context->cancelled)) {
bool blocked_by_count = queue_is_full(context->queue);
bool blocked_by_budget = false;
if (context->max_queue_bytes > 0) {
size_t budget = context->max_queue_bytes;
size_t used = context->queued_bytes;
if (used >= budget) {
blocked_by_budget = true;
} else if (file_bytes > budget - used) {
/* A single payload larger than the whole budget (not possible with
the per-file receive cap) is only admitted to an empty pipeline so
the wait can never deadlock. */
blocked_by_budget = used != 0;
}
}
if (!blocked_by_count && !blocked_by_budget)
break;
cnd_wait(&context->condition_not_full, &context->mutex);
}
if (atomic_load(&context->cancelled)) {
mtx_unlock(&context->mutex);
file_destroy(file);
return false;
}
if (!queue_enqueue(context->queue, file)) {
mtx_unlock(&context->mutex);
file_destroy(file);
return false;
}
context->queued_bytes += file_bytes;
cnd_signal(&context->condition_not_empty);
mtx_unlock(&context->mutex);
return true;
}
static bool receiver_enqueue_file(File* file, void* context_pointer) {
PipelineContextReceiver* context = (PipelineContextReceiver*)context_pointer;
return pipeline_context_receiver_enqueue_file(context, file);
}
static void receiver_thread_fail(PipelineContextReceiver* context) {
mtx_lock(&context->mutex);
atomic_store(&context->cancelled, true);
context->receiver_done = true;
cnd_broadcast(&context->condition_not_empty);
cnd_broadcast(&context->condition_not_full);
mtx_unlock(&context->mutex);
}
int receive_thread(void* pipeline_context) {
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
protocol_session_bind(&context->session);
mtx_lock(&context->mutex);
int file_descriptor = context->file_descriptor;
const Config* config = context->config;
mtx_unlock(&context->mutex);
ReceiverSink sink = {receiver_enqueue_file, context, false, false, NULL};
if (receiver_process_pending((Config*)config, file_descriptor, &sink,
&context->deferred_manifest) != 0) {
receiver_thread_fail(context);
protocol_session_unbind();
return thrd_error;
}
mtx_lock(&context->mutex);
context->receiver_done = true;
cnd_signal(&context->condition_not_empty);
mtx_unlock(&context->mutex);
protocol_session_unbind();
return thrd_success;
}
int write_thread(void* pipeline_context) {
PipelineContextReceiver* context = (PipelineContextReceiver*)pipeline_context;
protocol_session_bind(&context->session);
mtx_lock(&context->mutex);
bool save_to_disk = context->config->save_to_disk;
char* root_directory = str_dup(context->config->receive_root_directory);
mtx_unlock(&context->mutex);
if (save_to_disk && !root_directory) {
mtx_lock(&context->mutex);
atomic_store(&context->cancelled, true);
context->receiver_done = true;
cnd_broadcast(&context->condition_not_full);
cnd_broadcast(&context->condition_not_empty);
mtx_unlock(&context->mutex);
protocol_session_unbind();
return thrd_error;
}
while (true) {
File* file =
queue_dequeue_multithreaded(context->queue, &context->mutex, &context->condition_not_empty,
&context->condition_not_full, &context->receiver_done);
if (file == NULL) {
free(root_directory);
protocol_session_unbind();
return thrd_success;
}
size_t file_bytes = file->data ? file->data->size : 0;
FileSaveResult result = FILE_SAVE_SKIPPED;
/* Server-contacting --dry-run: never write. The receiver thread does not
enqueue anything on the dry-run path, but this keeps the writer thread
provably mutation-free if a data frame ever reached it. */
bool dry_run = context->config->dry_run;
if (save_to_disk && !dry_run) {
result = file_save_to_disk_full(root_directory, file, context->config);
if (result == FILE_SAVE_ERROR) {
file_destroy(file);
pipeline_context_receiver_note_bytes_released(context, file_bytes);
mtx_lock(&context->mutex);
atomic_store(&context->cancelled, true);
context->receiver_done = true;
cnd_broadcast(&context->condition_not_full);
cnd_broadcast(&context->condition_not_empty);
mtx_unlock(&context->mutex);
free(root_directory);
protocol_session_unbind();
return thrd_error;
}
}
/* P7 Wave D: a directory's times are never applied inline (a later child
write would clobber them); accumulate the metadata here and let the
caller apply it once every writer has drained. */
if (!dry_run && result != FILE_SAVE_ERROR && file->is_dir && file->metadata &&
dir_times_should_capture(context->config) &&
!dir_time_list_add(&context->dir_times, file->path, file->metadata)) {
file_destroy(file);
pipeline_context_receiver_note_bytes_released(context, file_bytes);
mtx_lock(&context->mutex);
atomic_store(&context->cancelled, true);
context->receiver_done = true;
cnd_broadcast(&context->condition_not_full);
cnd_broadcast(&context->condition_not_empty);
mtx_unlock(&context->mutex);
free(root_directory);
protocol_session_unbind();
return thrd_error;
}
/* Record the per-file outcome so a --remove-source-files sender learns
which sources were actually written versus skipped on the receiver.
Explicit directory entries and recreated device/special nodes have no
source and are never acknowledged (mirrors receiver.c). */
if (!dry_run && context->config->remove_source_files && !file->is_dir && !file->is_special &&
!file->skip && !receiver_outcomes_append(&context->outcomes, (unsigned char)result)) {
file_destroy(file);
pipeline_context_receiver_note_bytes_released(context, file_bytes);
mtx_lock(&context->mutex);
atomic_store(&context->cancelled, true);
context->receiver_done = true;
cnd_broadcast(&context->condition_not_full);
cnd_broadcast(&context->condition_not_empty);
mtx_unlock(&context->mutex);
free(root_directory);
protocol_session_unbind();
return thrd_error;
}
file_destroy(file);
pipeline_context_receiver_note_bytes_released(context, file_bytes);
}
}
-68
View File
@@ -1,68 +0,0 @@
#ifndef RECEIVER_PIPELINE_H
#define RECEIVER_PIPELINE_H
#include <stdatomic.h>
#include <stdbool.h>
#include <threads.h>
#include "config.h"
#include "file.h"
#include "file_receive.h"
#include "protocol.h"
#include "queue.h"
#include "receiver.h"
#include <openssl/ssl.h>
typedef struct PipelineContextReceiver {
Queue* queue;
Config* config;
int file_descriptor;
SSL* ssl;
ProtocolSession session;
ReceiverOutcomes outcomes;
mtx_t mutex;
cnd_t condition_not_full;
cnd_t condition_not_empty;
bool receiver_done;
atomic_bool cancelled;
/* Aggregate payload bytes that have been received but not yet released by
the disk writer (queued or in the writer's hand). Guarded by `mutex`.
When `max_queue_bytes` is non-zero the receiver blocks before enqueuing
once this total would exceed it, so decompressed/copied file payloads
buffered ahead of a slow disk writer respect the per-connection memory
budget instead of growing without bound. */
size_t queued_bytes;
size_t max_queue_bytes;
/* Keep-set manifest for the commit-style (late) deletion
(--delete/--delete-after/--delete-delay). receive_thread parses the whole
protocol stream but hands the manifest here instead of deleting while the
disk writer may still be draining; the caller (server.c) commits the
deletion after both threads have joined, so no extra is removed unless the
transfer truly succeeded. NULL in the early delete modes (which delete at
the manifest). */
DeleteManifest* deferred_manifest;
/* P7 Wave D: directory metadata collected by write_thread from received
directory entries. Only write_thread mutates it (before it joins); the
caller (server.c) applies it after the delete/delay-updates phase. */
DirTimeList dir_times;
} PipelineContextReceiver;
PipelineContextReceiver* pipeline_context_receiver_create(Config* config, Queue* queue_receiver,
int file_descriptor, SSL* ssl);
void pipeline_context_receiver_destroy(PipelineContextReceiver* context);
/* Bound the bytes buffered ahead of the disk writer (see max_queue_bytes). */
void pipeline_context_receiver_set_queue_byte_limit(PipelineContextReceiver* context,
size_t max_bytes);
/* Blocking enqueue used by the receive pipeline sink. Blocks while the queue
is full by element count or when adding `file` would push queued_bytes over
the configured byte limit; waits until the disk writer releases bytes.
Takes ownership of `file` on success and destroys it on failure/cancel. */
bool pipeline_context_receiver_enqueue_file(PipelineContextReceiver* context, File* file);
/* Account for `released_bytes` of payload memory that has been freed by the
disk writer, unblocking a receiver that is waiting on the byte limit. */
void pipeline_context_receiver_note_bytes_released(PipelineContextReceiver* context,
size_t released_bytes);
int receive_thread(void* pipeline_context);
int write_thread(void* pipeline_context);
#endif
+255 -1188
View File
File diff suppressed because it is too large Load Diff
-305
View File
@@ -1,305 +0,0 @@
#include "server_cli.h"
#include "charset.h"
#include "credentials.h"
#include "utils.h"
#include <limits.h>
#include <stdarg.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/socket.h>
static void set_error(char* err, size_t err_size, const char* fmt, ...) {
if (!err || err_size == 0)
return;
va_list args;
va_start(args, fmt);
vsnprintf(err, err_size, fmt, args);
va_end(args);
}
void server_cli_options_default(ServerCliOptions* opts) {
if (!opts)
return;
memset(opts, 0, sizeof(*opts));
opts->destination_root = ".";
opts->port = 8080;
opts->bind_family = AF_UNSPEC;
}
static bool arg_is(const char* arg, const char* name) {
return strcmp(arg, name) == 0;
}
/* Match "--opt" against "--opt=value" / separate-value forms; on the "=" form
* *value receives the inline value. Returns true when the argument is the
* named option in either form. */
static bool arg_has_value(const char* arg, const char* name, const char** value) {
if (strcmp(arg, name) == 0)
return true; /* separate form; caller takes the next argv slot */
size_t name_len = strlen(name);
if (strncmp(arg, name, name_len) == 0 && arg[name_len] == '=') {
*value = arg + name_len + 1;
return true;
}
return false;
}
static int parse_port_arg(const char* value, int* port, char* err, size_t err_size) {
char* end;
long p = strtol(value, &end, 10);
if (*end != '\0' || p <= 0 || p > 65535) {
char* escaped = output_escape(value, false);
set_error(err, err_size, "invalid port '%s' (must be 1-65535)",
escaped ? escaped : "<allocation failed>");
free(escaped);
return -1;
}
*port = (int)p;
return 0;
}
int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err, size_t err_size) {
if (err && err_size)
err[0] = '\0';
server_cli_options_default(opts);
for (int i = 1; i < argc; i++) {
const char* inline_value = NULL;
if (arg_is(argv[i], "--help")) {
opts->show_help = true;
return 1;
} else if (arg_is(argv[i], "--stdio")) {
opts->stdio_mode = true;
} else if (arg_is(argv[i], "--daemon")) {
opts->daemon_mode = true;
} else if (arg_is(argv[i], "--no-detach")) {
opts->no_detach = true;
} else if (arg_is(argv[i], "-v") || arg_is(argv[i], "--verbose")) {
opts->verbose = true;
} else if (arg_is(argv[i], "--tls")) {
opts->use_tls = true;
} else if (arg_is(argv[i], "--cert")) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --cert");
return -1;
}
opts->tls_cert = argv[++i];
} else if (arg_is(argv[i], "--key")) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --key");
return -1;
}
opts->tls_key = argv[++i];
} else if (arg_is(argv[i], "--ca")) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --ca");
return -1;
}
opts->tls_ca = argv[++i];
} else if (arg_is(argv[i], "--client-cn")) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --client-cn");
return -1;
}
opts->client_cn = argv[++i];
} else if (arg_is(argv[i], "--destination-root")) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --destination-root");
return -1;
}
opts->destination_root = argv[++i];
opts->destination_root_set = true;
} else if (arg_has_value(argv[i], "--password-file", &inline_value)) {
if (!inline_value) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --password-file");
return -1;
}
inline_value = argv[++i];
}
opts->password_file = inline_value;
} else if (arg_has_value(argv[i], "--early-input", &inline_value)) {
if (!inline_value) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --early-input");
return -1;
}
inline_value = argv[++i];
}
opts->early_input_file = inline_value;
} else if (arg_has_value(argv[i], "--hash-credentials", &inline_value)) {
if (!inline_value) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --hash-credentials");
return -1;
}
inline_value = argv[++i];
}
opts->hash_credentials_file = inline_value;
} else if (arg_has_value(argv[i], "--iterations", &inline_value)) {
if (!inline_value) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --iterations");
return -1;
}
inline_value = argv[++i];
}
char* end = NULL;
long n = strtol(inline_value, &end, 10);
if (!end || *end != '\0' || n < (long)CREDENTIAL_MIN_ITERS ||
n > (long)CREDENTIAL_MAX_ITERS) {
set_error(err, err_size, "--iterations must be in [%u,%u], got '%s'", CREDENTIAL_MIN_ITERS,
CREDENTIAL_MAX_ITERS, inline_value);
return -1;
}
opts->hash_iterations = (uint32_t)n;
opts->hash_iterations_set = true;
} else if (arg_is(argv[i], "--address")) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --address");
return -1;
}
opts->bind_address = argv[++i];
} else if (arg_is(argv[i], "-4") || arg_is(argv[i], "--ipv4")) {
if (opts->bind_family == AF_INET6) {
set_error(err, err_size, "--ipv4 and --ipv6 are mutually exclusive");
return -1;
}
opts->bind_family = AF_INET;
} else if (arg_is(argv[i], "-6") || arg_is(argv[i], "--ipv6")) {
if (opts->bind_family == AF_INET) {
set_error(err, err_size, "--ipv4 and --ipv6 are mutually exclusive");
return -1;
}
opts->bind_family = AF_INET6;
} else if (arg_is(argv[i], "--allow-delete")) {
opts->allow_delete = true;
} else if (arg_is(argv[i], "--trust-sender")) {
opts->trust_sender = true;
} else if (arg_is(argv[i], "--no-super")) {
opts->no_super = true;
} else if (arg_is(argv[i], "--allow-super")) {
opts->allow_super = true;
} else if (arg_is(argv[i], "--allow-unauthenticated")) {
opts->allow_unauthenticated = true;
} else if (arg_has_value(argv[i], "--iconv", &inline_value)) {
if (!inline_value) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --iconv");
return -1;
}
inline_value = argv[++i];
}
opts->iconv_spec = inline_value;
} else if (arg_is(argv[i], "-p")) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for -p");
return -1;
}
opts->port_set = true;
if (parse_port_arg(argv[++i], &opts->port, err, err_size) != 0)
return -1;
} else {
if (arg_has_value(argv[i], "--config", &inline_value)) {
if (!inline_value) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --config");
return -1;
}
inline_value = argv[++i];
}
opts->config_path = inline_value;
} else if (arg_has_value(argv[i], "--dparam", &inline_value)) {
if (!inline_value) {
if (i + 1 >= argc) {
set_error(err, err_size, "missing argument for --dparam");
return -1;
}
inline_value = argv[++i];
}
const char** grown =
realloc((char**)opts->dparams, (size_t)(opts->dparam_count + 1) * sizeof(const char*));
if (!grown) {
set_error(err, err_size, "out of memory parsing --dparam");
return -1;
}
opts->dparams = grown;
opts->dparams[opts->dparam_count++] = inline_value;
} else if (argv[i][0] == '-') {
char* escaped = output_escape(argv[i], false);
set_error(err, err_size, "unknown option: %s", escaped ? escaped : "<allocation failed>");
free(escaped);
return -1;
} else {
set_error(err, err_size, "unexpected argument '%s'", argv[i]);
return -1;
}
}
}
/* Cross-mode validation. */
if (opts->stdio_mode && opts->daemon_mode) {
set_error(err, err_size, "--stdio and --daemon are mutually exclusive");
return -1;
}
if (opts->daemon_mode && opts->destination_root_set) {
set_error(err, err_size,
"--destination-root cannot be combined with --daemon (module paths "
"replace it)");
return -1;
}
if (!opts->daemon_mode &&
(opts->config_path != NULL || opts->dparam_count > 0 || opts->no_detach ||
opts->password_file != NULL || opts->early_input_file != NULL)) {
set_error(err, err_size,
"--config, --dparam, --no-detach, --password-file, and --early-input require "
"--daemon");
return -1;
}
if (opts->hash_credentials_file != NULL && (opts->daemon_mode || opts->stdio_mode)) {
set_error(err, err_size, "--hash-credentials cannot be combined with --daemon or --stdio");
return -1;
}
if (opts->allow_super && opts->no_super) {
set_error(err, err_size, "--allow-super and --no-super are mutually exclusive");
return -1;
}
if (opts->allow_super && opts->daemon_mode) {
set_error(err, err_size,
"--allow-super is for a locally-launched standalone TCP server; daemon modules opt "
"in per module with 'client owner = yes'");
return -1;
}
/* --stdio is the SSH transport: the remote server argv is composed by the
* CLIENT (directly and via --remote-option), so a client could otherwise pass
* --allow-super to a root --stdio receiver and defeat the C3 secure default.
* Never honor it there; the super mode stays forced OFF. An operator who
* must keep the historical permissive behavior over SSH has to launch the
* receiver through a forced command, not via client-composed argv. */
if (opts->allow_super && opts->stdio_mode) {
set_error(err, err_size,
"--allow-super is not accepted with --stdio (the remote argv is client-composed; "
"use a forced command if the default must hold)");
return -1;
}
if (opts->hash_iterations_set && opts->hash_credentials_file == NULL) {
set_error(err, err_size, "--iterations requires --hash-credentials");
return -1;
}
/* --iconv: reject a malformed CONVERT_SPEC or an unsupported charset name at
startup (a probe iconv_open is attempted). */
if (opts->iconv_spec != NULL && !charset_spec_valid(opts->iconv_spec)) {
set_error(err, err_size, "--iconv requires LOCAL[,REMOTE] charset names supported by iconv");
return -1;
}
return 0;
}
void server_cli_options_free(ServerCliOptions* opts) {
if (!opts)
return;
free((char**)opts->dparams);
opts->dparams = NULL;
opts->dparam_count = 0;
}
-79
View File
@@ -1,79 +0,0 @@
#ifndef SERVER_CLI_H
#define SERVER_CLI_H
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
/* Parsed fastsync-server command line. All string members are borrowed
* pointers into the original argv (valid for the life of the argv array the
* caller passed to server_cli_parse); dparams points at the raw --dparam
* argument strings. No member owns heap memory. */
typedef struct ServerCliOptions {
bool stdio_mode; /* --stdio */
bool daemon_mode; /* --daemon */
bool no_detach; /* --no-detach */
bool verbose; /* -v / --verbose */
bool show_help; /* --help */
bool use_tls; /* --tls */
const char* tls_cert; /* --cert */
const char* tls_key; /* --key */
const char* tls_ca; /* --ca */
const char* client_cn; /* --client-cn */
bool destination_root_set; /* an explicit --destination-root was given */
const char* destination_root; /* --destination-root value ("." if unset) */
bool port_set; /* an explicit -p was given */
int port; /* -p value (default 8080 when unset) */
const char* config_path; /* --config value, or NULL */
const char* password_file; /* --password-file value, or NULL (daemon) */
const char* early_input_file; /* --early-input value, or NULL (daemon) */
/* --hash-credentials=FILE: read `user:password` lines from FILE and print
* new-format credential-store lines to stdout, then exit. Standalone mode
* (mutually exclusive with --daemon/--stdio). */
const char* hash_credentials_file;
bool hash_iterations_set; /* an explicit --iterations was given */
uint32_t hash_iterations; /* --iterations value (default CREDENTIAL_DEFAULT_ITERS) */
const char** dparams; /* raw --dparam override strings */
int dparam_count;
const char* bind_address; /* --address */
int bind_family; /* AF_UNSPEC / AF_INET / AF_INET6 */
bool allow_delete; /* --allow-delete */
bool trust_sender; /* --trust-sender */
bool allow_unauthenticated; /* --allow-unauthenticated */
/* --no-super: operator veto forcing SUPER_MODE_OFF for every connection, so
* the receiver never attempts super-user activities (ownership application,
* device-node creation) even when running as root. Applies to --stdio and
* --daemon alike; also makes the server refuse any client --copy-as. */
bool no_super; /* --no-super */
/* --allow-super: locally-launched standalone TCP listener opt-in that keeps
* the historical permissive behavior for a PRIVILEGED (root) receiver.
* Without it a root standalone server forces SUPER_MODE_OFF, so a client
* --devices / --write-devices / --super / ownership request cannot make it
* create device nodes, write raw devices, or apply client-chosen ownership.
* It is rejected for --stdio: that path's remote argv is composed by the
* client (directly and via --remote-option), so it must never opt a root
* receiver back into super mode. Non-root receivers are unaffected (the
* kernel refuses the confined attempts). The daemon path instead uses the
* per-module `client owner = yes` opt-in. */
bool allow_super; /* --allow-super */
/* --iconv=CONVERT_SPEC: the server's own LOCAL charset declaration. The
* client's full spec rides the wire config frame anyway; when the server is
* started with its own --iconv, its LOCAL half overrides the local charset
* the client assumed so the server converts received names to ITS charset.
* Borrowed pointer into argv (never owns heap). */
const char* iconv_spec; /* --iconv value, or NULL */
} ServerCliOptions;
/* Parse argc/argv into *opts. Zero-initialize *opts before calling (or use
* server_cli_options_default). Returns:
* 1 -- --help was requested (opts->show_help set; caller prints usage).
* 0 -- parsed successfully.
* -1 -- invalid arguments (err is filled with the reason).
*/
void server_cli_options_default(ServerCliOptions* opts);
int server_cli_parse(int argc, char* argv[], ServerCliOptions* opts, char* err, size_t err_size);
/* Release the only heap the parsed options own (the dparams pointer array; the
* strings it points at are borrowed from argv and are not freed). Safe to
* call on a zero-initialized/defaulted struct. */
void server_cli_options_free(ServerCliOptions* opts);
#endif
+4 -8
View File
@@ -1,19 +1,17 @@
#include "log.h" #include "log.h"
#include "array_list.h" #include "array_list.h"
#include "protocol.h"
#include <limits.h>
#include <stdio.h> #include <stdio.h>
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
ArrayList* array_list_create(void (*item_destroyer)(void* item)) { ArrayList* array_list_create(void (*item_destroyer)(void* item)) {
ArrayList* list = (ArrayList*)protocol_alloc(sizeof(ArrayList)); ArrayList* list = (ArrayList*)malloc(sizeof(ArrayList));
if (list == NULL) { if (list == NULL) {
log_perror("ERROR: Could not allocate memory for array list struct"); log_perror("ERROR: Could not allocate memory for array list struct");
return NULL; return NULL;
} }
list->items = protocol_alloc(INITIAL_ARRAY_SIZE * sizeof(void*)); list->items = malloc(INITIAL_ARRAY_SIZE * sizeof(void*));
if (list->items == NULL) { if (list->items == NULL) {
free(list); free(list);
return NULL; return NULL;
@@ -40,12 +38,10 @@ void array_list_delete(ArrayList* array_list) {
static bool array_list_extend(ArrayList* array_list) { static bool array_list_extend(ArrayList* array_list) {
if (array_list == NULL) if (array_list == NULL)
return false; return false;
if (array_list->capacity > INT_MAX / 2)
return false;
int new_capacity = array_list->capacity * 2; int new_capacity = array_list->capacity * 2;
if (new_capacity == 0) if (new_capacity == 0)
new_capacity = INITIAL_ARRAY_SIZE; new_capacity = INITIAL_ARRAY_SIZE;
void* new_items = protocol_realloc(array_list->items, new_capacity * sizeof(void*)); void* new_items = realloc(array_list->items, new_capacity * sizeof(void*));
if (new_items == NULL) { if (new_items == NULL) {
log_perror("ERROR: Could not reallocate memory for array list items"); log_perror("ERROR: Could not reallocate memory for array list items");
return false; return false;
@@ -71,7 +67,7 @@ void** array_list_to_array(const ArrayList* array_list) {
if (array_list == NULL) { if (array_list == NULL) {
return NULL; return NULL;
} }
void** array = protocol_alloc(array_list->size * sizeof(void*)); void** array = malloc(array_list->size * sizeof(void*));
if (array == NULL) { if (array == NULL) {
log_perror("Could not malloc space for array from array list!"); log_perror("Could not malloc space for array from array list!");
return NULL; return NULL;
-160
View File
@@ -1,160 +0,0 @@
#include "batch.h"
#include "data.h"
#include "file.h"
#include "file_receive.h"
#include "log.h"
#include <errno.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h>
/* Serialization metadata mode for the batch stream, captured from the config at
* batch_write_header time. The header persists it into the file so a batch is
* self-describing: batch_read_apply re-reads it from the file (not from the
* reading config), so a batch written with -M is applied identically by an
* invoking process regardless of its own -M setting. The batch driver is a
* single sequential scan pass within one thread, so this module-level flag is
* safe. */
static bool batch_metadata_mode = false;
static bool write_all_bytes(int fd, const void* data, size_t size) {
const unsigned char* p = (const unsigned char*)data;
size_t done = 0;
while (done < size) {
ssize_t n = write(fd, p + done, size - done);
if (n < 0 && errno == EINTR)
continue;
if (n <= 0)
return false;
done += (size_t)n;
}
return true;
}
bool batch_write_header(int fd, const Config* config) {
if (fd < 0)
return false;
batch_metadata_mode = config != NULL && config->use_metadata;
if (!write_all_bytes(fd, BATCH_MAGIC, BATCH_MAGIC_LEN))
return false;
unsigned char version = BATCH_FORMAT_VERSION;
if (!write_all_bytes(fd, &version, 1))
return false;
unsigned char mode = batch_metadata_mode ? 1 : 0;
return write_all_bytes(fd, &mode, 1);
}
bool batch_write_chunk(int fd, Chunk* chunk) {
if (fd < 0 || chunk == NULL)
return false;
Data* serialized = chunk_serialize(chunk, batch_metadata_mode);
if (serialized == NULL)
return false;
bool ok = false;
unsigned long long length = (unsigned long long)serialized->size;
if (length > BATCH_MAX_RECORD) {
log_message(LOG_LEVEL_ERROR, "batch: record size %llu exceeds the %llu-byte cap", length,
(unsigned long long)BATCH_MAX_RECORD);
} else if (write_all_bytes(fd, &length, sizeof(length)) &&
(length == 0 || write_all_bytes(fd, serialized->data, (size_t)length))) {
ok = true;
}
data_destroy(serialized);
return ok;
}
/* Read exactly `size` bytes. Returns true on success. On reaching EOF, sets
* *clean_eof only when no bytes had been read yet (a clean boundary) and returns
* that value, so a truncated record (EOF mid-read) yields false. */
static bool read_exact(int fd, void* data, size_t size, bool* clean_eof) {
unsigned char* p = (unsigned char*)data;
size_t done = 0;
while (done < size) {
ssize_t n = read(fd, p + done, size - done);
if (n < 0 && errno == EINTR)
continue;
if (n == 0) {
if (clean_eof)
*clean_eof = done == 0;
return done == 0;
}
if (n < 0)
return false;
done += (size_t)n;
}
if (clean_eof)
*clean_eof = false;
return true;
}
int batch_read_apply(int fd, const Config* config, const char* dest_root) {
if (fd < 0 || dest_root == NULL || dest_root[0] == '\0')
return -1;
char magic[BATCH_MAGIC_LEN];
bool eof = false;
if (!read_exact(fd, magic, BATCH_MAGIC_LEN, &eof) || eof ||
memcmp(magic, BATCH_MAGIC, BATCH_MAGIC_LEN) != 0) {
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad magic)");
return -1;
}
unsigned char version;
if (!read_exact(fd, &version, 1, &eof) || eof || version != BATCH_FORMAT_VERSION) {
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad or missing format version)");
return -1;
}
unsigned char mode;
if (!read_exact(fd, &mode, 1, &eof) || eof || (mode != 0 && mode != 1)) {
log_message(LOG_LEVEL_ERROR, "batch: malformed header (bad metadata flag)");
return -1;
}
bool use_metadata = mode == 1;
while (1) {
unsigned long long length;
if (!read_exact(fd, &length, sizeof(length), &eof)) {
log_message(LOG_LEVEL_ERROR, "batch: truncated length prefix");
return -1;
}
if (eof)
break; /* clean end of stream */
if (length == 0 || length > BATCH_MAX_RECORD) {
log_message(LOG_LEVEL_ERROR, "batch: rejected record length %llu (valid range 1..%llu)",
length, (unsigned long long)BATCH_MAX_RECORD);
return -1;
}
char* record = (char*)malloc((size_t)length);
if (record == NULL) {
log_message(LOG_LEVEL_ERROR, "batch: could not allocate a %llu-byte record", length);
return -1;
}
if (!read_exact(fd, record, (size_t)length, &eof) || eof) {
log_message(LOG_LEVEL_ERROR, "batch: truncated chunk record");
free(record);
return -1;
}
Data* data = data_create(record, (size_t)length);
if (data == NULL)
return -1; /* data_create frees `record` on failure */
Chunk* chunk = chunk_deserialize(data, use_metadata);
data_destroy(data);
if (chunk == NULL) {
log_message(LOG_LEVEL_ERROR, "batch: rejected malformed chunk record");
return -1;
}
for (int i = 0; i < chunk->element_count; i++) {
File* file = chunk->items[i];
chunk->items[i] = NULL;
if (file == NULL)
continue;
FileSaveResult result = file_save_to_disk_full(dest_root, file, config);
file_destroy(file);
if (result == FILE_SAVE_ERROR) {
chunk_destroy(chunk);
return -1;
}
}
chunk_destroy(chunk);
}
return 0;
}
-28
View File
@@ -1,28 +0,0 @@
#ifndef BATCH_H
#define BATCH_H
#include "chunk.h"
#include "config.h"
/* Phase 6 residual-batch codec. A residual batch is a self-contained
* single-file record of a whole source tree: a magic+format-version header
* followed by length-prefixed chunk blobs (each built with chunk_serialize),
* byte-identical by construction. The batch is a client-only driver feature:
* it never crosses the wire, so there is no PROTOCOL_VERSION bump and no server
* change. */
#define BATCH_MAGIC "FSTRESBATCH"
#define BATCH_MAGIC_LEN 11
#define BATCH_FORMAT_VERSION 1
/* Max size of a single length-prefixed record (a whole serialized chunk,
* which can span several files). A single source file near the 64 MB wire
* limit plus per-file headers can produce a record slightly over 64 MB, so a
* large file just under the wire cap may be refused by the batch writer; this
* is documented upstream and the failure is clean (the partial batch is
* unlinked), never a truncated/corrupt batch. */
#define BATCH_MAX_RECORD (64ULL * 1024 * 1024)
bool batch_write_header(int fd, const Config* config);
bool batch_write_chunk(int fd, Chunk* chunk);
int batch_read_apply(int fd, const Config* config, const char* dest_root);
#endif
-384
View File
@@ -1,384 +0,0 @@
#include "charset.h"
#include "log.h"
#include "protocol.h"
#include "utils.h"
#include <errno.h>
#include <iconv.h>
#include <stdlib.h>
#include <string.h>
typedef struct {
iconv_t cd;
} CharsetConversion;
/* Process-wide wire conversion descriptor (one direction per process: a client
* only sends, a server only receives). CONCURRENCY CONTRACT: iconv_t is not
* guaranteed thread-safe, so every conversion MUST run on a single thread at a
* time. This holds today -- on the client the conversions run on the sender
* thread (in the -m pipeline chunk_serialize/send happen on the sender thread
* only), on the server on the receive-loop thread; the descriptor is
* initialized on one thread before any transfer thread spawns and torn down
* (charset_wire_free) only after all threads have joined. Do not add a
* concurrent conversion path (e.g. parallel chunk serialization) without
* guarding access with a mutex. */
static CharsetConversion* g_wire_conv;
/* Grow *buf to double capacity, freeing it on failure. realloc preserves the
* already-written prefix, so the caller only tracks its write offset. */
static bool grow_charset_buffer(char** buf, size_t* cap) {
size_t new_cap = *cap * 2;
if (new_cap <= *cap) {
free(*buf);
*buf = NULL;
return false;
}
char* grown = realloc(*buf, new_cap);
if (!grown) {
free(*buf);
*buf = NULL;
return false;
}
*buf = grown;
*cap = new_cap;
return true;
}
/* Throw away any pending shift state so a subsequent conversion starts clean.
* The flush output is discarded; for the stateless single-byte/UTF charsets
* this feature targets it is a no-op. */
static void charset_conversion_reset(const CharsetConversion* conv) {
char scratch[64];
char* sp = scratch;
size_t sl = sizeof(scratch);
(void)iconv(conv->cd, NULL, NULL, &sp, &sl);
}
int charset_spec_parse(const char* spec, char** local_out, char** remote_out) {
if (!local_out || !remote_out)
return -1;
*local_out = NULL;
*remote_out = NULL;
if (!spec || spec[0] == '\0')
return -1;
char* dup = str_dup(spec);
if (!dup)
return -1;
char* comma = strchr(dup, ',');
if (comma) {
if (comma == dup || comma[1] == '\0') {
free(dup);
return -1;
}
*comma = '\0';
*local_out = str_dup(dup);
*remote_out = str_dup(comma + 1);
free(dup);
} else {
*local_out = str_dup(dup);
*remote_out = str_dup(dup);
free(dup);
}
if (!*local_out || !*remote_out) {
free(*local_out);
free(*remote_out);
*local_out = NULL;
*remote_out = NULL;
return -1;
}
return 0;
}
void* charset_conversion_open(const char* from_charset, const char* to_charset) {
if (!from_charset || !to_charset)
return NULL;
iconv_t cd = iconv_open(to_charset, from_charset);
if (cd == (iconv_t)-1)
return NULL;
CharsetConversion* conv = malloc(sizeof(CharsetConversion));
if (!conv) {
iconv_close(cd);
return NULL;
}
conv->cd = cd;
return conv;
}
void charset_conversion_close(void* conversion) {
if (!conversion)
return;
CharsetConversion* conv = (CharsetConversion*)conversion;
iconv_close(conv->cd);
free(conv);
}
/* Probe a single conversion direction: the from/to charsets both open AND a
* representative ASCII name converts to a byte string containing no embedded
* NUL (so a target charset like UTF-16 that emits NUL bytes for ordinary ASCII
* names is rejected up front -- such an output would be silently truncated by
* the C-string wire helpers). */
static bool direction_probe_valid(const char* from, const char* to) {
if (!from || !to)
return false;
void* conv = charset_conversion_open(from, to);
if (!conv)
return false;
bool ok = true;
char input = 'a';
char* in_ptr = &input;
size_t in_left = 1;
char out_buf[64];
char* out_ptr = out_buf;
size_t out_left = sizeof(out_buf);
if (iconv(((CharsetConversion*)conv)->cd, &in_ptr, &in_left, &out_ptr, &out_left) == (size_t)-1)
ok = false;
char flush_buf[64];
char* flush_ptr = flush_buf;
size_t flush_left = sizeof(flush_buf);
if (ok &&
iconv(((CharsetConversion*)conv)->cd, NULL, NULL, &flush_ptr, &flush_left) == (size_t)-1)
ok = false;
size_t produced = (size_t)(out_ptr - out_buf);
if (ok && produced > 0 && memchr(out_buf, '\0', produced) != NULL)
ok = false;
charset_conversion_close(conv);
return ok;
}
bool charset_pair_valid(const char* local, const char* remote) {
/* Both ends convert in opposite directions with the same two charsets, so a
* valid spec must open (and be NUL-free) in BOTH directions: the sender
* opens local->remote, the receiver opens remote->local. */
return direction_probe_valid(local, remote) && direction_probe_valid(remote, local);
}
bool charset_spec_valid(const char* spec) {
if (!spec)
return true;
char* local;
char* remote;
if (charset_spec_parse(spec, &local, &remote) != 0)
return false;
bool ok = charset_pair_valid(local, remote);
free(local);
free(remote);
return ok;
}
bool charset_spec_valid_direction(const char* from_charset, const char* to_charset) {
return direction_probe_valid(from_charset, to_charset);
}
/* The receiver's real conversion is wire(client REMOTE) -> server-local (the
* server's own --iconv LOCAL half, or the client's LOCAL half when the server
* has no --iconv). A dedicated pre-ack check so an impossible direction is
* rejected before the connection instead of refusing mid-transfer. */
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec) {
if (!spec)
return true;
char* local;
char* remote;
if (charset_spec_parse(spec, &local, &remote) != 0)
return false;
const char* wire = remote;
const char* target_local = local;
char* server_local = NULL;
char* server_remote = NULL;
if (server_spec) {
if (charset_spec_parse(server_spec, &server_local, &server_remote) != 0) {
free(local);
free(remote);
return false;
}
target_local = server_local;
}
bool ok = charset_spec_valid_direction(wire, target_local);
free(server_local);
free(server_remote);
free(local);
free(remote);
return ok;
}
char* charset_convert(const void* conversion, const char* in, int* err_out) {
if (!conversion || !in)
return NULL;
const CharsetConversion* conv = (const CharsetConversion*)conversion;
size_t in_len = strlen(in);
size_t cap = in_len + 16;
char* out = malloc(cap);
if (!out)
return NULL;
size_t in_left = in_len;
char* in_ptr = (char*)in;
size_t out_used = 0;
while (in_left > 0) {
char* out_ptr = out + out_used;
size_t out_left = cap - out_used;
if (iconv(conv->cd, &in_ptr, &in_left, &out_ptr, &out_left) == (size_t)-1) {
if (errno != E2BIG) {
if (err_out)
*err_out = errno;
charset_conversion_reset(conv);
free(out);
return NULL;
}
/* Output exhausted but input remains. E2BIG does not roll the output
pointer back: the bytes iconv already emitted before the failure must
be preserved, so advance out_used before growing. */
out_used = (size_t)(out_ptr - out);
if (!grow_charset_buffer(&out, &cap))
return NULL;
continue;
}
out_used = (size_t)(out_ptr - out);
}
/* Flush any pending shift state (a no-op for the stateless single-byte and
UTF charsets this feature targets, but keeps the descriptor clean). */
for (;;) {
char* out_ptr = out + out_used;
size_t out_left = cap - out_used;
if (iconv(conv->cd, NULL, NULL, &out_ptr, &out_left) == (size_t)-1) {
if (errno != E2BIG) {
if (err_out)
*err_out = errno;
charset_conversion_reset(conv);
free(out);
return NULL;
}
out_used = (size_t)(out_ptr - out);
if (!grow_charset_buffer(&out, &cap))
return NULL;
continue;
}
out_used = (size_t)(out_ptr - out);
break;
}
/* A successful iconv call may legitimately consume the whole buffer (output
exactly fills cap), leaving no room for the terminator: guarantee headroom
before the final write. */
if (out_used >= cap && !grow_charset_buffer(&out, &cap))
return NULL;
/* Defense in depth: a target charset that emits embedded NUL bytes would
truncate at the first NUL in the C-string wire helpers; fail cleanly
(validation already rejects such charsets up front). */
if (memchr(out, '\0', out_used) != NULL) {
if (err_out)
*err_out = EILSEQ;
charset_conversion_reset(conv);
free(out);
return NULL;
}
out[out_used] = '\0';
return out;
}
bool charset_wire_init_sender(const char* spec) {
charset_wire_free();
if (!spec)
return true;
char* local;
char* remote;
if (charset_spec_parse(spec, &local, &remote) != 0)
return false;
void* conv = charset_conversion_open(local, remote);
free(local);
free(remote);
if (!conv)
return false;
g_wire_conv = (CharsetConversion*)conv;
return true;
}
bool charset_wire_init_receiver(const char* spec, const char* server_spec) {
charset_wire_free();
if (!spec)
return true;
char* local;
char* remote;
if (charset_spec_parse(spec, &local, &remote) != 0)
return false;
/* The wire charset is the client spec's REMOTE half; the local charset is
* the client spec's LOCAL half unless the server was itself started with
* --iconv naming a different local charset (the server halves above never
* travel, so the server's own flag is the only way its local charset can
* differ from what the client assumed). */
const char* wire = remote;
const char* target_local = local;
char* server_local = NULL;
char* server_remote = NULL;
if (server_spec) {
if (charset_spec_parse(server_spec, &server_local, &server_remote) != 0) {
free(local);
free(remote);
return false;
}
target_local = server_local;
}
void* conv = charset_conversion_open(wire, target_local);
free(server_local);
free(server_remote);
free(local);
free(remote);
if (!conv)
return false;
g_wire_conv = (CharsetConversion*)conv;
return true;
}
void charset_wire_free(void) {
if (g_wire_conv) {
charset_conversion_close(g_wire_conv);
g_wire_conv = NULL;
}
}
bool charset_wire_active(void) {
return g_wire_conv != NULL;
}
char* charset_wire_apply(const char* path) {
if (!g_wire_conv)
return str_dup(path);
return charset_convert(g_wire_conv, path, NULL);
}
static void charset_convert_failure_log(const char* path) {
char* escaped = output_escape(path, false);
log_message(LOG_LEVEL_ERROR, "--iconv: cannot convert file name '%s' to the target charset",
escaped ? escaped : "<unprintable>");
free(escaped);
}
bool send_wire_str(int file_descriptor, const char* local_path) {
if (!g_wire_conv)
return send_str(file_descriptor, local_path);
char* wire = charset_wire_apply(local_path);
if (!wire) {
charset_convert_failure_log(local_path);
return false;
}
bool ok = send_str(file_descriptor, wire);
free(wire);
return ok;
}
char* receive_wire_str(int file_descriptor) {
char* raw = receive_str(file_descriptor);
if (!raw)
return NULL;
if (!g_wire_conv)
return raw;
char* local = charset_convert(g_wire_conv, raw, NULL);
if (!local) {
charset_convert_failure_log(raw);
free(raw);
return NULL;
}
free(raw);
return local;
}
-85
View File
@@ -1,85 +0,0 @@
#ifndef CHARSET_H
#define CHARSET_H
#include <stdbool.h>
#include <stddef.h>
/* --iconv=CONVERT_SPEC file-name charset conversion (rsync compatibility).
*
* CONVERT_SPEC is "LOCAL[,REMOTE]": LOCAL is the charset of our own file
* names, REMOTE is the charset of the remote side's file names and defaults
* to LOCAL when the comma half is omitted. The sender converts every local
* path from LOCAL to REMOTE before it goes on the wire; the receiver converts
* every received path back from REMOTE to LOCAL. A NULL/disabled spec means
* identity with zero overhead (the common path never consults iconv).
*
* All helpers are friendly to the strict cold path: the wire conversion state
* is process-global (one direction per process -- a client only sends, a
* server only receives) and is initialized once, before any path is
* serialized, so conversion compiles to a single non-NULL check when disabled.
*/
/* Parse CONVERT_SPEC into malloc'd LOCAL and REMOTE charset names (caller
* frees both). REMOTE is a separate copy of LOCAL when no comma is present.
* Returns 0 on success, -1 on a malformed spec (empty halves / missing value /
* allocation failure); nothing is allocated on the -1 path. Both output
* pointers are REQUIRED (non-NULL). */
int charset_spec_parse(const char* spec, char** local_out, char** remote_out);
/* True when a CONVERT_SPEC is well-formed AND its charsets are usable for this
* feature: each pair opens in a probe iconv_open in BOTH directions (a sender
* converts local->remote, the receiver converts remote->local) and converting
* a representative ASCII name emits no embedded NUL byte (a UTF-16-style NUL
* emitter would be silently truncated by the C-string wire helpers). A typo'd
* charset name is therefore rejected at startup, not mid-run. NULL (iconv
* disabled) is always valid. */
bool charset_spec_valid(const char* spec);
/* Probe a concrete from->to conversion pair without keeping the descriptor:
* both charsets open AND a representative ASCII name converts with no embedded
* NUL. Used for direction-specific validation (e.g. the receiver's exact
* wire->local direction including a server-side charset override). */
bool charset_spec_valid_direction(const char* from_charset, const char* to_charset);
bool charset_pair_valid(const char* local, const char* remote);
/* One-shot conversion of a NUL-terminated input to a malloc'd NUL-terminated
* result, or NULL on failure. On failure *err_out (when non-NULL) receives
* the iconv errno (EILSEQ/EINVAL = the input is not representable in the
* target charset). The caller must free the result. */
char* charset_convert(const void* conversion, const char* in, int* err_out);
/* Open a conversion descriptor for direction from_charset -> to_charset.
* Returns NULL (errno = EINVAL) when a charset name is unsupported. Freed
* with charset_conversion_close. */
void* charset_conversion_open(const char* from_charset, const char* to_charset);
void charset_conversion_close(void* conversion);
/* Process-wide wire conversion. charset_wire_init_sender (client side) opens
* LOCAL->REMOTE; charset_wire_init_receiver (server side) opens
* wire(REMOTE)->server-local. server_spec is the server's own --iconv, whose
* LOCAL half may override the local charset the client assumed; NULL reuses
* the client spec's LOCAL half. Both return false on an unsupported spec.
* The state is freed with charset_wire_free. */
bool charset_wire_init_sender(const char* spec);
bool charset_wire_init_receiver(const char* spec, const char* server_spec);
void charset_wire_free(void);
bool charset_wire_active(void);
/* Pre-ack receiver-direction sanity (see charset_wire_init_receiver): true
* when the exact wire->server-local conversion the receiver will use (client
* spec's REMOTE half into the server's own LOCAL half, or the client's LOCAL
* half when the server has no --iconv) opens and produces NUL-free output. */
bool charset_wire_receiver_spec_valid(const char* spec, const char* server_spec);
/* Convert a path across the wire in the process direction. Returns a malloc'd
* string, or NULL when the name cannot be represented in the target charset. */
char* charset_wire_apply(const char* path);
/* Convenience wire string I/O: encode+send_str / receive_str+decode. Both
* return false/NULL (logging a clear --iconv error) on conversion failure, so
* an unconvertible path FAILS the transfer cleanly instead of silently sending
* a mangled name. */
bool send_wire_str(int file_descriptor, const char* local_path);
char* receive_wire_str(int file_descriptor);
#endif
-75
View File
@@ -1,75 +0,0 @@
#include "checksum.h"
#include <openssl/evp.h>
#include <string.h>
#include <strings.h>
/* delta.c owns the single XXH_IMPLEMENTATION that provides the xxHash symbols
* for the whole binary; this TU only needs the declarations. */
#include <xxhash.h>
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
size_t out_capacity, size_t* out_len) {
if (!out || !out_len || out_capacity < CHECKSUM_MAX_DIGEST_LEN)
return false;
if (data == NULL && size != 0)
return false;
if (algo == CHECKSUM_ALGO_XXH64) {
uint64_t digest = XXH64(data, size, seed);
memcpy(out, &digest, sizeof(digest));
*out_len = sizeof(digest);
return true;
}
if (algo == CHECKSUM_ALGO_MD5) {
/* md5 takes no seed; the caller's seed is deliberately ignored (documented
* in RSYNC_COMPAT.md). OpenSSL's one-shot EVP_Digest needs a non-NULL
* buffer even for an empty input, so map a NULL data + size==0 to an empty
* buffer. */
static const uint8_t empty = 0;
const void* input = data ? data : &empty;
unsigned int digest_len = 0;
if (EVP_Digest(input, size, out, &digest_len, EVP_md5(), NULL) != 1)
return false;
if (digest_len > out_capacity)
return false;
*out_len = digest_len;
return true;
}
return false;
}
int checksum_algo_from_name(const char* name) {
if (!name)
return -1;
if (strcasecmp(name, "xxh64") == 0 || strcasecmp(name, "xxhash") == 0)
return (int)CHECKSUM_ALGO_XXH64;
if (strcasecmp(name, "md5") == 0)
return (int)CHECKSUM_ALGO_MD5;
return -1;
}
const char* checksum_algo_name(ChecksumAlgo algo) {
switch (algo) {
case CHECKSUM_ALGO_XXH64:
return "xxh64";
case CHECKSUM_ALGO_MD5:
return "md5";
}
return "<unknown>";
}
bool checksum_algo_valid(int algo) {
return algo == (int)CHECKSUM_ALGO_XXH64 || algo == (int)CHECKSUM_ALGO_MD5;
}
uint8_t checksum_digest_len(ChecksumAlgo algo) {
switch (algo) {
case CHECKSUM_ALGO_XXH64:
return 8;
case CHECKSUM_ALGO_MD5:
return 16;
}
return 0;
}
-45
View File
@@ -1,45 +0,0 @@
#ifndef CHECKSUM_H
#define CHECKSUM_H
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
/* Whole-file content-digest algorithms selectable with --checksum-choice and
* seeded with --checksum-seed. The ids are the values actually placed on the
* wire (config frame), so they must be kept stable and validated on receive.
* CHECKSUM_ALGO_XXH64 == 0 is the default and is byte-for-byte what FastSync
* computed before these options existed (xxHash64 with seed 0). */
typedef enum { CHECKSUM_ALGO_XXH64 = 0, CHECKSUM_ALGO_MD5 = 1 } ChecksumAlgo;
/* md5 digest is 16 bytes, the longest supported. */
#define CHECKSUM_MAX_DIGEST_LEN 16
/* Compute the whole-file digest of the first `size` bytes of `data`.
*
* - CHECKSUM_ALGO_XXH64: xxHash64(data, size, seed) (full 64-bit seed).
* - CHECKSUM_ALGO_MD5: md5(data, size) via OpenSSL EVP.
* md5 has no seed, so `seed` is ignored (documented).
* - `size == 0` hashes the empty input (plus its seed), not a NULL input.
*
* Writes up to `out_capacity` bytes into `out`, storing the digest length in
* *out_len. Returns false on NULL out* or when the digest would not fit.
* Never writes more than CHECKSUM_MAX_DIGEST_LEN bytes. */
bool checksum_digest(ChecksumAlgo algo, uint64_t seed, const void* data, size_t size, uint8_t* out,
size_t out_capacity, size_t* out_len);
/* Resolve a --checksum-choice string (case-insensitive) to an algorithm id.
* Accepts "xxh64" and "xxhash" (both map to CHECKSUM_ALGO_XXH64, rsync's
* xxhash spelling) and "md5". Returns -1 for any unsupported name. */
int checksum_algo_from_name(const char* name);
/* Canonical name of an algorithm (used in CLI error messages). */
const char* checksum_algo_name(ChecksumAlgo algo);
/* True when `algo` is a supported id (used by config receive validation). */
bool checksum_algo_valid(int algo);
/* Digest length in bytes for an algorithm (xxx64 = 8, md5 = 16). */
uint8_t checksum_digest_len(ChecksumAlgo algo);
#endif /* CHECKSUM_H */
-90
View File
@@ -1,90 +0,0 @@
#include "chmod.h"
#include <stddef.h>
#include <string.h>
static bool parse_clause(mode_t* mode, const char* begin, const char* end) {
const char* p = begin;
unsigned who = 0;
while (p < end && strchr("ugoa", *p)) {
if (*p == 'a')
who = 7;
else
who |= *p == 'u' ? 1U : (*p == 'g' ? 2U : 4U);
p++;
}
if (who == 0)
who = 7;
if (p == end || (*p != '+' && *p != '-' && *p != '='))
return false;
char operation = *p++;
mode_t bits = 0;
while (p < end) {
mode_t bit;
switch (*p++) {
case 'r':
bit = 4;
break;
case 'w':
bit = 2;
break;
case 'x':
bit = 1;
break;
default:
return false;
}
bits |= bit;
}
for (unsigned class_index = 0; class_index < 3; class_index++) {
unsigned class_bit = 1U << class_index;
if (!(who & class_bit))
continue;
mode_t shift = (mode_t)((2U - class_index) * 3U);
mode_t mask = (mode_t)(7U << shift);
mode_t class_bits = (mode_t)(bits << shift);
if (operation == '+')
*mode |= class_bits;
else if (operation == '-')
*mode &= ~class_bits;
else
*mode = (*mode & ~mask) | class_bits;
}
return true;
}
bool chmod_apply(mode_t mode, const char* spec, mode_t* result) {
if (!spec || !*spec || !result)
return false;
bool numeric = true;
size_t length = strlen(spec);
if (length > 4)
numeric = false;
for (size_t i = 0; i < length && numeric; i++)
numeric = spec[i] >= '0' && spec[i] <= '7';
if (numeric) {
if (length == 0 || length > 4)
return false;
mode_t parsed = 0;
for (size_t i = 0; i < length; i++)
parsed = (mode_t)((parsed << 3) | (spec[i] - '0'));
*result = parsed;
return true;
}
mode_t changed = mode;
const char* begin = spec;
while (*begin) {
const char* end = strchr(begin, ',');
if (!end)
end = begin + strlen(begin);
if (!parse_clause(&changed, begin, end))
return false;
if (*end == '\0')
break;
begin = end + 1;
if (!*begin)
return false;
}
*result = changed;
return true;
}
-10
View File
@@ -1,10 +0,0 @@
#ifndef CHMOD_H
#define CHMOD_H
#include <stdbool.h>
#include <sys/stat.h>
/* Apply the supported rsync --chmod syntax to a permission mode. */
bool chmod_apply(mode_t mode, const char* spec, mode_t* result);
#endif
+77 -279
View File
@@ -1,13 +1,11 @@
#include <stddef.h> #include <stddef.h>
#include <stdint.h> #include <stdint.h>
#include <limits.h> #include <limits.h>
#include <stdatomic.h>
#include <stdio.h> #include <stdio.h>
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
#include "array_list.h" #include "array_list.h"
#include "charset.h"
#include "chunk.h" #include "chunk.h"
#include "compression.h" #include "compression.h"
#include "data.h" #include "data.h"
@@ -21,37 +19,10 @@
#define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024) #define MAX_FILE_DATA_SIZE (64ULL * 1024 * 1024)
#define MAX_FILES_PER_CHUNK 65536U #define MAX_FILES_PER_CHUNK 65536U
/* Reserve `charge` against `session`'s connection budget. This mirrors the
static protocol_reserve_memory() in protocol.c: the receive-side call sites
only have the Data.owner pointer (a ProtocolSession*), and protocol.c is out
of scope for this fix, so the same atomic CAS accounting is reproduced here.
The matching release always goes through data_destroy()'s Data.owner path. */
static bool chunk_session_reserve(ProtocolSession* session, size_t charge) {
unsigned long long allocated = atomic_load(&session->total_allocated_bytes);
while (true) {
if (allocated > MAX_CONNECTION_MEMORY ||
(unsigned long long)charge > MAX_CONNECTION_MEMORY - allocated)
return false;
if (atomic_compare_exchange_weak(&session->total_allocated_bytes, &allocated,
allocated + (unsigned long long)charge))
return true;
}
}
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge) {
if (!data || charge == 0 || session == NULL)
return true;
if (!chunk_session_reserve(session, charge))
return false;
data->owner = session;
data->protocol_charge = charge;
return true;
}
Chunk* chunk_create(File** items, int element_count) { Chunk* chunk_create(File** items, int element_count) {
if (element_count < 0 || (element_count > 0 && items == NULL)) if (element_count < 0 || (element_count > 0 && items == NULL))
return NULL; return NULL;
Chunk* chunk = (Chunk*)protocol_alloc(sizeof(Chunk)); Chunk* chunk = (Chunk*)malloc(sizeof(Chunk));
if (chunk == NULL) { if (chunk == NULL) {
log_perror("ERROR: Could not allocate memory for chunk structure"); log_perror("ERROR: Could not allocate memory for chunk structure");
return NULL; return NULL;
@@ -64,7 +35,7 @@ Chunk* chunk_create(File** items, int element_count) {
free(chunk); free(chunk);
return NULL; return NULL;
} }
chunk->items = (File**)protocol_alloc((size_t)element_count * sizeof(File*)); chunk->items = (File**)malloc((size_t)element_count * sizeof(File*));
if (chunk->items == NULL) { if (chunk->items == NULL) {
free(chunk); free(chunk);
return NULL; return NULL;
@@ -92,22 +63,9 @@ void chunk_destroy(void* item) {
free(chunk); free(chunk);
} }
/* --iconv: a chunk blob carries wire-charset path/target bytes. Encode the
* sender-side path (a no-op copy when iconv is disabled) so the blob is in the
* same charset as every other wire string. */
static char* chunk_encode_wire(const char* path) {
if (!charset_wire_active())
return str_dup(path);
return charset_wire_apply(path);
}
static unsigned long long per_file_serialize_size(File* file, bool use_metadata) { static unsigned long long per_file_serialize_size(File* file, bool use_metadata) {
unsigned long long size = sizeof(size_t); unsigned long long size = sizeof(size_t);
char* wire_path = chunk_encode_wire(file_wire_path(file)); size_t path_len = strlen(file->path);
if (!wire_path)
return 0;
size_t path_len = strlen(wire_path);
free(wire_path);
unsigned long long metadata_size = unsigned long long metadata_size =
use_metadata ? sizeof(int) + (file->metadata ? FILE_METADATA_WIRE_SIZE : 0) : 0; use_metadata ? sizeof(int) + (file->metadata ? FILE_METADATA_WIRE_SIZE : 0) : 0;
if ((unsigned long long)path_len > ULLONG_MAX - size) if ((unsigned long long)path_len > ULLONG_MAX - size)
@@ -116,39 +74,12 @@ static unsigned long long per_file_serialize_size(File* file, bool use_metadata)
if (metadata_size > ULLONG_MAX - size) if (metadata_size > ULLONG_MAX - size)
return 0; return 0;
size += metadata_size; size += metadata_size;
/* Entry type marker: 0 = regular file, 1 = explicit directory entry,
2 = symlink entry (carries its target string), 3 = special/device node
(recreated by the receiver). */
if (sizeof(int) > ULLONG_MAX - size)
return 0;
size += sizeof(int);
/* A special node also carries its rdev major/minor. */
if (file->is_special) {
if (2 * sizeof(int32_t) > ULLONG_MAX - size)
return 0;
size += 2 * sizeof(int32_t);
}
if (sizeof(size_t) > ULLONG_MAX - size) if (sizeof(size_t) > ULLONG_MAX - size)
return 0; return 0;
size += sizeof(size_t); size += sizeof(size_t);
if ((unsigned long long)file->data->size > ULLONG_MAX - size) if ((unsigned long long)file->data->size > ULLONG_MAX - size)
return 0; return 0;
size += file->data->size; return size + file->data->size;
/* Symlink entries append the target string (length-prefixed). */
if (file->is_symlink) {
char* wire_target = chunk_encode_wire(file->symlink_target ? file->symlink_target : "");
if (!wire_target)
return 0;
size_t target_len = strlen(wire_target);
free(wire_target);
if (sizeof(size_t) > ULLONG_MAX - size)
return 0;
size += sizeof(size_t);
if ((unsigned long long)target_len > ULLONG_MAX - size)
return 0;
size += target_len;
}
return size;
} }
Data* chunk_serialize(Chunk* chunk, bool use_metadata) { Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
@@ -158,8 +89,7 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
for (int i = 0; i < chunk->element_count; i++) { for (int i = 0; i < chunk->element_count; i++) {
if (!chunk->items[i] || !chunk->items[i]->path || !chunk->items[i]->data || if (!chunk->items[i] || !chunk->items[i]->path || !chunk->items[i]->data ||
(chunk->items[i]->data->size > 0 && !chunk->items[i]->data->data) || (chunk->items[i]->data->size > 0 && !chunk->items[i]->data->data) ||
chunk->items[i]->path[0] == '\0' || has_path_traversal(chunk->items[i]->path) || chunk->items[i]->path[0] == '\0' || has_path_traversal(chunk->items[i]->path))
(file_wire_path(chunk->items[i]))[0] == '\0')
return NULL; return NULL;
unsigned long long file_size = per_file_serialize_size(chunk->items[i], use_metadata); unsigned long long file_size = per_file_serialize_size(chunk->items[i], use_metadata);
if (file_size == 0 || file_size > ULLONG_MAX - data_size || data_size + file_size > SIZE_MAX) if (file_size == 0 || file_size > ULLONG_MAX - data_size || data_size + file_size > SIZE_MAX)
@@ -174,30 +104,11 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
char* data_pointer = data->data; char* data_pointer = data->data;
for (int i = 0; i < chunk->element_count; i++) { for (int i = 0; i < chunk->element_count; i++) {
File* file = chunk->items[i]; File* file = chunk->items[i];
char* wire_path = chunk_encode_wire(file_wire_path(file)); size_t path_len = strlen(file->path);
if (wire_path == NULL) {
data_destroy(data);
return NULL;
}
size_t path_len = strlen(wire_path);
memcpy(data_pointer, &path_len, sizeof(size_t)); memcpy(data_pointer, &path_len, sizeof(size_t));
data_pointer += sizeof(size_t); data_pointer += sizeof(size_t);
memcpy(data_pointer, wire_path, path_len); memcpy(data_pointer, file->path, path_len);
data_pointer += path_len; data_pointer += path_len;
free(wire_path);
int entry_type = file->is_dir ? 1 : (file->is_symlink ? 2 : (file->is_special ? 3 : 0));
memcpy(data_pointer, &entry_type, sizeof(int));
data_pointer += sizeof(int);
if (file->is_special) {
int32_t special_major = file->rdev_major;
int32_t special_minor = file->rdev_minor;
memcpy(data_pointer, &special_major, sizeof(special_major));
data_pointer += sizeof(special_major);
memcpy(data_pointer, &special_minor, sizeof(special_minor));
data_pointer += sizeof(special_minor);
}
if (use_metadata) if (use_metadata)
metadata_to_buf(&data_pointer, file->metadata); metadata_to_buf(&data_pointer, file->metadata);
@@ -205,24 +116,8 @@ Data* chunk_serialize(Chunk* chunk, bool use_metadata) {
size_t file_data_size = file->data->size; size_t file_data_size = file->data->size;
memcpy(data_pointer, &file_data_size, sizeof(size_t)); memcpy(data_pointer, &file_data_size, sizeof(size_t));
data_pointer += sizeof(size_t); data_pointer += sizeof(size_t);
if (file_data_size > 0) memcpy(data_pointer, file->data->data, file_data_size);
memcpy(data_pointer, file->data->data, file_data_size);
data_pointer += file_data_size; data_pointer += file_data_size;
if (file->is_symlink) {
char* wire_target = chunk_encode_wire(file->symlink_target ? file->symlink_target : "");
if (wire_target == NULL) {
data_destroy(data);
return NULL;
}
size_t target_len = strlen(wire_target);
memcpy(data_pointer, &target_len, sizeof(size_t));
data_pointer += sizeof(size_t);
if (target_len > 0)
memcpy(data_pointer, wire_target, target_len);
data_pointer += target_len;
free(wire_target);
}
} }
return data; return data;
} }
@@ -235,20 +130,17 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
return NULL; return NULL;
char* data_pointer = data->data; char* data_pointer = data->data;
size_t remaining_size = data->size; size_t remaining_size = data->size;
/* The element currently being parsed is owned by `files` only after the
* array_list_add() at the end of the iteration; until then the error
* epilogue destroys it directly. Keeping this one pointer nulled after the
* hand-off makes the single cleanup path correct for every failure. */
File* file = NULL;
while (remaining_size > 0) { while (remaining_size > 0) {
if ((unsigned int)files->size >= MAX_FILES_PER_CHUNK) { if ((unsigned int)files->size >= MAX_FILES_PER_CHUNK) {
log_message(LOG_LEVEL_ERROR, "Chunk contains too many files"); log_message(LOG_LEVEL_ERROR, "Chunk contains too many files");
goto error; array_list_delete(files);
return NULL;
} }
if (remaining_size < sizeof(size_t)) { if (remaining_size < sizeof(size_t)) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path length"); log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path length");
goto error; array_list_delete(files);
return NULL;
} }
size_t path_len; size_t path_len;
@@ -258,117 +150,77 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
if (path_len > SIZE_MAX - 1 || remaining_size < path_len) { if (path_len > SIZE_MAX - 1 || remaining_size < path_len) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path"); log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for path");
goto error; array_list_delete(files);
return NULL;
} }
char* path = protocol_alloc(path_len + 1); if (path_len == SIZE_MAX) {
array_list_delete(files);
return NULL;
}
char* path = malloc(path_len + 1);
if (path == NULL) { if (path == NULL) {
log_perror("Could not allocate memory for file path"); log_perror("Could not allocate memory for file path");
goto error; array_list_delete(files);
return NULL;
} }
memcpy(path, data_pointer, path_len); memcpy(path, data_pointer, path_len);
path[path_len] = '\0'; path[path_len] = '\0';
if (memchr(path, '\0', path_len) != NULL) { if (memchr(path, '\0', path_len) != NULL) {
free(path); free(path);
goto error; array_list_delete(files);
return NULL;
} }
data_pointer += path_len; data_pointer += path_len;
remaining_size -= path_len; remaining_size -= path_len;
/* --iconv: the blob holds the wire charset; translate it to the receiver's
local charset before validation and creation so the destination gets the
local name. A name that cannot be decoded fails the file cleanly. */
if (charset_wire_active()) {
char* local_path = charset_wire_apply(path);
free(path);
if (local_path == NULL) {
log_message(LOG_LEVEL_ERROR,
"--iconv: received chunk file name cannot be converted to the local charset");
goto error;
}
path = local_path;
path_len = strlen(path);
}
if (path_len == 0 || has_path_traversal(path)) { if (path_len == 0 || has_path_traversal(path)) {
free(path); free(path);
goto error; array_list_delete(files);
return NULL;
} }
file = file_create(path); File* file = file_create(path);
free(path); free(path);
if (file == NULL) if (file == NULL) {
goto error; array_list_delete(files);
return NULL;
if (remaining_size < sizeof(int)) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for entry type");
goto error;
}
int entry_type;
memcpy(&entry_type, data_pointer, sizeof(int));
if (entry_type != 0 && entry_type != 1 && entry_type != 2 && entry_type != 3) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad entry type");
goto error;
}
file->is_dir = entry_type == 1;
file->is_symlink = entry_type == 2;
file->is_special = entry_type == 3;
data_pointer += sizeof(int);
remaining_size -= sizeof(int);
if (file->is_special) {
if (remaining_size < 2 * (int32_t)sizeof(int32_t)) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for special rdev");
goto error;
}
int32_t special_major, special_minor;
memcpy(&special_major, data_pointer, sizeof(special_major));
data_pointer += sizeof(special_major);
memcpy(&special_minor, data_pointer, sizeof(special_minor));
data_pointer += sizeof(special_minor);
remaining_size -= 2 * sizeof(int32_t);
/* Reject an out-of-range/negative rdev here as a malformed chunk (the
same 0xffff / 0x00ffffff bounds file_special_rdev_valid uses), so a
bogus large-but-positive rdev is refused cleanly instead of being
deferred to the creation site where it would abort after the frame. */
if (special_major < 0 || special_minor < 0 || special_major > 0xffff ||
special_minor > 0x00ffffff) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: out-of-range special rdev");
goto error;
}
file->rdev_major = special_major;
file->rdev_minor = special_minor;
} }
if (use_metadata) { if (use_metadata) {
if (remaining_size < sizeof(int)) { if (remaining_size < sizeof(int)) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata"); log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata");
goto error; file_destroy(file);
array_list_delete(files);
return NULL;
} }
/* Peek at the present flag to determine the total record size before // Peek at present flag to determine total size needed before reading
decoding. metadata_from_buf() independently bounds-checks every read
against remaining_size, so a short body can never over-read. */
int present_flag; int present_flag;
memcpy(&present_flag, data_pointer, sizeof(int)); memcpy(&present_flag, data_pointer, sizeof(int));
if ((present_flag != 0 && present_flag != 1) || if ((present_flag != 0 && present_flag != 1) ||
(present_flag == 1 && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE)) { (present_flag == 1 && remaining_size < sizeof(int) + FILE_METADATA_WIRE_SIZE)) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata body"); log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for metadata body");
goto error; file_destroy(file);
array_list_delete(files);
return NULL;
} }
file->metadata = metadata_from_buf((const uint8_t*)data_pointer, remaining_size); file->metadata = metadata_from_buf(&data_pointer);
size_t metadata_consumed = sizeof(int); remaining_size -= sizeof(int);
if (present_flag == 1) { if (present_flag == 1) {
if (file->metadata == NULL) if (file->metadata == NULL) {
goto error; file_destroy(file);
metadata_consumed += FILE_METADATA_WIRE_SIZE; array_list_delete(files);
return NULL;
}
remaining_size -= FILE_METADATA_WIRE_SIZE;
} }
data_pointer += metadata_consumed;
remaining_size -= metadata_consumed;
} }
if (remaining_size < sizeof(size_t)) { if (remaining_size < sizeof(size_t)) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for data size"); log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for data size");
goto error; file_destroy(file);
array_list_delete(files);
return NULL;
} }
size_t file_data_size; size_t file_data_size;
@@ -378,120 +230,75 @@ Chunk* chunk_deserialize(Data* data, bool use_metadata) {
if (remaining_size < file_data_size) { if (remaining_size < file_data_size) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for file content"); log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for file content");
goto error; file_destroy(file);
array_list_delete(files);
return NULL;
} }
// Reject individual file data larger than the maximum allowed size. // Reject individual file data larger than the maximum allowed size.
if (file_data_size > MAX_FILE_DATA_SIZE) { if (file_data_size > MAX_FILE_DATA_SIZE) {
log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size, log_message(LOG_LEVEL_ERROR, "File data size %zu exceeds maximum %llu", file_data_size,
(unsigned long long)MAX_FILE_DATA_SIZE); (unsigned long long)MAX_FILE_DATA_SIZE);
goto error; file_destroy(file);
array_list_delete(files);
return NULL;
} }
size_t allocation_size = file_data_size > 0 ? file_data_size : 1; size_t allocation_size = file_data_size > 0 ? file_data_size : 1;
void* file_data = protocol_alloc(allocation_size); void* file_data = malloc(allocation_size);
if (file_data == NULL) { if (file_data == NULL) {
log_perror("Could not allocate memory for file data"); log_perror("Could not allocate memory for file data");
goto error; file_destroy(file);
array_list_delete(files);
return NULL;
} }
memcpy(file_data, data_pointer, file_data_size); memcpy(file_data, data_pointer, file_data_size);
Data* replacement = data_create(file_data, file_data_size); Data* replacement = data_create(file_data, file_data_size);
if (replacement == NULL) if (replacement == NULL) {
goto error; file_destroy(file);
/* Charge the retained per-file copy to the connection budget (when the array_list_delete(files);
inbound chunk carries an owning session) so the queued copies are not return NULL;
held outside MAX_CONNECTION_MEMORY (B6). A NULL owner (e.g. a local
batch apply) leaves the copy uncharged. */
if (!data_charge_session(replacement, data->owner, allocation_size)) {
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for chunk file data");
data_destroy(replacement);
goto error;
} }
data_destroy(file->data); data_destroy(file->data);
file->data = replacement; file->data = replacement;
data_pointer += file_data_size; data_pointer += file_data_size;
remaining_size -= file_data_size; remaining_size -= file_data_size;
if (file->is_symlink) { if (!array_list_add(files, file)) {
if (remaining_size < sizeof(size_t)) { file_destroy(file);
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: not enough data for symlink target"); array_list_delete(files);
goto error; return NULL;
}
size_t target_len;
memcpy(&target_len, data_pointer, sizeof(size_t));
data_pointer += sizeof(size_t);
remaining_size -= sizeof(size_t);
if (target_len == 0 || remaining_size < target_len) {
log_message(LOG_LEVEL_ERROR, "Invalid chunk format: bad symlink target");
goto error;
}
char* target = protocol_alloc(target_len + 1);
if (!target) {
log_perror("Could not allocate memory for symlink target");
goto error;
}
memcpy(target, data_pointer, target_len);
target[target_len] = '\0';
if (memchr(target, '\0', target_len) != NULL) {
free(target);
goto error;
}
/* The symlink target also rides the wire charset; decode it to the local
charset like the path (a target is a path). */
if (charset_wire_active()) {
char* local_target = charset_wire_apply(target);
free(target);
if (local_target == NULL) {
log_message(LOG_LEVEL_ERROR,
"--iconv: received chunk symlink target cannot be converted to the local "
"charset");
goto error;
}
target = local_target;
}
file->symlink_target = target;
data_pointer += target_len;
remaining_size -= target_len;
} }
if (!array_list_add(files, file))
goto error;
file = NULL;
} }
File** file_array = (File**)array_list_to_array(files); File** file_array = (File**)array_list_to_array(files);
if (files->size > 0 && file_array == NULL) if (files->size > 0 && file_array == NULL) {
goto error; array_list_delete(files);
return NULL;
}
Chunk* chunk = chunk_create(file_array, files->size); Chunk* chunk = chunk_create(file_array, files->size);
free(file_array); free(file_array);
if (chunk == NULL) if (chunk == NULL) {
goto error; array_list_delete(files);
return NULL;
}
files->item_destroyer = NULL; files->item_destroyer = NULL;
array_list_delete(files); array_list_delete(files);
return chunk;
error: return chunk;
if (file)
file_destroy(file);
array_list_delete(files);
return NULL;
} }
Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) { Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata) {
return chunk_compress_with_threads(chunk, compression_level, use_metadata, 0);
}
Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_metadata,
int compression_threads) {
log_message(LOG_LEVEL_DEBUG, "Starting to compress chunk"); log_message(LOG_LEVEL_DEBUG, "Starting to compress chunk");
Data* serialized = chunk_serialize(chunk, use_metadata); Data* serialized = chunk_serialize(chunk, use_metadata);
if (serialized == NULL) if (serialized == NULL)
return NULL; return NULL;
Data* compressed = data_compress_with_threads(serialized, compression_level, compression_threads); Data* compressed = data_compress(serialized, compression_level);
data_destroy(serialized); data_destroy(serialized);
if (compressed == NULL) if (compressed == NULL)
return NULL; return NULL;
log_debug_message(LOG_DEBUG_PACK, "Chunk successfully compressed"); log_message(LOG_LEVEL_DEBUG, "Chunk successfully compressed");
return compressed; return compressed;
} }
@@ -503,21 +310,12 @@ Chunk* receive_chunk_data(int fd, const Config* config) {
} }
Data* data_to_process = chunk_data; Data* data_to_process = chunk_data;
if (config->use_compression) { if (config->use_compression) {
/* Preserve the inbound session across decompression so the (larger)
decompressed chunk is charged to the same connection budget; the
compressed buffer's own charge is released by data_destroy below. */
ProtocolSession* owner = chunk_data->owner;
data_to_process = data_decompress_limited(chunk_data, MAX_CHUNK_SIZE); data_to_process = data_decompress_limited(chunk_data, MAX_CHUNK_SIZE);
data_destroy(chunk_data); data_destroy(chunk_data);
if (data_to_process == NULL) { if (data_to_process == NULL) {
log_message(LOG_LEVEL_ERROR, "Failed to decompress chunk"); log_message(LOG_LEVEL_ERROR, "Failed to decompress chunk");
return NULL; return NULL;
} }
if (!data_charge_session(data_to_process, owner, data_to_process->size)) {
log_message(LOG_LEVEL_ERROR, "Per-connection memory limit exceeded for decompressed chunk");
data_destroy(data_to_process);
return NULL;
}
} }
// Reject chunks larger than the maximum allowed size to prevent OOM. // Reject chunks larger than the maximum allowed size to prevent OOM.
-13
View File
@@ -19,19 +19,6 @@ void chunk_destroy(void* chunk);
Data* chunk_serialize(Chunk* chunk, bool use_metadata); Data* chunk_serialize(Chunk* chunk, bool use_metadata);
Chunk* chunk_deserialize(Data* data, bool use_metadata); Chunk* chunk_deserialize(Data* data, bool use_metadata);
Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata); Data* chunk_compress(Chunk* chunk, int compression_level, bool use_metadata);
Data* chunk_compress_with_threads(Chunk* chunk, int compression_level, bool use_metadata,
int compression_threads);
Chunk* receive_chunk_data(int fd, const Config* config); Chunk* receive_chunk_data(int fd, const Config* config);
/* Charge `charge` retained bytes of `data` against `session`'s per-connection
* budget (MAX_CONNECTION_MEMORY), mirroring the protocol layer's accounting, and
* record them on `data` so data_destroy() returns the charge through the
* Data.owner path. Returns false (leaving `data` uncharged) when the ceiling
* would be exceeded. A NULL/zero-size charge or a NULL session is a no-op
* success. The receive-side decompression and chunk-copy paths know the owning
* session only through the Data.owner of the buffer they are processing, so
* this is the entry point that lets them participate in the connection budget
* without a session handle (B6). */
bool data_charge_session(Data* data, ProtocolSession* session, size_t charge);
#endif #endif
+47 -242
View File
@@ -1,238 +1,73 @@
#include "compression.h" #include "compression.h"
#include "data.h" #include "data.h"
#include "log.h" #include "log.h"
#include "protocol.h" #include <stdlib.h>
#include <limits.h> #include <limits.h>
#include <stdint.h> #include <stdint.h>
#include <stdlib.h>
#include <string.h> #include <string.h>
#include <strings.h> #include <strings.h>
#include <threads.h>
#include <unistd.h>
#include <zstd.h> #include <zstd.h>
#define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024) #define INITIAL_DECOMPRESS_BUF_SIZE (1024 * 1024)
#define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */ #define MAX_DECOMPRESSED_SIZE (100ULL * 1024 * 1024) /* 100 MB hard ceiling */
static char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv", static const char* SKIP_COMPRESSION_EXTENSIONS[] = {".jpg", ".jpeg", ".png", ".gif", ".mp4", ".mkv",
".zip", ".gz", ".xz", ".zst", NULL}; ".zip", ".gz", ".xz", ".zst", NULL};
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count) { bool compression_should_skip(const char* path) {
if (!path) if (!path)
return false; return false;
const char* dot = strrchr(path, '.'); const char* dot = strrchr(path, '.');
if (!dot) if (!dot)
return false; return false;
if (count < 0) { for (int i = 0; SKIP_COMPRESSION_EXTENSIONS[i]; i++) {
suffixes = SKIP_COMPRESSION_EXTENSIONS; if (strcasecmp(dot, SKIP_COMPRESSION_EXTENSIONS[i]) == 0)
count = 0;
while (SKIP_COMPRESSION_EXTENSIONS[count])
count++;
}
for (int i = 0; i < count; i++) {
if (strcasecmp(dot, suffixes[i]) == 0)
return true; return true;
} }
return false; return false;
} }
/* Per-thread cache of zstd contexts plus the grow-only compression scratch
* buffer. zstd contexts are stateful and not safe to share between threads,
* so each thread keeps its own (see compression_get_thread_ctx). The cache is
* stored in a C11 thread-specific storage slot whose destructor releases the
* contexts when the thread exits; this keeps LeakSanitizer clean for the
* short-lived sender/receiver/scanner worker threads without every worker
* entry point having to remember to call compression_free_thread_contexts().
* The main thread's slot is not torn down by tss at process exit, so an atexit
* hook releases it (and compression_free_thread_contexts allows eager
* release). */
typedef struct {
ZSTD_CCtx* cctx;
ZSTD_DCtx* dctx;
void* out_buf; /* reusable ZSTD_compressBound-sized output scratch */
size_t out_cap; /* bytes currently allocated for out_buf */
int level; /* compression level currently applied to cctx */
int workers; /* nbWorkers currently applied to cctx */
bool params_set;
bool cached; /* false when the TSS slot could not be used: caller owns */
} CompressionThreadCtx;
static once_flag compression_tls_once = ONCE_FLAG_INIT;
static tss_t compression_tls_key;
static bool compression_tls_ready;
static void compression_tls_make_key(void);
static void compression_ctx_free(CompressionThreadCtx* ctx) {
if (!ctx)
return;
if (ctx->cctx)
ZSTD_freeCCtx(ctx->cctx);
if (ctx->dctx)
ZSTD_freeDCtx(ctx->dctx);
free(ctx->out_buf);
free(ctx);
}
static void compression_tls_destructor(void* value) {
compression_ctx_free((CompressionThreadCtx*)value);
}
void compression_free_thread_contexts(void) {
call_once(&compression_tls_once, compression_tls_make_key);
if (!compression_tls_ready)
return;
CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key);
if (!ctx)
return;
/* Clear the slot first so the thread-exit destructor cannot free it twice. */
tss_set(compression_tls_key, NULL);
compression_ctx_free(ctx);
}
static void compression_atexit_cleanup(void) {
compression_free_thread_contexts();
}
static void compression_tls_make_key(void) {
if (tss_create(&compression_tls_key, compression_tls_destructor) == thrd_success) {
compression_tls_ready = true;
atexit(compression_atexit_cleanup);
}
}
static CompressionThreadCtx* compression_get_thread_ctx(void) {
call_once(&compression_tls_once, compression_tls_make_key);
if (!compression_tls_ready) {
/* Extremely unlikely: fall back to an uncached context the caller frees. */
return (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx));
}
CompressionThreadCtx* ctx = (CompressionThreadCtx*)tss_get(compression_tls_key);
if (ctx)
return ctx;
ctx = (CompressionThreadCtx*)calloc(1, sizeof(CompressionThreadCtx));
if (!ctx)
return NULL;
ctx->cached = true;
if (tss_set(compression_tls_key, ctx) != thrd_success)
ctx->cached = false;
return ctx;
}
/* Release an uncached context immediately; cached contexts are owned by the
* thread's TSS slot and freed on thread exit / compression_free_thread_contexts. */
static void compression_ctx_put(CompressionThreadCtx* ctx) {
if (ctx && !ctx->cached)
compression_ctx_free(ctx);
}
Data* data_compress(Data* data_to_compress, int compression_level) { Data* data_compress(Data* data_to_compress, int compression_level) {
return data_compress_with_threads(data_to_compress, compression_level, 0);
}
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
int compression_threads) {
if (!data_to_compress || (!data_to_compress->data && data_to_compress->size != 0) ||
compression_threads < 0 || compression_threads > COMPRESSION_MAX_THREADS)
return NULL;
log_message(LOG_LEVEL_DEBUG, "Starting to compress data"); log_message(LOG_LEVEL_DEBUG, "Starting to compress data");
size_t dst_size = ZSTD_compressBound(data_to_compress->size); size_t dst_size = ZSTD_compressBound(data_to_compress->size);
Data* compressed_data = data_create_empty(dst_size);
if (compressed_data == NULL)
return NULL;
CompressionThreadCtx* ctx = compression_get_thread_ctx(); ZSTD_CCtx* cctx = ZSTD_createCCtx();
if (ctx == NULL) { if (!cctx) {
log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD compression context"); log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context");
data_destroy(compressed_data);
return NULL; return NULL;
} }
Data* compressed_data = NULL;
if (!ctx->cctx) { size_t zret = ZSTD_CCtx_setParameter(cctx, ZSTD_c_compressionLevel, compression_level);
ctx->cctx = ZSTD_createCCtx(); if (ZSTD_isError(zret)) {
if (!ctx->cctx) { log_message(LOG_LEVEL_ERROR, "Failed to set compression level: %s", ZSTD_getErrorName(zret));
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD compression context"); ZSTD_freeCCtx(cctx);
goto cleanup; data_destroy(compressed_data);
} return NULL;
ctx->params_set = false;
}
/* Reset only the session: parameters (and any already-allocated zstd worker
* pool) stay attached to the context, so compressing the next file does not
* rebuild the pool. */
ZSTD_CCtx_reset(ctx->cctx, ZSTD_reset_session_only);
if (!ctx->params_set || ctx->level != compression_level) {
size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_compressionLevel, compression_level);
if (ZSTD_isError(zret)) {
log_message(LOG_LEVEL_ERROR, "Failed to set compression level: %s", ZSTD_getErrorName(zret));
goto cleanup;
}
ctx->level = compression_level;
}
int available_threads = 0;
if (compression_threads > 0) {
long online_cpus = sysconf(_SC_NPROCESSORS_ONLN);
available_threads = online_cpus > 0 && online_cpus < compression_threads ? (int)online_cpus
: compression_threads;
}
if (!ctx->params_set || ctx->workers != available_threads) {
size_t zret = ZSTD_CCtx_setParameter(ctx->cctx, ZSTD_c_nbWorkers, available_threads);
if (ZSTD_isError(zret)) {
log_message(LOG_LEVEL_ERROR, "Failed to set compression threads: %s",
ZSTD_getErrorName(zret));
goto cleanup;
}
ctx->workers = available_threads;
}
ctx->params_set = true;
if (available_threads > 0) {
/* Streaming compression needs the source size before threaded mode can end a frame. */
size_t zret = ZSTD_CCtx_setPledgedSrcSize(ctx->cctx, data_to_compress->size);
if (ZSTD_isError(zret)) {
log_message(LOG_LEVEL_ERROR, "Failed to set compression source size: %s",
ZSTD_getErrorName(zret));
goto cleanup;
}
}
if (ctx->out_cap < dst_size) {
void* grown = protocol_realloc(ctx->out_buf, dst_size);
if (grown == NULL) {
log_message(LOG_LEVEL_ERROR, "Failed to allocate compression buffer");
goto cleanup;
}
ctx->out_buf = grown;
ctx->out_cap = dst_size;
} }
ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0}; ZSTD_inBuffer input = {data_to_compress->data, data_to_compress->size, 0};
ZSTD_outBuffer output = {ctx->out_buf, dst_size, 0}; ZSTD_outBuffer output = {compressed_data->data, dst_size, 0};
size_t ret; size_t ret;
do { do {
ret = ZSTD_compressStream2(ctx->cctx, &output, &input, ZSTD_e_end); ret = ZSTD_compressStream2(cctx, &output, &input, ZSTD_e_end);
if (ZSTD_isError(ret)) { if (ZSTD_isError(ret)) {
log_message(LOG_LEVEL_ERROR, "Compression failed: %s", ZSTD_getErrorName(ret)); log_message(LOG_LEVEL_ERROR, "Compression failed: %s", ZSTD_getErrorName(ret));
goto cleanup; ZSTD_freeCCtx(cctx);
data_destroy(compressed_data);
return NULL;
} }
} while (ret > 0); } while (ret > 0);
/* Hand off an exactly-sized copy; the scratch buffer stays cached so the next
* call does not reallocate a ZSTD_compressBound-sized block. */
compressed_data = data_create_empty(output.pos);
if (compressed_data == NULL) {
log_message(LOG_LEVEL_ERROR, "Failed to allocate compressed data");
goto cleanup;
}
if (output.pos > 0)
memcpy(compressed_data->data, ctx->out_buf, output.pos);
compressed_data->size = output.pos; compressed_data->size = output.pos;
ZSTD_freeCCtx(cctx);
log_debug_message(LOG_DEBUG_UTIL, "Data succesfully compressed from %zu to %zu", log_message(LOG_LEVEL_DEBUG, "Data succesfully compressed from %zu to %zu",
data_to_compress->size, compressed_data->size); data_to_compress->size, compressed_data->size);
cleanup:
compression_ctx_put(ctx);
return compressed_data; return compressed_data;
} }
@@ -240,16 +75,12 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) || if (!compressed_data || (!compressed_data->data && compressed_data->size != 0) ||
maximum_size == 0) maximum_size == 0)
return NULL; return NULL;
log_debug_message(LOG_DEBUG_UTIL, "Start to decompress data"); log_message(LOG_LEVEL_DEBUG, "Start to decompress data");
unsigned long long dst_size = unsigned long long dst_size =
ZSTD_getFrameContentSize(compressed_data->data, compressed_data->size); ZSTD_getFrameContentSize(compressed_data->data, compressed_data->size);
/* ZSTD_isError() is also true for ZSTD_CONTENTSIZE_ERROR and if (ZSTD_isError(dst_size)) {
* ZSTD_CONTENTSIZE_UNKNOWN (both are encoded near (size_t)-1), so test the log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: %s",
* sentinels explicitly instead of blanket-rejecting every error-ish value: ZSTD_getErrorName(dst_size));
* only CONTENTSIZE_ERROR means an unreadable header, while CONTENTSIZE_UNKNOWN
* must reach the estimate fallback below. */
if (dst_size == ZSTD_CONTENTSIZE_ERROR) {
log_message(LOG_LEVEL_ERROR, "Failed to get decompressed size: invalid zstd frame");
return NULL; return NULL;
} }
@@ -269,30 +100,20 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
return NULL; return NULL;
} }
CompressionThreadCtx* ctx = compression_get_thread_ctx(); ZSTD_DCtx* dctx = ZSTD_createDCtx();
if (ctx == NULL) { if (!dctx) {
log_message(LOG_LEVEL_ERROR, "Failed to allocate ZSTD decompression context"); log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context");
return NULL; return NULL;
} }
Data* uncompressed_data = NULL;
if (!ctx->dctx) {
ctx->dctx = ZSTD_createDCtx();
if (!ctx->dctx) {
log_message(LOG_LEVEL_ERROR, "Failed to create ZSTD decompression context");
goto cleanup;
}
}
/* Reset only the session; decompression parameters are sticky. */
ZSTD_DCtx_reset(ctx->dctx, ZSTD_reset_session_only);
size_t buf_size = (dst_size > 0) ? (size_t)dst_size : INITIAL_DECOMPRESS_BUF_SIZE; size_t buf_size = (dst_size > 0) ? (size_t)dst_size : INITIAL_DECOMPRESS_BUF_SIZE;
if (buf_size > maximum_size) if (buf_size > maximum_size)
buf_size = maximum_size; buf_size = maximum_size;
uncompressed_data = data_create_empty(buf_size); Data* uncompressed_data = data_create_empty(buf_size);
if (!uncompressed_data) { if (!uncompressed_data) {
log_message(LOG_LEVEL_ERROR, "Failed to allocate decompression buffer"); log_message(LOG_LEVEL_ERROR, "Failed to allocate decompression buffer");
goto cleanup; ZSTD_freeDCtx(dctx);
return NULL;
} }
ZSTD_inBuffer input = {compressed_data->data, compressed_data->size, 0}; ZSTD_inBuffer input = {compressed_data->data, compressed_data->size, 0};
@@ -300,57 +121,41 @@ Data* data_decompress_limited(Data* compressed_data, size_t maximum_size) {
size_t ret; size_t ret;
do { do {
ret = ZSTD_decompressStream(ctx->dctx, &output, &input); ret = ZSTD_decompressStream(dctx, &output, &input);
if (ZSTD_isError(ret)) { if (ZSTD_isError(ret)) {
log_message(LOG_LEVEL_ERROR, "Decompression failed: %s", ZSTD_getErrorName(ret)); log_message(LOG_LEVEL_ERROR, "Decompression failed: %s", ZSTD_getErrorName(ret));
ZSTD_freeDCtx(dctx);
data_destroy(uncompressed_data); data_destroy(uncompressed_data);
uncompressed_data = NULL; return NULL;
goto cleanup;
} }
if (ret > 0 && output.pos == output.size) { if (ret > 0 && output.pos == output.size) {
if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) { if (buf_size >= hard_limit || buf_size > SIZE_MAX / 2) {
log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes", log_message(LOG_LEVEL_ERROR, "Decompressed data exceeds %llu bytes",
(unsigned long long)MAX_DECOMPRESSED_SIZE); (unsigned long long)MAX_DECOMPRESSED_SIZE);
ZSTD_freeDCtx(dctx);
data_destroy(uncompressed_data); data_destroy(uncompressed_data);
uncompressed_data = NULL; return NULL;
goto cleanup;
} }
buf_size *= 2; buf_size *= 2;
if (buf_size > hard_limit) if (buf_size > hard_limit)
buf_size = (size_t)hard_limit; buf_size = (size_t)hard_limit;
void* new_data = protocol_realloc(uncompressed_data->data, buf_size); void* new_data = realloc(uncompressed_data->data, buf_size);
if (!new_data) { if (!new_data) {
log_message(LOG_LEVEL_ERROR, "Failed to grow decompression buffer"); log_message(LOG_LEVEL_ERROR, "Failed to grow decompression buffer");
ZSTD_freeDCtx(dctx);
data_destroy(uncompressed_data); data_destroy(uncompressed_data);
uncompressed_data = NULL; return NULL;
goto cleanup;
} }
uncompressed_data->data = new_data; uncompressed_data->data = new_data;
output.dst = new_data; output.dst = new_data;
output.size = buf_size; output.size = buf_size;
/* Re-attempt with the larger output buffer; the truncated-frame check
* below must not reject a complete frame that merely filled the previous
* buffer exactly. */
continue;
}
/* A positive hint with all input consumed means the frame is incomplete: a
* truncated stream would otherwise spin here forever (ZSTD_decompressStream
* keeps returning the same hint). Fail instead of burning CPU. */
if (ret != 0 && input.pos == input.size) {
log_message(LOG_LEVEL_ERROR,
"Truncated zstd frame: input exhausted with %zu bytes still expected", ret);
data_destroy(uncompressed_data);
uncompressed_data = NULL;
goto cleanup;
} }
} while (ret > 0); } while (ret > 0);
uncompressed_data->size = output.pos; uncompressed_data->size = output.pos;
ZSTD_freeDCtx(dctx);
log_debug_message(LOG_DEBUG_UTIL, "Decompressed data successfully"); log_message(LOG_LEVEL_DEBUG, "Decompressed data successfully");
cleanup:
compression_ctx_put(ctx);
return uncompressed_data; return uncompressed_data;
} }
+1 -13
View File
@@ -4,21 +4,9 @@
#include "data.h" #include "data.h"
#include <stdbool.h> #include <stdbool.h>
#define COMPRESSION_MAX_THREADS 64
Data* data_compress(Data* data_to_compress, int compression_level); Data* data_compress(Data* data_to_compress, int compression_level);
Data* data_compress_with_threads(Data* data_to_compress, int compression_level,
int compression_threads);
Data* data_decompress(Data* compressed_data); Data* data_decompress(Data* compressed_data);
Data* data_decompress_limited(Data* compressed_data, size_t maximum_size); Data* data_decompress_limited(Data* compressed_data, size_t maximum_size);
bool compression_should_skip_with_suffixes(const char* path, char* const* suffixes, int count); bool compression_should_skip(const char* path);
/* Release the calling thread's cached zstd contexts (compressor, decompressor
* and scratch buffer). The cache is thread-local and is also released
* automatically when a worker thread exits (via a C11 tss destructor) and for
* the main thread at process exit; this explicit entry point exists so tests
* and long-lived callers can drop the cache deterministically. Safe to call
* when no context has been created, and idempotent. */
void compression_free_thread_contexts(void);
#endif #endif
+188 -1152
View File
File diff suppressed because it is too large Load Diff
+73 -866
View File
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
-203
View File
@@ -1,203 +0,0 @@
#ifndef CREDENTIALS_H
#define CREDENTIALS_H
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
#include <stdio.h>
/* Daemon password authentication (A7 remediation, protocol 2.19.0).
*
* FastSync authenticates a daemon connection with a SCRAM-SHA-256-style
* challenge/response handshake. The daemon stores only a salted PBKDF2
* verifier (never the password, and never a value that can be replayed as a
* bearer credential): the client proves knowledge of the password against a
* per-connection server nonce, and the server proves the same shared secret
* back. See credentials.c for the exact derivation.
*
* Server credential store format (--password-file and --early-input): one line
* per entry,
* user:$fastsync$1$pbkdf2-sha256$<iters>$<salt_b64>$<stored_key_b64>$<server_key_b64>
* with standard base64, a 16-byte salt and 32-byte keys, and iters in
* [CREDENTIAL_MIN_ITERS, CREDENTIAL_MAX_ITERS]. Blank lines and lines whose
* first non-space character is '#' or ';' are comments. The parser is STRICT:
* a malformed line fails the whole load so a typo can never silently change who
* may log in. A line holding the legacy (unsalted SHA-256 hex) secret is
* hard-rejected with an actionable "legacy" error; there is no auto-upgrade.
* Use `fastsync-server --hash-credentials` to generate new-format lines.
*
* Alongside the store, credentials_load maintains an exact-mode-0600
* `<store_path>.dummykey` sidecar holding the store-wide random dummy key. It
* is auto-created on first load and MUST be preserved across restarts: it makes
* the dummy challenge for an unknown user stable for the life of the store, so
* a daemon restart cannot be used as a username-enumeration oracle. A sidecar
* that is not an exact-mode-0600 regular file of exactly 32 bytes fails the load
* (fail closed); creation forces exact 0600 with fchmod (so a restrictive umask
* cannot leave the sidecar unreadable), and only a create/write/fsync/link or
* fchmod failure degrades to a transient per-run key with a warning. NOTE: the
* sidecar requires EXACT 0600, whereas the store / password files only reject
* group/other bits (a deliberate difference).
*
* Client --password-file format: the FIRST meaningful (non-comment, non-blank)
* line is `user:password`, holding the literal password. The client keeps it
* only for the duration of the handshake and wipes it at teardown; the file
* should be mode 0600 and readable only by its owner. */
/* Longest accepted credential-file line (excluding the trailing newline). */
#define CREDENTIAL_MAX_LINE 4096
/* Upper bound on a username in a credential file and on the wire. Kept well
* below MAX_STRING_SIZE so a wire username can never exhaust anything. */
#define CREDENTIAL_MAX_USER_LEN 256
/* Upper bound on a client-file password (before derivation). */
#define CREDENTIAL_MAX_PASSWORD_LEN 1024
/* SCRAM-SHA-256 parameters. Salt and client nonce sizes are fixed by the
* shared-auth-message framing; keys are always 32 bytes (SHA-256). */
#define CREDENTIAL_SALT_LEN 16
#define CREDENTIAL_NONCE_LEN 32
#define CREDENTIAL_KEY_LEN 32
#define CREDENTIAL_DEFAULT_ITERS 600000u
#define CREDENTIAL_MIN_ITERS 100000u
#define CREDENTIAL_MAX_ITERS 10000000u
/* Buffer size for the full AuthMessage (prefix + three length-prefixed fields).
* Worst case: 16 + 4 + 256 + 4 + 32 + 4 + 32. */
#define CREDENTIAL_AUTH_MESSAGE_MAX \
(16 + 4 + CREDENTIAL_MAX_USER_LEN + 4 + CREDENTIAL_NONCE_LEN + 4 + CREDENTIAL_NONCE_LEN)
typedef struct CredentialStore CredentialStore;
/* One resolved verifier. `found` is false for an unknown user or a user not on
* a module's auth list; the remaining fields then hold a deterministic dummy
* salt (HMAC of the store-wide dummy key over the username), the store-wide
* uniform iteration count (default for an empty store) and fixed dummy keys, so
* the server can run the same challenge/response math with no enumeration or
* timing oracle. */
typedef struct {
uint8_t salt[CREDENTIAL_SALT_LEN];
uint32_t iters;
uint8_t stored_key[CREDENTIAL_KEY_LEN];
uint8_t server_key[CREDENTIAL_KEY_LEN];
bool found;
} CredentialVerifier;
/* Load the daemon credential store.
*
* password_file and early_input_file are both NULL-or-path, matching the
* server's --password-file and --early-input options. A file that cannot be
* opened or that fails the strict grammar is a hard error (err filled, NULL
* returned) -- the daemon fails CLOSED rather than serving an auth-required
* module with a partial store. Both files may be NULL, which yields an empty
* store (every auth-required module then refuses connections). Every entry in
* the resulting store must agree on the iteration count; entries that disagree
* (within one file or across the two layered sources) are rejected. When both
* are given, the --early-input file is layered over --password-file: a duplicate
* username whose verifier matches is deduplicated; one whose verifier differs
* is an error (the two sources disagree), never a silent pick.
*
* The returned store is heap-owned; free it with credentials_free. */
CredentialStore* credentials_load(const char* password_file, const char* early_input_file,
char* err, size_t err_size);
/* Wipe every stored key/salt and free the store. */
void credentials_free(CredentialStore* store);
/* True when `user` is a single bounded token free of whitespace/control bytes
* (the rule applied to store users, client-file users and the module list). */
bool credentials_username_valid(const char* user);
/* Standard base64. encode writes NUL-terminated output to out (size out_sz).
* decode writes the raw bytes to out (capacity out_sz) and stores the length;
* the input must be a well-formed padded base64 string. Both return false on
* NULL arguments, a bad character/length, or insufficient output space. */
bool credentials_b64_encode(const uint8_t* in, size_t n, char* out, size_t out_sz);
bool credentials_b64_decode(const char* in, uint8_t* out, size_t out_sz, size_t* out_len);
/* Fill out[0..n) from the CSPRNG (RAND_bytes). Returns false on failure. */
bool credentials_random_bytes(uint8_t* out, size_t n);
/* Resolve `user` against the store AND the module's auth-user list. The list
* scan is a constant-time full-length comparison with no early break. On a
* miss, *out is filled with a dummy verifier (a deterministic per-username salt
* derived from the store's dummy key, the store-wide uniform iteration count,
* fixed dummy keys, found=false). Returns false on invalid arguments or an
* HMAC/crypto primitive failure. */
bool credentials_get_verifier(const CredentialStore* store, const char* user,
const char* const* module_users, int n, CredentialVerifier* out);
/* Derive the SCRAM keys from a plaintext password:
* K = PBKDF2-HMAC-SHA256(password, salt, iters, 32)
* ClientKey = HMAC-SHA256(K, "Client Key"); StoredKey = SHA256(ClientKey)
* ServerKey = HMAC-SHA256(K, "Server Key")
* Any of client_key/stored_key/server_key may be NULL when not needed.
* `iters` must lie in [CREDENTIAL_MIN_ITERS, CREDENTIAL_MAX_ITERS]. */
bool credentials_compute_keys(const char* password, const uint8_t salt[CREDENTIAL_SALT_LEN],
uint32_t iters, uint8_t client_key[CREDENTIAL_KEY_LEN],
uint8_t stored_key[CREDENTIAL_KEY_LEN],
uint8_t server_key[CREDENTIAL_KEY_LEN]);
/* Serialize the shared AuthMessage:
* "FastSync-Auth-v1" || be32(len(user)) || user
* || be32(32) || server_nonce
* || be32(32) || client_nonce
* out must hold at least CREDENTIAL_AUTH_MESSAGE_MAX bytes. *out_len receives
* the number of bytes written. */
bool credentials_build_auth_message(const char* user, const uint8_t* snonce, const uint8_t* cnonce,
uint8_t* out, size_t out_sz, size_t* out_len);
/* Client side: ClientProof = ClientKey XOR HMAC(StoredKey, AuthMessage), and
* the expected ServerSignature = HMAC(ServerKey, AuthMessage). */
bool credentials_client_proof(const uint8_t client_key[CREDENTIAL_KEY_LEN],
const uint8_t stored_key[CREDENTIAL_KEY_LEN],
const uint8_t server_key[CREDENTIAL_KEY_LEN], const uint8_t* auth_msg,
size_t msg_len, uint8_t proof[CREDENTIAL_KEY_LEN],
uint8_t server_sig[CREDENTIAL_KEY_LEN]);
/* Server side: recompute ClientSig' = HMAC(StoredKey, AuthMessage) and
* ClientKey' = proof XOR ClientSig', then accept iff v->found AND
* SHA256(ClientKey') equals StoredKey (constant-time over the 32-byte keys).
* Always computes server_sig_out = HMAC(ServerKey, AuthMessage). Returns the
* accept decision. */
bool credentials_verify_response(const CredentialVerifier* v, const char* user,
const uint8_t* snonce, const uint8_t* cnonce,
const uint8_t proof[CREDENTIAL_KEY_LEN],
uint8_t server_sig_out[CREDENTIAL_KEY_LEN]);
/* Derive a new-format store line for `user`/`password` and write it (without a
* trailing newline) into out. A random 16-byte salt is used. On failure err is
* filled. Used by --hash-credentials and by tests. */
bool credentials_hash_store_line(const char* user, const char* password, uint32_t iters, char* out,
size_t out_sz, char* err, size_t err_size);
/* Read `user:password` lines from `path` (the same no-group/other-bits check as
* the other secret files) and write one new-format store line per entry to
* `out`.
* Blank/comment lines are skipped; a malformed line fails the whole run.
* Returns 0 on success, -1 on error (err filled). Used by
* `--hash-credentials`. */
int credentials_hash_file(const char* path, uint32_t iters, FILE* out, char* err, size_t err_size);
/* Read the CLIENT-side secret file: the first meaningful line is
* `user:password` (the literal password). *user_out and *password_out are
* freshly allocated on success (password is plaintext -- the caller derives the
* proof and then burns/frees it); both are NULL on error. Returns 0 on
* success, -1 on failure (err filled: the path is named, never the credential
* itself). Only the line's trailing CR/LF are stripped: the password's bytes
* are otherwise preserved exactly, so a password with leading/trailing
* whitespace (after the ':') is kept usable. The username is trimmed of
* surrounding space/tabs. */
int credentials_read_secret_file(const char* path, char** user_out, char** password_out, char* err,
size_t err_size);
/* Constant-time equality over exactly len bytes. */
bool credentials_secure_equal(const char* a, const char* b, size_t len);
/* Overwrite secret[0..len) with zeros (best-effort wipe). */
void credentials_burn(char* secret, size_t len);
/* Number of entries currently in the store (tests/introspection). */
int credentials_store_size(const CredentialStore* store);
/* Whether the store contains an entry for `user` (tests/introspection). */
bool credentials_store_has(const CredentialStore* store, const char* user);
#endif
-794
View File
@@ -1,794 +0,0 @@
#include "daemon_conf.h"
#include "credentials.h"
#include "utils.h"
#include <arpa/inet.h>
#include <ctype.h>
#include <errno.h>
#include <limits.h>
#include <netinet/in.h>
#include <stdarg.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <strings.h>
/* ------------------------------------------------------------------ */
/* helpers */
/* ------------------------------------------------------------------ */
static void set_error(char* err, size_t err_size, const char* fmt, ...) {
if (!err || err_size == 0)
return;
va_list args;
va_start(args, fmt);
vsnprintf(err, err_size, fmt, args);
va_end(args);
}
/* Trim leading and trailing ASCII space/tab in place; returns the new start. */
static char* trim_ws(char* s) {
while (*s == ' ' || *s == '\t')
s++;
size_t len = strlen(s);
while (len > 0 && (s[len - 1] == ' ' || s[len - 1] == '\t'))
s[--len] = '\0';
return s;
}
/* Case-insensitive equality of a parsed key against a canonical key name. */
static bool key_equals(const char* key, const char* canonical) {
return strcasecmp(key, canonical) == 0;
}
static bool parse_bool_value(const char* value, bool* out) {
if (strcasecmp(value, "yes") == 0 || strcasecmp(value, "true") == 0 || strcmp(value, "1") == 0) {
*out = true;
return true;
}
if (strcasecmp(value, "no") == 0 || strcasecmp(value, "false") == 0 || strcmp(value, "0") == 0) {
*out = false;
return true;
}
return false;
}
/* Parse an IPv4/IPv6 CIDR "addr/prefix" into `bytes`/`*family`. Returns false
* for a malformed address, a missing/oversized prefix, or a prefix that does
* not fit the address family. */
static bool parse_cidr(const char* cidr, int* prefix_out, uint8_t* bytes, int* family_out) {
const char* slash = strchr(cidr, '/');
if (!slash)
return false;
size_t addr_len = (size_t)(slash - cidr);
if (addr_len == 0 || addr_len >= INET6_ADDRSTRLEN)
return false;
char addr[INET6_ADDRSTRLEN];
memcpy(addr, cidr, addr_len);
addr[addr_len] = '\0';
char* end = NULL;
long prefix = strtol(slash + 1, &end, 10);
if (end == slash + 1 || *end != '\0')
return false;
struct in_addr v4;
struct in6_addr v6;
if (inet_pton(AF_INET, addr, &v4) == 1) {
if (prefix < 0 || prefix > 32)
return false;
memcpy(bytes, &v4, sizeof(v4));
*prefix_out = (int)prefix;
*family_out = AF_INET;
return true;
}
if (inet_pton(AF_INET6, addr, &v6) == 1) {
if (prefix < 0 || prefix > 128)
return false;
memcpy(bytes, &v6, sizeof(v6));
*prefix_out = (int)prefix;
*family_out = AF_INET6;
return true;
}
return false;
}
/* A host pattern is valid when it is `*`, a valid IPv4/IPv6 literal, or a valid
* CIDR. Peer addresses reaching the matcher are always numeric, so hostname
* globs are rejected at parse time: accepting one would create a deny rule that
* silently never matches (fail-open). */
static bool host_pattern_valid(const char* pattern) {
if (!pattern || *pattern == '\0')
return false;
if (strcmp(pattern, "*") == 0)
return true;
if (strchr(pattern, '/')) {
uint8_t bytes[16];
int prefix;
int family;
return parse_cidr(pattern, &prefix, bytes, &family);
}
struct in_addr v4;
struct in6_addr v6;
return inet_pton(AF_INET, pattern, &v4) == 1 || inet_pton(AF_INET6, pattern, &v6) == 1;
}
/* Append every comma- and/or whitespace-separated host pattern in `value` to
* the heap-owned list (or replace the list when `replace` is set, which --dparam
* uses so an override can narrow access rather than only widen it). Returns
* false (err filled) on an invalid pattern or an allocation failure. */
static bool store_host_list(char*** list, int* count, const char* value, const char* key,
const char* module_name, bool replace, char* err, size_t err_size) {
if (replace) {
for (int i = 0; i < *count; i++)
free((*list)[i]);
free(*list);
*list = NULL;
*count = 0;
}
char* copy = str_dup(value);
if (!copy) {
if (module_name)
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
else
set_error(err, err_size, "out of memory parsing '%s'", key);
return false;
}
char* save = NULL;
int added = 0;
for (char* token = strtok_r(copy, ", \t", &save); token; token = strtok_r(NULL, ", \t", &save)) {
if (!host_pattern_valid(token)) {
if (module_name)
set_error(err, err_size, "module '%s': invalid host pattern '%s' in '%s'", module_name,
token, key);
else
set_error(err, err_size, "invalid host pattern '%s' in '%s'", token, key);
free(copy);
return false;
}
char** grown = realloc(*list, (size_t)(*count + 1) * sizeof(char*));
if (!grown) {
if (module_name)
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
else
set_error(err, err_size, "out of memory parsing '%s'", key);
free(copy);
return false;
}
*list = grown;
char* dup = str_dup(token);
if (!dup) {
if (module_name)
set_error(err, err_size, "out of memory parsing '%s' for module '%s'", key, module_name);
else
set_error(err, err_size, "out of memory parsing '%s'", key);
free(copy);
return false;
}
(*list)[(*count)++] = dup;
added++;
}
free(copy);
/* A present key with an empty (or separator-only) value would otherwise
* install a zero-length list, i.e. no ACL at all: a strict-parse config must
* never silently turn a restrictive directive into "allow everyone". */
if (added == 0) {
if (module_name)
set_error(err, err_size, "module '%s': '%s' must list at least one host pattern", module_name,
key);
else
set_error(err, err_size, "'%s' must list at least one host pattern", key);
return false;
}
return true;
}
/* Parse a `max connections` value: a positive integer (0/negative/garbage are
* rejected because they would silently disable the cap or admit nothing). */
static bool store_max_connections(int* slot, const char* value, const char* module_name, char* err,
size_t err_size) {
char* end = NULL;
errno = 0;
long n = strtol(value, &end, 10);
if (*value == '\0' || errno != 0 || *end != '\0' || n <= 0 || n > INT_MAX) {
if (module_name)
set_error(err, err_size,
"module '%s': invalid 'max connections' '%s' (must be a positive "
"integer)",
module_name, value);
else
set_error(err, err_size, "invalid 'max connections' '%s' (must be a positive integer)",
value);
return false;
}
*slot = (int)n;
return true;
}
/* Parse a non-negative concurrency cap where 0 means unlimited/disabled
* (per-module `max connections`, `max connections per host`,
* `auth lockout threshold`). Negative/garbage/oversized values are rejected. */
static bool store_optional_cap(int* slot, const char* value, int max_value, const char* key,
const char* module_name, char* err, size_t err_size) {
char* end = NULL;
errno = 0;
long n = strtol(value, &end, 10);
if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 || n > max_value) {
if (module_name)
set_error(err, err_size, "module '%s': invalid '%s' '%s' (must be 0-%d)", module_name, key,
value, max_value);
else
set_error(err, err_size, "invalid '%s' '%s' (must be 0-%d)", key, value, max_value);
return false;
}
*slot = (int)n;
return true;
}
/* Parse an `auth failure delay` value: 0 (disabled) through the configured cap. */
static bool store_auth_failure_delay(int* slot, const char* value, char* err, size_t err_size) {
char* end = NULL;
errno = 0;
long n = strtol(value, &end, 10);
if (*value == '\0' || errno != 0 || *end != '\0' || n < 0 ||
n > DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS) {
set_error(err, err_size, "invalid 'auth failure delay' '%s' (must be 0-%d milliseconds)", value,
DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS);
return false;
}
*slot = (int)n;
return true;
}
bool daemon_module_name_valid(const char* name) {
if (!name || *name == '\0')
return false;
size_t len = strlen(name);
if (len > DAEMON_MAX_MODULE_NAME)
return false;
for (size_t i = 0; i < len; i++) {
unsigned char c = (unsigned char)name[i];
bool alnum = (c >= 'a' && c <= 'z') || (c >= 'A' && c <= 'Z') || (c >= '0' && c <= '9');
if (!alnum && c != '.' && c != '_' && c != '-')
return false;
}
return true;
}
DaemonConf* daemon_conf_create(void) {
DaemonConf* conf = calloc(1, sizeof(DaemonConf));
if (!conf)
return NULL;
conf->global.port = DAEMON_CONF_DEFAULT_PORT;
conf->global.max_connections = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS;
conf->global.auth_failure_delay_ms = DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS;
conf->global.max_connections_per_host = DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST;
conf->global.auth_lockout_threshold = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD;
conf->global.auth_lockout_duration_sec = DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC;
return conf;
}
/* Free a heap-owned pattern list of `count` entries. */
static void free_string_list(char** list, int count) {
for (int i = 0; i < count; i++)
free(list[i]);
free(list);
}
void daemon_conf_free(DaemonConf* conf) {
if (!conf)
return;
free(conf->global.motd_file);
free(conf->global.address);
free_string_list(conf->global.hosts_allow, conf->global.hosts_allow_count);
free_string_list(conf->global.hosts_deny, conf->global.hosts_deny_count);
for (int i = 0; i < conf->module_count; i++) {
DaemonModule* m = &conf->modules[i];
free(m->name);
free(m->path);
for (int j = 0; j < m->auth_user_count; j++)
free(m->auth_users[j]);
free(m->auth_users);
free_string_list(m->hosts_allow, m->hosts_allow_count);
free_string_list(m->hosts_deny, m->hosts_deny_count);
}
free(conf->modules);
free(conf);
}
const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char* name) {
if (!conf || !name)
return NULL;
for (int i = 0; i < conf->module_count; i++) {
if (strcmp(conf->modules[i].name, name) == 0)
return &conf->modules[i];
}
return NULL;
}
/* Replace *slot with a str_dup of value; returns false on allocation failure. */
static bool store_string(char** slot, const char* value) {
char* dup = str_dup(value);
if (!dup)
return false;
free(*slot);
*slot = dup;
return true;
}
static bool store_port(int* slot, const char* value, char* err, size_t err_size) {
char* end;
errno = 0;
long p = strtol(value, &end, 10);
if (errno != 0 || *end != '\0' || *value == '\0' || p <= 0 || p > 65535) {
set_error(err, err_size, "invalid port '%s' (must be 1-65535)", value);
return false;
}
*slot = (int)p;
return true;
}
/* Apply a global scalar key/value. Keys are case-insensitive. Returns false
* (err filled) on an unknown key or an invalid value. */
static bool apply_global_key(DaemonConf* conf, char* key, const char* value, bool replace_hosts,
char* err, size_t err_size) {
if (key_equals(key, "port"))
return store_port(&conf->global.port, value, err, err_size);
if (key_equals(key, "motd file")) {
if (!store_string(&conf->global.motd_file, value)) {
set_error(err, err_size, "out of memory parsing 'motd file'");
return false;
}
return true;
}
if (key_equals(key, "address")) {
if (!store_string(&conf->global.address, value)) {
set_error(err, err_size, "out of memory parsing 'address'");
return false;
}
return true;
}
if (key_equals(key, "max connections"))
return store_max_connections(&conf->global.max_connections, value, NULL, err, err_size);
if (key_equals(key, "max connections per host"))
return store_optional_cap(&conf->global.max_connections_per_host, value,
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "max connections per host", NULL,
err, err_size);
if (key_equals(key, "auth failure delay"))
return store_auth_failure_delay(&conf->global.auth_failure_delay_ms, value, err, err_size);
if (key_equals(key, "auth lockout threshold"))
return store_optional_cap(&conf->global.auth_lockout_threshold, value,
DAEMON_CONF_MAX_CONCURRENCY_LIMIT, "auth lockout threshold", NULL,
err, err_size);
if (key_equals(key, "auth lockout duration"))
return store_optional_cap(&conf->global.auth_lockout_duration_sec, value,
DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC, "auth lockout duration",
NULL, err, err_size);
if (key_equals(key, "hosts allow"))
return store_host_list(&conf->global.hosts_allow, &conf->global.hosts_allow_count, value,
"hosts allow", NULL, replace_hosts, err, err_size);
if (key_equals(key, "hosts deny"))
return store_host_list(&conf->global.hosts_deny, &conf->global.hosts_deny_count, value,
"hosts deny", NULL, replace_hosts, err, err_size);
set_error(err, err_size, "unknown global key '%s'", key);
return false;
}
/* Apply a module key/value to the currently-open module. Returns false (err
* filled) on an unknown module key or an invalid value. */
static bool apply_module_key(DaemonModule* module, char* key, char* value, char* err,
size_t err_size) {
if (key_equals(key, "path")) {
if (*value == '\0') {
set_error(err, err_size, "module '%s': 'path' must not be empty", module->name);
return false;
}
if (!store_string(&module->path, value)) {
set_error(err, err_size, "out of memory parsing 'path' for module '%s'", module->name);
return false;
}
return true;
}
if (key_equals(key, "read only")) {
bool parsed;
if (!parse_bool_value(value, &parsed)) {
set_error(err, err_size,
"module '%s': 'read only' must be yes/no (or true/false/1/0), got '%s'",
module->name, value);
return false;
}
module->read_only = parsed;
return true;
}
if (key_equals(key, "client owner")) {
bool parsed;
if (!parse_bool_value(value, &parsed)) {
set_error(err, err_size,
"module '%s': 'client owner' must be yes/no (or true/false/1/0), got '%s'",
module->name, value);
return false;
}
module->client_owner = parsed;
return true;
}
if (key_equals(key, "auth users")) {
char* list = str_dup(value);
if (!list) {
set_error(err, err_size, "out of memory parsing 'auth users' for module '%s'", module->name);
return false;
}
char* save = NULL;
int added = 0;
for (char* token = strtok_r(list, ",", &save); token; token = strtok_r(NULL, ",", &save)) {
const char* user = trim_ws(token);
if (*user == '\0')
continue;
if (!credentials_username_valid(user)) {
set_error(err, err_size, "module '%s': invalid 'auth users' entry '%s'", module->name,
user);
free(list);
return false;
}
char** grown =
realloc(module->auth_users, (size_t)(module->auth_user_count + 1) * sizeof(char*));
if (!grown) {
free(list);
set_error(err, err_size, "out of memory parsing 'auth users' for module '%s'",
module->name);
return false;
}
module->auth_users = grown;
char* dup = str_dup(user);
if (!dup) {
free(list);
set_error(err, err_size, "out of memory parsing 'auth users' for module '%s'",
module->name);
return false;
}
module->auth_users[module->auth_user_count++] = dup;
added++;
}
free(list);
/* An empty/separator-only value must not silently disable authentication:
* the key's presence is an explicit request for an allow-list. */
if (added == 0) {
set_error(err, err_size, "module '%s': 'auth users' must list at least one user",
module->name);
return false;
}
return true;
}
if (key_equals(key, "max connections"))
return store_optional_cap(&module->max_connections, value, DAEMON_CONF_MAX_CONCURRENCY_LIMIT,
"max connections", module->name, err, err_size);
if (key_equals(key, "hosts allow"))
return store_host_list(&module->hosts_allow, &module->hosts_allow_count, value, "hosts allow",
module->name, false, err, err_size);
if (key_equals(key, "hosts deny"))
return store_host_list(&module->hosts_deny, &module->hosts_deny_count, value, "hosts deny",
module->name, false, err, err_size);
set_error(err, err_size, "unknown key '%s' in module '%s'", key, module->name);
return false;
}
static bool module_open_valid(const DaemonModule* module, char* err, size_t err_size) {
if (module->path == NULL) {
set_error(err, err_size, "module '%s' has no 'path'", module->name);
return false;
}
return true;
}
/* Validate a [section] header line body (text between the brackets) and set
* *name to the module name. Returns false on a malformed header. */
static bool parse_section_name(char* body, const char** name_out, char* err, size_t err_size) {
char* name = trim_ws(body);
if (!daemon_module_name_valid(name)) {
set_error(err, err_size, "invalid module name '%s' (must be 1-%d chars of [A-Za-z0-9._-])",
name, DAEMON_MAX_MODULE_NAME);
return false;
}
*name_out = name;
return true;
}
/* Open (or switch to) a module section. Closes any previously open module
* (validating it has a path) and appends the new one. */
static int open_module(DaemonConf* conf, int* current_module, const char* name, char* err,
size_t err_size) {
if (*current_module >= 0) {
if (!module_open_valid(&conf->modules[*current_module], err, err_size))
return -1;
}
if (daemon_conf_find_module(conf, name)) {
set_error(err, err_size, "duplicate module '%s'", name);
return -1;
}
if (conf->module_count >= DAEMON_CONF_MAX_MODULES) {
set_error(err, err_size, "too many modules (limit %d); module '%s' rejected",
DAEMON_CONF_MAX_MODULES, name);
return -1;
}
DaemonModule* grown =
realloc(conf->modules, (size_t)(conf->module_count + 1) * sizeof(DaemonModule));
if (!grown) {
set_error(err, err_size, "out of memory adding module '%s'", name);
return -1;
}
conf->modules = grown;
memset(&conf->modules[conf->module_count], 0, sizeof(DaemonModule));
conf->modules[conf->module_count].name = str_dup(name);
if (!conf->modules[conf->module_count].name) {
set_error(err, err_size, "out of memory adding module '%s'", name);
return -1;
}
conf->module_count++;
*current_module = conf->module_count - 1;
return 0;
}
/* Split a "key = value" line (value pointer returned in *value, pointing into
* line). Returns false when there is no '='. */
static bool split_key_value(char* line, char** key, char** value) {
char* eq = strchr(line, '=');
if (!eq)
return false;
*eq = '\0';
*key = trim_ws(line);
*value = trim_ws(eq + 1);
return true;
}
/* Strip one layer of surrounding double quotes from a trimmed value. A value
* that starts with '"' but does not end with '"' is an error. */
static bool unquote_value(char* value, char* err, size_t err_size) {
size_t len = strlen(value);
if (len == 0 || value[0] != '"')
return true;
if (len < 2 || value[len - 1] != '"') {
set_error(err, err_size, "unterminated quoted value");
return false;
}
memmove(value, value + 1, len - 2);
value[len - 2] = '\0';
return true;
}
DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size) {
if (err && err_size)
err[0] = '\0';
if (!path) {
set_error(err, err_size, "no daemon config path");
return NULL;
}
FILE* fp = fopen(path, "r");
if (!fp) {
set_error(err, err_size, "cannot open daemon config '%s': %s", path, strerror(errno));
return NULL;
}
DaemonConf* conf = daemon_conf_create();
if (!conf) {
fclose(fp);
set_error(err, err_size, "out of memory allocating daemon config");
return NULL;
}
int current_module = -1;
int line_no = 0;
char line[DAEMON_CONF_MAX_LINE + 2];
bool ok = true;
while (ok && fgets(line, sizeof(line), fp)) {
line_no++;
size_t len = strlen(line);
if (len == DAEMON_CONF_MAX_LINE + 1 && line[len - 1] != '\n') {
/* The read stopped at the buffer edge without a newline and there is
* more file to come: the line exceeds the bound. */
if (!feof(fp)) {
set_error(err, err_size, "line %d exceeds the %d-byte limit", line_no,
DAEMON_CONF_MAX_LINE);
ok = false;
break;
}
}
if (len > 0 && line[len - 1] == '\n')
line[--len] = '\0';
if (len > 0 && line[len - 1] == '\r')
line[--len] = '\0';
char* cursor = line;
while (*cursor == ' ' || *cursor == '\t')
cursor++;
if (*cursor == '\0' || *cursor == '#' || *cursor == ';')
continue; /* blank or comment line */
if (*cursor == '[') {
char* close = strchr(cursor, ']');
if (!close) {
set_error(err, err_size, "line %d: unterminated module header", line_no);
ok = false;
break;
}
*close = '\0';
char* trailing = close + 1;
const char* rest = trim_ws(trailing);
if (*rest != '\0') {
set_error(err, err_size, "line %d: unexpected text after module header", line_no);
ok = false;
break;
}
const char* name = NULL;
if (!parse_section_name(cursor + 1, &name, err, err_size)) {
ok = false;
break;
}
if (open_module(conf, &current_module, name, err, err_size) != 0) {
ok = false;
break;
}
continue;
}
char* key;
char* value;
if (!split_key_value(cursor, &key, &value)) {
set_error(err, err_size, "line %d: expected 'key = value'", line_no);
ok = false;
break;
}
if (*key == '\0') {
set_error(err, err_size, "line %d: empty key", line_no);
ok = false;
break;
}
if (!unquote_value(value, err, err_size)) {
ok = false;
break;
}
if (current_module >= 0) {
if (!apply_module_key(&conf->modules[current_module], key, value, err, err_size)) {
ok = false;
break;
}
} else {
if (!apply_global_key(conf, key, value, false, err, err_size)) {
ok = false;
break;
}
}
}
if (ok && ferror(fp)) {
set_error(err, err_size, "error reading daemon config '%s': %s", path, strerror(errno));
ok = false;
}
fclose(fp);
if (ok && current_module >= 0 &&
!module_open_valid(&conf->modules[current_module], err, err_size)) {
ok = false;
}
if (!ok) {
daemon_conf_free(conf);
return NULL;
}
return conf;
}
int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size) {
if (err && err_size)
err[0] = '\0';
if (!conf || !assignment || *assignment == '\0') {
set_error(err, err_size, "--dparam requires a KEY=VALUE override");
return -1;
}
char* copy = str_dup(assignment);
if (!copy) {
set_error(err, err_size, "out of memory parsing --dparam");
return -1;
}
char* eq = strchr(copy, '=');
if (!eq) {
free(copy);
set_error(err, err_size, "--dparam '%s' has no '=' (expected KEY=VALUE)", assignment);
return -1;
}
*eq = '\0';
char* key = trim_ws(copy);
const char* value = trim_ws(eq + 1);
if (*key == '\0') {
free(copy);
set_error(err, err_size, "--dparam '%s' has an empty key", assignment);
return -1;
}
if (*value == '\0') {
free(copy);
set_error(err, err_size, "--dparam '%s' has an empty value", assignment);
return -1;
}
bool ok = apply_global_key(conf, key, value, true, err, err_size);
free(copy);
return ok ? 0 : -1;
}
/* Compare the first `prefix` bits of two 16-byte address buffers. */
static bool bit_prefix_match(const uint8_t* a, const uint8_t* b, int prefix) {
int whole = prefix / 8;
if (whole > 0 && memcmp(a, b, (size_t)whole) != 0)
return false;
int remainder = prefix % 8;
if (remainder == 0)
return true;
uint8_t mask = (uint8_t)(0xffu << (8 - remainder));
return (a[whole] & mask) == (b[whole] & mask);
}
/* Case-insensitive glob match used for hostname patterns. Falls back to the
* shared case-sensitive matcher when an operand is too long for the stack
* buffers. */
static bool host_glob_match(const char* pattern, const char* str) {
char pbuf[256];
char sbuf[256];
size_t plen = strlen(pattern);
size_t slen = strlen(str);
if (plen >= sizeof(pbuf) || slen >= sizeof(sbuf))
return glob_match(pattern, str);
for (size_t i = 0; i <= plen; i++)
pbuf[i] = (char)tolower((unsigned char)pattern[i]);
for (size_t i = 0; i <= slen; i++)
sbuf[i] = (char)tolower((unsigned char)str[i]);
return glob_match(pbuf, sbuf);
}
bool daemon_host_pattern_match(const char* pattern, const char* peer_ip) {
if (!pattern || *pattern == '\0' || !peer_ip || *peer_ip == '\0')
return false;
if (strcmp(pattern, "*") == 0)
return true;
if (strchr(pattern, '/')) {
uint8_t pattern_bytes[16];
uint8_t peer_bytes[16];
int prefix = 0;
int family = AF_UNSPEC;
if (!parse_cidr(pattern, &prefix, pattern_bytes, &family))
return false;
if (inet_pton(family, peer_ip, peer_bytes) != 1)
return false;
return bit_prefix_match(pattern_bytes, peer_bytes, prefix);
}
struct in_addr pattern_v4;
struct in_addr peer_v4;
if (inet_pton(AF_INET, pattern, &pattern_v4) == 1)
return inet_pton(AF_INET, peer_ip, &peer_v4) == 1 && pattern_v4.s_addr == peer_v4.s_addr;
struct in6_addr pattern_v6;
struct in6_addr peer_v6;
if (inet_pton(AF_INET6, pattern, &pattern_v6) == 1)
return inet_pton(AF_INET6, peer_ip, &peer_v6) == 1 &&
memcmp(&pattern_v6, &peer_v6, sizeof(pattern_v6)) == 0;
/* Not a literal: a hostname/glob pattern. */
return host_glob_match(pattern, peer_ip);
}
bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count,
char* const* deny, int deny_count) {
if (!peer_ip)
return false;
for (int i = 0; i < deny_count; i++) {
if (daemon_host_pattern_match(deny[i], peer_ip))
return false;
}
if (allow_count > 0) {
for (int i = 0; i < allow_count; i++) {
if (daemon_host_pattern_match(allow[i], peer_ip))
return true;
}
return false;
}
return true;
}
bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny,
int deny_count) {
(void)allow;
(void)deny;
return allow_count > 0 || deny_count > 0;
}
-182
View File
@@ -1,182 +0,0 @@
#ifndef DAEMON_CONF_H
#define DAEMON_CONF_H
#include <stdbool.h>
#include <stddef.h>
/* FastSync-native daemon configuration (a FastSync analog of rsyncd.conf).
*
* This is the config the fastsync-server --daemon listener consumes. It is
* line-based with an implicit global section followed by zero or more
* [module] sections. The full grammar is documented in RSYNC_COMPAT.md
* ("Daemon Mode") and summarized below; the parser lives entirely in
* daemon_conf.c so it can be unit tested without any socket code.
*
* The parser is STRICT: an unknown key, a malformed line, a value that does
* not parse, a module without a `path`, or a line longer than
* DAEMON_CONF_MAX_LINE all fail the whole load with a clear, line-numbered
* error instead of being silently ignored. This keeps a typo from silently
* changing what a module serves.
*/
/* A daemon module's configured root is used exactly like the standalone
* server's --destination-root: the daemon confines every connection that
* selects this module to this path (file_open_secure_parent /
* has_path_traversal / path_is_within all keep the existing confinement, just
* per-module). There is never any client-chosen root: a module path always
* stays confined. A daemon REFUSES every client-chosen ownership / super-user
* request by default -- --numeric-ids, --chown, --usermap/--groupmap,
* --fake-super, --copy-as and an explicit --super -- because there is no
* per-module opt-in unless the operator adds one. An operator opts a single
* module in with `client owner = yes` (DaemonModule.client_owner), which allows
* that client to choose ownership within that module's root (the standalone/SSH
* server honors such requests for its single operator-authorized root). The
* operator-level --no-super veto additionally forces super-user activities off
* for every daemon connection, even an opted-in module. See server_module_gate
* in server.c and RSYNC_COMPAT.md.
*
* `auth_users` is honored by Wave B daemon authentication: a module that
* declares auth users accepts a connection only when the presented username is
* on this list AND verifies against the daemon's credential store
* (--password-file / --early-input). An auth-required module with no usable
* store refuses (fail closed) rather than falling open; see server.c. Auth is
* never bypassed by ignoring the list. */
typedef struct DaemonModule {
char* name; /* module name, as the client requests it */
char* path; /* module root (daemon-side authorized root) */
bool read_only; /* `read only = yes/no`; default no */
bool client_owner; /* `client owner = yes/no`; default no. Per-module opt-in
that lets this module's clients choose ownership
(--numeric-ids/--chown/--usermap/--groupmap/--fake-super/
--copy-as) and request explicit --super super-user
activities. Without it the daemon refuses all of them. */
char** auth_users; /* `auth users = a,b`; Wave B credential list */
int auth_user_count;
/* `max connections = N` (optional per-module cap). 0 means unlimited. The
* per-connection child records the selected module in the shared registry
* (daemon_limits.c) once the config frame names it, so the cap is enforced
* across all forked children; the parent reclaims the slot on SIGCHLD. */
int max_connections;
char** hosts_allow; /* `hosts allow = a,b`; host access allow patterns */
int hosts_allow_count;
char** hosts_deny; /* `hosts deny = a,b`; host access deny patterns */
int hosts_deny_count;
} DaemonModule;
/* Global (pre-module) scalar keys. `motd file` is parsed and stored but has
* no wire effect yet (MOTD display is Wave C). */
typedef struct DaemonConfGlobals {
int port; /* `port`, default DAEMON_CONF_DEFAULT_PORT (873) */
char* motd_file; /* `motd file`, may be NULL */
char* address; /* `address` (optional bind address), may be NULL */
int max_connections; /* `max connections`, default
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS (100) */
int auth_failure_delay_ms; /* `auth failure delay`, milliseconds; default
DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS */
int max_connections_per_host; /* `max connections per host`, concurrent cap per
source IP; default
DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST (0 =
unlimited) */
int auth_lockout_threshold; /* `auth lockout threshold`, failed attempts from
one source before lockout; default
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD (0
disables) */
int auth_lockout_duration_sec; /* `auth lockout duration`, seconds; default
DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC
(0 disables) */
char** hosts_allow; /* `hosts allow`; global host access allow patterns */
int hosts_allow_count;
char** hosts_deny; /* `hosts deny`; global host access deny patterns */
int hosts_deny_count;
} DaemonConfGlobals;
typedef struct DaemonConf {
DaemonConfGlobals global;
DaemonModule* modules;
int module_count;
} DaemonConf;
#define DAEMON_CONF_DEFAULT_PORT 873
/* Default global connection cap when `max connections` is absent. Matches the
* historical hardcoded listener value. */
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS 100
/* Default `auth failure delay` in milliseconds (0 disables the throttle). */
#define DAEMON_CONF_DEFAULT_AUTH_FAILURE_DELAY_MS 500
/* Default `max connections per host` (0 = unlimited). */
#define DAEMON_CONF_DEFAULT_MAX_CONNECTIONS_PER_HOST 0
/* Default cross-process auth lockout: 10 failed attempts from one source lock
* it out for 300 s (0 disables either knob). */
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_THRESHOLD 10
#define DAEMON_CONF_DEFAULT_AUTH_LOCKOUT_DURATION_SEC 300
/* Upper bound on a `max connections per host` or `auth lockout threshold`
* value, so a typo cannot size the shared registry absurdly. */
#define DAEMON_CONF_MAX_CONCURRENCY_LIMIT 1000000
/* Upper bound on `auth lockout duration` (7 days). */
#define DAEMON_CONF_MAX_AUTH_LOCKOUT_DURATION_SEC 604800
/* Largest accepted `auth failure delay`, so a typo cannot pin a connection
* child in nanosleep for an absurd time. */
/* Bounded well below the socket I/O timeout so a failed-auth child cannot hold
* a connection slot for long enough to amplify connection-cap exhaustion. */
#define DAEMON_CONF_MAX_AUTH_FAILURE_DELAY_MS 5000
/* Upper bound on the number of [module] sections, so the shared registry's
* per-module counter array stays fixed-size. The parser rejects the next
* section past this bound. */
#define DAEMON_CONF_MAX_MODULES 256
/* Longest accepted config line (excluding the trailing newline). Longer lines
* are rejected rather than buffered unboundedly. */
#define DAEMON_CONF_MAX_LINE 4096
/* Upper bound on a module name. Kept far below MAX_STRING_SIZE so a wire
* module name can never exhaust anything by being long. */
#define DAEMON_MAX_MODULE_NAME 200
/* Allocate an empty daemon config with defaulted globals (port 873, no
* modules, no motd/address). Never fails for an allocation failure; callers
* must still NULL-check. */
DaemonConf* daemon_conf_create(void);
/* Parse `path` into a freshly allocated DaemonConf. Returns NULL on any error
* and fills `err` (err_size bytes) with a clear, line-numbered message. The
* returned object is heap-owned; free it with daemon_conf_free. */
DaemonConf* daemon_conf_load(const char* path, char* err, size_t err_size);
void daemon_conf_free(DaemonConf* conf);
/* Case-sensitive exact module lookup by name. Returns the module or NULL.
* Module names are matched exactly (rsync semantics). */
const DaemonModule* daemon_conf_find_module(const DaemonConf* conf, const char* name);
/* Module-name syntax check: non-empty, at most DAEMON_MAX_MODULE_NAME chars,
* and only [A-Za-z0-9._-]. Used by the config parser, the client's
* host::module/path destination parser, and (implicitly) by the daemon lookup
* (a name that fails this can never match a parsed module). */
bool daemon_module_name_valid(const char* name);
/* Parse one --dparam=KEY=VALUE (or "--dparam KEY=VALUE") override string and
* apply it to the global keys only. Keys are case-insensitive and limited to
* the global keys defined by the grammar (port, motd file, address,
* max connections, max connections per host, auth failure delay,
* auth lockout threshold, auth lockout duration, hosts allow, hosts deny).
* Returns 0 on success, -1 on error (err filled). */
int daemon_conf_apply_dparam(DaemonConf* conf, const char* assignment, char* err, size_t err_size);
/* Host access-control matching (pure; no I/O). `daemon_host_pattern_match`
* matches one configured pattern against a numeric peer IP string. Supported
* patterns: `*` (match anything), an IPv4/IPv6 literal, an IPv4/IPv6 CIDR
* (`10.0.0.0/8`, `2001:db8::/32`), or a glob (`*.example.com`) evaluated with
* the same matcher as file globs; a glob only matches a peer string of the
* same shape, so a numeric peer never matches a hostname glob. */
bool daemon_host_pattern_match(const char* pattern, const char* peer_ip);
/* rsync-like combined decision over a deny list and an allow list: a matching
* deny rejects (deny takes precedence); otherwise, when any allow entries
* exist, a peer that matches none is rejected; with no allow entries every
* peer not denied is accepted. An empty/unset pair returns true. */
bool daemon_hosts_allowed(const char* peer_ip, char* const* allow, int allow_count,
char* const* deny, int deny_count);
/* True when at least one allow or deny pattern is configured (i.e. an
* unprovable peer must fail closed rather than being treated as unrestricted). */
bool daemon_hosts_restricted(char* const* allow, int allow_count, char* const* deny,
int deny_count);
#endif
-494
View File
@@ -1,494 +0,0 @@
#include "daemon_limits.h"
#include "daemon_conf.h"
#include "log.h"
#include <arpa/inet.h>
#include <netinet/in.h>
#include <stdatomic.h>
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
#include <sys/mman.h>
#include <time.h>
/* The two module-count bounds must agree: the daemon config parser never
* produces more than DAEMON_CONF_MAX_MODULES modules, so the shared registry's
* per-module counter array is sized from the same bound. */
_Static_assert(DAEMON_LIMITS_MAX_MODULES == DAEMON_CONF_MAX_MODULES,
"daemon_limits module bound must match daemon_conf");
/* Slot lifecycle states (stored in slot_state). */
enum {
SLOT_FREE = 0,
SLOT_CLAIMED = 1,
SLOT_REGISTERED = 2,
};
/* The registry header lives at the base of the shared mapping; the pointer
* fields point at the arrays carved out of the same mapping. Absolute pointers
* remain valid in a forked child because fork() clones the address space and
* mapping, so parent and child observe the same virtual addresses. */
struct DaemonLimitRegistry {
int max_slots;
int module_count;
int host_slots; /* power of two; 1 when no per-source tracking is needed */
int per_host_cap;
int lockout_threshold;
int lockout_duration_sec;
size_t map_size;
_Atomic long long host_full_warn; /* last "table full" warning epoch */
_Atomic int* slot_state;
_Atomic int* slot_pid;
_Atomic int* slot_module;
_Atomic int* slot_host; /* per-source table bucket, or -1 */
_Atomic int* module_active;
_Atomic uint64_t* host_key; /* 0 == empty bucket */
_Atomic int* host_active;
_Atomic int* host_fail;
_Atomic long long* host_until; /* epoch seconds the lockout expires */
_Atomic long long* host_last_use; /* epoch seconds the bucket was last touched */
};
static size_t round_up(size_t n, size_t align) {
return (n + align - 1) & ~(align - 1);
}
static size_t next_pow2(size_t n) {
size_t p = 1;
while (p < n)
p <<= 1;
return p;
}
/* Parse a numeric IPv4/IPv6 peer string into family + raw bytes. */
static bool parse_peer_ip(const char* peer_ip, int* family, unsigned char* bytes) {
if (!peer_ip || *peer_ip == '\0')
return false;
struct in_addr v4;
if (inet_pton(AF_INET, peer_ip, &v4) == 1) {
memcpy(bytes, &v4, sizeof(v4));
*family = AF_INET;
return true;
}
struct in6_addr v6;
if (inet_pton(AF_INET6, peer_ip, &v6) == 1) {
memcpy(bytes, &v6, sizeof(v6));
*family = AF_INET6;
return true;
}
return false;
}
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok) {
if (ok)
*ok = false;
unsigned char bytes[16];
int family = AF_UNSPEC;
if (!parse_peer_ip(peer_ip, &family, bytes))
return 0;
uint64_t hash = 14695981039346656037ULL ^ (uint64_t)(uint32_t)family;
size_t length = family == AF_INET ? 4 : 16;
for (size_t i = 0; i < length; i++) {
hash ^= bytes[i];
hash *= 1099511628211ULL;
}
if (hash == 0)
hash = 0x9e3779b97f4a7c15ULL;
if (ok)
*ok = true;
return hash;
}
/* True when the registry must maintain per-source buckets: either the per-host
* cap is configured, or the auth lockout is (threshold AND duration > 0). A
* lockout threshold without a duration is a no-op, so it must not size or intern
* the table. create(), register() and the lockout paths all agree on this. */
static bool registry_tracks_hosts(const DaemonLimitRegistry* registry) {
return registry->per_host_cap > 0 ||
(registry->lockout_threshold > 0 && registry->lockout_duration_sec > 0);
}
/* Find the bucket holding `peer_ip`, or -1 when it has no entry. Finding a
* bucket refreshes its last-use time so the eviction policy sees it as live. */
static int host_lookup(DaemonLimitRegistry* registry, const char* peer_ip) {
bool ok = false;
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
if (!ok)
return -1;
size_t mask = (size_t)registry->host_slots - 1;
size_t start = (size_t)(key & mask);
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
size_t idx = (start + i) & mask;
uint64_t current = atomic_load_explicit(&registry->host_key[idx], memory_order_acquire);
if (current == key) {
atomic_store_explicit(&registry->host_last_use[idx], (long long)time(NULL),
memory_order_relaxed);
return (int)idx;
}
if (current == 0)
return -1; /* no tombstones: an empty bucket ends the probe chain */
}
return -1;
}
/* A bucket with no live connection may be repurposed: immediately when its
* lockout deadline has already passed (the review's "expired" case), or after an
* idle window when it holds no pending lockout. A bucket with a future lockout
* deadline is retained so the lockout actually lasts its configured duration. */
static bool host_bucket_reclaimable(DaemonLimitRegistry* registry, size_t idx, long long now) {
if (atomic_load_explicit(&registry->host_active[idx], memory_order_relaxed) != 0)
return false;
long long until = atomic_load_explicit(&registry->host_until[idx], memory_order_relaxed);
if (until != 0)
return until <= now;
long long last_use = atomic_load_explicit(&registry->host_last_use[idx], memory_order_relaxed);
/* A bucket whose key is published but whose last_use has not yet been stamped
* (last_use == 0) must be treated as live: reclaiming it here would steal a
* bucket a racing child just claimed. The claim path also stamps last_use
* before publishing the key, so this window cannot persist. */
return last_use != 0 && now - last_use >= DAEMON_LIMITS_HOST_EVICT_IDLE_SEC;
}
/* Emit at most one "per-source table full" warning per
* DAEMON_LIMITS_HOST_FULL_WARN_SEC across all forked children. Called from a
* normal (non-signal) child path, so logging is safe here. */
static void host_warn_table_full(DaemonLimitRegistry* registry, long long now) {
long long last = atomic_load_explicit(&registry->host_full_warn, memory_order_relaxed);
if (last != 0 && now - last < DAEMON_LIMITS_HOST_FULL_WARN_SEC)
return;
if (atomic_compare_exchange_strong_explicit(&registry->host_full_warn, &last, now,
memory_order_relaxed, memory_order_relaxed)) {
log_message(LOG_LEVEL_WARNING,
"daemon: per-source registry is full (%d slots) and no bucket can be reclaimed; "
"'max connections per host' and the auth lockout are temporarily not enforced for "
"new sources (the per-module cap and host ACLs still apply)",
registry->host_slots);
}
}
/* Find or insert the bucket for `peer_ip`. Insertion is a lock-free CAS so two
* forked children racing on the same source converge on one bucket.
*
* When the probe finds no empty bucket it reclaims, via a key CAS, the first
* bucket that is reclaimable (expired lockout or idle, and no active
* connection) and resets its counters. This bounds the table's lifetime so it
* cannot fill permanently and stay fail-open. Returns -1 only when the address
* is unparseable or the table is genuinely full of live/locked buckets
* (callers fail open: the global/module caps and ACLs still apply). */
static int host_intern(DaemonLimitRegistry* registry, const char* peer_ip) {
bool ok = false;
uint64_t key = daemon_limits_host_hash(peer_ip, &ok);
if (!ok)
return -1;
long long now = (long long)time(NULL);
size_t mask = (size_t)registry->host_slots - 1;
size_t start = (size_t)(key & mask);
/* A couple of passes bound the work: the first normally claims/seeds a bucket;
* a lost eviction CAS retries once against the freshly observed table. */
for (int pass = 0; pass < 2; pass++) {
int evict = -1;
uint64_t evict_key = 0;
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
size_t idx = (start + i) & mask;
uint64_t current = atomic_load_explicit(&registry->host_key[idx], memory_order_acquire);
if (current == key) {
atomic_store_explicit(&registry->host_last_use[idx], now, memory_order_relaxed);
return (int)idx;
}
if (current == 0) {
/* Stamp last_use *before* publishing the key so a reclaimer racing the
* claim can never observe a claimed bucket with last_use == 0 and
* evict it. A pre-stamp is harmless if the CAS loses: the bucket is
* either still empty (never inspected for reclaim) or has just been
* taken by another source that wants a fresh timestamp anyway. */
atomic_store_explicit(&registry->host_last_use[idx], now, memory_order_relaxed);
uint64_t expected = 0;
if (atomic_compare_exchange_strong_explicit(&registry->host_key[idx], &expected, key,
memory_order_acq_rel, memory_order_acquire)) {
return (int)idx;
}
if (atomic_load_explicit(&registry->host_key[idx], memory_order_acquire) == key) {
return (int)idx;
}
continue; /* another child won this empty bucket; keep probing */
}
if (evict < 0 && host_bucket_reclaimable(registry, idx, now)) {
evict = (int)idx;
evict_key = current;
}
}
if (evict >= 0) {
/* Refresh the timestamp before the key changes hands so the reused bucket
* is not seen as immediately idle by a racing reclaimer. */
atomic_store_explicit(&registry->host_last_use[evict], now, memory_order_relaxed);
uint64_t expected = evict_key;
if (atomic_compare_exchange_strong_explicit(&registry->host_key[evict], &expected, key,
memory_order_acq_rel, memory_order_acquire)) {
/* The bucket now belongs to the new source; clear the evicted source's
* stale lockout/failure state. */
atomic_store_explicit(&registry->host_active[evict], 0, memory_order_relaxed);
atomic_store_explicit(&registry->host_fail[evict], 0, memory_order_relaxed);
atomic_store_explicit(&registry->host_until[evict], 0, memory_order_relaxed);
/* Two children can race to intern the same brand-new key into different
* eviction targets, leaving the table with duplicate buckets for `key`.
* Re-scan for the first (canonical) bucket holding `key`; when it
* precedes `evict`, drop our duplicate's occupancy and hand back the
* canonical bucket so per-source counts are not orphaned on the
* duplicate. The duplicate keeps its key, so no tombstone hole is
* created and probe chains stay intact; it ages out normally. */
for (size_t i = 0; i < (size_t)registry->host_slots; i++) {
size_t candidate = (start + i) & mask;
uint64_t found =
atomic_load_explicit(&registry->host_key[candidate], memory_order_acquire);
if (found == key) {
if (candidate != (size_t)evict) {
atomic_store_explicit(&registry->host_active[evict], 0, memory_order_relaxed);
return (int)candidate;
}
break;
}
if (found == 0)
break; /* the key is present at `evict`, so this cannot happen first */
}
return evict;
}
continue; /* lost the race; re-probe with fresh observations */
}
break; /* no free and no reclaimable bucket: genuinely full */
}
host_warn_table_full(registry, now);
return -1;
}
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
int lockout_threshold, int lockout_duration_sec) {
if (max_slots < DAEMON_LIMITS_MIN_SLOTS)
max_slots = DAEMON_LIMITS_MIN_SLOTS;
if (max_slots > DAEMON_LIMITS_MAX_SLOTS)
max_slots = DAEMON_LIMITS_MAX_SLOTS;
if (module_count < 1)
module_count = 1;
if (module_count > DAEMON_LIMITS_MAX_MODULES)
module_count = DAEMON_LIMITS_MAX_MODULES;
if (per_host_cap < 0)
per_host_cap = 0;
if (lockout_threshold < 0)
lockout_threshold = 0;
if (lockout_duration_sec < 0)
lockout_duration_sec = 0;
bool need_hosts = per_host_cap > 0 || (lockout_threshold > 0 && lockout_duration_sec > 0);
int host_slots = 1;
if (need_hosts) {
size_t want = (size_t)max_slots * 4;
if (want < 64)
want = 64;
if (want > DAEMON_LIMITS_MAX_HOST_SLOTS)
want = DAEMON_LIMITS_MAX_HOST_SLOTS;
host_slots = (int)next_pow2(want);
}
size_t header = round_up(sizeof(DaemonLimitRegistry), 16);
size_t slot_bytes =
round_up((size_t)max_slots * sizeof(_Atomic int), 16) * 4; /* state,pid,module,host */
size_t module_bytes = round_up((size_t)module_count * sizeof(_Atomic int), 16);
size_t host_key_bytes = round_up((size_t)host_slots * sizeof(_Atomic uint64_t), 16);
size_t host_int_bytes = round_up((size_t)host_slots * sizeof(_Atomic int), 16) * 2;
size_t host_until_bytes = round_up((size_t)host_slots * sizeof(_Atomic long long), 16) * 2;
size_t total =
header + slot_bytes + module_bytes + host_key_bytes + host_int_bytes + host_until_bytes + 16;
void* map = mmap(NULL, total, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_ANONYMOUS, -1, 0);
if (map == MAP_FAILED)
return NULL;
memset(map, 0, total);
DaemonLimitRegistry* registry = (DaemonLimitRegistry*)map;
registry->max_slots = max_slots;
registry->module_count = module_count;
registry->host_slots = host_slots;
registry->per_host_cap = per_host_cap;
registry->lockout_threshold = lockout_threshold;
registry->lockout_duration_sec = lockout_duration_sec;
registry->map_size = total;
unsigned char* cursor = (unsigned char*)map + header;
registry->slot_state = (atomic_int*)cursor;
cursor += (size_t)max_slots * sizeof(_Atomic int);
registry->slot_pid = (atomic_int*)cursor;
cursor += (size_t)max_slots * sizeof(_Atomic int);
registry->slot_module = (atomic_int*)cursor;
cursor += (size_t)max_slots * sizeof(_Atomic int);
registry->slot_host = (atomic_int*)cursor;
cursor += (size_t)max_slots * sizeof(_Atomic int);
registry->module_active = (atomic_int*)cursor;
cursor += (size_t)module_count * sizeof(_Atomic int);
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
registry->host_key = (_Atomic uint64_t*)cursor;
cursor += (size_t)host_slots * sizeof(_Atomic uint64_t);
registry->host_active = (atomic_int*)cursor;
cursor += (size_t)host_slots * sizeof(_Atomic int);
registry->host_fail = (atomic_int*)cursor;
cursor += (size_t)host_slots * sizeof(_Atomic int);
cursor = (unsigned char*)round_up((size_t)(uintptr_t)cursor, 16);
registry->host_until = (atomic_llong*)cursor;
cursor += (size_t)host_slots * sizeof(_Atomic long long);
registry->host_last_use = (atomic_llong*)cursor;
for (int i = 0; i < max_slots; i++) {
atomic_store(&registry->slot_module[i], -1);
atomic_store(&registry->slot_host[i], -1);
}
return registry;
}
void daemon_limits_destroy(DaemonLimitRegistry* registry) {
if (!registry)
return;
munmap(registry, registry->map_size);
}
int daemon_limits_claim_slot(DaemonLimitRegistry* registry) {
if (!registry)
return DAEMON_LIMITS_NO_SLOT;
for (int i = 0; i < registry->max_slots; i++) {
int expected = SLOT_FREE;
if (atomic_compare_exchange_strong(&registry->slot_state[i], &expected, SLOT_CLAIMED)) {
atomic_store(&registry->slot_pid[i], 0);
atomic_store(&registry->slot_module[i], -1);
atomic_store(&registry->slot_host[i], -1);
return i;
}
}
return DAEMON_LIMITS_NO_SLOT;
}
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid) {
if (!registry || slot < 0 || slot >= registry->max_slots)
return;
atomic_store(&registry->slot_pid[slot], (int)pid);
}
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot) {
if (!registry || slot < 0 || slot >= registry->max_slots)
return;
atomic_exchange_explicit(&registry->slot_state[slot], SLOT_FREE, memory_order_acq_rel);
atomic_store_explicit(&registry->slot_pid[slot], 0, memory_order_relaxed);
/* The module/host occupancy arrays are derived from the slot table; do not
* decrement here or a SIGKILL between a child's increment and its REGISTERED
* publish would leak a count. Callers that need the derived counts call
* daemon_limits_recompute. */
}
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid) {
if (!registry || pid <= 0)
return;
for (int i = 0; i < registry->max_slots; i++) {
if (atomic_load(&registry->slot_state[i]) == SLOT_FREE)
continue;
if (atomic_load(&registry->slot_pid[i]) == (int)pid) {
daemon_limits_reclaim_slot(registry, i);
return;
}
}
}
void daemon_limits_recompute(DaemonLimitRegistry* registry) {
if (!registry)
return;
/* Zero the derived arrays, then re-derive solely from the REGISTERED slots.
* A child that was SIGKILLed after incrementing a counter but before
* publishing REGISTERED is not counted, and its leaked increment is erased by
* the zeroing, so the leak cannot persist. */
for (int m = 0; m < registry->module_count; m++)
atomic_store_explicit(&registry->module_active[m], 0, memory_order_relaxed);
for (int h = 0; h < registry->host_slots; h++)
atomic_store_explicit(&registry->host_active[h], 0, memory_order_relaxed);
for (int i = 0; i < registry->max_slots; i++) {
if (atomic_load_explicit(&registry->slot_state[i], memory_order_acquire) != SLOT_REGISTERED)
continue;
int module = atomic_load_explicit(&registry->slot_module[i], memory_order_relaxed);
if (module >= 0 && module < registry->module_count)
atomic_fetch_add_explicit(&registry->module_active[module], 1, memory_order_relaxed);
int host = atomic_load_explicit(&registry->slot_host[i], memory_order_relaxed);
if (host >= 0 && host < registry->host_slots)
atomic_fetch_add_explicit(&registry->host_active[host], 1, memory_order_relaxed);
}
}
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
const char* peer_ip, int module_cap) {
if (!registry || slot < 0 || slot >= registry->max_slots)
return DAEMON_LIMIT_UNAVAILABLE;
if (module_index < 0 || module_index >= registry->module_count)
return DAEMON_LIMIT_UNAVAILABLE;
if (atomic_load_explicit(&registry->slot_state[slot], memory_order_acquire) != SLOT_CLAIMED)
return DAEMON_LIMIT_UNAVAILABLE;
int host = -1;
if (registry_tracks_hosts(registry))
host = host_intern(registry, peer_ip);
int module_count = atomic_fetch_add(&registry->module_active[module_index], 1) + 1;
if (module_cap > 0 && module_count > module_cap) {
atomic_fetch_sub(&registry->module_active[module_index], 1);
return DAEMON_LIMIT_MODULE_FULL;
}
if (host >= 0) {
int host_count = atomic_fetch_add(&registry->host_active[host], 1) + 1;
if (registry->per_host_cap > 0 && host_count > registry->per_host_cap) {
atomic_fetch_sub(&registry->host_active[host], 1);
atomic_fetch_sub(&registry->module_active[module_index], 1);
return DAEMON_LIMIT_HOST_FULL;
}
}
atomic_store(&registry->slot_module[slot], module_index);
atomic_store(&registry->slot_host[slot], host);
atomic_store_explicit(&registry->slot_state[slot], SLOT_REGISTERED, memory_order_release);
return DAEMON_LIMIT_OK;
}
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
int* seconds_remaining) {
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
return false;
int bucket = host_lookup(registry, peer_ip);
if (bucket < 0)
return false;
long long until = atomic_load(&registry->host_until[bucket]);
long long now = (long long)time(NULL);
if (until > now) {
if (seconds_remaining)
*seconds_remaining = (int)(until - now);
return true;
}
if (until != 0) {
/* The previous lockout has expired: clear the stale counter so the source
* gets a fresh allowance. */
atomic_store(&registry->host_fail[bucket], 0);
atomic_store(&registry->host_until[bucket], 0);
}
return false;
}
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip) {
if (!registry || registry->lockout_threshold <= 0 || registry->lockout_duration_sec <= 0)
return;
int bucket = host_intern(registry, peer_ip);
if (bucket < 0)
return;
int failures = atomic_fetch_add(&registry->host_fail[bucket], 1) + 1;
if (failures >= registry->lockout_threshold) {
long long now = (long long)time(NULL);
atomic_store(&registry->host_until[bucket], now + (long long)registry->lockout_duration_sec);
}
}
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip) {
if (!registry)
return;
int bucket = host_lookup(registry, peer_ip);
if (bucket < 0)
return;
atomic_store(&registry->host_fail[bucket], 0);
atomic_store(&registry->host_until[bucket], 0);
}
-147
View File
@@ -1,147 +0,0 @@
#ifndef DAEMON_LIMITS_H
#define DAEMON_LIMITS_H
#include <stdbool.h>
#include <stddef.h>
#include <stdint.h>
/* Cross-process daemon connection registry.
*
* The daemon listener forks ONE child per accepted connection, so any
* per-module / per-source accounting must live in state shared across the
* forked children. This module owns a fixed-size registry carved out of an
* anonymous shared mapping (mmap(MAP_SHARED | MAP_ANONYMOUS)) created by the
* accept-loop PARENT before it forks; every child inherits the mapping (and the
* pointer to it) across fork().
*
* Rules:
* - ONLY C11 atomics (atomic_*); never mtx_t/pthread locks, which can deadlock
* in a forked child if another thread held them at fork time.
* - No heap allocation after fork: the mapping is fixed-size and all access is
* atomic load/store/CAS over preallocated arrays.
*
* Slot lifecycle (the parent reclaims even when a child is SIGKILLed):
* FREE --(parent claim_slot)--> CLAIMED
* CLAIMED --(child register)--> REGISTERED
* any --(parent reclaim)--> FREE
* The child records its module index and per-source bucket into the slot before
* publishing REGISTERED; the parent's SIGCHLD handler matches the reaped pid to
* the slot and, when REGISTERED, decrements the module/per-source counters.
* A child killed before registering holds no counts, so reclaiming a CLAIMED
* slot only frees the slot.
*
* Per-source identity is the normalized numeric peer IP (IPv4-mapped IPv6 is
* already collapsed to IPv4 by utils_fd_peer_ip); it is interned into an
* open-addressed, linear-probing table keyed by a 64-bit hash. The same table
* also carries the cross-process auth-failure counter and lockout deadline.
*
* Per-source table lifetime: a bucket's key is never cleared back to empty (that
* would break every later probe chain that passed through it). Instead the
* table has a bounded-lifetime eviction policy: when no empty bucket exists, the
* first bucket that is reclaimable -- no active connection AND (its lockout
* deadline has passed OR it has been idle for
* DAEMON_LIMITS_HOST_EVICT_IDLE_SEC) -- is atomically repurposed for the new
* source via a CAS of its key, and its counters are reset. The table therefore
* cannot fill permanently, and a full table degrades to fail-open for the
* per-source cap/lockout of new sources (the per-module cap and host ACLs still
* apply) instead of staying fail-open forever. A rate-limited warning is logged
* on the fail-open path. The eviction race with a concurrent
* registration/reclaim on the same bucket is benign: it can at worst lose one
* source's counter (fail-open), never corrupt memory or the module caps.
*/
typedef struct DaemonLimitRegistry DaemonLimitRegistry;
/* Result of a per-connection admission check. */
typedef enum {
DAEMON_LIMIT_OK = 0, /* admitted; slot is now REGISTERED */
DAEMON_LIMIT_MODULE_FULL, /* module's `max connections` cap reached */
DAEMON_LIMIT_HOST_FULL, /* global `max connections per host` cap reached */
DAEMON_LIMIT_UNAVAILABLE, /* registry/slot unusable (caller fails open) */
} DaemonLimitResult;
/* Bounds for registry sizing. A slot is one concurrently live child. */
#define DAEMON_LIMITS_MIN_SLOTS 16
#define DAEMON_LIMITS_MAX_SLOTS 65536
#define DAEMON_LIMITS_MAX_HOST_SLOTS 65536
#define DAEMON_LIMITS_NO_SLOT (-1)
/* Upper bound on `module_count`, matching daemon_conf.h's DAEMON_CONF_MAX_MODULES
* (asserted in daemon_limits.c) so a caller can never size the per-module counter
* array larger than the config parser can produce. */
#define DAEMON_LIMITS_MAX_MODULES 256
/* Per-source table lifetime: a bucket with no active connection and no pending
* lockout is reclaimable once it has been idle this long, so a flood of distinct
* sources cannot pin the table full forever. A bucket whose lockout deadline
* has passed is reclaimable immediately (independent of this idle window). */
#define DAEMON_LIMITS_HOST_EVICT_IDLE_SEC 300
/* Minimum spacing between "per-source table is full" warnings, so a table-full
* attack cannot flood the log. */
#define DAEMON_LIMITS_HOST_FULL_WARN_SEC 60
/* Create the shared registry in the calling (parent) process. `max_slots` is
* the number of concurrently live children to track (clamped to
* [DAEMON_LIMITS_MIN_SLOTS, DAEMON_LIMITS_MAX_SLOTS]); `module_count` is the
* number of daemon modules (clamped to
* [1, DAEMON_LIMITS_MAX_MODULES]); `per_host_cap` and the lockout pair come
* from the daemon config (0 disables). Returns NULL on failure (e.g. mmap
* allocation); callers must degrade gracefully (global cap + ACLs still
* apply). */
DaemonLimitRegistry* daemon_limits_create(int max_slots, int module_count, int per_host_cap,
int lockout_threshold, int lockout_duration_sec);
/* Unmap the registry. Only the creating process may call this. */
void daemon_limits_destroy(DaemonLimitRegistry* registry);
/* Parent side: reserve a slot for the next fork. Returns the slot index or
* DAEMON_LIMITS_NO_SLOT when every slot is in use. */
int daemon_limits_claim_slot(DaemonLimitRegistry* registry);
/* Parent side: record the forked child's pid in a claimed slot. */
void daemon_limits_set_slot_pid(DaemonLimitRegistry* registry, int slot, long pid);
/* Parent side: release a slot. The slot becomes FREE; the module/per-source
* occupancy arrays are DERIVED state and are only refreshed by
* daemon_limits_recompute, which callers must invoke afterwards when they rely
* on the derived counts (the SIGCHLD handler batches one recompute for the whole
* reap). Idempotent. */
void daemon_limits_reclaim_slot(DaemonLimitRegistry* registry, int slot);
/* Parent SIGCHLD side: release the slot owned by `pid` (no-op when not found).
* Like reclaim_slot this does not touch the derived occupancy arrays; call
* daemon_limits_recompute after a batch of releases. */
void daemon_limits_reclaim_pid(DaemonLimitRegistry* registry, long pid);
/* Parent side (async-signal-safe; atomics only, no malloc/log): rebuild
* module_active[] / host_active[] from scratch by scanning the REGISTERED slots.
* The slot table is the single source of truth, so this self-heals any
* count leaked by a child that was SIGKILLed mid-registration (it zeroes the
* arrays and re-derives them). Bounded by max_slots + host_slots. A
* registration racing this call can be transiently undercounted until the next
* recompute, which can only relax a cap briefly -- never corrupt memory. */
void daemon_limits_recompute(DaemonLimitRegistry* registry);
/* Child side: admit the connection for `module_index` from `peer_ip`. Always
* tracks the module/per-source occupancy (so the parent's reclaim is
* symmetric); when `module_cap` > 0 it additionally enforces the per-module
* cap. A NULL/empty or non-numeric `peer_ip` skips the per-source track (the
* callers use that to exempt a trusted loopback peer from the per-host cap; the
* per-module cap still applies). Returns DAEMON_LIMIT_OK and publishes the
* slot, or a refusal reason. */
DaemonLimitResult daemon_limits_register(DaemonLimitRegistry* registry, int slot, int module_index,
const char* peer_ip, int module_cap);
/* Child side: true when `peer_ip` is currently locked out after too many failed
* authentications. `seconds_remaining` may be NULL. */
bool daemon_limits_auth_locked(DaemonLimitRegistry* registry, const char* peer_ip,
int* seconds_remaining);
/* Child side: count one failed authentication for `peer_ip`; once the threshold
* is reached the source is locked out for the configured duration. */
void daemon_limits_auth_record_failure(DaemonLimitRegistry* registry, const char* peer_ip);
/* Child side: clear the failure counter/lockout for a source that authenticated
* successfully (no-op when the source has no table entry). */
void daemon_limits_auth_record_success(DaemonLimitRegistry* registry, const char* peer_ip);
/* Pure helper: 64-bit FNV-1a hash of a numeric peer IP plus its family, used to
* index the per-source table. *ok is set false (and 0 returned) for a NULL or
* non-numeric address. Exposed for unit testing. */
uint64_t daemon_limits_host_hash(const char* peer_ip, bool* ok);
#endif
+5 -12
View File
@@ -1,12 +1,11 @@
#include "data.h" #include "data.h"
#include "log.h" #include "log.h"
#include "protocol.h"
#include <stdlib.h> #include <stdlib.h>
Data* data_create_empty(size_t data_size) { Data* data_create_empty(size_t data_size) {
/* malloc(0) is UB; allocate at least 1 byte but preserve requested size */ /* malloc(0) is UB; allocate at least 1 byte but preserve requested size */
size_t alloc_size = data_size > 0 ? data_size : 1; size_t alloc_size = data_size > 0 ? data_size : 1;
void* data = protocol_alloc(alloc_size); void* data = malloc(alloc_size);
if (data == NULL) { if (data == NULL) {
log_message(LOG_LEVEL_ERROR, "Could not allocate memory for empty data"); log_message(LOG_LEVEL_ERROR, "Could not allocate memory for empty data");
return NULL; return NULL;
@@ -15,7 +14,7 @@ Data* data_create_empty(size_t data_size) {
} }
Data* data_create_reserve(size_t size) { Data* data_create_reserve(size_t size) {
Data* d = protocol_alloc(sizeof(Data)); Data* d = malloc(sizeof(Data));
if (d == NULL) { if (d == NULL) {
log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data"); log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data");
return NULL; return NULL;
@@ -23,12 +22,11 @@ Data* data_create_reserve(size_t size) {
d->data = NULL; d->data = NULL;
d->size = size; d->size = size;
d->protocol_charge = 0; d->protocol_charge = 0;
d->owner = NULL;
return d; return d;
} }
Data* data_create(void* data, size_t data_size) { Data* data_create(void* data, size_t data_size) {
Data* new_data = protocol_alloc(sizeof(Data)); Data* new_data = malloc(sizeof(Data));
if (new_data == NULL) { if (new_data == NULL) {
log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data"); log_message(LOG_LEVEL_ERROR, "Could not allocate memory for data");
free(data); free(data);
@@ -37,19 +35,14 @@ Data* data_create(void* data, size_t data_size) {
new_data->data = data; new_data->data = data;
new_data->size = data_size; new_data->size = data_size;
new_data->protocol_charge = 0; new_data->protocol_charge = 0;
new_data->owner = NULL;
return new_data; return new_data;
} }
void data_destroy(Data* data) { void data_destroy(Data* data) {
if (data == NULL) if (data == NULL)
return; return;
if (data->protocol_charge != 0) { if (data->protocol_charge != 0)
if (data->owner != NULL) protocol_release_memory(data->protocol_charge);
protocol_release_memory_for_session(data->owner, data->protocol_charge);
else
protocol_release_memory(data->protocol_charge);
}
free(data->data); free(data->data);
free(data); free(data);
} }
-18
View File
@@ -3,25 +3,11 @@
#include <stdlib.h> #include <stdlib.h>
/* Forward declaration for the connection budget a received Data is charged
* against; defined in protocol.h (which includes this header). */
typedef struct ProtocolSession ProtocolSession;
typedef struct { typedef struct {
void* data; void* data;
size_t size; size_t size;
/* Non-zero only for a buffer charged to the protocol connection budget. */ /* Non-zero only for a buffer charged to the protocol connection budget. */
size_t protocol_charge; size_t protocol_charge;
/* Session whose budget `protocol_charge` was reserved from. When non-NULL,
* the charge is returned to this session directly, regardless of which
* session (if any) is bound to the destroying thread. owner is not
* guaranteed to be set whenever protocol_charge is non-zero: it is NULL for
* uncharged Data and for Data that has no recorded owner, in which case any
* charge falls back to the session bound at destroy time.
*
* Lifetime contract: a Data with a non-NULL owner must not outlive that
* ProtocolSession -- data_destroy dereferences owner to return the charge. */
ProtocolSession* owner;
} Data; } Data;
Data* data_create_empty(size_t data_size); Data* data_create_empty(size_t data_size);
@@ -29,9 +15,5 @@ Data* data_create_reserve(size_t size);
Data* data_create(void* data, size_t data_size); Data* data_create(void* data, size_t data_size);
void data_destroy(Data* data); void data_destroy(Data* data);
void protocol_release_memory(size_t charge); void protocol_release_memory(size_t charge);
/* Release `charge` against `session` directly instead of the thread-local bound
* session. Used by data_destroy to honor Data.owner; `session` must outlive
* the Data whose charge is being returned. A NULL session is a no-op. */
void protocol_release_memory_for_session(ProtocolSession* session, size_t charge);
#endif #endif
-338
View File
@@ -1,338 +0,0 @@
#include "delay_updates.h"
#include "config.h"
#include "file.h"
#include "log.h"
#include "utils.h"
#include <dirent.h>
#include <errno.h>
#include <fcntl.h>
#include <libgen.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/file.h>
#include <sys/stat.h>
#include <unistd.h>
DelayUpdatesContext* delay_updates_context_create(const char* root_directory) {
if (!root_directory)
return NULL;
DelayUpdatesContext* context = calloc(1, sizeof(DelayUpdatesContext));
if (!context)
return NULL;
context->root_directory = str_dup(root_directory);
if (!context->root_directory) {
free(context);
return NULL;
}
context->staging_root = path_cat(root_directory, DELAY_UPDATES_STAGING_DIR);
if (!context->staging_root) {
free(context->root_directory);
free(context);
return NULL;
}
context->entries = NULL;
context->count = 0;
context->capacity = 0;
context->prepared = false;
context->lock_fd = -1;
if (mtx_init(&context->mutex, mtx_plain) != thrd_success) {
free(context->staging_root);
free(context->root_directory);
free(context);
return NULL;
}
return context;
}
void delay_updates_context_destroy(DelayUpdatesContext* context) {
if (!context)
return;
mtx_destroy(&context->mutex);
if (context->lock_fd >= 0)
close(context->lock_fd);
context->lock_fd = -1;
free(context->staging_root);
free(context->root_directory);
for (size_t i = 0; i < context->count; i++) {
free(context->entries[i].staged_path);
free(context->entries[i].final_path);
free(context->entries[i].file_path);
}
free(context->entries);
free(context);
}
bool delay_updates_staging_name_conflict(const char* dir) {
if (!dir || !*dir)
return false;
size_t length = strlen(dir);
while (length > 0 && dir[length - 1] == '/')
length--;
size_t reserved_length = strlen(DELAY_UPDATES_STAGING_DIR);
if (length != reserved_length)
return false;
return strncmp(dir, DELAY_UPDATES_STAGING_DIR, length) == 0;
}
/* Recursively delete every entry inside an open directory (never following
symlinks). The directory itself is left in place. Mirrors the fd-relative
walk used by the delete code so a symlink planted inside the staging tree
can never redirect removal outside of it. */
static bool delay_wipe_dir_fd(int dirfd) {
int scanfd = dup(dirfd);
if (scanfd < 0)
return false;
DIR* dir = fdopendir(scanfd);
if (!dir) {
close(scanfd);
return false;
}
bool operation_ok = true;
const struct dirent* entry;
while ((entry = readdir(dir)) != NULL) {
if (strcmp(entry->d_name, ".") == 0 || strcmp(entry->d_name, "..") == 0)
continue;
struct stat st;
if (fstatat(dirfd, entry->d_name, &st, AT_SYMLINK_NOFOLLOW) != 0) {
if (errno != ENOENT)
operation_ok = false;
continue;
}
if (S_ISDIR(st.st_mode)) {
int childfd = openat(dirfd, entry->d_name, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
bool child_removed = false;
if (childfd >= 0) {
child_removed = delay_wipe_dir_fd(childfd);
close(childfd);
} else if (errno != ENOENT) {
operation_ok = false;
}
if (child_removed && unlinkat(dirfd, entry->d_name, AT_REMOVEDIR) != 0 && errno != ENOENT)
operation_ok = false;
} else {
if (unlinkat(dirfd, entry->d_name, 0) != 0 && errno != ENOENT)
operation_ok = false;
}
}
closedir(dir);
return operation_ok;
}
bool delay_updates_prepare(DelayUpdatesContext* context) {
if (!context)
return false;
if (context->prepared)
return true;
int fd = file_open_private_dir(context->staging_root);
if (fd < 0) {
int saved_errno = errno;
char* escaped = output_escape(context->staging_root, false);
log_message(LOG_LEVEL_ERROR, "could not create --delay-updates staging directory '%s': %s",
escaped ? escaped : "<allocation failed>", strerror(saved_errno));
free(escaped);
return false;
}
/* Hold an exclusive advisory lock on the staging directory for the whole
transfer. The staging directory name is fixed, so two simultaneous
delayed transfers to the same destination root would otherwise share it
and destroy each other's staged files. The lock makes the second session
fail cleanly instead of corrupting the first. The lock is released when
the context (and its file descriptor) is destroyed. */
if (flock(fd, LOCK_EX | LOCK_NB) != 0) {
int saved_errno = errno;
close(fd);
if (saved_errno == EWOULDBLOCK || saved_errno == EAGAIN) {
char* escaped = output_escape(context->staging_root, false);
log_message(LOG_LEVEL_ERROR,
"another --delay-updates transfer to '%s' is already in progress; refusing to "
"share the staging directory",
escaped ? escaped : "<allocation failed>");
free(escaped);
} else {
log_message(LOG_LEVEL_ERROR, "could not lock --delay-updates staging directory '%s': %s",
context->staging_root, strerror(saved_errno));
}
return false;
}
context->lock_fd = fd;
/* Only now, with exclusive ownership, wipe leftovers from an interrupted
earlier transfer; this can never race with a live session. */
bool ok = delay_wipe_dir_fd(fd);
if (!ok) {
log_message(LOG_LEVEL_ERROR, "could not clear stale --delay-updates staging files under '%s'",
context->staging_root);
close(context->lock_fd);
context->lock_fd = -1;
return false;
}
context->prepared = true;
return true;
}
bool delay_updates_record(DelayUpdatesContext* context, const char* staged_path,
const char* final_path, const char* file_path) {
if (!context || !staged_path || !final_path || !file_path)
return false;
char* staged_copy = str_dup(staged_path);
char* final_copy = str_dup(final_path);
char* file_copy = str_dup(file_path);
if (!staged_copy || !final_copy || !file_copy) {
free(staged_copy);
free(final_copy);
free(file_copy);
return false;
}
mtx_lock(&context->mutex);
bool ok = true;
if (context->count == context->capacity) {
size_t new_capacity = context->capacity == 0 ? 64 : context->capacity * 2;
if (new_capacity < context->capacity) {
ok = false;
} else {
StagedFileEntry* grown = realloc(context->entries, new_capacity * sizeof(StagedFileEntry));
if (!grown) {
ok = false;
} else {
context->entries = grown;
context->capacity = new_capacity;
}
}
}
if (ok) {
context->entries[context->count].staged_path = staged_copy;
context->entries[context->count].final_path = final_copy;
context->entries[context->count].file_path = file_copy;
context->count++;
}
mtx_unlock(&context->mutex);
if (!ok) {
free(staged_copy);
free(final_copy);
free(file_copy);
}
return ok;
}
/* Move an existing final destination file aside before the staged replacement
is installed. Deferred from stage time so the final destination is not
modified until publication. Mirrors the immediate-mode backup logic. */
static bool delay_publish_backup(const DelayUpdatesContext* context, const Config* config,
const StagedFileEntry* entry) {
bool backup_enabled = config && config->backup && !config->ignore_existing;
if (!backup_enabled)
return true;
const char* backup_suffix = (config && config->suffix) ? config->suffix : "~";
struct stat backup_stat;
if (!file_stat_secure(entry->final_path, &backup_stat))
return true; /* nothing to back up */
char* backup_path = NULL;
if (config->backup_dir) {
char* confined_backup = path_cat(context->root_directory, config->backup_dir);
if (!confined_backup)
return false;
backup_path = path_cat(confined_backup, entry->file_path);
free(confined_backup);
} else {
size_t path_len = strlen(entry->final_path);
size_t suffix_len = strlen(backup_suffix);
if (path_len > SIZE_MAX - suffix_len - 1)
return false;
backup_path = malloc(path_len + suffix_len + 1);
if (backup_path) {
memcpy(backup_path, entry->final_path, path_len);
memcpy(backup_path + path_len, backup_suffix, suffix_len + 1);
}
}
if (!backup_path)
return false;
char* parent_copy = str_dup(backup_path);
if (!parent_copy || !file_ensure_directory_secure(dirname(parent_copy))) {
free(parent_copy);
free(backup_path);
return false;
}
free(parent_copy);
bool ok = file_rename_secure(entry->final_path, backup_path);
free(backup_path);
return ok;
}
static bool delay_publish_entry(DelayUpdatesContext* context, const Config* config,
const StagedFileEntry* entry) {
if (!delay_publish_backup(context, config, entry))
return false;
if (!file_rename_secure(entry->staged_path, entry->final_path)) {
if (errno == EXDEV) {
char* escaped = output_escape(entry->final_path, false);
log_message(LOG_LEVEL_ERROR,
"staging directory is on a different filesystem than the destination; cannot "
"atomically install file (EXDEV): %s",
escaped ? escaped : "<allocation failed>");
free(escaped);
} else {
char* escaped = output_escape(entry->final_path, false);
log_message(LOG_LEVEL_ERROR, "could not install staged file '%s': %s",
escaped ? escaped : "<allocation failed>", strerror(errno));
free(escaped);
}
return false;
}
return true;
}
/* Remove the staging tree (contents plus the directory itself). Returns true
when nothing is left behind (including the case where it never existed). */
static bool delay_updates_remove_staging_tree(DelayUpdatesContext* context) {
int fd = open(context->staging_root, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
if (fd < 0)
return errno == ENOENT;
bool ok = delay_wipe_dir_fd(fd);
if (close(fd) != 0)
ok = false;
if (ok && rmdir(context->staging_root) != 0 && errno != ENOENT)
ok = false;
return ok;
}
bool delay_updates_publish(DelayUpdatesContext* context, const Config* config) {
if (!context)
return false;
mtx_lock(&context->mutex);
bool ok = true;
for (size_t i = 0; i < context->count; i++) {
if (!delay_publish_entry(context, config, &context->entries[i])) {
ok = false;
break;
}
}
mtx_unlock(&context->mutex);
/* Renaming files out of the staging tree leaves the mirrored directories
behind, and a mid-publish failure leaves the remaining staged files.
Remove whatever is left so a later run starts from a clean staging area
and no staged content can linger after a failed publish. If that cleanup
fails, tell the operator: a stale staging directory would otherwise
silently accumulate and make the next transfer's prepare-wipe fail. */
if (!delay_updates_remove_staging_tree(context)) {
log_message(LOG_LEVEL_WARNING,
"could not fully remove --delay-updates staging directory '%s' after publish; a "
"later --delay-updates transfer to this destination will try to clear it",
context->staging_root);
}
return ok;
}
void delay_updates_cleanup(DelayUpdatesContext* context) {
if (!context)
return;
/* Only a context that gained exclusive ownership may touch the shared
staging directory. If prepare never succeeded (e.g. lock contention with
another live session) the directory belongs to that other session and must
be left alone. */
if (!context->prepared)
return;
delay_updates_remove_staging_tree(context);
}
-69
View File
@@ -1,69 +0,0 @@
#ifndef DELAY_UPDATES_H
#define DELAY_UPDATES_H
#include <stdbool.h>
#include <stddef.h>
#include <threads.h>
/* Forward-declared in config.h; full type needed by file_save_to_disk. */
typedef struct Config Config;
/* One staged file awaiting publication. */
typedef struct {
char* staged_path; /* full path inside the staging tree */
char* final_path; /* full final destination path */
char* file_path; /* the file path as received on the wire */
} StagedFileEntry;
/* Receiver-side --delay-updates staging registry. All successfully written
files land under a private staging directory inside the receive root and are
atomically renamed into their final destination only at the very end of the
transfer. A single receiver pipeline (see src/server/receiver_pipeline.h)
has exactly one writer thread, but the registry is still mutex-protected so
the same object can be safely
shared with the publish/cleanup phase that runs after the threads join. */
typedef struct DelayUpdatesContext {
char* root_directory; /* receive root the staging dir lives under */
char* staging_root; /* root_directory/<staging dir name> */
mtx_t mutex;
StagedFileEntry* entries;
size_t count;
size_t capacity;
bool prepared; /* staging dir created, wiped, and exclusively locked */
int lock_fd; /* advisory exclusive flock held on the staging dir, or -1 */
} DelayUpdatesContext;
/* Name of the private staging subdirectory created under the receive root. */
#define DELAY_UPDATES_STAGING_DIR ".fastsync-stage"
/* True when `dir` (ignoring a trailing "/") is the reserved staging directory
name. Used to reject a --backup-dir that would collide with the internal
staging area. */
bool delay_updates_staging_name_conflict(const char* dir);
/* Create an empty staging context rooted below root_directory. Does not touch
the filesystem yet. */
DelayUpdatesContext* delay_updates_context_create(const char* root_directory);
void delay_updates_context_destroy(DelayUpdatesContext* context);
/* Create the private 0700 staging directory (on first call) and wipe any
leftovers from a previously interrupted delayed transfer. Idempotent. */
bool delay_updates_prepare(DelayUpdatesContext* context);
/* Record a fully-written staged file for later publication. Copies all three
paths. Returns false on allocation failure. */
bool delay_updates_record(DelayUpdatesContext* context, const char* staged_path,
const char* final_path, const char* file_path);
/* Atomically rename every staged file into its final destination. Deferred
--backup handling runs immediately before each rename. On any failure the
remaining staged files are removed (best effort); already-published files
are not rolled back. Afterwards the staging tree is removed so a successful
or failed publish leaves no staging leftovers. */
bool delay_updates_publish(DelayUpdatesContext* context, const Config* config);
/* Best-effort removal of every staged file and the staging directory itself.
Safe to call when nothing was staged or after a successful publish. */
void delay_updates_cleanup(DelayUpdatesContext* context);
#endif
+43 -191
View File
@@ -1,6 +1,5 @@
#include "delta.h" #include "delta.h"
#include "log.h" #include "log.h"
#include "protocol.h"
#include <stdint.h> #include <stdint.h>
#include <limits.h> #include <limits.h>
#include <stdlib.h> #include <stdlib.h>
@@ -29,21 +28,12 @@ uint32_t delta_xxhash32(const void* data, uint32_t len) {
return XXH32(data, len, 0); return XXH32(data, len, 0);
} }
uint32_t delta_xxhash32_seeded(const void* data, uint32_t len, uint32_t seed) {
return XXH32(data, len, seed);
}
uint64_t delta_xxhash64(const void* data, size_t len) { uint64_t delta_xxhash64(const void* data, size_t len) {
return XXH64(data, len, 0); return XXH64(data, len, 0);
} }
DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size, DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size,
uint32_t block_size) { uint32_t block_size) {
return delta_signature_create_seeded(old_file_data, old_file_size, block_size, 0);
}
DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_t old_file_size,
uint32_t block_size, uint32_t seed) {
if (old_file_data == NULL || old_file_size == 0 || block_size == 0) if (old_file_data == NULL || old_file_size == 0 || block_size == 0)
return NULL; return NULL;
@@ -53,7 +43,7 @@ DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_
uint32_t block_count = (uint32_t)((old_file_size + block_size - 1) / block_size); uint32_t block_count = (uint32_t)((old_file_size + block_size - 1) / block_size);
DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature)); DeltaSignature* sig = malloc(sizeof(DeltaSignature));
if (!sig) if (!sig)
return NULL; return NULL;
@@ -64,7 +54,7 @@ DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_
free(sig); free(sig);
return NULL; return NULL;
} }
sig->blocks = protocol_alloc((size_t)block_count * sizeof(DeltaBlockSig)); sig->blocks = malloc((size_t)block_count * sizeof(DeltaBlockSig));
if (!sig->blocks) { if (!sig->blocks) {
free(sig); free(sig);
return NULL; return NULL;
@@ -76,7 +66,7 @@ DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_
uint32_t len = uint32_t len =
(uint32_t)((old_file_size - offset < block_size) ? (old_file_size - offset) : block_size); (uint32_t)((old_file_size - offset < block_size) ? (old_file_size - offset) : block_size);
sig->blocks[i].adler32 = delta_adler32(data + offset, len); sig->blocks[i].adler32 = delta_adler32(data + offset, len);
sig->blocks[i].xxhash = delta_xxhash32_seeded(data + offset, len, seed); sig->blocks[i].xxhash = delta_xxhash32(data + offset, len);
} }
return sig; return sig;
@@ -92,7 +82,7 @@ Data* delta_signature_serialize(const DeltaSignature* sig) {
total > SIZE_MAX) total > SIZE_MAX)
return NULL; return NULL;
uint8_t* buf = protocol_alloc((size_t)total); uint8_t* buf = malloc((size_t)total);
if (!buf) if (!buf)
return NULL; return NULL;
@@ -121,7 +111,7 @@ DeltaSignature* delta_signature_deserialize(const Data* data) {
const uint8_t* buf = (const uint8_t*)data->data; const uint8_t* buf = (const uint8_t*)data->data;
size_t pos = 0; size_t pos = 0;
DeltaSignature* sig = protocol_alloc(sizeof(DeltaSignature)); DeltaSignature* sig = malloc(sizeof(DeltaSignature));
if (!sig) if (!sig)
return NULL; return NULL;
@@ -159,7 +149,7 @@ DeltaSignature* delta_signature_deserialize(const Data* data) {
free(sig); free(sig);
return NULL; return NULL;
} }
sig->blocks = protocol_alloc((size_t)blocks_size); sig->blocks = malloc((size_t)blocks_size);
if (!sig->blocks) { if (!sig->blocks) {
free(sig); free(sig);
return NULL; return NULL;
@@ -188,7 +178,7 @@ static bool ensure_capacity(DeltaInstruction** instrs, uint32_t* capacity, uint3
if (*capacity > MAX_DELTA_INSTRUCTIONS / 2) if (*capacity > MAX_DELTA_INSTRUCTIONS / 2)
return false; return false;
uint32_t new_cap = *capacity * 2; uint32_t new_cap = *capacity * 2;
DeltaInstruction* tmp = protocol_realloc(*instrs, (size_t)new_cap * sizeof(DeltaInstruction)); DeltaInstruction* tmp = realloc(*instrs, (size_t)new_cap * sizeof(DeltaInstruction));
if (!tmp) if (!tmp)
return false; return false;
*instrs = tmp; *instrs = tmp;
@@ -205,7 +195,7 @@ static bool flush_literal(DeltaInstruction** instrs, uint32_t* capacity, uint32_
uint32_t lit_len = (uint32_t)(end - start); uint32_t lit_len = (uint32_t)(end - start);
if (!ensure_capacity(instrs, capacity, *count)) if (!ensure_capacity(instrs, capacity, *count))
return false; return false;
uint8_t* lit_data = protocol_alloc(lit_len); uint8_t* lit_data = malloc(lit_len);
if (!lit_data) if (!lit_data)
return false; return false;
memcpy(lit_data, data + start, lit_len); memcpy(lit_data, data + start, lit_len);
@@ -225,126 +215,8 @@ static void free_instructions(DeltaInstruction* instrs, uint32_t count) {
free(instrs); free(instrs);
} }
/* Sentinel meaning "no signature block" in the lookup index chains. Block
* counts are bounded well below UINT32_MAX, so it doubles as a null link. */
#define DELTA_NO_BLOCK UINT32_MAX
/* Avalanche mix for the rolling checksum so blocks do not cluster in the
* bucket table when the weak checksum has little entropy (e.g. all-zero or
* patterned files). */
static uint32_t delta_adler_mix(uint32_t h) {
h ^= h >> 16;
h *= 0x7feb352dU;
h ^= h >> 15;
h *= 0x846ca68bU;
h ^= h >> 16;
return h;
}
/* Smallest power of two >= v. v must be non-zero. */
static uint32_t delta_next_pow2(uint32_t v) {
v--;
v |= v >> 1;
v |= v >> 2;
v |= v >> 4;
v |= v >> 8;
v |= v >> 16;
return v + 1;
}
/* Build a hash index over sig->blocks keyed by the (mixed) rolling checksum.
* All blocks sharing an Adler-32 value land in the same bucket; collisions
* are chained through a single contiguous allocation:
*
* [0, bucket_count) heads (first block per bucket)
* [bucket_count, 2*bucket_count) tails (last block per bucket)
* [2*bucket_count, ...) per-block chain links
*
* Blocks are inserted in ascending index order so every bucket chain is
* ordered exactly like the historical linear scan. Returns the base pointer
* (also the heads array) or NULL when no index could be allocated; callers
* then fall back to the linear scan. */
static uint32_t* delta_build_index(const DeltaSignature* sig, uint32_t bucket_count) {
if (sig->block_count == 0 || bucket_count == 0)
return NULL;
size_t entries = (size_t)2 * bucket_count + sig->block_count;
if (entries > SIZE_MAX / sizeof(uint32_t))
return NULL;
uint32_t* index = protocol_alloc(entries * sizeof(uint32_t));
if (!index)
return NULL;
uint32_t* heads = index;
uint32_t* tails = index + bucket_count;
uint32_t* next = index + 2 * bucket_count;
uint32_t mask = bucket_count - 1;
memset(heads, 0xFF, (size_t)bucket_count * sizeof(uint32_t));
memset(tails, 0xFF, (size_t)bucket_count * sizeof(uint32_t));
for (uint32_t j = 0; j < sig->block_count; j++) {
uint32_t b = delta_adler_mix(sig->blocks[j].adler32) & mask;
if (heads[b] == DELTA_NO_BLOCK)
heads[b] = j;
else
next[tails[b]] = j;
tails[b] = j;
next[j] = DELTA_NO_BLOCK;
}
return index;
}
/* Locate the signature block matching the byte window at new_data[i].
*
* Mirrors the original per-window behaviour exactly: only a full block_size
* window can match, candidates are accepted only when the weak (Adler-32) and
* strong (xxHash32) checksums both agree, and the lowest block index wins so
* the emitted op stream is byte-identical to the linear scan. When heads is
* non-NULL the candidate set is reached through the bucket index (expected
* O(1) per window); otherwise an exact linear scan is used. */
static uint32_t delta_find_match(const uint8_t* window, uint32_t window_len, uint32_t adler,
bool full_window, const DeltaSignature* sig, const uint32_t* heads,
const uint32_t* next, uint32_t mask, uint32_t seed) {
if (!full_window || sig->block_count == 0)
return DELTA_NO_BLOCK;
if (heads) {
uint32_t b = delta_adler_mix(adler) & mask;
uint32_t window_xxh = 0;
bool have_xxh = false;
for (uint32_t j = heads[b]; j != DELTA_NO_BLOCK; j = next[j]) {
if (sig->blocks[j].adler32 != adler)
continue;
if (!have_xxh) {
window_xxh = delta_xxhash32_seeded(window, window_len, seed);
have_xxh = true;
}
if (window_xxh == sig->blocks[j].xxhash)
return j;
}
return DELTA_NO_BLOCK;
}
/* Fallback used when the index could not be allocated. */
for (uint32_t j = 0; j < sig->block_count; j++) {
if (sig->blocks[j].adler32 == adler) {
uint32_t window_xxh = delta_xxhash32_seeded(window, window_len, seed);
if (window_xxh == sig->blocks[j].xxhash)
return j;
}
}
return DELTA_NO_BLOCK;
}
Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig, Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig,
uint32_t block_size) { uint32_t block_size) {
return delta_compute_seeded(new_file_data, new_file_size, sig, block_size, 0);
}
Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
const DeltaSignature* sig, uint32_t block_size, uint32_t seed) {
if (!new_file_data || !sig || !sig->blocks || new_file_size == 0 || block_size == 0 || if (!new_file_data || !sig || !sig->blocks || new_file_size == 0 || block_size == 0 ||
block_size > DELTA_BLOCK_SIZE_MAX || sig->block_size != block_size) block_size > DELTA_BLOCK_SIZE_MAX || sig->block_size != block_size)
return NULL; return NULL;
@@ -353,29 +225,10 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
uint32_t capacity = 64; uint32_t capacity = 64;
uint32_t count = 0; uint32_t count = 0;
DeltaInstruction* instrs = protocol_alloc((size_t)capacity * sizeof(DeltaInstruction)); DeltaInstruction* instrs = malloc((size_t)capacity * sizeof(DeltaInstruction));
if (!instrs) if (!instrs)
return NULL; return NULL;
/* Build a one-time bucket index over the signature blocks keyed by the weak
* checksum. This turns the per-byte-window candidate lookup from an
* O(block_count) linear scan into an expected O(1) probe, which dominates
* the cost for large mostly-matching files (the diff steps one byte at a
* time through changed regions). On allocation failure the probe falls back
* to the original linear scan, so behaviour is unchanged under memory
* pressure. */
uint32_t* index = NULL;
const uint32_t* chain_next = NULL;
uint32_t mask = 0;
if (sig->block_count > 0) {
uint32_t bucket_count = delta_next_pow2(sig->block_count);
index = delta_build_index(sig, bucket_count);
if (index) {
chain_next = index + 2 * bucket_count;
mask = bucket_count - 1;
}
}
uint64_t literal_start = 0; uint64_t literal_start = 0;
bool has_literal = false; bool has_literal = false;
@@ -410,32 +263,34 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
} }
bool matched = false; bool matched = false;
uint32_t match_block = delta_find_match(new_data + i, window_len, adler, full_window, sig, for (uint32_t j = 0; j < sig->block_count; j++) {
index, chain_next, mask, seed); if (adler == sig->blocks[j].adler32 && full_window) {
if (match_block != DELTA_NO_BLOCK) { uint32_t xxh = delta_xxhash32(new_data + i, window_len);
if (has_literal) { if (xxh == sig->blocks[j].xxhash) {
if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, i)) { if (has_literal) {
free_instructions(instrs, count); if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, i)) {
free(index); free_instructions(instrs, count);
return NULL; return NULL;
}
has_literal = false;
}
if (!ensure_capacity(&instrs, &capacity, count)) {
free_instructions(instrs, count);
return NULL;
}
instrs[count].type = DELTA_INSTR_BLOCK_MATCH;
instrs[count].match.block_index = j;
instrs[count].match.block_offset = 0;
instrs[count].match.length = window_len;
count++;
i += window_len;
rolling_valid = false;
matched = true;
break;
} }
has_literal = false;
} }
if (!ensure_capacity(&instrs, &capacity, count)) {
free_instructions(instrs, count);
free(index);
return NULL;
}
instrs[count].type = DELTA_INSTR_BLOCK_MATCH;
instrs[count].match.block_index = match_block;
instrs[count].match.block_offset = 0;
instrs[count].match.length = window_len;
count++;
i += window_len;
rolling_valid = false;
matched = true;
} }
if (!matched) { if (!matched) {
@@ -447,8 +302,6 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
} }
} }
free(index);
if (has_literal) { if (has_literal) {
if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, new_file_size)) { if (!flush_literal(&instrs, &capacity, &count, new_data, literal_start, new_file_size)) {
free_instructions(instrs, count); free_instructions(instrs, count);
@@ -456,7 +309,7 @@ Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
} }
} }
Delta* delta = protocol_alloc(sizeof(Delta)); Delta* delta = malloc(sizeof(Delta));
if (!delta) { if (!delta) {
free_instructions(instrs, count); free_instructions(instrs, count);
return NULL; return NULL;
@@ -502,7 +355,7 @@ Data* delta_serialize(const Delta* delta) {
if (delta->delta_size > UINT64_MAX - header_size || header_size + delta->delta_size > SIZE_MAX) if (delta->delta_size > UINT64_MAX - header_size || header_size + delta->delta_size > SIZE_MAX)
return NULL; return NULL;
uint64_t total = header_size + delta->delta_size; uint64_t total = header_size + delta->delta_size;
uint8_t* buf = protocol_alloc((size_t)total); uint8_t* buf = malloc((size_t)total);
if (!buf) if (!buf)
return NULL; return NULL;
@@ -542,7 +395,7 @@ Delta* delta_deserialize(const Data* data) {
const uint8_t* buf = (const uint8_t*)data->data; const uint8_t* buf = (const uint8_t*)data->data;
size_t pos = 0; size_t pos = 0;
Delta* delta = protocol_alloc(sizeof(Delta)); Delta* delta = malloc(sizeof(Delta));
if (!delta) if (!delta)
return NULL; return NULL;
@@ -559,10 +412,9 @@ Delta* delta_deserialize(const Data* data) {
return NULL; return NULL;
} }
delta->instructions = delta->instructions = delta->instruction_count == 0
delta->instruction_count == 0 ? NULL
? NULL : malloc((size_t)delta->instruction_count * sizeof(DeltaInstruction));
: protocol_alloc((size_t)delta->instruction_count * sizeof(DeltaInstruction));
if (delta->instruction_count > 0 && !delta->instructions) { if (delta->instruction_count > 0 && !delta->instructions) {
free(delta); free(delta);
return NULL; return NULL;
@@ -613,7 +465,7 @@ Delta* delta_deserialize(const Data* data) {
free(delta); free(delta);
return NULL; return NULL;
} }
delta->instructions[i].literal.data = protocol_alloc(lit_len ? lit_len : 1); delta->instructions[i].literal.data = malloc(lit_len ? lit_len : 1);
if (!delta->instructions[i].literal.data) { if (!delta->instructions[i].literal.data) {
log_message(LOG_LEVEL_ERROR, "Failed to allocate %u bytes for literal data", lit_len); log_message(LOG_LEVEL_ERROR, "Failed to allocate %u bytes for literal data", lit_len);
free_instructions(delta->instructions, i); free_instructions(delta->instructions, i);
@@ -640,7 +492,7 @@ void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta,
delta->new_file_size > DELTA_MAX_FILE_SIZE || delta->new_file_size > SIZE_MAX) delta->new_file_size > DELTA_MAX_FILE_SIZE || delta->new_file_size > SIZE_MAX)
return NULL; return NULL;
void* output = protocol_alloc(delta->new_file_size ? (size_t)delta->new_file_size : 1); void* output = malloc(delta->new_file_size ? (size_t)delta->new_file_size : 1);
if (!output) if (!output)
return NULL; return NULL;
-11
View File
@@ -56,22 +56,12 @@ typedef struct {
DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size, DeltaSignature* delta_signature_create(const void* old_file_data, uint64_t old_file_size,
uint32_t block_size); uint32_t block_size);
/* Seeded equivalent of delta_signature_create: the per-block strong (xxHash32)
* checksum uses `seed` (the low 32 bits of --checksum-seed). Passing seed 0 is
* identical to the unseeded function. */
DeltaSignature* delta_signature_create_seeded(const void* old_file_data, uint64_t old_file_size,
uint32_t block_size, uint32_t seed);
Data* delta_signature_serialize(const DeltaSignature* sig); Data* delta_signature_serialize(const DeltaSignature* sig);
DeltaSignature* delta_signature_deserialize(const Data* data); DeltaSignature* delta_signature_deserialize(const Data* data);
void delta_signature_destroy(DeltaSignature* sig); void delta_signature_destroy(DeltaSignature* sig);
Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig, Delta* delta_compute(const void* new_file_data, uint64_t new_file_size, const DeltaSignature* sig,
uint32_t block_size); uint32_t block_size);
/* Seeded equivalent of delta_compute: the per-window strong (xxHash32) check
* uses `seed` (the low 32 bits of --checksum-seed). The receiver's signature
* must have been built with the same seed for matching. */
Delta* delta_compute_seeded(const void* new_file_data, uint64_t new_file_size,
const DeltaSignature* sig, uint32_t block_size, uint32_t seed);
Data* delta_serialize(const Delta* delta); Data* delta_serialize(const Delta* delta);
Delta* delta_deserialize(const Data* data); Delta* delta_deserialize(const Data* data);
void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta, uint32_t block_size); void* delta_apply(const void* old_data, uint64_t old_size, const Delta* delta, uint32_t block_size);
@@ -82,7 +72,6 @@ bool delta_is_worthwhile(const Delta* delta, uint64_t new_file_size);
uint32_t delta_adler32(const void* data, uint32_t len); uint32_t delta_adler32(const void* data, uint32_t len);
uint32_t delta_xxhash32(const void* data, uint32_t len); uint32_t delta_xxhash32(const void* data, uint32_t len);
uint32_t delta_xxhash32_seeded(const void* data, uint32_t len, uint32_t seed);
uint64_t delta_xxhash64(const void* data, size_t len); uint64_t delta_xxhash64(const void* data, size_t len);
#endif #endif
+71 -1039
View File
File diff suppressed because it is too large Load Diff
+5 -109
View File
@@ -4,7 +4,6 @@
#include "file_send.h" #include "file_send.h"
#include "file_receive.h" #include "file_receive.h"
#include "file_types.h" #include "file_types.h"
#include "checksum.h"
#include <stdbool.h> #include <stdbool.h>
#include <stdint.h> #include <stdint.h>
#include <sys/stat.h> #include <sys/stat.h>
@@ -15,126 +14,23 @@
File* file_create(const char* path); File* file_create(const char* path);
void file_destroy(void* item); void file_destroy(void* item);
bool file_load_data(File* file); bool file_load_data(File* file);
/* Compute the whole-file content digest of `file` with the negotiated bool file_checksum(File* file, uint64_t* checksum);
* --checksum-choice algorithm and --checksum-seed. Writes the digest into
* `out` (capacity `out_capacity`) and its length into `*out_len`. Returns
* false on read/allocation failure or when the digest would not fit. */
bool file_checksum(File* file, ChecksumAlgo algo, uint64_t seed, uint8_t* out, size_t out_capacity,
size_t* out_len);
size_t file_content_to_buffer(File* file); size_t file_content_to_buffer(File* file);
FileMetadata* file_metadata_create(const char* path, const struct stat* stats, bool capture_atime, FileMetadata* file_metadata_create(const struct stat* stats);
bool capture_crtime);
void file_metadata_destroy(void* metadata); void file_metadata_destroy(void* metadata);
/* --open-noatime process-wide sender policy; see file.c. */
void file_set_open_noatime(bool enable);
bool file_get_open_noatime(void);
/* Open `path` read-only for transfer, honouring --open-noatime when set. */
int file_open_for_read(const char* path);
bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size, bool file_write_to_disk(const char* path, const void* data, unsigned long long data_size,
bool inplace, bool sparse); bool inplace, bool sparse);
/* Symlink trust-boundary helpers (Phase 4, symlink wave). --munge-links /* A configured fd without a canonical identity deliberately rejects paths. */
* sender-side marker: every transmitted symlink target is prefixed with this bool file_set_authorized_root(int fd, const char* canonical_path);
* while the flag is on; the receiver strips it to restore the real target. */
#define SYMLINK_MUNGE_PREFIX "#SYMLINK/"
char* file_symlink_munge(const char* target);
/* True when a lexical target is relative and contains no ".." component, so it
* can never escape the receive root once created beneath it. */
bool file_symlink_target_contained(const char* target);
/* Strip a leading SYMLINK_MUNGE_PREFIX from `target` (mutable, in place);
* returns true when a marker was removed. */
bool file_symlink_unmunge(char* target);
/* Create a symlink at `path` -> `target`, confined below the authorized root
* (O_NOFOLLOW parent walk, symlinkat; the target is never followed). Returns
* false when a directory already occupies `path`. */
bool file_symlink_at_secure(const char* path, const char* target);
/* --keep-dirlinks (-K) receiver process-wide policy: allow an in-root existing
* symlink-to-directory to be followed as a directory. */
void file_set_keep_dirlinks(bool enable);
/* --trust-sender receiver process-wide policy (Phase 5). When set, the
* receiver trusts that the sender already produced a clean file list and skips
* its own redundant up-front re-validation of incoming paths (the empty/".."
* rejection and the escaping-symlink-target containment). The low-level
* fd-relative confinement primitives below are deliberately NOT disabled by
* this flag, so a hostile sender still cannot escape the authorized root. */
void file_set_trust_sender(bool enable);
bool file_get_trust_sender(void);
/* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */ /* Secure path/filesystem primitives (symlink-safe, O_NOFOLLOW, root-confined). */
bool file_path_exists_secure(const char* path); bool file_path_exists_secure(const char* path);
bool file_stat_secure(const char* path, struct stat* st); bool file_stat_secure(const char* path, struct stat* st);
bool file_destination_is_newer_secure(const char* path, const FileMetadata* metadata);
int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs); int file_open_secure_parent(const char* path, char** leaf_out, bool create_dirs);
bool file_ensure_directory_secure(const char* path); bool file_ensure_directory_secure(const char* path);
bool file_directory_exists_secure(const char* path);
bool file_rename_secure(const char* old_path, const char* new_path); bool file_rename_secure(const char* old_path, const char* new_path);
/* Remove the whole directory tree at `path` (confined, symlink-safe). Used by
--force to clear a non-empty destination directory that blocks an incoming
regular file. See the .c for the exact success semantics. */
bool file_remove_tree_secure(const char* path);
/* Open a private 0700 directory (creating it on demand) that must live below
the authorized root. Used for the --temp-dir scratch directory and the
--delay-updates staging directory. */
int file_open_private_dir(const char* dir_path);
/* The file_to_disk_secure* variants write a temporary copy in the destination
directory and atomically rename it over `path`. temp_dir is an absolute,
root-confined scratch directory (already validated by the caller): when it
is non-NULL the temporary copy is instead created there (with a name unique
across the whole scratch directory) and atomically renamed into the
destination directory once fully written and fsynced. A rename across
filesystems (EXDEV) fails the write with an error; the file is never
silently copied into place. Pass NULL for the historical same-directory
behavior. --inplace writes never use temp_dir. */
bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size, bool file_to_disk_secure(const char* path, const void* data, unsigned long long data_size,
bool inplace, bool sparse, bool preallocate, const FileMetadata* metadata, bool inplace, bool sparse, const FileMetadata* metadata);
bool preserve_executability, const char* temp_dir);
bool file_to_disk_secure_with_fsync(const char* path, const void* data,
unsigned long long data_size, bool inplace, bool sparse,
bool preallocate, const FileMetadata* metadata,
bool preserve_executability, bool use_fsync,
const char* temp_dir);
/* With update enabled, an existing newer destination is left untouched. The
check is descriptor-based for inplace writes; atomic replacement still has
an unavoidable final rename race without filesystem locking. */
bool file_to_disk_secure_update(const char* path, const void* data, unsigned long long data_size,
bool inplace, bool sparse, bool preallocate,
const FileMetadata* metadata, bool preserve_executability,
const char* temp_dir);
bool file_to_disk_secure_no_replace(const char* path, const void* data,
unsigned long long data_size, bool sparse, bool preallocate,
const FileMetadata* metadata, bool preserve_executability,
const char* temp_dir);
/* Receiver write-path variant that also applies per-file xattrs (-X/-A) and the
* --fake-super stat xattr fd-relative before the final rename. `update` /
* `no_replace` / `use_fsync` mirror the plain wrappers above; `keep_partial`
* enables --partial best-effort retention of a failed write's temp. */
bool file_to_disk_secure_attrs(const char* path, const void* data, unsigned long long data_size,
bool inplace, bool sparse, bool preallocate,
const FileMetadata* metadata, bool preserve_executability,
bool update, bool no_replace, bool use_fsync,
const FileXattrList* xattrs, bool fake_super, bool keep_partial,
const char* temp_dir);
/* Atomic --link-dest install: replace `path` with a hard link to `basis_path`
(via a temp name + rename); fall back to a byte-identical local copy from
`data` when the link is impossible (EXDEV/EPERM/unsupported filesystem).
`metadata` is applied only on the copy fallback. `preallocate` applies to
that copy fallback only (a hard-linked file shares the basis inode and is
never re-allocated). */
bool file_to_disk_secure_link(const char* path, const char* basis_path, const void* data,
unsigned long long data_size, bool preallocate,
const FileMetadata* metadata, bool preserve_executability,
bool use_fsync, const char* temp_dir);
/* Like file_to_disk_secure_link, but the byte-copy fallback also applies the
* per-file xattrs (-X/-A) and --fake-super stat xattr (fd-relative). On a
* successful hard link no attributes are applied (the shared inode already
* carries the basis's). */
bool file_to_disk_secure_link_attrs(const char* path, const char* basis_path, const void* data,
unsigned long long data_size, bool preallocate,
const FileMetadata* metadata, bool preserve_executability,
bool use_fsync, const FileXattrList* xattrs, bool fake_super,
const char* temp_dir);
#endif #endif
-242
View File
@@ -1,242 +0,0 @@
#include "file_list.h"
#include "log.h"
#include "utils.h"
#include <errno.h>
#include <limits.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
typedef struct {
char** items;
int count;
int capacity;
} StringList;
static void string_list_destroy(StringList* list) {
if (!list)
return;
for (int i = 0; i < list->count; i++)
free(list->items[i]);
free(list->items);
}
static bool string_list_add(StringList* list, const char* text) {
if (list->count == list->capacity) {
if (list->capacity > INT_MAX / 2)
return false;
int new_cap = list->capacity > 0 ? list->capacity * 2 : 16;
char** grown = realloc(list->items, (size_t)new_cap * sizeof(char*));
if (!grown)
return false;
list->items = grown;
list->capacity = new_cap;
}
list->items[list->count] = str_dup(text);
if (!list->items[list->count])
return false;
list->count++;
return true;
}
/* Validate and normalize one entry. Returns:
* 1 -> added to `out`
* 0 -> blank entry, skip
* -1 -> invalid (message set in `err`)
* `strip_line_endings` trims a trailing CR/LF (line mode only); NUL mode keeps
* the entry bytes verbatim so names ending in CR/LF survive. */
static int normalize_entry(const char* raw, size_t len, bool strip_line_endings, StringList* out,
char* err, size_t err_size) {
if (strip_line_endings) {
while (len > 0 && (raw[len - 1] == '\n' || raw[len - 1] == '\r'))
len--;
}
if (len == 0)
return 0;
if (raw[0] == '/') {
int print_len = len > (size_t)INT_MAX ? INT_MAX : (int)len;
snprintf(err, err_size, "absolute path entries are not allowed: '%.*s'", print_len, raw);
return -1;
}
/* Reject NUL bytes inside a token defensively. In NUL-delimited mode the
* delimiter itself is the final byte and is expected; in line mode any NUL is
* embedded garbage (strlen-based parsing would otherwise silently truncate). */
size_t scan_len = strip_line_endings ? len : len - 1;
if (memchr(raw, '\0', scan_len)) {
snprintf(err, err_size, "entry contains an embedded NUL byte");
return -1;
}
char* dup = malloc(len + 1);
if (!dup) {
snprintf(err, err_size, "memory allocation failed");
return -1;
}
memcpy(dup, raw, len);
dup[len] = '\0';
/* Rebuild the path token-by-token: skip '.' and empty segments, reject '..'. */
size_t out_len = 0;
for (const char* part = dup;;) {
const char* slash = strchr(part, '/');
size_t part_len = slash ? (size_t)(slash - part) : strlen(part);
if (part_len == 1 && part[0] == '.') {
/* skip "." segment */
} else if (part_len == 2 && part[0] == '.' && part[1] == '.') {
snprintf(err, err_size, "path traversal entry is not allowed: '%s'", dup);
free(dup);
return -1;
} else if (part_len > 0) {
if (out_len > 0)
dup[out_len++] = '/';
memmove(dup + out_len, part, part_len);
out_len += part_len;
}
if (!slash)
break;
part = slash + 1;
}
dup[out_len] = '\0';
int result;
if (out_len == 0) {
/* "." / "./" lists the source root: the whole tree is transferred. */
result = string_list_add(out, "") ? 1 : -1;
if (result < 0)
snprintf(err, err_size, "memory allocation failed");
} else {
result = string_list_add(out, dup) ? 1 : -1;
if (result < 0)
snprintf(err, err_size, "memory allocation failed");
}
free(dup);
return result;
}
/* Build the membership index over the exact entries only. `file_list_affects`
combines the exact/descendant lookups with a walk of the query's own ancestor
prefixes, so no ancestor prefix is ever materialized as a copy and the index
stays O(entry count) memory regardless of path depth. An empty entry (the
source root) sets whole_tree and short-circuits every query. */
static bool file_list_index_build(FileListSet* set, char* err, size_t err_size) {
if (!path_index_build(&set->index, (const char* const*)set->entries, (size_t)set->count)) {
snprintf(err, err_size, "memory allocation failed");
return false;
}
for (int i = 0; i < set->count; i++) {
if (set->entries[i][0] == '\0') {
set->whole_tree = true;
break;
}
}
return true;
}
static FileListSet* string_list_to_set(StringList* raw, char* err, size_t err_size) {
FileListSet* set = calloc(1, sizeof(FileListSet));
if (!set) {
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
set->count = raw->count;
set->entries = raw->items;
raw->items = NULL;
raw->count = 0;
if (!file_list_index_build(set, err, err_size)) {
file_list_destroy(set);
return NULL;
}
return set;
}
FileListSet* file_list_load(const char* path, bool null_separated, char* err, size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
if (!path || !*path) {
snprintf(err, err_size, "no file given");
return NULL;
}
FILE* fp = fopen(path, "r");
if (!fp) {
char* escaped = output_escape(path, false);
snprintf(err, err_size, "could not open '%s': %s", escaped ? escaped : path, strerror(errno));
free(escaped);
return NULL;
}
StringList raw = {0};
char* line = NULL;
size_t line_cap = 0;
bool ok = true;
char delim = null_separated ? '\0' : '\n';
while (ok) {
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, delim, UTILS_MAX_LINE_LEN);
if (n < 0) {
if (errno == EFBIG)
snprintf(err, err_size, "entry in file list exceeds %d bytes", (int)UTILS_MAX_LINE_LEN);
else
snprintf(err, err_size, "error reading file list: %s", strerror(errno));
ok = false;
break;
}
if (n == 0)
break;
int r = normalize_entry(line, (size_t)n, !null_separated, &raw, err, err_size);
if (r < 0) {
ok = false;
break;
}
}
free(line);
fclose(fp);
if (!ok) {
string_list_destroy(&raw);
return NULL;
}
FileListSet* set = string_list_to_set(&raw, err, err_size);
if (!set)
string_list_destroy(&raw);
return set;
}
void file_list_destroy(FileListSet* set) {
if (!set)
return;
path_index_free(&set->index);
for (int i = 0; i < set->count; i++)
free(set->entries[i]);
free(set->entries);
free(set);
}
bool file_list_affects(const FileListSet* set, const char* rel) {
if (!set)
return true;
if (!rel)
return false;
if (set->whole_tree)
return true; /* whole tree listed */
/* An exact entry match means `rel` itself is listed. */
if (path_index_contains(&set->index, rel))
return true;
/* Otherwise `rel` is affected when a listed entry is an ancestor directory of
it; walk rel's own directory prefixes (which preserve path-boundary
semantics) and test each for an exact entry. No prefixes are stored. */
size_t len = strlen(rel);
while (len > 0) {
const char* slash = NULL;
for (size_t i = len; i-- > 0;) {
if (rel[i] == '/') {
slash = rel + i;
break;
}
}
if (!slash)
break;
len = (size_t)(slash - rel);
if (path_index_contains_n(&set->index, rel, len))
return true;
}
/* Finally `rel` is affected when it is an ancestor directory of a listed
entry (binary search for the first entry at or after `rel` + '/'). */
return path_index_has_descendant(&set->index, rel);
}
-43
View File
@@ -1,43 +0,0 @@
#ifndef FILE_LIST_H
#define FILE_LIST_H
#include "utils.h"
#include <stdbool.h>
#include <stddef.h>
/* --files-from allow-set. The file lists source paths RELATIVE to the source
* root. A listed regular file is transferred; a listed directory transfers its
* whole subtree (FastSync's recursion is always on). Blank lines are ignored.
*
* Entries are normalized: leading "./" and duplicate "/" are removed, an entry
* of "." means the whole tree, absolute entries and ".." traversal are
* rejected at parse time. The set is immutable and shared read-only across
* scanner worker threads.
*
* Membership is answered from `index`, built once at load time over the exact
* entries only: `index.exact` matches a listed path, the sorted view detects an
* ancestor directory of a listed entry, and `rel`'s own directory prefixes are
* matched against the exact set while descending. No ancestor prefix is stored
* as a separate string, so the index is O(entry count) memory however deep the
* paths are, and each query is O(path length) comparisons. */
typedef struct {
char** entries; /* normalized rel paths; "" means the whole tree */
int count;
PathIndex index;
bool whole_tree; /* an entry of "" lists the source root */
} FileListSet;
/* Load and validate a --files-from file. When `null_separated` (-0/--from0)
* entries are delimited by NUL instead of newlines. Returns NULL with a message
* in `err` on open/validation failure. An empty file yields an empty set
* (nothing is transferred). */
FileListSet* file_list_load(const char* path, bool null_separated, char* err, size_t err_size);
void file_list_destroy(FileListSet* set);
/* True when `rel` (path relative to the source root, "" == root) is a listed
* entry, lives under a listed directory, or is an ancestor directory of a
* listed entry. Used to prune scanning: directories are descended only when
* this returns true, files are transferred only when it returns true. */
bool file_list_affects(const FileListSet* set, const char* rel);
#endif
+212 -2692
View File
File diff suppressed because it is too large Load Diff
+1 -107
View File
@@ -7,115 +7,9 @@
/* Server-side file receive/save path. */ /* Server-side file receive/save path. */
/* Cumulative caps for the deferred directory-time accumulator. The sender may
* legitimately split a large tree across repeated STATUS_DIR_TIMES frames, so a
* per-frame bound is not enough: the receiver must bound the TOTAL it retains
* against a hostile sender. Mirror the delete-manifest limits
* (MAX_MANIFEST_ENTRIES / MAX_MANIFEST_BYTES): the entry count bounds the
* metadata array and the byte budget bounds the concatenated path strings. */
#define MAX_DIR_TIME_ENTRIES (1024 * 1024)
#define MAX_DIR_TIME_BYTES (16ULL * 1024 * 1024)
File* file_receive(const Config* config, int file_descriptor); File* file_receive(const Config* config, int file_descriptor);
File* file_receive_directory(int file_descriptor, const Config* config);
File* file_receive_dir_time(int file_descriptor, const Config* config);
File* file_receive_hardlink(int file_descriptor);
File* file_receive_symlink(int file_descriptor, const Config* config);
File* file_receive_special(int file_descriptor);
bool file_special_rdev_valid(int32_t major, int32_t minor, mode_t mode);
File* receive_incremental_check(int fd, const Config* config, bool* skipped); File* receive_incremental_check(int fd, const Config* config, bool* skipped);
/* Extended variant used by the receiver. `would_transfer` (may be NULL) is set int receive_manifest(int fd, const Config* config, int* next_status);
* true only on the server-contacting --dry-run path when the file is not up to
* date: the receiver has already sent STATUS_DRY_RUN_TRANSFER and returns NULL
* without storing anything. On that path `*skipped` is true for an up-to-date
* (STATUS_OK) file and both flags are false for a genuine error. */
File* receive_incremental_check_ex(int fd, const Config* config, bool* skipped,
bool* would_transfer);
/* P7 Wave D directory-time accumulator. The receiver collects the metadata of
* every directory it creates/receives (STATUS_MKDIR with metadata and/or the
* trailing STATUS_DIR_TIMES frame(s)) and applies the times only at the END of the
* transfer, after all children have been written and after the delete /
* --delay-updates phases have committed (writing or removing a child bumps the
* parent's mtime). -O/--omit-dir-times skips the application entirely. The
* list owns deep copies of the paths and metadata; freed on every path. */
typedef struct {
char** paths; /* owned, destination-relative wire paths */
FileMetadata* entries; /* owned, parallel to paths */
size_t count;
size_t capacity;
size_t bytes; /* cumulative strlen of every retained path */
} DirTimeList;
/* Capture gate shared by the sender-side and receiver-side sinks: directory
* metadata is accumulated only when --times/--metadata is in effect and
* -O/--omit-dir-times does not suppress it. Kept here, next to the accumulator
* it guards, so both call sites express the same condition. */
bool dir_times_should_capture(const Config* config);
void dir_time_list_init(DirTimeList* list);
void dir_time_list_free(DirTimeList* list);
/* Deep-copy one directory's path + metadata into the list. Returns false on
* allocation failure OR when the cumulative entry/byte caps would be exceeded
* (the caller fails the transfer). */
bool dir_time_list_add(DirTimeList* list, const char* wire_path, const FileMetadata* metadata);
/* Apply every accumulated directory's mtime (and atime when captured) beneath
* `root_directory`, confined fd-relative. Best-effort per entry: an absent
* directory (an empty/pruned source dir that was deliberately not created) or a
* non-directory at the path is skipped QUIETLY, an unreachable one with a
* warning, and never fatal. */
void dir_time_list_apply(const DirTimeList* list, const char* root_directory);
/* A received delete-manifest frame: the keep-set (`keeps`, destination-relative
paths the sender transferred/keeps) plus `protected`, destination-relative
prefixes the sender asks the receiver never to delete (paths excluded on the
source, protected at any depth). When --delete-excluded is given the sender
transmits an empty protected list so excluded destination mirrors are treated
as ordinary extras. With --delete-missing-args a third section (`missing`)
carries the destination mirrors of explicitly-listed source entries that do
not exist: each is an exact deletion request, independent of the ordinary
extras walk (never blocked by the protected prefixes) and processed when the
manifest is committed. */
typedef struct DeleteManifest {
ArrayList* keeps;
ArrayList* protected;
ArrayList* missing;
} DeleteManifest;
void delete_manifest_free(DeleteManifest* manifest);
/* Read a delete-manifest frame: keep count + keeps, then protected count +
protected prefixes, then missing count + missing paths (self-delimiting; the
leading STATUS_MANIFEST code has been consumed). Returns an owned
DeleteManifest, or NULL after signalling STATUS_ERROR on a malformed frame. */
DeleteManifest* receive_manifest_entries(int fd);
/* Remove destination entries under config->receive_root_directory that are not
in `manifest` (bounded, all-or-nothing walk; staging-dir, basis-dir and
protected-prefix skips). `--max-delete` and `--force` are honored here. The
caller decides WHEN to run it based on the negotiated delete timing. Returns
false (and the transfer fails) when the deletion cannot be committed. */
bool manifest_delete_extras(const Config* config, DeleteManifest* manifest);
/* --delete-missing-args exact-path deletions: remove each destination mirror
in `manifest->missing` (never blocked by the protected prefixes, staging dir
and basis dirs excluded). A regular file/symlink is unlinked; an empty
directory is removed; a NON-empty directory is removed recursively only when
--delete or --force is in effect, otherwise it is left with a warning (rsync
parity). A missing path is a no-op. Returns false only on a genuine
confinement or I/O error (the run then fails); tolerated per-path cases are
reported and skipped. */
bool manifest_delete_missing_args(const Config* config, DeleteManifest* manifest);
/* Run every deletion family the manifest carries: the --delete-missing-args
exact-path deletions first (user requests are not blocked by exclusion
protection), then the ordinary extras walk when --delete is active. Returns
true when nothing to do or everything committed. */
bool manifest_delete_all(const Config* config, DeleteManifest* manifest);
/* Outcome of a single file_save_to_disk operation. The receiver needs to
distinguish "written" from "skipped" so --remove-source-files can be told
which sources were actually stored. */
typedef enum { FILE_SAVE_ERROR = 0, FILE_SAVE_WRITTEN = 1, FILE_SAVE_SKIPPED = 2 } FileSaveResult;
FileSaveResult file_save_to_disk_full(const char* root_directory, const File* file,
const Config* config);
bool file_save_to_disk(const char* root_directory, const File* file, const Config* config); bool file_save_to_disk(const char* root_directory, const File* file, const Config* config);
#endif #endif
+8 -53
View File
@@ -10,59 +10,28 @@
#include <time.h> #include <time.h>
#include <unistd.h> #include <unistd.h>
#include "charset.h"
#include "compression.h" #include "compression.h"
#include "data.h" #include "data.h"
#include "file.h" #include "file.h"
#include "log.h" #include "log.h"
#include "metadata.h" #include "metadata.h"
#include "protocol.h" #include "protocol.h"
#include "xattr.h"
/* Transmit a device/special node (--devices / --specials) as a STATUS_SPECIAL
* frame: the destination path, the metadata frame (whose mode's S_IFMT bits
* carry the node kind) and the device rdev major/minor. The receiver validates
* the kind and rdev and recreates the node (privilege-gating the mknod). */
bool file_send_special(const File* file, int file_descriptor, bool use_metadata) {
if (!file || !file_wire_path(file))
return false;
if (!send_status(file_descriptor, STATUS_SPECIAL))
return false;
if (!send_wire_str(file_descriptor, file_wire_path(file)))
return false;
if (use_metadata && !metadata_send(file_descriptor, file->metadata))
return false;
int32_t major = file->rdev_major;
int32_t minor = file->rdev_minor;
return send_n_data(file_descriptor, &major, sizeof(major)) &&
send_n_data(file_descriptor, &minor, sizeof(minor));
}
bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata, bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata,
int compression_level, bool send_path) { int compression_level, bool send_path) {
return file_send_single_calls_with_skip(file, file_descriptor, use_metadata, compression_level,
send_path, NULL, -1, 0, false);
}
bool file_send_single_calls_with_skip(File* file, int file_descriptor, bool use_metadata,
int compression_level, bool send_path,
char* const* skip_suffixes, int skip_count,
int compression_threads, bool send_xattrs) {
if (!file || !file->path || !file->data || (file->data->size != 0 && !file->data->data)) if (!file || !file->path || !file->data || (file->data->size != 0 && !file->data->data))
return false; return false;
const Data* data_to_send = file->data; const Data* data_to_send = file->data;
Data* compressed_data = NULL; Data* compressed_data = NULL;
if (compression_level > 0 && if (compression_level > 0 && !compression_should_skip(file->path)) {
!compression_should_skip_with_suffixes(file->path, skip_suffixes, skip_count)) { compressed_data = data_compress(file->data, compression_level);
compressed_data =
data_compress_with_threads(file->data, compression_level, compression_threads);
if (compressed_data == NULL) { if (compressed_data == NULL) {
log_message(LOG_LEVEL_ERROR, "Failed to compress file data"); log_message(LOG_LEVEL_ERROR, "Failed to compress file data");
return false; return false;
} }
data_to_send = compressed_data; data_to_send = compressed_data;
} }
if (send_path && !send_wire_str(file_descriptor, file_wire_path(file))) { if (send_path && !send_str(file_descriptor, file->path)) {
data_destroy(compressed_data); data_destroy(compressed_data);
return false; return false;
} }
@@ -70,10 +39,6 @@ bool file_send_single_calls_with_skip(File* file, int file_descriptor, bool use_
data_destroy(compressed_data); data_destroy(compressed_data);
return false; return false;
} }
if (send_xattrs && !xattr_send(file_descriptor, file ? file->xattrs : NULL)) {
data_destroy(compressed_data);
return false;
}
if (!send_data(file_descriptor, data_to_send)) { if (!send_data(file_descriptor, data_to_send)) {
data_destroy(compressed_data); data_destroy(compressed_data);
return false; return false;
@@ -84,28 +49,18 @@ bool file_send_single_calls_with_skip(File* file, int file_descriptor, bool use_
bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level, bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level,
bool send_path) { bool send_path) {
return file_send_sendfile_with_skip(file, file_descriptor, use_metadata, compression_level,
send_path, NULL, -1, 0, false);
}
bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_metadata,
int compression_level, bool send_path, char* const* skip_suffixes,
int skip_count, int compression_threads, bool send_xattrs) {
if (!file || !file->path || !file->data) if (!file || !file->path || !file->data)
return false; return false;
if (compression_level > 0) if (compression_level > 0)
return file_send_single_calls_with_skip(file, file_descriptor, use_metadata, compression_level, return file_send_single_calls(file, file_descriptor, use_metadata, compression_level,
send_path, skip_suffixes, skip_count, send_path);
compression_threads, send_xattrs);
if (send_path && !send_wire_str(file_descriptor, file_wire_path(file))) if (send_path && !send_str(file_descriptor, file->path))
return false; return false;
if (use_metadata && !metadata_send(file_descriptor, file->metadata)) if (use_metadata && !metadata_send(file_descriptor, file->metadata))
return false; return false;
if (send_xattrs && !xattr_send(file_descriptor, file ? file->xattrs : NULL))
return false;
int fd = file_open_for_read(file->path); int fd = open(file->path, O_RDONLY);
if (fd == -1) { if (fd == -1) {
log_perror("Could not open file for sendfile"); log_perror("Could not open file for sendfile");
return false; return false;
@@ -145,7 +100,7 @@ bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_meta
off_t offset = 0; off_t offset = 0;
struct timespec deadline; struct timespec deadline;
clock_gettime(CLOCK_MONOTONIC, &deadline); clock_gettime(CLOCK_MONOTONIC, &deadline);
deadline.tv_sec += protocol_get_io_timeout_sec(); deadline.tv_sec += 60;
while ((unsigned long long)offset < file_size) { while ((unsigned long long)offset < file_size) {
struct timespec now; struct timespec now;
clock_gettime(CLOCK_MONOTONIC, &now); clock_gettime(CLOCK_MONOTONIC, &now);
-8
View File
@@ -6,17 +6,9 @@
/* Client-side file send path. */ /* Client-side file send path. */
bool file_send_special(const File* file, int file_descriptor, bool use_metadata);
bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata, bool file_send_single_calls(File* file, int file_descriptor, bool use_metadata,
int compression_level, bool send_path); int compression_level, bool send_path);
bool file_send_single_calls_with_skip(File* file, int file_descriptor, bool use_metadata,
int compression_level, bool send_path,
char* const* skip_suffixes, int skip_count,
int compression_threads, bool send_xattrs);
bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level, bool file_send_sendfile(File* file, int file_descriptor, bool use_metadata, int compression_level,
bool send_path); bool send_path);
bool file_send_sendfile_with_skip(File* file, int file_descriptor, bool use_metadata,
int compression_level, bool send_path, char* const* skip_suffixes,
int skip_count, int compression_threads, bool send_xattrs);
#endif #endif
+175 -34
View File
@@ -1,7 +1,129 @@
#include <errno.h> #include <errno.h>
#include <fcntl.h>
#include <libgen.h>
#include <limits.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <unistd.h> #include <unistd.h>
#include "file_store.h" #include "file_store.h"
#include "metadata.h"
#include "utils.h"
static int authorized_root_fd = -1;
static char* authorized_root_path;
static bool path_is_within_root(const char* root, const char* path) {
size_t root_length = strlen(root);
return strncmp(root, path, root_length) == 0 &&
(path[root_length] == '\0' || path[root_length] == '/');
}
bool file_store_set_authorized_root(int fd, const char* canonical_path) {
char* new_path = canonical_path ? str_dup(canonical_path) : NULL;
if (canonical_path && !new_path) {
authorized_root_fd = -1;
free(authorized_root_path);
authorized_root_path = NULL;
return false;
}
free(authorized_root_path);
authorized_root_path = new_path;
authorized_root_fd = fd;
return true;
}
int file_store_open_secure_parent(const char* path, char** leaf_out) {
char* copy = str_dup(path);
if (!copy)
return -1;
char* parent = dirname(copy);
const char* slash = strrchr(path, '/');
char* leaf = str_dup(slash ? slash + 1 : path);
if (!leaf) {
free(copy);
return -1;
}
int fd;
if (authorized_root_fd >= 0) {
if (!authorized_root_path || path[0] != '/' ||
!path_is_within_root(authorized_root_path, path)) {
free(copy);
free(leaf);
return -1;
}
fd = dup(authorized_root_fd);
if (fd < 0) {
free(copy);
free(leaf);
return -1;
}
size_t root_length = strlen(authorized_root_path);
char* relative = str_dup(path + root_length);
if (!relative) {
free(copy);
free(leaf);
close(fd);
return -1;
}
free(copy);
copy = relative;
parent = dirname(copy);
} else {
fd = (parent[0] == '/') ? open("/", O_RDONLY | O_DIRECTORY | O_CLOEXEC)
: open(".", O_RDONLY | O_DIRECTORY | O_CLOEXEC);
}
if (fd < 0) {
free(copy);
free(leaf);
return -1;
}
char* save = NULL;
char* component = strtok_r(parent, "/", &save);
while (component) {
if (strcmp(component, "..") == 0) {
close(fd);
free(copy);
free(leaf);
return -1;
}
if (strcmp(component, ".") != 0) {
int next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
if (next < 0 && errno == ENOENT) {
if (mkdirat(fd, component, 0755) == 0 || errno == EEXIST)
next = openat(fd, component, O_RDONLY | O_DIRECTORY | O_NOFOLLOW | O_CLOEXEC);
}
if (next < 0) {
close(fd);
free(copy);
free(leaf);
return -1;
}
close(fd);
fd = next;
}
component = strtok_r(NULL, "/", &save);
}
free(copy);
*leaf_out = leaf;
return fd;
}
bool file_store_rename_secure(const char* old_path, const char* new_path) {
char *old_leaf = NULL, *new_leaf = NULL;
int old_parent = file_store_open_secure_parent(old_path, &old_leaf);
int new_parent = file_store_open_secure_parent(new_path, &new_leaf);
bool ok = old_parent >= 0 && new_parent >= 0 &&
renameat(old_parent, old_leaf, new_parent, new_leaf) == 0;
if (old_parent >= 0)
close(old_parent);
if (new_parent >= 0)
close(new_parent);
free(old_leaf);
free(new_leaf);
return ok;
}
static bool write_all(int fd, const void* data, unsigned long long size) { static bool write_all(int fd, const void* data, unsigned long long size) {
const unsigned char* p = data; const unsigned char* p = data;
@@ -17,40 +139,59 @@ static bool write_all(int fd, const void* data, unsigned long long size) {
return true; return true;
} }
/* A run of NUL bytes at least this long is emitted as a hole (lseek) rather bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size,
* than written, so the resulting file is genuinely sparse on the filesystem. */ bool inplace, bool sparse, const FileMetadata* metadata) {
#define SPARSE_HOLE_MIN 4096U char* leaf = NULL;
int dirfd = file_store_open_secure_parent(path, &leaf);
/* Sparse-aware writer (--sparse/-S). Walks `data`; any all-zero run of at if (dirfd < 0)
* least SPARSE_HOLE_MIN bytes is skipped with lseek(SEEK_CUR) so the block is return false;
* never allocated (a real hole on the destination); every other byte is written int fd = -1;
* normally. The file is pre-sized with ftruncate by the callers before this bool ok = false;
* runs, so holes are guaranteed and the offset bookkeeping stays correct if (inplace) {
* (each lseek advances the fd offset exactly as a write of that many bytes fd = openat(dirfd, leaf, O_WRONLY | O_CREAT | O_TRUNC | O_CLOEXEC | O_NOFOLLOW, 0644);
* would). After the final run, ftruncate(size) guarantees the logical size is if (fd >= 0) {
* exactly `size` even when the tail was a hole. The full file image is in if (!sparse || data_size == 0 || ftruncate(fd, (off_t)data_size) == 0)
* memory, so no wire change is needed. Returns false on I/O error. */ ok = write_all(fd, data, data_size);
bool file_store_write_sparse(int fd, const unsigned char* data, unsigned long long size) { if (ok && metadata)
unsigned long long i = 0; ok = file_restore_metadata_fd(fd, metadata);
while (i < size) {
if (data[i] == 0) {
unsigned long long run_start = i;
while (i < size && data[i] == 0)
i++;
unsigned long long run_len = i - run_start;
if (run_len >= SPARSE_HOLE_MIN) {
if (lseek(fd, (off_t)run_len, SEEK_CUR) < 0)
return false;
} else if (!write_all(fd, data + run_start, run_len)) {
return false;
}
} else {
unsigned long long run_start = i;
while (i < size && data[i] != 0)
i++;
if (!write_all(fd, data + run_start, i - run_start))
return false;
} }
} else {
int tmp_size = snprintf(NULL, 0, ".%s.tmp.%ld.%u", leaf, (long)getpid(), 99U);
if (tmp_size < 0) {
close(dirfd);
free(leaf);
return false;
}
char* tmp = malloc((size_t)tmp_size + 1);
if (!tmp) {
close(dirfd);
free(leaf);
return false;
}
for (unsigned int i = 0; i < 100 && !ok; ++i) {
snprintf(tmp, (size_t)tmp_size + 1, ".%s.tmp.%ld.%u", leaf, (long)getpid(), i);
fd = openat(dirfd, tmp, O_WRONLY | O_CREAT | O_EXCL | O_CLOEXEC | O_NOFOLLOW, 0600);
if (fd < 0)
continue;
if (sparse && data_size > 0)
ok = ftruncate(fd, (off_t)data_size) == 0;
if (ok || (!sparse || data_size == 0))
ok = write_all(fd, data, data_size);
if (ok && metadata)
ok = file_restore_metadata_fd(fd, metadata);
if (close(fd) != 0)
ok = false;
fd = -1;
if (ok && renameat(dirfd, tmp, dirfd, leaf) != 0)
ok = false;
if (!ok)
unlinkat(dirfd, tmp, 0);
}
free(tmp);
} }
return ftruncate(fd, (off_t)size) == 0; if (fd >= 0)
close(fd);
close(dirfd);
free(leaf);
return ok;
} }
+6 -7
View File
@@ -1,14 +1,13 @@
#ifndef FILE_STORE_H #ifndef FILE_STORE_H
#define FILE_STORE_H #define FILE_STORE_H
#include "file.h"
#include <stdbool.h> #include <stdbool.h>
/* Sparse-aware write (--sparse/-S): every all-zero run of at least bool file_store_set_authorized_root(int fd, const char* canonical_path);
* SPARSE_HOLE_MIN bytes is skipped with lseek(SEEK_CUR) so it becomes a real int file_store_open_secure_parent(const char* path, char** leaf_out);
* hole; every other byte is written. The caller pre-sizes the file with bool file_store_rename_secure(const char* old_path, const char* new_path);
* ftruncate; this function also ftruncate()s to `size` at the end so a trailing bool file_store_write_secure(const char* path, const void* data, unsigned long long data_size,
* hole keeps the exact logical length. Shared by the file_store and file write bool inplace, bool sparse, const FileMetadata* metadata);
* paths. Returns false on write/lseek/ftruncate error. */
bool file_store_write_sparse(int fd, const unsigned char* data, unsigned long long size);
#endif #endif
-72
View File
@@ -2,7 +2,6 @@
#define FILE_TYPES_H #define FILE_TYPES_H
#include "data.h" #include "data.h"
#include "xattr.h"
#include <stdbool.h> #include <stdbool.h>
#include <sys/stat.h> #include <sys/stat.h>
@@ -14,84 +13,13 @@ typedef struct {
gid_t gid; gid_t gid;
time_t mtime_sec; time_t mtime_sec;
long mtime_nsec; long mtime_nsec;
/* Optional access time (-U/--atimes) and creation/birth time (-N/--crtimes),
* appended for protocol 2.12.0. The SENDER sets the corresponding *_valid
* flag only when the preserve option is active (and, for crtime, only when
* the source platform exposed a birth time via statx STATX_BTIME). The wire
* always carries the fields and the flags; a false flag tells the receiver to
* ignore the value. */
bool atime_valid;
time_t atime_sec;
long atime_nsec;
bool crtime_valid;
time_t crtime_sec;
long crtime_nsec;
} FileMetadata; } FileMetadata;
typedef struct { typedef struct {
char* path; char* path;
/* Sender-side override for the path transmitted on the wire (and used for
* the delete manifest / change output). NULL means "use `path`". With
* -R + --files-from this holds the entry's bare relative destination path,
* while `path` stays the absolute local source path the client reads from.
* Never populated on the receiver. */
char* send_path;
Data* data; Data* data;
FileMetadata* metadata; FileMetadata* metadata;
bool skip; bool skip;
/* True when this entry is an explicit directory entry (--dirs mode): the
* receiver creates the directory instead of writing a regular file. */
bool is_dir;
/* Receiver-only (P7 Wave D): this is a STATUS_DIR_TIMES entry. It carries a
* traversed source directory's metadata for DEFERRED application, but must
* NEVER create the directory: the scanner captures every traversed directory
* (including empty ones whose parents no child write created), so creation
* would resurrect the empty dirs that FastSync deliberately never transfers.
* file_save_to_disk_full short-circuits such an entry as FILE_SAVE_SKIPPED,
* and the sink still accumulates the metadata into its DirTimeList. */
bool dir_time_only;
/* Receiver-only, --link-dest: when set, install the destination entry as a
* hard link to this absolute (root-confined) path instead of writing
* `data`. The matching code has already verified the link target's content
* equals the incoming file, and `data` is kept as the cross-filesystem
* fallback (a local copy) if the hard link cannot be created. */
char* basis_link;
/* --hard-links (-H), sender + receiver wire state. link_group is a run-local
* id shared by every member of one source inode (0 = not part of a group).
* The FIRST member (link_first == true) carries its data on the wire and is
* written normally; every sibling (link_first == false) carries NO data and
* hardlink_target holds the first member's wire path so the receiver can link
* to (or copy from) the already-installed first member. */
int link_group;
bool link_first;
char* hardlink_target;
/* Symlink-type entry (-l/--links, or -k/--copy-dirlinks' keep-as-symlink
* branch). When true, `symlink_target` holds the (sender-munged, if
* --munge-links) target string that is carried on the wire; the receiver
* creates a symlink to (an unmunged) target instead of writing regular-file
* data. `data` is empty for a symlink entry. Sender + receiver state. */
bool is_symlink;
char* symlink_target;
/* Phase 4 special/devices: when `is_special` is true this entry is a device
* or special node to be RECREATED on the destination (mknod/mkfifo) rather
* than written from `data`. The concrete node kind is derived from the
* metadata mode's S_IFMT bits (receiver-validated), and rdev_major/minor
* carry the device major/minor numbers for char/block devices. CROSSES the
* wire (protocol 2.13.0). */
bool is_special;
int32_t rdev_major;
int32_t rdev_minor;
/* Phase-4 xattrs (-X/--xattrs, -A/--acls). Sender: captured from the source
* file when use_xattrs is set; transmitted in the per-file metadata frame.
* Receiver: parsed off the wire, attached here, and applied fd-relative on
* the written file. NULL/0 == the file carries no xattrs. */
FileXattrList* xattrs;
} File; } File;
/* The path that should be sent on the wire and used for the receiver-side
* destination layout (see send_path). */
static inline const char* file_wire_path(const File* file) {
return file && file->send_path ? file->send_path : (file ? file->path : NULL);
}
#endif #endif
-443
View File
@@ -1,443 +0,0 @@
#include "filter.h"
#include "log.h"
#include "utils.h"
#include <errno.h>
#include <limits.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
/* ---- Single rule parsing ---- */
static bool rule_text_is_unsupported_word(const char* p, size_t len) {
static const char* const words[] = {"merge", "dir-merge", "hide", "show",
"protect", "risk", "clear"};
for (size_t i = 0; i < sizeof(words) / sizeof(words[0]); i++) {
size_t wl = strlen(words[i]);
if (len == wl && strncmp(p, words[i], wl) == 0)
return true;
}
return false;
}
/* rsync include/exclude rule modifiers we do NOT implement. A rule whose +/- is
* immediately followed by one of these is rejected instead of being silently
* parsed as a literal pattern. */
static bool is_unsupported_rule_modifier(char c) {
return c == '!' || c == 'C' || c == 's' || c == 'r' || c == 'p' || c == 'x';
}
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
if (!line)
return NULL;
char* text = str_dup(line);
if (!text) {
if (err)
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
size_t len = strlen(text);
while (len > 0 && (text[len - 1] == '\n' || text[len - 1] == '\r'))
text[--len] = '\0';
const char* p = text;
while (*p == ' ' || *p == '\t')
p++;
if (*p == '\0') {
snprintf(err, err_size, "empty filter rule");
free(text);
return NULL;
}
FilterAction action = FILTER_ACTION_EXCLUDE;
if (*p == '+' || *p == '-') {
action = *p == '+' ? FILTER_ACTION_INCLUDE : FILTER_ACTION_EXCLUDE;
p++;
/* rsync attaches rule modifiers directly to the +/- (e.g. "-s foo"). Only
* the '/' anchor modifier is supported; anything else is a clear error
* rather than a silently-ignored literal. */
if (*p != ' ' && *p != '\t' && *p != '\0' && is_unsupported_rule_modifier(*p)) {
snprintf(err, err_size,
"filter rule modifier '%c' is not supported (only the '/' anchor after +/- "
"is implemented; put a space between +/- and the pattern)",
*p);
free(text);
return NULL;
}
while (*p == ' ' || *p == '\t')
p++;
} else {
/* ':' (dir-merge) and '.' (merge) are rsync filter-rule shorthands. At the
* start of a rule they mean "merge this file", so reject them instead of
* silently turning them into inert exclude patterns. */
if (*p == ':' || *p == '.' || *p == '!') {
snprintf(err, err_size,
"filter rule starting with '%c' is not supported (merge/dir-merge/list-clear "
"shorthands are not implemented; use +/- include/exclude rules)",
*p);
free(text);
return NULL;
}
const char* sp = p;
while (*sp != '\0' && *sp != ' ' && *sp != '\t')
sp++;
size_t word_len = (size_t)(sp - p);
if (rule_text_is_unsupported_word(p, word_len)) {
snprintf(err, err_size,
"'%.*s' filter directives are not supported (only +/- include/exclude rules "
"with an optional '/' anchor and trailing '/' dir marker)",
(int)word_len, p);
free(text);
return NULL;
}
if (word_len == strlen("include") && strncmp(p, "include", word_len) == 0) {
action = FILTER_ACTION_INCLUDE;
p = sp;
} else if (word_len == strlen("exclude") && strncmp(p, "exclude", word_len) == 0) {
action = FILTER_ACTION_EXCLUDE;
p = sp;
}
while (*p == ' ' || *p == '\t')
p++;
}
if (*p == '\0') {
snprintf(err, err_size, "filter rule has no pattern");
free(text);
return NULL;
}
/* A pattern beginning with '/' is anchored (either as "-/foo" or "- /foo"). */
bool anchored = false;
if (*p == '/') {
anchored = true;
p++;
while (*p == ' ' || *p == '\t')
p++;
}
if (*p == '\0') {
snprintf(err, err_size, "filter rule has no pattern after '/' anchor");
free(text);
return NULL;
}
/* Pattern runs to the end of the rule; a single trailing '/' marks dir-only. */
size_t pat_len = strlen(p);
bool dir_only = false;
if (pat_len > 1 && p[pat_len - 1] == '/') {
dir_only = true;
pat_len--;
} else if (pat_len == 1 && p[0] == '/') {
/* "//" anchored with nothing after: meaningless. */
snprintf(err, err_size, "filter rule has no pattern");
free(text);
return NULL;
}
FilterRule* rule = calloc(1, sizeof(FilterRule));
if (!rule) {
snprintf(err, err_size, "memory allocation failed");
free(text);
return NULL;
}
rule->pattern = malloc(pat_len + 1);
if (!rule->pattern) {
free(rule);
snprintf(err, err_size, "memory allocation failed");
free(text);
return NULL;
}
memcpy(rule->pattern, p, pat_len);
rule->pattern[pat_len] = '\0';
rule->action = action;
rule->anchored = anchored;
rule->dir_only = dir_only;
rule->owner = NULL;
free(text);
return rule;
}
void filter_rule_free(FilterRule* rule) {
if (!rule)
return;
free(rule->pattern);
free(rule->owner);
free(rule);
}
/* ---- Ordered rule lists ---- */
FilterRuleList* filter_rule_list_create(void) {
return calloc(1, sizeof(FilterRuleList));
}
bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule) {
if (!list || !rule)
return false;
if (list->count == list->capacity) {
if (list->capacity > INT_MAX / 2)
return false;
int new_cap = list->capacity > 0 ? list->capacity * 2 : 8;
FilterRule** grown = realloc(list->items, (size_t)new_cap * sizeof(FilterRule*));
if (!grown)
return false;
list->items = grown;
list->capacity = new_cap;
}
list->items[list->count++] = rule;
return true;
}
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err,
size_t err_size) {
FilterRule* rule = filter_rule_parse(line, err, err_size);
if (!rule)
return false;
if (!filter_rule_list_add(list, rule)) {
filter_rule_free(rule);
snprintf(err, err_size, "memory allocation failed");
return false;
}
return true;
}
void filter_rule_list_free(FilterRuleList* list) {
if (!list)
return;
for (int i = 0; i < list->count; i++)
filter_rule_free(list->items[i]);
free(list->items);
free(list);
}
static bool set_rule_owner(FilterRule* rule, const char* owner) {
char* dup = str_dup(owner ? owner : "");
if (!dup)
return false;
free(rule->owner);
rule->owner = dup;
return true;
}
/* ---- CVS default excludes (-C) ---- */
typedef struct {
const char* pattern;
bool dir_only;
} CvsDefaultRule;
static const CvsDefaultRule CVS_DEFAULTS[] = {
{"RCS", false}, {"SCCS", false}, {"CVS", false}, {"CVS.adm", false},
{"RCSLOG", false}, {"cvslog.*", false}, {"tags", false}, {"TAGS", false},
{".make.state", false}, {".nse_depinfo", false}, {"*~", false}, {"#*", false},
{".#*", false}, {",*", false}, {"_$*", false}, {"*$", false},
{"*.old", false}, {"*.bak", false}, {"*.BAK", false}, {"*.orig", false},
{"*.rej", false}, {".del-*", false}, {"*.a", false}, {"*.olb", false},
{"*.o", false}, {"*.obj", false}, {"*.so", false}, {"*.exe", false},
{"*.Z", false}, {"*.elc", false}, {"*.ln", false}, {"core", false},
{".svn/", true}, {".git/", true}, {".hg/", true}, {".bzr/", true},
};
static bool cvs_rule_list_append(FilterRuleList* list) {
for (size_t i = 0; i < sizeof(CVS_DEFAULTS) / sizeof(CVS_DEFAULTS[0]); i++) {
FilterRule* rule = calloc(1, sizeof(FilterRule));
if (!rule)
return false;
rule->action = FILTER_ACTION_EXCLUDE;
rule->dir_only = CVS_DEFAULTS[i].dir_only;
size_t plen = strlen(CVS_DEFAULTS[i].pattern);
if (rule->dir_only && plen > 0 && CVS_DEFAULTS[i].pattern[plen - 1] == '/')
plen--; /* keep the cleaned pattern, matching filter_rule_parse */
rule->pattern = malloc(plen + 1);
if (!rule->pattern) {
free(rule);
return false;
}
memcpy(rule->pattern, CVS_DEFAULTS[i].pattern, plen);
rule->pattern[plen] = '\0';
if (!set_rule_owner(rule, "")) {
filter_rule_free(rule);
return false;
}
if (!filter_rule_list_add(list, rule)) {
filter_rule_free(rule);
return false;
}
}
return true;
}
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
char* err, size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
FilterRuleList* list = filter_rule_list_create();
if (!list) {
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
for (int i = 0; i < rule_count; i++) {
if (!rule_texts || !rule_texts[i])
continue;
FilterRule* rule = filter_rule_parse(rule_texts[i], err, err_size);
if (!rule) {
filter_rule_list_free(list);
return NULL;
}
if (!set_rule_owner(rule, "")) {
filter_rule_free(rule);
filter_rule_list_free(list);
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
if (!filter_rule_list_add(list, rule)) {
filter_rule_free(rule);
filter_rule_list_free(list);
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
}
if (cvs_exclude && !cvs_rule_list_append(list)) {
filter_rule_list_free(list);
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
return list;
}
/* ---- Per-directory .rsync-filter files ---- */
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
char* err, size_t err_size) {
if (err && err_size > 0)
err[0] = '\0';
if (exists)
*exists = false;
char* filter_path = path_cat(dir_path, ".rsync-filter");
if (!filter_path) {
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
FILE* fp = fopen(filter_path, "r");
free(filter_path);
if (!fp) {
if (errno == ENOENT || errno == ENOTDIR)
return filter_rule_list_create();
char* escaped_dir = output_escape(dir_path, log_get_8_bit_output());
log_message(LOG_LEVEL_WARNING, "Could not read .rsync-filter in %s: %s",
escaped_dir ? escaped_dir : "<allocation failed>", strerror(errno));
free(escaped_dir);
return filter_rule_list_create();
}
if (exists)
*exists = true;
FilterRuleList* list = filter_rule_list_create();
if (!list) {
fclose(fp);
snprintf(err, err_size, "memory allocation failed");
return NULL;
}
char* line = NULL;
size_t line_cap = 0;
bool ok = true;
while (true) {
ssize_t n = utils_getdelim_bounded(fp, &line, &line_cap, '\n', UTILS_MAX_LINE_LEN);
if (n < 0) {
if (errno == EFBIG) {
snprintf(err, err_size, "line in .rsync-filter exceeds %d bytes", (int)UTILS_MAX_LINE_LEN);
} else {
snprintf(err, err_size, "error reading .rsync-filter: %s", strerror(errno));
}
ok = false;
break;
}
if (n == 0)
break;
const char* p = line;
while (*p == ' ' || *p == '\t')
p++;
if (*p == '\0' || *p == '\n' || *p == '\r' || *p == '#')
continue;
FilterRule* rule = filter_rule_parse(p, err, err_size);
if (!rule) {
ok = false;
break;
}
if (!set_rule_owner(rule, owner_rel)) {
filter_rule_free(rule);
snprintf(err, err_size, "memory allocation failed");
ok = false;
break;
}
if (!filter_rule_list_add(list, rule)) {
filter_rule_free(rule);
snprintf(err, err_size, "memory allocation failed");
ok = false;
break;
}
}
free(line);
fclose(fp);
if (!ok) {
filter_rule_list_free(list);
return NULL;
}
return list;
}
/* ---- Rule matching ---- */
/* Match a pattern that contains '/' (non-anchored) against the end of the
* relative path, starting at any path-component boundary. */
static bool glob_suffix_match(const char* pattern, const char* str) {
if (glob_match(pattern, str))
return true;
for (const char* slash = strchr(str, '/'); slash; slash = strchr(slash + 1, '/')) {
if (glob_match(pattern, slash + 1))
return true;
}
return false;
}
static FilterAction rule_matches(const FilterRule* rule, const char* rel_path, const char* leaf,
bool is_dir) {
if (!rule || !rule->pattern)
return FILTER_ACTION_NONE;
if (rule->dir_only && !is_dir)
return FILTER_ACTION_NONE;
/* A rule applies only to entries below its owner directory. */
const char* rel2 = rel_path;
if (rule->owner && rule->owner[0] != '\0') {
size_t owner_len = strlen(rule->owner);
if (strncmp(rule->owner, rel_path, owner_len) != 0)
return FILTER_ACTION_NONE;
if (rel_path[owner_len] != '/')
return FILTER_ACTION_NONE;
rel2 = rel_path + owner_len + 1;
}
if (rel2[0] == '\0')
return FILTER_ACTION_NONE;
bool matched;
if (rule->anchored) {
matched = glob_match(rule->pattern, rel2);
} else if (strchr(rule->pattern, '/') != NULL) {
matched = glob_suffix_match(rule->pattern, rel2);
} else {
matched = glob_match(rule->pattern, leaf);
}
return matched ? rule->action : FILTER_ACTION_NONE;
}
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
bool is_dir) {
if (!list)
return FILTER_ACTION_NONE;
for (int i = 0; i < list->count; i++) {
FilterAction action = rule_matches(list->items[i], rel_path, leaf, is_dir);
if (action != FILTER_ACTION_NONE)
return action;
}
return FILTER_ACTION_NONE;
}
-83
View File
@@ -1,83 +0,0 @@
#ifndef FILTER_H
#define FILTER_H
#include <stdbool.h>
#include <stddef.h>
/* rsync-style filter rule engine (client-side file selection).
*
* Supported rule syntax (documented subset):
* [+|-] [anchored '/' prefix] pattern [trailing '/' for dir-only]
*
* "+ PATTERN" include rule (first match wins)
* "- PATTERN" exclude rule
* "PATTERN" implicit exclude rule (rsync default)
* "include PATTERN" / "exclude PATTERN" word forms
* leading '/' after the +/- anchors the pattern to its owner directory
* (the transfer root for command-line/-C rules, the directory that
* contains a .rsync-filter file for per-directory rules)
* a trailing '/' makes the rule match directories only
*
* Rejected explicitly (no silent no-ops): the rsync merge/dir-merge/list-clear
* shorthands written as a rule that starts with ':' or '.' or '!', the
* merge/dir-merge/hide/show/protect/risk/clear words, and every include/exclude
* rule modifier other than '/' (! C s r p x). The pattern must be separated
* from +/- by a space (or a single '/' anchor), exactly like rsync's
* "-s foo"/"-p ..." modifier syntax is refused.
*/
typedef enum {
FILTER_ACTION_NONE = 0, /* no rule matched */
FILTER_ACTION_EXCLUDE = -1,
FILTER_ACTION_INCLUDE = 1
} FilterAction;
typedef struct {
FilterAction action;
bool anchored; /* pattern anchored to the rule's owner directory */
bool dir_only; /* pattern had a trailing '/': matches directories only */
char* owner; /* owning directory rel path ("" == transfer root) */
char* pattern; /* cleaned glob pattern (no leading '/', no trailing '/') */
} FilterRule;
typedef struct {
FilterRule** items; /* owned array of rule pointers */
int count;
int capacity;
} FilterRuleList;
/* Parse a single filter-rule line (no trailing newline required). Returns an
* owned rule, or NULL on unsupported/invalid syntax with a message in `err`. */
FilterRule* filter_rule_parse(const char* line, char* err, size_t err_size);
void filter_rule_free(FilterRule* rule);
FilterRuleList* filter_rule_list_create(void);
/* Append a fully-parsed rule (takes ownership). Returns false on OOM. */
bool filter_rule_list_add(FilterRuleList* list, FilterRule* rule);
/* Parse `line` and append it. Returns false and fills `err` on bad syntax. */
bool filter_rule_list_parse_append(FilterRuleList* list, const char* line, char* err,
size_t err_size);
void filter_rule_list_free(FilterRuleList* list);
/* Build the command-line filter set: `rule_texts` (--filter=RULE in the order
* given, 0..rule_count) followed by the -C CVS default excludes when
* cvs_exclude is true. All rules are owned by "" (the transfer root).
* Returns NULL on unsupported rule text (message in `err`). */
FilterRuleList* filter_base_build(const char* const* rule_texts, int rule_count, bool cvs_exclude,
char* err, size_t err_size);
/* Read "<dir_path>/.rsync-filter" and return its rules, each owned by
* `owner_rel`. A missing file yields an empty list with *exists=false; an
* unreadable file is treated as missing. Returns NULL only on parse or
* allocation failure (message in `err`). */
FilterRuleList* filter_file_read(const char* dir_path, const char* owner_rel, bool* exists,
char* err, size_t err_size);
/* Evaluate an entry against one ordered rule list. Returns FILTER_ACTION_NONE
* when no rule matched, otherwise the first matching rule's action.
* `rel_path` is the entry's path relative to the transfer root ("" == root),
* `leaf` its final name, `is_dir` whether it is a directory. */
FilterAction filter_rules_apply(const FilterRuleList* list, const char* rel_path, const char* leaf,
bool is_dir);
#endif
-127
View File
@@ -1,127 +0,0 @@
#include "hardlink.h"
#include <stdint.h>
#include <stdlib.h>
#include <string.h>
#include "log.h"
#include "utils.h"
/* ---- Sender-side detection table ---- */
HardLinkTable* hardlink_table_create(void) {
HardLinkTable* table = calloc(1, sizeof(HardLinkTable));
if (!table)
return NULL;
if (mtx_init(&table->mutex, mtx_plain) != thrd_success) {
free(table);
return NULL;
}
table->next_gid = 1;
return table;
}
static void hardlink_item_destroy(HardLinkItem* item) {
if (!item)
return;
free(item->first_path);
item->first_path = NULL;
}
void hardlink_table_destroy(HardLinkTable* table) {
if (!table)
return;
for (size_t i = 0; i < table->count; i++)
hardlink_item_destroy(&table->items[i]);
free(table->items);
table->items = NULL;
table->count = 0;
table->capacity = 0;
mtx_destroy(&table->mutex);
free(table);
}
static HardLinkItem* hardlink_table_find_locked(HardLinkTable* table, dev_t dev, ino_t ino) {
for (size_t i = 0; i < table->count; i++) {
if (table->items[i].dev == dev && table->items[i].ino == ino)
return &table->items[i];
}
return NULL;
}
static bool hardlink_table_add_locked(HardLinkTable* table, dev_t dev, ino_t ino, const char* path,
int gid, HardLinkItem** out) {
if (table->count == table->capacity) {
size_t new_capacity = table->capacity == 0 ? 8 : table->capacity * 2;
if (new_capacity < table->capacity)
return false;
HardLinkItem* grown = realloc(table->items, new_capacity * sizeof(HardLinkItem));
if (!grown)
return false;
table->items = grown;
table->capacity = new_capacity;
}
HardLinkItem* item = &table->items[table->count];
char* dup = str_dup(path);
if (!dup)
return false;
memset(item, 0, sizeof(*item));
item->dev = dev;
item->ino = ino;
item->gid = gid;
item->first_path = dup;
table->count++;
*out = item;
return true;
}
bool hardlink_table_assign(HardLinkTable* table, const char* wire_path, dev_t dev, ino_t ino,
int* gid, bool* is_first, char** first_path_out) {
if (!table || !wire_path || !gid || !is_first || !first_path_out)
return false;
if (mtx_lock(&table->mutex) != thrd_success)
return false;
bool ok = true;
const HardLinkItem* item = hardlink_table_find_locked(table, dev, ino);
int next_gid;
if (item) {
*is_first = false;
char* dup = str_dup(item->first_path);
if (!dup) {
ok = false;
} else {
*gid = item->gid;
*first_path_out = dup;
}
next_gid = -1;
} else {
if (table->next_gid <= 0) {
ok = false;
next_gid = -1;
} else {
next_gid = table->next_gid;
HardLinkItem* created = NULL;
if (!hardlink_table_add_locked(table, dev, ino, wire_path, next_gid, &created)) {
ok = false;
} else {
char* dup = str_dup(wire_path);
if (!dup) {
hardlink_item_destroy(created);
table->count--;
ok = false;
} else {
*is_first = true;
*gid = next_gid;
*first_path_out = dup;
}
}
}
}
if (ok && next_gid > 0)
table->next_gid++;
mtx_unlock(&table->mutex);
if (!ok) {
log_message(LOG_LEVEL_ERROR, "memory allocation failed while detecting hard links");
}
return ok;
}
-66
View File
@@ -1,66 +0,0 @@
#ifndef HARDLINK_H
#define HARDLINK_H
#include <stdbool.h>
#include <stddef.h>
#include <sys/types.h>
#include <threads.h>
/*
* --hard-links / -H support.
*
* Sender side: a HardLinkTable detects regular files on the source that share
* an (st_dev, st_ino) identity (a `cp -al`-style hard-linked tree) and assigns
* each distinct inode a stable, run-local link-group id. The first member
* encountered carries the file data; every later member is marked as a sibling
* (no data payload) that the receiver creates as a hard link to the first
* member's destination file. Grouping is scoped by st_dev so inode reuse
* across different filesystems is never conflated. The table is mutex-guarded
* so the parallel (multi-threaded) scanner COULD share one instance across its
* worker threads; the first-thread-to-call designates the data-carrying member,
* which is safe because a hard-link group's members are byte-identical. (In
* practice the sender forces the sequential scanner whenever -H is on; the
* mutex guards the shared table for any path that supplies one.)
*
* ORDERING (why there is no receiver-side handshake): the receiver stores every
* file - including a hard-link group's first member - through a SINGLE writer
* thread draining a single FIFO queue driven by a single receive thread, so
* wire order == write order and every sibling is processed AFTER its group's
* first member. The sender additionally forces the sequential scanner with -H
* so the first-member frame always precedes its siblings on the wire. Sibling
* install therefore needs no present/wait registry: it hard-links to the first
* member (or copies it) knowing that path is already installed - or that, if
* the first member was skipped (already up to date), its destination still
* exists. This guarantee is REQUIRED; do not introduce a concurrent
* multi-writer receiver for -H without re-adding an ordering mechanism.
*/
typedef struct HardLinkItem {
dev_t dev;
ino_t ino;
int gid;
char* first_path; /* wire path of the group's data-carrying first member */
} HardLinkItem;
typedef struct HardLinkTable {
mtx_t mutex;
HardLinkItem* items;
size_t count;
size_t capacity;
int next_gid;
} HardLinkTable;
HardLinkTable* hardlink_table_create(void);
void hardlink_table_destroy(HardLinkTable* table);
/* Assign a link-group id to the regular file at `wire_path` with (dev, ino).
* On the first encounter the file becomes the group's first (data-carrying)
* member (*is_first = true) and a fresh gid is allocated. On a later member
* *is_first = false and *first_path_out is set to a malloc'd copy of the first
* member's wire path (the caller stores it and owns it; on the first member
* path the returned *first_path_out is a malloc'd copy of its own wire path).
* Returns false on allocation failure (transfer should abort). */
bool hardlink_table_assign(HardLinkTable* table, const char* wire_path, dev_t dev, ino_t ino,
int* gid, bool* is_first, char** first_path_out);
#endif
-748
View File
@@ -1,748 +0,0 @@
#include "identity.h"
#include "log.h"
#include "utils.h"
#include <errno.h>
#include <fcntl.h>
#include <grp.h>
#include <limits.h>
#include <pwd.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#include <sys/stat.h>
#include <unistd.h>
/* The active identity snapshot lives in a per-process global. The TCP server
* forks one child process per connection, so a connection never shares this
* with another; within a connection the multithreaded receiver reads it without
* mutation. This is what lets the fd-relative metadata path consult the
* negotiated policy without threading a Config through every write helper. */
typedef struct {
bool numeric_ids;
bool chown_uid_set;
int32_t chown_uid;
bool chown_gid_set;
int32_t chown_gid;
IdentityMap* usermap;
int usermap_count;
IdentityMap* groupmap;
int groupmap_count;
/* --super / --no-super tri-state (SUPER_MODE_AUTO when unset). Snapshotted
* per connection so privilege_super_permitted() can gate super-user
* activities without a Config argument. */
SuperMode super_mode;
/* --copy-as=USER[:GROUP]: snapshotted so the ownership resolver can force the
* target ids without a Config argument. */
bool copy_as_set;
int32_t copy_as_uid;
int32_t copy_as_gid;
bool set;
} IdentityActive;
static IdentityActive g_identity;
static void identity_active_reset(void) {
free(g_identity.usermap);
free(g_identity.groupmap);
g_identity.usermap = NULL;
g_identity.groupmap = NULL;
g_identity.usermap_count = 0;
g_identity.groupmap_count = 0;
g_identity.numeric_ids = false;
g_identity.chown_uid_set = false;
g_identity.chown_uid = 0;
g_identity.chown_gid_set = false;
g_identity.chown_gid = 0;
g_identity.super_mode = SUPER_MODE_AUTO;
g_identity.copy_as_set = false;
g_identity.copy_as_uid = 0;
g_identity.copy_as_gid = 0;
g_identity.set = false;
}
void identity_clear_active(void) {
identity_active_reset();
}
bool identity_set_active(const Config* config) {
identity_active_reset();
if (!config)
return true;
g_identity.numeric_ids = config->numeric_ids;
g_identity.chown_uid_set = config->chown_uid_set;
g_identity.chown_uid = config->chown_uid;
g_identity.chown_gid_set = config->chown_gid_set;
g_identity.chown_gid = config->chown_gid;
g_identity.super_mode = config->super_mode;
g_identity.copy_as_set = config->copy_as_set;
g_identity.copy_as_uid = config->copy_as_uid;
g_identity.copy_as_gid = config->copy_as_gid;
if (config->usermap_count > 0) {
g_identity.usermap = calloc((size_t)config->usermap_count, sizeof(IdentityMap));
if (!g_identity.usermap)
goto alloc_failed;
memcpy(g_identity.usermap, config->usermap,
(size_t)config->usermap_count * sizeof(IdentityMap));
g_identity.usermap_count = config->usermap_count;
}
if (config->groupmap_count > 0) {
g_identity.groupmap = calloc((size_t)config->groupmap_count, sizeof(IdentityMap));
if (!g_identity.groupmap)
goto alloc_failed;
memcpy(g_identity.groupmap, config->groupmap,
(size_t)config->groupmap_count * sizeof(IdentityMap));
g_identity.groupmap_count = config->groupmap_count;
}
g_identity.set = true;
/* A root receiver would honor any client-supplied ownership request (a
--usermap/--groupmap/--chown/--copy-as, or raw ids under --numeric-ids).
Surface that prominently; a privileged daemon applying arbitrary client
ownership is a deliberate, opt-in choice the operator should be aware of. */
if (geteuid() == 0)
log_message(LOG_LEVEL_WARNING,
"identity mapping active and running as root: client-supplied "
"ownership (usermap/groupmap/chown/numeric-ids) will be honored; "
"run the daemon as an unprivileged user unless intended");
/* --super explicitly requests super-user activities, but FastSync never
elevates privileges: when the receiver is not already root the kernel will
refuse those confined attempts and each is skipped per entry. Warn exactly
once at activation time (never abort) so the operator knows the flag cannot
succeed on this host. */
if (g_identity.super_mode == SUPER_MODE_ON && geteuid() != 0)
log_message(LOG_LEVEL_WARNING,
"--super requested but the receiver is not privileged; super-user "
"activities (ownership, device nodes) will be attempted but refused "
"by the kernel and skipped per entry");
return true;
alloc_failed:
/* Never proceed with a partial (count-left-zero) map: that would silently
apply the WRONG ownership policy. Fail closed and let the caller refuse
the connection. */
log_message(LOG_LEVEL_ERROR, "memory allocation failed while activating identity policy");
identity_active_reset();
return false;
}
bool privilege_super_permitted(void) {
return privilege_super_mode_permitted(g_identity.super_mode);
}
bool privilege_super_mode_permitted(SuperMode mode) {
/* AUTO and ON both attempt the confined operation; OFF forbids it even for a
* root receiver. AUTO is the historical FastSync behavior (always attempt
* and let the kernel refuse an unprivileged call, which the caller skips), so
* it must stay permissive or a group-only chown that a non-root receiver is
* allowed to make would regress. */
return mode != SUPER_MODE_OFF;
}
bool identity_active_enabled(void) {
/* numeric_ids is included: this set only gates identity_apply_ownership,
which runs only when metadata is present (a -M/--preserve transfer). A
standalone --numeric-ids (no ownership-affecting flag) carries no
metadata, never reaches identity_apply_ownership, and therefore correctly
stays inert; combined with -M it activates raw-id application. --super /
--no-super does NOT enable ownership: it only permits or forbids the
already-requested super-user activities, so a --super with no explicit
identity flag must never silently apply client-chosen ownership. */
return g_identity.set &&
(g_identity.numeric_ids || g_identity.chown_uid_set || g_identity.chown_gid_set ||
g_identity.usermap_count > 0 || g_identity.groupmap_count > 0 || g_identity.copy_as_set);
}
bool identity_ownership_requested(const Config* config) {
if (!config)
return false;
/* Every value that makes the receiver act on a client-chosen owner, plus an
* explicit --super (super-user device-node activities). Pure config, so the
* daemon gate can evaluate it before identity_set_active(). */
return config->numeric_ids || config->chown_uid_set || config->chown_gid_set ||
config->usermap_count > 0 || config->groupmap_count > 0 || config->copy_as_set ||
config->fake_super || config->super_mode == SUPER_MODE_ON;
}
bool identity_copy_as_active(void) {
return g_identity.set && g_identity.copy_as_set;
}
bool identity_copy_as_refused(const Config* config) {
if (!config || !config->copy_as_set)
return false;
/* The safe-subset --copy-as needs a privileged (root) receiver, and an
* operator/--no-super veto forbids the ownership change even for root. This
* is deliberately a pure function of the config and the current effective uid
* (never the active snapshot) because the server evaluates it at the
* pre-STATUS_OK config gate, before identity_set_active() has run. */
return geteuid() != 0 || config->super_mode == SUPER_MODE_OFF;
}
bool identity_wire_valid(const Config* config) {
if (!config)
return false;
if (config->usermap_count < 0 || config->usermap_count > MAX_IDENTITY_MAP ||
config->groupmap_count < 0 || config->groupmap_count > MAX_IDENTITY_MAP)
return false;
if (config->chown_uid_set && config->chown_uid < IDENTITY_MATCH_ANY)
return false;
if (config->chown_gid_set && config->chown_gid < IDENTITY_MATCH_ANY)
return false;
for (int i = 0; i < config->usermap_count; i++) {
if (config->usermap[i].from < IDENTITY_MATCH_ANY || config->usermap[i].to < IDENTITY_CURRENT)
return false;
}
for (int i = 0; i < config->groupmap_count; i++) {
if (config->groupmap[i].from < IDENTITY_MATCH_ANY || config->groupmap[i].to < IDENTITY_CURRENT)
return false;
}
/* Defense-in-depth: a --copy-as block must never carry a negative (sentinel)
* id into the ownership path. receive_copy_as_options already rejects them,
* but identity_wire_valid is the shared validation used by both the receiver
* and unit tests, so re-assert it here. */
if (config->copy_as_set && (config->copy_as_uid < 0 || config->copy_as_gid < 0))
return false;
return true;
}
/* ---- CLI-time name/number resolution ---- */
/* Parse a single FROM/TO token into an int32 id. Returns 0 on success, -1 on a
* malformed or unresolvable token. When is_group, name lookups use the group
* database; otherwise the user database. A `*` token returns IDENTITY_MATCH_ANY
* / IDENTITY_CURRENT (the same -1 value, disambiguated by the caller's
* position). An `@`-prefixed or bare-decimal token is a numeric id. */
static int identity_resolve_token(const char* token, bool is_group, int32_t* out) {
if (!token || *token == '\0')
return -1;
if (strcmp(token, "*") == 0) {
*out = IDENTITY_MATCH_ANY;
return 0;
}
const char* num = (token[0] == '@') ? token + 1 : token;
if (*num != '\0') {
bool all_digits = true;
for (const char* p = num; *p; p++)
if (*p < '0' || *p > '9')
all_digits = false;
if (all_digits) {
char* endptr = NULL;
errno = 0;
long val = strtol(num, &endptr, 10);
if (errno == 0 && endptr && *endptr == '\0' && val >= 0 && val <= INT32_MAX) {
*out = (int32_t)val;
return 0;
}
return -1;
}
}
/* A name (or a name-like numeric that failed strict numeric parse). */
if (is_group) {
struct group* gr = getgrnam(token);
if (!gr)
return -1;
*out = (int32_t)gr->gr_gid;
return 0;
}
struct passwd* pw = getpwnam(token);
if (!pw)
return -1;
*out = (int32_t)pw->pw_uid;
return 0;
}
static int identity_append_rule(IdentityMap** map, int* count, int32_t from, int32_t to) {
if (*count >= MAX_IDENTITY_MAP)
return -1;
IdentityMap* grown = realloc(*map, (size_t)(*count + 1) * sizeof(IdentityMap));
if (!grown)
return -1;
*map = grown;
(*map)[*count].from = from;
(*map)[*count].to = to;
(*count)++;
return 0;
}
int identity_parse_map(Config* config, const char* value, bool is_group) {
if (!config || !value || *value == '\0') {
log_message(LOG_LEVEL_ERROR, "%smap requires a value", is_group ? "--group" : "--user");
return -1;
}
char* list = str_dup(value);
if (!list)
return -1;
const char* optname = is_group ? "--groupmap" : "--usermap";
char* saveptr = NULL;
for (char* rule = strtok_r(list, ",", &saveptr); rule; rule = strtok_r(NULL, ",", &saveptr)) {
char* colon = strchr(rule, ':');
if (!colon || colon == rule) {
/* Log before freeing: `rule` points into the str_dup'd list. */
log_message(LOG_LEVEL_ERROR, "%s rules must be FROM:TO (got '%s')", optname, rule);
free(list);
return -1;
}
*colon = '\0';
char* from_token = rule;
char* to_token = colon + 1;
if (*to_token == '\0') {
free(list);
log_message(LOG_LEVEL_ERROR, "%s rule 'FROM:' is missing the TO value (got '%s')", optname,
value);
return -1;
}
int32_t from_id, to_id;
if (identity_resolve_token(from_token, is_group, &from_id) != 0 ||
identity_resolve_token(to_token, is_group, &to_id) != 0) {
free(list);
log_message(LOG_LEVEL_ERROR,
"%s could not resolve '%s' (name must exist on the source; use "
"@N for a numeric id)",
optname, value);
return -1;
}
if (identity_append_rule(is_group ? &config->groupmap : &config->usermap,
is_group ? &config->groupmap_count : &config->usermap_count, from_id,
to_id) != 0) {
free(list);
log_message(LOG_LEVEL_ERROR, "%s has too many rules (max %d)", optname, MAX_IDENTITY_MAP);
return -1;
}
}
free(list);
return 0;
}
/* Split --chown=USER:GROUP on the first UNESCAPED colon, honoring backslash
* escapes (a `\:` is a literal colon inside a name; a lone backslash before any
* other character is kept verbatim). Both sides are returned as malloc'd
* strings (the absent side is NULL). */
static int identity_split_chown(const char* value, char** puser, char** pgroup) {
size_t len = strlen(value);
char* user = malloc(len + 1);
char* group = malloc(len + 1);
if (!user || !group) {
free(user);
free(group);
return -1;
}
const char* p = value;
size_t ui = 0;
bool split_seen = false;
size_t gi = 0;
while (*p) {
if (*p == '\\' && p[1] == ':') {
/* an escaped colon: a literal ':' in the current side's name */
if (split_seen)
group[gi++] = ':';
else
user[ui++] = ':';
p += 2;
continue;
}
if (*p == ':') {
split_seen = true;
p++;
continue;
}
if (split_seen)
group[gi++] = *p;
else
user[ui++] = *p;
p++;
}
user[ui] = '\0';
group[gi] = '\0';
char* u = str_dup(user);
char* g = str_dup(group);
free(user);
free(group);
if (!u || !g) {
free(u);
free(g);
return -1;
}
*puser = u;
*pgroup = g;
return 0;
}
int identity_parse_chown(Config* config, const char* value) {
if (!config || !value || *value == '\0') {
log_message(LOG_LEVEL_ERROR, "--chown requires a value (USER:GROUP, USER, or :GROUP)");
return -1;
}
/* Reject more than one UNESCAPED colon (a name or group may not contain an
* unescaped ':' in the spec). The scan is escape-aware: a `\:` is a literal
* colon inside a name, not a field separator. */
int colons = 0;
bool saw_colon = false;
const char* p = value;
while (*p) {
if (*p == '\\' && p[1] == ':') {
p += 2;
continue;
}
if (*p == ':') {
colons++;
saw_colon = true;
}
p++;
}
if (colons > 1) {
log_message(LOG_LEVEL_ERROR, "--chown must have at most one ':' (got '%s')", value);
return -1;
}
char *user = NULL, *group = NULL;
if (identity_split_chown(value, &user, &group) != 0) {
log_message(LOG_LEVEL_ERROR, "memory allocation failed for --chown");
return -1;
}
int ret = 0;
if (!saw_colon) {
/* --chown=USER: owner only. */
if (*user == '\0') {
log_message(LOG_LEVEL_ERROR, "--chown requires a user or group (got '%s')", value);
ret = -1;
} else if (identity_resolve_token(user, false, &config->chown_uid) != 0) {
log_message(LOG_LEVEL_ERROR,
"--chown could not resolve user '%s' (use a name that exists "
"on the source, '*', or @N)",
value);
ret = -1;
} else {
config->chown_uid_set = true;
}
} else {
/* --chown=USER:GROUP, --chown=:GROUP, --chown=USER: */
if (*user != '\0') {
if (identity_resolve_token(user, false, &config->chown_uid) != 0) {
log_message(LOG_LEVEL_ERROR, "--chown could not resolve user '%s'", value);
ret = -1;
goto done;
}
config->chown_uid_set = true;
}
if (*group != '\0') {
if (identity_resolve_token(group, true, &config->chown_gid) != 0) {
log_message(LOG_LEVEL_ERROR, "--chown could not resolve group '%s'", value);
ret = -1;
goto done;
}
config->chown_gid_set = true;
}
if (!*user && !*group) {
log_message(LOG_LEVEL_ERROR, "--chown must set a user, a group, or both (got '%s')", value);
ret = -1;
}
}
done:
free(user);
free(group);
return ret;
}
/* uid_t/gid_t are unsigned and may hold a value wider than the signed int32 the
* wire (and the identity policy) uses. Reject such an id instead of truncating
* it to an out-of-range (possibly negative sentinel) value. */
static bool identity_id_fits_int32(unsigned long id) {
return id <= (unsigned long)INT32_MAX;
}
/* Resolve one --copy-as id token. A '*' token means the caller's current
* effective uid (user) or gid (group). Returns 0 on success. On failure sets
* *overflow when a '*' id was wider than int32 so the caller can log the
* specific message; otherwise the token was simply unresolvable. */
static int identity_resolve_copy_as_id(const char* token, bool is_group, int32_t* out,
bool* overflow) {
*overflow = false;
if (strcmp(token, "*") == 0) {
unsigned long current = is_group ? (unsigned long)getegid() : (unsigned long)geteuid();
if (!identity_id_fits_int32(current)) {
*overflow = true;
return -1;
}
*out = (int32_t)current;
return 0;
}
return identity_resolve_token(token, is_group, out);
}
int identity_parse_copy_as(Config* config, const char* value) {
if (!config || !value || *value == '\0') {
log_message(LOG_LEVEL_ERROR, "--copy-as requires USER[:GROUP]");
return -1;
}
/* --copy-as=USER[:GROUP] is the whole grammar: at most one field separator.
* (Unlike --chown there is no escaped-colon form; a name containing ':' is
* simply not expressible, and the extra colon is a clear parse error.) */
int colons = 0;
for (const char* p = value; *p; p++)
if (*p == ':')
colons++;
if (colons > 1) {
char* escaped = output_escape(value, false);
log_message(LOG_LEVEL_ERROR, "--copy-as must be USER[:GROUP] (got '%s')",
escaped ? escaped : "<allocation failed>");
free(escaped);
return -1;
}
char* spec = str_dup(value);
if (!spec) {
log_message(LOG_LEVEL_ERROR, "memory allocation failed for --copy-as");
return -1;
}
const char* user_token = spec;
const char* group_token = NULL;
char* colon = strchr(spec, ':');
if (colon) {
*colon = '\0';
group_token = colon + 1;
}
/* The spec is untrusted user input echoed back in error paths: escape it once
* (8-bit-safe) so a control byte cannot forge a log line. */
char* escaped_spec = output_escape(value, false);
const char* shown = escaped_spec ? escaped_spec : "<allocation failed>";
int ret = -1;
if (*user_token == '\0') {
log_message(LOG_LEVEL_ERROR, "--copy-as is missing the user (got '%s')", shown);
goto done;
}
bool overflow = false;
int32_t uid;
if (identity_resolve_copy_as_id(user_token, false, &uid, &overflow) != 0) {
if (overflow)
log_message(LOG_LEVEL_ERROR, "--copy-as: current user id %lu exceeds INT32_MAX",
(unsigned long)geteuid());
else
log_message(LOG_LEVEL_ERROR,
"--copy-as could not resolve user (use a name that exists on the "
"source, '*', or @N): %s",
shown);
goto done;
}
int32_t gid;
if (group_token) {
if (*group_token == '\0') {
log_message(LOG_LEVEL_ERROR, "--copy-as group is empty (got '%s')", shown);
goto done;
}
if (identity_resolve_copy_as_id(group_token, true, &gid, &overflow) != 0) {
if (overflow)
log_message(LOG_LEVEL_ERROR, "--copy-as: current group id %lu exceeds INT32_MAX",
(unsigned long)getegid());
else
log_message(LOG_LEVEL_ERROR, "--copy-as could not resolve group (got '%s')", shown);
goto done;
}
} else {
/* Group omitted: use the user's primary gid. A numeric id with no local
* passwd entry has no primary gid to look up, so fall back to gid == uid
* (the rsync-style numeric convention; documented divergence). */
struct passwd* pw = getpwuid((uid_t)uid);
if (pw) {
if (!identity_id_fits_int32((unsigned long)pw->pw_gid)) {
log_message(LOG_LEVEL_ERROR,
"--copy-as: primary group id %lu for the requested user exceeds INT32_MAX",
(unsigned long)pw->pw_gid);
goto done;
}
gid = (int32_t)pw->pw_gid;
} else {
gid = uid;
}
}
/* The group-default and gid==uid fallbacks must never store a negative
* (sentinel) value; the explicit numeric path is already capped by
* identity_resolve_token. */
if (uid < 0 || gid < 0) {
log_message(LOG_LEVEL_ERROR, "--copy-as resolved id does not fit in int32 (got '%s')", shown);
goto done;
}
config->copy_as_set = true;
config->copy_as_uid = uid;
config->copy_as_gid = gid;
/* Ownership application needs the metadata path (the source uid/gid must be
* transmitted); imply it exactly like --chown/--usermap/--groupmap. */
config->use_metadata = true;
ret = 0;
done:
free(escaped_spec);
free(spec);
return ret;
}
/* ---- Receiver-side ownership application ---- */
static bool identity_map_lookup(const IdentityMap* map, int count, int32_t source_id,
int32_t* out_to) {
for (int i = 0; i < count; i++) {
if (map[i].from == IDENTITY_MATCH_ANY || map[i].from == source_id) {
*out_to = map[i].to;
return true;
}
}
return false;
}
/* Resolve the target ownership from the negotiated policy against the entry's
* current stat. Shared by the fd (regular file) and no-follow (symlink) apply
* paths. Returns false when no side is to be changed. */
static bool identity_resolve_targets(const struct stat* st, int32_t source_uid, int32_t source_gid,
uid_t* out_uid, gid_t* out_gid) {
bool set_uid = false;
bool set_gid = false;
uid_t uid = 0;
gid_t gid = 0;
/* --copy-as (P7 Wave E) has the highest priority: it forces BOTH the owner
* and group of every written entry to the requested ids, beating usermap /
* groupmap / --chown / --numeric-ids and the best-effort name lookup. Only
* skip when the entry already carries exactly those ids. */
if (g_identity.copy_as_set) {
uid = (uid_t)g_identity.copy_as_uid;
gid = (gid_t)g_identity.copy_as_gid;
if (st->st_uid == uid && st->st_gid == gid)
return false;
*out_uid = uid;
*out_gid = gid;
return true;
}
int32_t target;
if (identity_map_lookup(g_identity.usermap, g_identity.usermap_count, source_uid, &target)) {
uid = target == IDENTITY_CURRENT ? geteuid() : (uid_t)target;
set_uid = true;
} else if (g_identity.chown_uid_set) {
uid = g_identity.chown_uid == IDENTITY_CURRENT ? geteuid() : (uid_t)g_identity.chown_uid;
set_uid = true;
} else if (g_identity.numeric_ids) {
uid = (uid_t)source_uid;
set_uid = true;
} else {
/* Best-effort name mapping against the receiver's own database: if the
* transmitted (numeric) id resolves to a name present on this machine,
* re-resolve it. On a shared-account host this is the identity operation;
* when the id has no name here, the user side is left alone. */
struct passwd* pw = getpwuid((uid_t)source_uid);
if (pw) {
const struct passwd* mapped = getpwnam(pw->pw_name);
if (mapped) {
uid = mapped->pw_uid;
set_uid = true;
}
}
}
if (identity_map_lookup(g_identity.groupmap, g_identity.groupmap_count, source_gid, &target)) {
gid = target == IDENTITY_CURRENT ? getegid() : (gid_t)target;
set_gid = true;
} else if (g_identity.chown_gid_set) {
gid = g_identity.chown_gid == IDENTITY_CURRENT ? getegid() : (gid_t)g_identity.chown_gid;
set_gid = true;
} else if (g_identity.numeric_ids) {
gid = (gid_t)source_gid;
set_gid = true;
} else {
struct group* gr = getgrgid((gid_t)source_gid);
if (gr) {
const struct group* mapped = getgrnam(gr->gr_name);
if (mapped) {
gid = mapped->gr_gid;
set_gid = true;
}
}
}
if (!set_uid && !set_gid)
return false;
/* An unset side keeps the file's current id so the other side can change. */
if (!set_uid)
uid = st->st_uid;
if (!set_gid)
gid = st->st_gid;
/* Only change ownership when the target differs (avoid needless syscalls and
* any chance of clearing setuid/setgid on an already-correct entry). */
if (st->st_uid == uid && st->st_gid == gid)
return false;
*out_uid = uid;
*out_gid = gid;
return true;
}
static void identity_log_chown_failure(const char* what, uid_t uid, gid_t gid) {
/* EPERM/EACCES are expected when the receiver is not privileged (e.g. the CI
* `nobody` user): warn and continue, never abort the transfer. Any other
* error (EIO/EROFS/ENOSPC/...) is a real failure and must not be silently
* downgraded to a warning.
*
* --copy-as is different: the whole point of the flag is that the target
* ownership is REQUIRED (the pre-flight gate already refused an unprivileged
* receiver). If the chown still fails with EPERM/EACCES (a capability-
* restricted root, root-squash, or a read-only mount) the run would be
* silently producing the WRONG ownership, so surface it at ERROR. The
* caller (identity_apply_ownership*) then reports the ENTRY as failed rather
* than as written, which becomes a FILE_SAVE_ERROR and fails the transfer
* (fail-fast) instead of reporting overall success with the wrong owner. */
if (errno == EPERM || errno == EACCES) {
if (identity_copy_as_active())
log_message(LOG_LEVEL_ERROR,
"could not apply --copy-as ownership on %s (uid=%ld gid=%ld): %s; "
"entry was written with the wrong owner",
what, (long)uid, (long)gid, strerror(errno));
else
log_message(LOG_LEVEL_WARNING,
"could not apply ownership (uid=%ld gid=%ld): %s; leaving as-is", (long)uid,
(long)gid, strerror(errno));
} else {
log_message(LOG_LEVEL_ERROR, "failed to apply ownership on %s (uid=%ld gid=%ld): %s", what,
(long)uid, (long)gid, strerror(errno));
}
}
bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid) {
/* Ownership application is OFF unless the client requested an identity flag.
* This is the controlled gate: a default (or plain -M) transfer never changes
* ownership, byte-for-byte preserving FastSync's existing behavior. --no-super
* additionally forbids it even when the receiver is root. */
if (!identity_active_enabled() || !privilege_super_permitted() || fd < 0)
return true;
struct stat st;
if (fstat(fd, &st) != 0)
return !identity_copy_as_active();
uid_t uid;
gid_t gid;
if (!identity_resolve_targets(&st, source_uid, source_gid, &uid, &gid))
return true;
if (fchown(fd, uid, gid) != 0) {
identity_log_chown_failure("file", uid, gid);
/* A required --copy-as ownership that did not land is a per-entry failure;
* every other policy stays best-effort (rsync parity). */
return !identity_copy_as_active();
}
return true;
}
bool identity_apply_ownership_link(int parent_fd, const char* leaf, int32_t source_uid,
int32_t source_gid) {
if (!identity_active_enabled() || !privilege_super_permitted() || parent_fd < 0 || !leaf)
return true;
struct stat st;
if (fstatat(parent_fd, leaf, &st, AT_SYMLINK_NOFOLLOW) != 0)
return !identity_copy_as_active();
uid_t uid;
gid_t gid;
if (!identity_resolve_targets(&st, source_uid, source_gid, &uid, &gid))
return true;
if (fchownat(parent_fd, leaf, uid, gid, AT_SYMLINK_NOFOLLOW) != 0) {
identity_log_chown_failure("no-follow entry", uid, gid);
return !identity_copy_as_active();
}
return true;
}
-133
View File
@@ -1,133 +0,0 @@
#ifndef IDENTITY_H
#define IDENTITY_H
#include "config.h"
#include <stdbool.h>
#include <stdint.h>
#include <sys/types.h>
/*
* Identity mapping: --numeric-ids / --usermap / --groupmap / --chown / --copy-as.
*
* FastSync transmits uid/gid numerically (int32 on the wire) and, by design,
* NEVER applies client-supplied ownership unless a user explicitly opts in with
* an identity flag below. This module is the controlled, opt-in,
* privilege-gated path for applying ownership on the receiver: the wire config
* is snapshotted once per connection via identity_set_active() and applied
* through an fd-relative fchown() in the receiver's metadata-restore path.
*
* Because only numeric ids cross the wire, name-based values are resolved to
* numbers at CLI parse time using the CLIENT (sender) machine's databases. On
* a shared-account source/destination this reproduces rsync's semantics; a
* genuinely different destination database is a documented divergence (see
* RSYNC_COMPAT.md).
*/
/* Parse one --usermap= / --groupmap= value (comma-separated FROM:TO rules,
* first match wins) into config->usermap / config->groupmap. is_group selects
* the group tables and name databases. Returns 0 on success, -1 on a
* malformed spec or an unresolvable name (never a silent no-op). */
int identity_parse_map(Config* config, const char* value, bool is_group);
/* Parse --chown=USER:GROUP. Supports USER:GROUP, USER (owner only), :GROUP
* (group only), '*' (current/root as appropriate) and numeric ids. Returns 0
* on success, -1 on a malformed spec / unresolvable name. */
int identity_parse_chown(Config* config, const char* value);
/* Parse --copy-as=USER[:GROUP] (P7 Wave E). USER is resolved with the same
* user-database rules as --chown (a name, @N/bare N numeric id, or '*' meaning
* the client's current euid); when ':GROUP' is present the group is resolved
* with the group database ('*' meaning the client's egid). When the group is
* omitted, the user's primary gid is used (getpwuid(uid)->pw_gid); if the
* resolved user is a numeric id with no local passwd entry, gid falls back to
* uid. On success sets copy_as_set/copy_as_uid/copy_as_gid and forces
* metadata transmission (ownership application needs the metadata path).
* Returns 0 on success, -1 on a malformed / empty / unresolvable spec (never a
* silent no-op). */
int identity_parse_copy_as(Config* config, const char* value);
/* True when a --copy-as request is active but the receiver is not permitted to
* perform the privileged ownership application it needs. This is the up-front
* refusal predicate: the server rejects the whole transfer at the config
* handshake rather than silently ignoring the requested ownership. It is a
* pure function of the config mode and the current effective uid (it does NOT
* read the active snapshot, so it is valid at the pre-STATUS_OK gate, before
* identity_set_active() has run). `super_mode` is the EFFECTIVE mode after any
* server-side policy veto. */
bool identity_copy_as_refused(const Config* config);
/* True when the CURRENT per-connection snapshot has a --copy-as active (i.e.
* identity_set_active() has run against a config with copy_as_set). The
* --fake-super owner replay consults this so a copy-as run never lets the
* recorded source owner overwrite the forced target owner. Reads the active
* snapshot, so call identity_set_active() first (the receiver does, before any
* write). */
bool identity_copy_as_active(void);
/* Receiver-side snapshot of the negotiated identity config. The server calls
* identity_set_active() once per connection (before any file write) using the
* config received over the wire; the snapshot is a deep copy so the caller may
* free its Config immediately. identity_clear_active() releases it.
*
* Returns true on success. On an allocation failure while deep-copying a
* requested usermap/groupmap it logs a LOG_LEVEL_ERROR, leaves the snapshot
* cleared (never a partial/wrong policy) and returns false; the caller must
* refuse the connection. */
bool identity_set_active(const Config* config);
void identity_clear_active(void);
/* True when any ownership-affecting identity option is present in the active
* snapshot. Ownership stays OFF ("do not apply") for every transfer that
* requests none of them, preserving FastSync's existing behavior. --super /
* --no-super alone does NOT enable ownership; an explicit identity flag
* (--numeric-ids / --chown / --usermap / --groupmap / --copy-as) is required. */
bool identity_active_enabled(void);
/* Pure, config-only predicate: true when the client requested ANY
* client-chosen ownership or super-user activity (--numeric-ids, --chown,
* --usermap/--groupmap, --copy-as, --fake-super, or an explicit --super). Used
* by the daemon module gate to decide whether a module's per-module opt-in is
* required; it never reads the per-connection snapshot. */
bool identity_ownership_requested(const Config* config);
/* Apply the negotiated ownership to an already-written file descriptor.
* source_uid/source_gid are the transmitted numeric ids. Resolution order:
* --copy-as (highest priority, forces both ids), then a matching
* usermap/groupmap rule, then --chown, then --numeric-ids (raw), then a
* best-effort name lookup on the receiver's own databases (skipped when the
* transmitted id has no name on this system). Only calls fchown() when the
* result differs from the current value.
*
* Returns false ONLY when an active --copy-as ownership application failed: its
* forced ownership is REQUIRED, so the caller must treat the entry as failed
* rather than reporting success with the wrong owner. For every other identity
* policy an fchown EPERM/EACCES is logged and ignored and true is returned
* (rsync parity: the transfer must not abort). A no-op when no identity policy
* is active returns true. */
bool identity_apply_ownership(int fd, int32_t source_uid, int32_t source_gid);
/* P7 Wave D: the no-follow (symlink) counterpart. Resolves the same
* usermap/groupmap/chown/numeric-ids/copy-as policy but applies it with
* fchownat(..., AT_SYMLINK_NOFOLLOW) so a symlink's own ownership is changed
* without ever dereferencing it. A no-op unless an identity flag is active.
* The return value follows identity_apply_ownership(): false only when an
* active --copy-as application failed. */
bool identity_apply_ownership_link(int parent_fd, const char* leaf, int32_t source_uid,
int32_t source_gid);
/* Receiver-side wire validation of the resolved identity fields. */
bool identity_wire_valid(const Config* config);
/* P7 Wave E receiver-side permission gate for super-user activities (ownership
* application and char/block device-node creation). `privilege_super_permitted`
* consults the per-connection snapshot (call identity_set_active() first);
* `privilege_super_mode_permitted` is the pure mode predicate and is what
* callers holding a Config use (the config-frame gate, file_receive). Both
* return false only for SUPER_MODE_OFF; SUPER_MODE_ON and SUPER_MODE_AUTO (the
* default) permit a confined attempt, matching FastSync's historical
* best-effort behavior where an unprivileged attempt is refused by the kernel
* and skipped. Neither EVER elevates privileges. */
bool privilege_super_permitted(void);
bool privilege_super_mode_permitted(SuperMode mode);
#endif
+13 -145
View File
@@ -1,124 +1,29 @@
#include "log.h" #include "log.h"
#include <errno.h> #include <errno.h>
#include <stdbool.h>
#include <stdarg.h> #include <stdarg.h>
#include <stdio.h> #include <stdio.h>
#include <stdlib.h>
#include <string.h> #include <string.h>
#include <threads.h>
#include <time.h> #include <time.h>
static const char* log_level_strings[] = {"DEBUG", "INFO", "WARN", "ERROR"}; static const char* log_level_strings[] = {"DEBUG", "INFO", "WARN", "ERROR"};
static LogLevel current_log_level = LOG_LEVEL_WARNING; static LogLevel current_log_level = LOG_LEVEL_WARNING;
static uint32_t current_debug_flags = 0;
static uint32_t info_flags = 0;
static bool info_flags_explicit = false;
static FILE* log_fp = NULL; static FILE* log_fp = NULL;
static _Thread_local bool eight_bit_output;
static LogStderrMode stderr_mode = LOG_STDERR_ERRORS;
/* Serializes access to log_fp and makes each emitted line atomic: the
* timestamp prefix, formatted body, and trailing newline are written as one
* critical section so concurrent threads cannot interleave partial lines.
* Initialized lazily (matching the protocol.c bw_mutex idiom) because logging
* can happen before main() installs any synchronization. */
static mtx_t log_mutex;
static once_flag log_mutex_once = ONCE_FLAG_INIT;
static void log_mutex_init(void) {
mtx_init(&log_mutex, mtx_plain);
}
void set_log_level(LogLevel level) { void set_log_level(LogLevel level) {
current_log_level = level; current_log_level = level;
} }
void set_log_debug_flags(uint32_t flags) {
current_debug_flags = flags;
}
uint32_t get_log_debug_flags(void) {
return current_debug_flags;
}
bool log_debug_enabled(LogDebugFlag flag) {
return current_log_level <= LOG_LEVEL_DEBUG && (current_debug_flags & flag) != 0;
}
void set_log_info_flags(uint32_t flags) {
info_flags = flags;
info_flags_explicit = true;
}
uint32_t get_log_info_flags(void) {
return info_flags;
}
void log_set_file(FILE* fp) { void log_set_file(FILE* fp) {
call_once(&log_mutex_once, log_mutex_init);
mtx_lock(&log_mutex);
log_fp = fp; log_fp = fp;
mtx_unlock(&log_mutex);
} }
void log_set_8_bit_output(bool enabled) { static inline void write_message(FILE* dest_io, LogLevel log_level, struct tm t, const char* format,
eight_bit_output = enabled; va_list args) {
} fprintf(dest_io, "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t.tm_year + 1900, t.tm_mon + 1,
t.tm_mday, t.tm_hour, t.tm_min, t.tm_sec, log_level_strings[log_level]);
bool log_get_8_bit_output(void) { vfprintf(dest_io, format, args);
return eight_bit_output; fprintf(dest_io, "\n");
}
void log_set_stderr_mode(LogStderrMode mode) {
stderr_mode = mode;
}
LogStderrMode log_get_stderr_mode(void) {
return stderr_mode;
}
/* Format one complete log line (timestamp prefix + body + newline) into a
* freshly allocated buffer. This is pure CPU/malloc work and must happen
* OUTSIDE the log mutex: the mutex only guards the log_fp pointer, so a
* stalled stderr/stdout pipe cannot block every logging thread. Returns NULL
* on allocation/formatting failure. */
static char* format_log_line(LogLevel log_level, const struct tm* t, const char* format,
va_list args) {
char prefix[64];
int prefix_len = snprintf(
prefix, sizeof(prefix), "%04d-%02d-%02d %02d:%02d:%02d [%s]: ", t->tm_year + 1900,
t->tm_mon + 1, t->tm_mday, t->tm_hour, t->tm_min, t->tm_sec, log_level_strings[log_level]);
if (prefix_len < 0 || prefix_len >= (int)sizeof(prefix))
return NULL;
va_list copy;
va_copy(copy, args);
int body_len = vsnprintf(NULL, 0, format, copy);
va_end(copy);
if (body_len < 0)
return NULL;
size_t total = (size_t)prefix_len + (size_t)body_len;
char* line = malloc(total + 2); /* body bytes + '\n' + NUL */
if (!line)
return NULL;
memcpy(line, prefix, (size_t)prefix_len);
vsnprintf(line + prefix_len, (size_t)body_len + 1, format, args);
line[total] = '\n';
line[total + 1] = '\0';
return line;
}
/* Write an already-formatted line to the console and, if configured, the log
* file. Only the log_fp pointer is read under the mutex (so log_set_file /
* config_delete cannot free it while it is in use); the single console fputs
* runs unlocked but is internally atomic per stdio stream. */
static void emit_log_line(FILE* console, const char* line) {
fputs(line, console);
call_once(&log_mutex_once, log_mutex_init);
mtx_lock(&log_mutex);
FILE* file = log_fp;
if (file)
fputs(line, file);
mtx_unlock(&log_mutex);
} }
void log_message(LogLevel log_level, const char* format, ...) { void log_message(LogLevel log_level, const char* format, ...) {
@@ -132,57 +37,20 @@ void log_message(LogLevel log_level, const char* format, ...) {
return; return;
FILE* dest_io = stdout; FILE* dest_io = stdout;
if (stderr_mode == LOG_STDERR_ALL || log_level == LOG_LEVEL_ERROR) { if (log_level == LOG_LEVEL_ERROR) {
dest_io = stderr; dest_io = stderr;
} }
va_list args; va_list args;
va_start(args, format); va_start(args, format);
char* line = format_log_line(log_level, &t, format, args); write_message(dest_io, log_level, t, format, args);
va_end(args); va_end(args);
if (!line)
return;
emit_log_line(dest_io, line);
free(line);
}
void log_debug_message(LogDebugFlag flag, const char* format, ...) { if (log_fp) {
if (current_log_level > LOG_LEVEL_DEBUG || !(current_debug_flags & flag)) va_start(args, format);
return; write_message(log_fp, log_level, t, format, args);
va_end(args);
time_t now = time(NULL); }
struct tm t;
if (!localtime_r(&now, &t))
return;
va_list args;
va_start(args, format);
char* line = format_log_line(LOG_LEVEL_DEBUG, &t, format, args);
va_end(args);
if (!line)
return;
emit_log_line(stdout, line);
free(line);
}
void log_info_message(LogInfoFlag flag, const char* format, ...) {
if ((info_flags_explicit && (info_flags & flag) == 0) ||
(!info_flags_explicit && current_log_level > LOG_LEVEL_DEBUG))
return;
time_t now = time(NULL);
struct tm t;
if (!localtime_r(&now, &t))
return;
va_list args;
va_start(args, format);
char* line = format_log_line(LOG_LEVEL_INFO, &t, format, args);
va_end(args);
if (!line)
return;
emit_log_line(stdout, line);
free(line);
} }
void log_perror(const char* context) { void log_perror(const char* context) {
-33
View File
@@ -2,45 +2,12 @@
#define LOG_H #define LOG_H
#include <stdio.h> #include <stdio.h>
#include <stdbool.h>
#include <stdint.h>
typedef enum { LOG_LEVEL_DEBUG, LOG_LEVEL_INFO, LOG_LEVEL_WARNING, LOG_LEVEL_ERROR } LogLevel; typedef enum { LOG_LEVEL_DEBUG, LOG_LEVEL_INFO, LOG_LEVEL_WARNING, LOG_LEVEL_ERROR } LogLevel;
typedef enum { LOG_STDERR_ERRORS, LOG_STDERR_ALL } LogStderrMode;
typedef enum {
LOG_DEBUG_IO = 1u << 0,
LOG_DEBUG_PROTO = 1u << 1,
LOG_DEBUG_PACK = 1u << 2,
LOG_DEBUG_UTIL = 1u << 3,
LOG_DEBUG_ALL = (1u << 4) - 1,
} LogDebugFlag;
typedef enum {
LOG_INFO_COPY = 1u << 0,
LOG_INFO_MISC = 1u << 1,
LOG_INFO_SKIP = 1u << 2,
LOG_INFO_STATS = 1u << 3,
LOG_INFO_ALL = LOG_INFO_COPY | LOG_INFO_MISC | LOG_INFO_SKIP | LOG_INFO_STATS,
} LogInfoFlag;
void log_message(LogLevel log_level, const char* message, ...); void log_message(LogLevel log_level, const char* message, ...);
void log_perror(const char* context); void log_perror(const char* context);
void set_log_level(LogLevel level); void set_log_level(LogLevel level);
void set_log_debug_flags(uint32_t flags);
uint32_t get_log_debug_flags(void);
/* True when a log_debug_message() call with the same flag would actually emit:
* the debug log level is enabled AND the flag is selected. Hot paths use this
* to skip expensive message formatting/escaping when the line is filtered. */
bool log_debug_enabled(LogDebugFlag flag);
void log_debug_message(LogDebugFlag flag, const char* message, ...);
void set_log_info_flags(uint32_t flags);
uint32_t get_log_info_flags(void);
void log_info_message(LogInfoFlag flag, const char* message, ...);
void log_set_file(FILE* fp); void log_set_file(FILE* fp);
void log_set_8_bit_output(bool enabled);
bool log_get_8_bit_output(void);
void log_set_stderr_mode(LogStderrMode mode);
LogStderrMode log_get_stderr_mode(void);
#endif #endif
+85 -214
View File
@@ -1,9 +1,7 @@
#include "metadata.h" #include "metadata.h"
#include "file.h" #include "file.h"
#include "identity.h"
#include "log.h" #include "log.h"
#include "protocol.h" #include "protocol.h"
#include "utils.h"
#include <errno.h> #include <errno.h>
#include <fcntl.h> #include <fcntl.h>
#include <stdint.h> #include <stdint.h>
@@ -26,29 +24,6 @@ typedef char static_assert_mode_t_fits[(sizeof(mode_t) <= sizeof(int32_t)) ? 1 :
typedef char static_assert_uid_t_fits[(sizeof(uid_t) <= sizeof(int32_t)) ? 1 : -1]; typedef char static_assert_uid_t_fits[(sizeof(uid_t) <= sizeof(int32_t)) ? 1 : -1];
typedef char static_assert_gid_t_fits[(sizeof(gid_t) <= sizeof(int32_t)) ? 1 : -1]; typedef char static_assert_gid_t_fits[(sizeof(gid_t) <= sizeof(int32_t)) ? 1 : -1];
bool metadata_mtime_matches(time_t left_sec, long left_nsec, time_t right_sec, long right_nsec,
int modify_window) {
int64_t left = (int64_t)left_sec;
int64_t right = (int64_t)right_sec;
int64_t seconds;
int64_t nanoseconds;
if (left > right || (left == right && left_nsec >= right_nsec)) {
seconds = left - right;
nanoseconds = (int64_t)left_nsec - (int64_t)right_nsec;
} else {
seconds = right - left;
nanoseconds = (int64_t)right_nsec - (int64_t)left_nsec;
}
if (nanoseconds < 0) {
seconds--;
nanoseconds += 1000000000LL;
}
if (modify_window == 0)
return left == right;
return seconds < modify_window || (seconds == modify_window && nanoseconds == 0);
}
void metadata_to_buf(char** buf, const FileMetadata* m) { void metadata_to_buf(char** buf, const FileMetadata* m) {
int32_t present = (m != NULL) ? 1 : 0; int32_t present = (m != NULL) ? 1 : 0;
memcpy(*buf, &present, sizeof(present)); memcpy(*buf, &present, sizeof(present));
@@ -70,87 +45,41 @@ void metadata_to_buf(char** buf, const FileMetadata* m) {
int64_t mtime_nsec = (int64_t)m->mtime_nsec; int64_t mtime_nsec = (int64_t)m->mtime_nsec;
memcpy(*buf, &mtime_nsec, sizeof(mtime_nsec)); memcpy(*buf, &mtime_nsec, sizeof(mtime_nsec));
*buf += sizeof(mtime_nsec); *buf += sizeof(mtime_nsec);
int32_t atime_valid = m->atime_valid ? 1 : 0;
memcpy(*buf, &atime_valid, sizeof(atime_valid));
*buf += sizeof(atime_valid);
int64_t atime_sec = (int64_t)m->atime_sec;
memcpy(*buf, &atime_sec, sizeof(atime_sec));
*buf += sizeof(atime_sec);
int64_t atime_nsec = (int64_t)m->atime_nsec;
memcpy(*buf, &atime_nsec, sizeof(atime_nsec));
*buf += sizeof(atime_nsec);
int32_t crtime_valid = m->crtime_valid ? 1 : 0;
memcpy(*buf, &crtime_valid, sizeof(crtime_valid));
*buf += sizeof(crtime_valid);
int64_t crtime_sec = (int64_t)m->crtime_sec;
memcpy(*buf, &crtime_sec, sizeof(crtime_sec));
*buf += sizeof(crtime_sec);
int64_t crtime_nsec = (int64_t)m->crtime_nsec;
memcpy(*buf, &crtime_nsec, sizeof(crtime_nsec));
*buf += sizeof(crtime_nsec);
} }
FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len) { FileMetadata* metadata_from_buf(char** buf) {
if (buf == NULL || len < sizeof(int32_t))
return NULL;
int32_t present; int32_t present;
memcpy(&present, buf, sizeof(present)); memcpy(&present, *buf, sizeof(present));
if (present != 1) *buf += sizeof(present);
if (present != 0 && present != 1)
return NULL; return NULL;
if (len < sizeof(int32_t) + FILE_METADATA_WIRE_SIZE) if (!present)
return NULL; return NULL;
const uint8_t* cursor = buf + sizeof(int32_t); FileMetadata* m = malloc(sizeof(FileMetadata));
FileMetadata* m = protocol_alloc(sizeof(FileMetadata));
if (m == NULL) if (m == NULL)
return NULL; return NULL;
int32_t mode; int32_t mode;
memcpy(&mode, cursor, sizeof(mode)); memcpy(&mode, *buf, sizeof(mode));
cursor += sizeof(mode); *buf += sizeof(mode);
m->mode = (mode_t)mode; m->mode = (mode_t)mode;
int32_t uid; int32_t uid;
memcpy(&uid, cursor, sizeof(uid)); memcpy(&uid, *buf, sizeof(uid));
cursor += sizeof(uid); *buf += sizeof(uid);
m->uid = (uid_t)uid; m->uid = (uid_t)uid;
int32_t gid; int32_t gid;
memcpy(&gid, cursor, sizeof(gid)); memcpy(&gid, *buf, sizeof(gid));
cursor += sizeof(gid); *buf += sizeof(gid);
m->gid = (gid_t)gid; m->gid = (gid_t)gid;
int64_t mtime_sec; int64_t mtime_sec;
memcpy(&mtime_sec, cursor, sizeof(mtime_sec)); memcpy(&mtime_sec, *buf, sizeof(mtime_sec));
cursor += sizeof(mtime_sec); *buf += sizeof(mtime_sec);
m->mtime_sec = (time_t)mtime_sec; m->mtime_sec = (time_t)mtime_sec;
int64_t mtime_nsec; int64_t mtime_nsec;
memcpy(&mtime_nsec, cursor, sizeof(mtime_nsec)); memcpy(&mtime_nsec, *buf, sizeof(mtime_nsec));
cursor += sizeof(mtime_nsec); *buf += sizeof(mtime_nsec);
m->mtime_nsec = (long)mtime_nsec; m->mtime_nsec = (long)mtime_nsec;
int32_t atime_valid; if (present != 1 || mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 ||
memcpy(&atime_valid, cursor, sizeof(atime_valid)); gid < 0) {
cursor += sizeof(atime_valid);
int64_t atime_sec;
memcpy(&atime_sec, cursor, sizeof(atime_sec));
cursor += sizeof(atime_sec);
int64_t atime_nsec;
memcpy(&atime_nsec, cursor, sizeof(atime_nsec));
cursor += sizeof(atime_nsec);
int32_t crtime_valid;
memcpy(&crtime_valid, cursor, sizeof(crtime_valid));
cursor += sizeof(crtime_valid);
int64_t crtime_sec;
memcpy(&crtime_sec, cursor, sizeof(crtime_sec));
cursor += sizeof(crtime_sec);
int64_t crtime_nsec;
memcpy(&crtime_nsec, cursor, sizeof(crtime_nsec));
cursor += sizeof(crtime_nsec);
m->atime_valid = atime_valid != 0;
m->atime_sec = (time_t)atime_sec;
m->atime_nsec = (long)atime_nsec;
m->crtime_valid = crtime_valid != 0;
m->crtime_sec = (time_t)crtime_sec;
m->crtime_nsec = (long)crtime_nsec;
if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0 ||
atime_valid < 0 || atime_valid > 1 || crtime_valid < 0 || crtime_valid > 1 ||
(atime_valid && (atime_nsec < 0 || atime_nsec >= 1000000000LL)) ||
(crtime_valid && (crtime_nsec < 0 || crtime_nsec >= 1000000000LL))) {
free(m); free(m);
return NULL; return NULL;
} }
@@ -159,17 +88,21 @@ FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len) {
bool metadata_send(int file_descriptor, const FileMetadata* m) { bool metadata_send(int file_descriptor, const FileMetadata* m) {
if (m == NULL) { if (m == NULL) {
int32_t absent = 0; int32_t zero = 0;
return send_n_data(file_descriptor, &absent, sizeof(absent)); return send_n_data(file_descriptor, &zero, sizeof(zero));
} }
/* One packed frame (protocol 2.20.0): the int32 present flag followed by the int32_t present = 1;
fixed FILE_METADATA_WIRE_SIZE-byte field record. metadata_to_buf() emits int32_t mode = (int32_t)m->mode;
exactly that layout (present + fields), so build it once and write the int32_t uid = (int32_t)m->uid;
whole record in a single call instead of one frame per field. */ int32_t gid = (int32_t)m->gid;
char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE]; int64_t mtime_sec = (int64_t)m->mtime_sec;
char* cursor = packed; int64_t mtime_nsec = (int64_t)m->mtime_nsec;
metadata_to_buf(&cursor, m); return send_n_data(file_descriptor, &present, sizeof(present)) &&
return send_n_data(file_descriptor, packed, sizeof(packed)); send_n_data(file_descriptor, &mode, sizeof(mode)) &&
send_n_data(file_descriptor, &uid, sizeof(uid)) &&
send_n_data(file_descriptor, &gid, sizeof(gid)) &&
send_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec)) &&
send_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec));
} }
FileMetadata* metadata_receive(int file_descriptor, int* ok) { FileMetadata* metadata_receive(int file_descriptor, int* ok) {
@@ -189,17 +122,54 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) {
*ok = 0; *ok = 0;
return NULL; return NULL;
} }
/* Rebuild the packed record metadata_from_buf() expects: the present flag we FileMetadata* m = malloc(sizeof(FileMetadata));
just read, followed by exactly FILE_METADATA_WIRE_SIZE field bytes. */ if (m == NULL) {
char packed[sizeof(int32_t) + FILE_METADATA_WIRE_SIZE];
memcpy(packed, &present, sizeof(present));
if (!receive_n_data(file_descriptor, packed + sizeof(present), FILE_METADATA_WIRE_SIZE)) {
if (ok) if (ok)
*ok = 0; *ok = 0;
return NULL; return NULL;
} }
FileMetadata* m = metadata_from_buf((const uint8_t*)packed, sizeof(packed)); int32_t mode;
if (m == NULL) { if (!receive_n_data(file_descriptor, &mode, sizeof(mode))) {
free(m);
if (ok)
*ok = 0;
return NULL;
}
m->mode = (mode_t)mode;
int32_t uid;
if (!receive_n_data(file_descriptor, &uid, sizeof(uid))) {
free(m);
if (ok)
*ok = 0;
return NULL;
}
m->uid = (uid_t)uid;
int32_t gid;
if (!receive_n_data(file_descriptor, &gid, sizeof(gid))) {
free(m);
if (ok)
*ok = 0;
return NULL;
}
m->gid = (gid_t)gid;
int64_t mtime_sec;
if (!receive_n_data(file_descriptor, &mtime_sec, sizeof(mtime_sec))) {
free(m);
if (ok)
*ok = 0;
return NULL;
}
m->mtime_sec = (time_t)mtime_sec;
int64_t mtime_nsec;
if (!receive_n_data(file_descriptor, &mtime_nsec, sizeof(mtime_nsec))) {
free(m);
if (ok)
*ok = 0;
return NULL;
}
m->mtime_nsec = (long)mtime_nsec;
if (mtime_nsec < 0 || mtime_nsec >= 1000000000LL || mode < 0 || uid < 0 || gid < 0) {
free(m);
if (ok) if (ok)
*ok = 0; *ok = 0;
return NULL; return NULL;
@@ -209,27 +179,12 @@ FileMetadata* metadata_receive(int file_descriptor, int* ok) {
return m; return m;
} }
static mode_t metadata_mode(const FileMetadata* metadata, mode_t current_mode, void file_restore_metadata(const char* path, const FileMetadata* metadata) {
bool preserve_executability) {
const mode_t execute_bits = S_IXUSR | S_IXGRP | S_IXOTH;
if (preserve_executability)
return (current_mode & 0777 & ~execute_bits) | (metadata->mode & execute_bits);
return metadata->mode & 0777 & ~(S_IWGRP | S_IWOTH);
}
void file_restore_metadata(const char* path, const FileMetadata* metadata,
bool preserve_executability) {
if (metadata == NULL) if (metadata == NULL)
return; return;
struct stat current; mode_t safe_mode = metadata->mode & 0777 & ~(S_IWGRP | S_IWOTH);
mode_t current_mode = stat(path, &current) == 0 ? current.st_mode : 0; if (chmod(path, safe_mode) != 0)
mode_t safe_mode = metadata_mode(metadata, current_mode, preserve_executability); log_message(LOG_LEVEL_WARNING, "Failed to chmod %s: %s", path, strerror(errno));
if (chmod(path, safe_mode) != 0) {
char* escaped_path = output_escape(path, log_get_8_bit_output());
log_message(LOG_LEVEL_WARNING, "Failed to chmod %s: %s",
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
free(escaped_path);
}
/* Never apply client-supplied ownership. The descriptor API below is the /* Never apply client-supplied ownership. The descriptor API below is the
receiver write path; retain this legacy API only for compatibility. */ receiver write path; retain this legacy API only for compatibility. */
struct timespec times[2]; struct timespec times[2];
@@ -237,104 +192,20 @@ void file_restore_metadata(const char* path, const FileMetadata* metadata,
times[0].tv_nsec = UTIME_OMIT; times[0].tv_nsec = UTIME_OMIT;
times[1].tv_sec = metadata->mtime_sec; times[1].tv_sec = metadata->mtime_sec;
times[1].tv_nsec = metadata->mtime_nsec; times[1].tv_nsec = metadata->mtime_nsec;
if (metadata->atime_valid) { if (utimensat(AT_FDCWD, path, times, 0) != 0)
times[0].tv_sec = metadata->atime_sec; log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s", path, strerror(errno));
times[0].tv_nsec = metadata->atime_nsec;
}
if (metadata->crtime_valid) {
log_message(LOG_LEVEL_DEBUG,
"crtime (birth time) %lld.%09ld transmitted for %s but not applied: no portable "
"setter exists",
(long long)metadata->crtime_sec, metadata->crtime_nsec, path);
}
if (utimensat(AT_FDCWD, path, times, 0) != 0) {
char* escaped_path = output_escape(path, log_get_8_bit_output());
log_message(LOG_LEVEL_WARNING, "Failed to set timestamps on %s: %s",
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
free(escaped_path);
}
} }
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata, bool file_restore_metadata_fd(int fd, const FileMetadata* metadata) {
bool omit_link_times) {
if (path == NULL || metadata == NULL)
return !identity_copy_as_active();
char* leaf = NULL;
int parent_fd = file_open_secure_parent(path, &leaf, false);
if (parent_fd < 0)
return !identity_copy_as_active();
/* Ownership (only when the identity policy is active) via lchown semantics:
fchownat with AT_SYMLINK_NOFOLLOW never dereferences the link. A failed
REQUIRED --copy-as ownership marks the entry failed; every other policy is
best-effort. */
bool owned = identity_apply_ownership_link(parent_fd, leaf, (int32_t)metadata->uid,
(int32_t)metadata->gid);
/* Symlink mode: not settable on Linux (fchmodat AT_SYMLINK_NOFOLLOW returns
EOPNOTSUPP/ENOTSUP); attempt it for platforms that support it and quietly
ignore the unsupported case so the transfer never fails over it. */
mode_t link_mode = metadata->mode & 0777;
if (fchmodat(parent_fd, leaf, link_mode, AT_SYMLINK_NOFOLLOW) != 0 && errno != EOPNOTSUPP &&
errno != ENOTSUP && errno != ENOSYS) {
log_message(LOG_LEVEL_DEBUG, "Could not set symlink mode on %s: %s", path, strerror(errno));
}
if (!omit_link_times) {
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
if (metadata->atime_valid) {
times[0].tv_sec = metadata->atime_sec;
times[0].tv_nsec = metadata->atime_nsec;
}
if (utimensat(parent_fd, leaf, times, AT_SYMLINK_NOFOLLOW) != 0) {
char* escaped_path = output_escape(path, log_get_8_bit_output());
log_message(LOG_LEVEL_WARNING, "Failed to set symlink timestamps on %s: %s",
escaped_path ? escaped_path : "<allocation failed>", strerror(errno));
free(escaped_path);
}
}
close(parent_fd);
free(leaf);
return owned;
}
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability) {
if (fd < 0 || metadata == NULL) if (fd < 0 || metadata == NULL)
return metadata == NULL; return metadata == NULL;
bool ok = true; bool ok = true;
struct stat current; mode_t safe_mode = metadata->mode & 0777 & ~(S_IWGRP | S_IWOTH);
if (fstat(fd, &current) != 0)
return false;
mode_t safe_mode = metadata_mode(metadata, current.st_mode, preserve_executability);
if (fchmod(fd, safe_mode) != 0) if (fchmod(fd, safe_mode) != 0)
ok = false; ok = false;
/* Client uid/gid values are deliberately not authoritative UNLESS the client /* Client uid/gid values are deliberately not authoritative. */
explicitly opted in with an identity flag (--numeric-ids / --usermap /
--groupmap / --chown). identity_apply_ownership is the controlled,
privilege-gated path: it consults the negotiated policy, resolves the
target ids, and applies them via an fd-relative fchown() that is confined
to the just-written file (EPERM/EACCES are logged, never fatal) -- EXCEPT
for an active --copy-as, whose forced ownership is REQUIRED: a failure
marks this entry as failed instead of reporting a wrong-owner write as
success. With no identity flag set it is a no-op, so a default or plain -M
transfer keeps FastSync's existing behavior of never applying client
ownership. */
if (!identity_apply_ownership(fd, (int32_t)metadata->uid, (int32_t)metadata->gid))
ok = false;
struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT}, struct timespec times[2] = {{.tv_sec = 0, .tv_nsec = UTIME_OMIT},
{.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}}; {.tv_sec = metadata->mtime_sec, .tv_nsec = metadata->mtime_nsec}};
if (metadata->atime_valid) {
times[0].tv_sec = metadata->atime_sec;
times[0].tv_nsec = metadata->atime_nsec;
}
/* --crtimes captures and transmits the source birth time, but there is no
* portable way to set a birth time (utimensat can only set atime/mtime), so
* the receiver deliberately does NOT apply it. This is explicit, honest
* unsupported-attribute handling: log a debug note and continue — never fail
* the transfer and never pretend the crtime was applied. */
if (metadata->crtime_valid) {
log_message(LOG_LEVEL_DEBUG,
"crtime (birth time) %lld.%09ld transmitted but not applied: no portable setter",
(long long)metadata->crtime_sec, metadata->crtime_nsec);
}
if (futimens(fd, times) != 0) if (futimens(fd, times) != 0)
ok = false; ok = false;
return ok; return ok;
+5 -40
View File
@@ -3,10 +3,8 @@
#include "file.h" #include "file.h"
#include <stdbool.h> #include <stdbool.h>
#include <stddef.h>
#include <stdint.h> #include <stdint.h>
#include <sys/stat.h> #include <sys/stat.h>
#include <time.h>
/* /*
* Wire format (introduced in protocol version 2.0.0): * Wire format (introduced in protocol version 2.0.0):
@@ -16,12 +14,6 @@
* int32_t gid (was gid_t, platform-dependent) * int32_t gid (was gid_t, platform-dependent)
* int64_t mtime_sec (was time_t, platform-dependent) * int64_t mtime_sec (was time_t, platform-dependent)
* int64_t mtime_nsec (was long, platform-dependent) * int64_t mtime_nsec (was long, platform-dependent)
* int32_t atime_valid (-U/--atimes; protocol 2.12.0)
* int64_t atime_sec
* int64_t atime_nsec
* int32_t crtime_valid (-N/--crtimes; protocol 2.12.0)
* int64_t crtime_sec
* int64_t crtime_nsec
* *
* Prior to 2.0.0 the wire format used the raw platform-dependent types, * Prior to 2.0.0 the wire format used the raw platform-dependent types,
* which broke compatiblity across different systems. All fields are now * which broke compatiblity across different systems. All fields are now
@@ -30,41 +22,14 @@
/* Size of metadata fields on wire, excluding the int32_t `present` field that /* Size of metadata fields on wire, excluding the int32_t `present` field that
* is always sent first. The total wire size for present metadata is * is always sent first. The total wire size for present metadata is
* sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (68 bytes on most platforms). * sizeof(int32_t) + FILE_METADATA_WIRE_SIZE (32 bytes on most platforms). */
* #define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 3 + sizeof(int64_t) * 2)
* metadata_send()/metadata_receive() (protocol 2.20.0) frame the metadata as a
* single packed record: one int32 present flag (0 = absent) followed, when
* present, by exactly FILE_METADATA_WIRE_SIZE bytes of field data. This is the
* same present+fields byte layout metadata_to_buf()/metadata_from_buf() use, so
* the wire metadata is now one frame instead of one frame per field. */
#define FILE_METADATA_WIRE_SIZE (sizeof(int32_t) * 5 + sizeof(int64_t) * 6)
void metadata_to_buf(char** buf, const FileMetadata* m); void metadata_to_buf(char** buf, const FileMetadata* m);
/* Decode one packed metadata record (an int32 present flag followed, when FileMetadata* metadata_from_buf(char** buf);
* present, by FILE_METADATA_WIRE_SIZE field bytes) from `buf`, which has `len`
* readable bytes. Every read is bounds-checked against `len`, so the function
* can never over-read the caller's buffer: a too-short record, an absent
* (present == 0) record and a malformed record all return NULL. A successful
* decode returns a heap-allocated FileMetadata owned by the caller. */
FileMetadata* metadata_from_buf(const uint8_t* buf, size_t len);
bool metadata_send(int file_descriptor, const FileMetadata* m); bool metadata_send(int file_descriptor, const FileMetadata* m);
FileMetadata* metadata_receive(int file_descriptor, int* ok); FileMetadata* metadata_receive(int file_descriptor, int* ok);
void file_restore_metadata(const char* path, const FileMetadata* metadata, void file_restore_metadata(const char* path, const FileMetadata* metadata);
bool preserve_executability); bool file_restore_metadata_fd(int fd, const FileMetadata* metadata);
bool file_restore_metadata_fd(int fd, const FileMetadata* metadata, bool preserve_executability);
/* P7 Wave D: apply a SYMLINK's own metadata using no-follow primitives only
* (utimensat/lchown/fchmodat with AT_SYMLINK_NOFOLLOW), confined fd-relative
* under the authorized root. `omit_link_times` (-J/--omit-link-times)
* suppresses the timestamps; the link's mode/ownership are still attempted
* (ownership stays gated by the identity policy and by default is not applied).
* A null metadata or an unfollowable parent is a harmless no-op. Returns false
* only when a REQUIRED --copy-as ownership application failed, so the caller can
* report the entry as failed instead of claiming a wrong-owner success. */
bool file_restore_symlink_metadata(const char* path, const FileMetadata* metadata,
bool omit_link_times);
/* Compare timestamps using rsync's whole-second modification window. */
bool metadata_mtime_matches(time_t left_sec, long left_nsec, time_t right_sec, long right_nsec,
int modify_window);
#endif #endif
-76
View File
@@ -1,76 +0,0 @@
#include "motd.h"
#include "protocol.h"
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
char* motd_read_file(const char* path) {
if (!path || path[0] == '\0')
return NULL;
FILE* fp = fopen(path, "rb");
if (!fp)
return NULL;
char* buffer = malloc(MOTD_MAX_BYTES + 1);
if (!buffer) {
fclose(fp);
return NULL;
}
/* fread stops at the bound; a larger file is truncated rather than read
* unbounded. ferror distinguishes a truncated read from an I/O failure. */
size_t total = fread(buffer, 1, MOTD_MAX_BYTES, fp);
if (ferror(fp)) {
free(buffer);
fclose(fp);
return NULL;
}
fclose(fp);
buffer[total] = '\0';
return buffer;
}
char* motd_render(const char* motd, bool eight_bit_output) {
if (!motd)
return NULL;
size_t length = strlen(motd);
if (length > (SIZE_MAX - 1) / 5)
return NULL;
char* rendered = malloc(length * 5 + 1);
if (!rendered)
return NULL;
size_t out = 0;
for (size_t i = 0; i < length; i++) {
unsigned char byte = (unsigned char)motd[i];
if (byte == '\n' || byte == '\t') {
rendered[out++] = (char)byte;
} else if ((byte >= 32 && byte <= 126) || (eight_bit_output && byte >= 128)) {
rendered[out++] = (char)byte;
} else {
rendered[out++] = '\\';
rendered[out++] = '#';
rendered[out++] = (char)('0' + ((byte >> 6) & 7));
rendered[out++] = (char)('0' + ((byte >> 3) & 7));
rendered[out++] = (char)('0' + (byte & 7));
}
}
rendered[out] = '\0';
return rendered;
}
bool motd_send(int file_descriptor, const char* motd) {
return send_str(file_descriptor, motd ? motd : "");
}
char* motd_receive(int file_descriptor) {
char* motd = receive_str(file_descriptor);
if (!motd)
return NULL;
/* Guard against a hostile/oversized peer: receive_str already bounded the
* frame at MAX_STRING_SIZE and consumed it, so discarding an over-bound
* body here keeps the stream framed while refusing to display it. */
if (strlen(motd) > MOTD_MAX_BYTES) {
free(motd);
return NULL;
}
return motd;
}
-54
View File
@@ -1,54 +0,0 @@
#ifndef MOTD_H
#define MOTD_H
#include <stdbool.h>
/* Daemon Message-Of-The-Day (Wave C).
*
* The daemon listener (fastsync-server --daemon) may advertise a `motd file`
* configured in its globals. When a client connects with a host::module/path
* destination and the module gate accepts the connection, the server sends the
* MOTD as a single string frame BEFORE any transfer data (rsync sends its MOTD
* as the first thing from the server at the start of a daemon connection).
* The client reads that frame right after the config/status handshake and
* displays it on stdout unless --no-motd was given.
*
* The MOTD is ordinary display text, never a secret, so it uses the normal
* (non-redacted) string primitive. The exchange is strictly server->client
* and happens on the daemon listener path only; the --stdio SSH path has no
* MOTD.
*
* No PROTOCOL_VERSION bump is involved: the frame is sent and read
* symmetrically by every 2.15.0 daemon build (the strict same-version
* handshake rejects any other version before the frame), so it cannot
* desynchronize a peer. */
/* Upper bound on the MOTD bytes the server will read from disk and put on the
* wire. Kept far below MAX_STRING_SIZE (64 KB) so a huge/hostile motd file
* can never produce an unbounded frame or allocation. */
#define MOTD_MAX_BYTES 4096
/* Read a daemon MOTD file, bounded to MOTD_MAX_BYTES. Returns a malloc'd
* NUL-terminated copy of the file content (bytes beyond the bound are
* truncated) or NULL when path is NULL/empty, the file cannot be opened or
* read, or allocation fails. An absent or unreadable motd file is NOT an
* error: the caller simply sends an empty MOTD frame and continues. */
char* motd_read_file(const char* path);
/* Render MOTD text for terminal display. Newlines and tabs are preserved so
* a multi-line motd still reads naturally, while every other non-printable /
* control byte (ESC included) is escaped with FastSync's `\NNN` octal
* convention, so a hostile server cannot inject terminal escape sequences
* through the MOTD. eight_bit_output keeps bytes >= 0x80 verbatim (matching
* --8-bit-output). Returns a malloc'd string or NULL on allocation failure. */
char* motd_render(const char* motd, bool eight_bit_output);
/* Send/receive the MOTD string frame. These wrap the normal string
* primitive: the MOTD is not a credential, so no redaction is used. The
* receiver additionally rejects an over-bound frame (> MOTD_MAX_BYTES) as a
* hostile input guard; the frame itself is always fully consumed first, so the
* stream stays framed. */
bool motd_send(int file_descriptor, const char* motd);
char* motd_receive(int file_descriptor);
#endif

Some files were not shown because too many files have changed in this diff Show More